Move the voice pipelines into core

Capture processing, the jitter buffer and decoding were desktop classes,
so any other platform would have had to copy them. They now live in core
and reach the platform only through two interfaces:

- AudioIo lists devices and opens capture/playback lines; the desktop's
  AudioDevices implements it over PipeWire and Java Sound.
- OpusCodec creates encoders and decoders; the desktop binds libopus
  through the FFM API as before.

DesktopVoiceInput becomes CaptureVoiceInput unchanged in behaviour.
DesktopVoiceOutput splits into VoiceStream, one speaker's jitter buffer
and decoder, paced by whoever pulls it, and StreamingVoiceOutput around
it. Speakers reach the device either on a line each (the desktop, so
each shows up in the PipeWire mixer) or mixed onto one shared line, as
mobile audio APIs want.

The settings dialog's device lists and microphone test now go through
the AudioBackend instead of desktop classes.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-25 08:01:28 +00:00
parent 3c389de36b
commit 7396c4ca40
36 changed files with 1290 additions and 531 deletions

View File

@@ -1,26 +0,0 @@
package com.ts3client.audio.desktop;
/**
* An open microphone line, delivering 16-bit little-endian PCM at
* {@link AudioDevices#SAMPLE_RATE}.
*
* <p>Modelled on {@link javax.sound.sampled.TargetDataLine} so the capture pipeline does
* not care whether the audio comes from PipeWire or Java Sound.
*/
public interface AudioCapture extends AutoCloseable {
/** Channel count actually negotiated with the device. */
int channels();
/** Begins capturing; audio read before this call is not delivered. */
void start();
/**
* Blocks until {@code length} bytes are captured. Returns fewer bytes only when the
* line is closing or the calling thread was interrupted.
*/
int read(byte[] buffer, int offset, int length);
@Override
void close();
}

View File

@@ -1,5 +1,10 @@
package com.ts3client.audio.desktop;
import com.ts3client.audio.AudioCapture;
import com.ts3client.audio.AudioDevice;
import com.ts3client.audio.AudioIo;
import com.ts3client.audio.AudioPlayback;
import com.ts3client.audio.VoiceFormat;
import com.ts3client.audio.desktop.pipewire.PipeWire;
import javax.sound.sampled.AudioFormat;
@@ -36,11 +41,9 @@ import java.util.List;
* {@link #openCapture} / {@link #openPlayback}, which fall back through the supported
* channel counts.
*/
public final class AudioDevices {
public final class AudioDevices implements AudioIo {
public static final int SAMPLE_RATE = 48_000;
public static final int FRAME_SIZE = 960; // 20 ms @ 48 kHz
public static final int MAX_CHANNELS = 2;
public static final AudioDevices INSTANCE = new AudioDevices();
/** Buffer size, in 20 ms frames, requested when opening a Java Sound line. */
private static final int BUFFER_FRAMES = 8;
@@ -48,42 +51,31 @@ public final class AudioDevices {
/** Marks a device id as a PipeWire node name rather than a Java Sound mixer name. */
private static final String PIPEWIRE_PREFIX = "pw:";
/** A selectable audio device. {@code id} is what gets persisted in the settings. */
public record Device(String id, String label) {
/** The implicit entry that lets the platform (PipeWire, Pulse, ALSA) choose. */
public static final Device DEFAULT = new Device("", "(System default)");
@Override
public String toString() {
return label;
}
}
private AudioDevices() {
}
/** 48 kHz signed 16-bit little-endian PCM with the given channel count. */
public static AudioFormat format(int channels) {
return new AudioFormat(SAMPLE_RATE, 16, channels, true, false);
return new AudioFormat(VoiceFormat.SAMPLE_RATE, 16, channels, true, false);
}
/** Devices that can provide microphone (capture) lines, system default first. */
public static List<Device> inputDevices() {
@Override
public List<AudioDevice> inputDevices() {
return devices(TargetDataLine.class, false);
}
/** Devices that can provide speaker (playback) lines, system default first. */
public static List<Device> outputDevices() {
@Override
public List<AudioDevice> outputDevices() {
return devices(SourceDataLine.class, true);
}
private static List<Device> devices(Class<?> lineClass, boolean sinks) {
List<Device> devices = new ArrayList<>();
devices.add(Device.DEFAULT);
private static List<AudioDevice> devices(Class<?> lineClass, boolean sinks) {
List<AudioDevice> devices = new ArrayList<>();
devices.add(AudioDevice.DEFAULT);
for (PipeWire.Node node : PipeWire.nodes()) {
if (node.sink() == sinks) {
devices.add(new Device(PIPEWIRE_PREFIX + node.name(), node.description()));
devices.add(new AudioDevice(PIPEWIRE_PREFIX + node.name(), node.description()));
}
}
@@ -95,7 +87,7 @@ public final class AudioDevices {
if (isDefaultMixer(name)) continue;
if (!AudioSystem.getMixer(mi).isLineSupported(anyFormat)) continue;
if (devices.stream().anyMatch(d -> d.id().equals(name))) continue;
devices.add(new Device(name, "ALSA: " + name));
devices.add(new AudioDevice(name, "ALSA: " + name));
}
return devices;
}
@@ -109,17 +101,13 @@ public final class AudioDevices {
return mixerName.endsWith("[default]");
}
/**
* Opens a capture line, preferring {@code preferredChannels} and falling back to the
* other channel count if the device won't take it. Inspect {@link AudioCapture#channels()}
* for what was actually opened.
*/
public static AudioCapture openCapture(String deviceId, int preferredChannels)
@Override
public AudioCapture openCapture(String deviceId, int preferredChannels)
throws LineUnavailableException {
if (usePipeWire(deviceId)) {
try {
return PipeWire.openCapture("TS3J Microphone", pipeWireNode(deviceId),
SAMPLE_RATE, clampChannels(preferredChannels));
VoiceFormat.SAMPLE_RATE, clampChannels(preferredChannels));
} catch (Exception e) {
if (isPipeWireDevice(deviceId)) throw unavailable(true, deviceId, e);
// The default device: Java Sound may still reach it through ALSA.
@@ -128,13 +116,13 @@ public final class AudioDevices {
return new JavaSoundCapture(open(TargetDataLine.class, deviceId, preferredChannels));
}
/** Opens a playback line; see {@link #openCapture} for the channel negotiation. */
public static AudioPlayback openPlayback(String deviceId, int preferredChannels)
@Override
public AudioPlayback openPlayback(String deviceId, int preferredChannels)
throws LineUnavailableException {
if (usePipeWire(deviceId)) {
try {
return PipeWire.openPlayback("TS3J Playback", pipeWireNode(deviceId),
SAMPLE_RATE, clampChannels(preferredChannels));
VoiceFormat.SAMPLE_RATE, clampChannels(preferredChannels));
} catch (Exception e) {
if (isPipeWireDevice(deviceId)) throw unavailable(false, deviceId, e);
}
@@ -167,7 +155,7 @@ public final class AudioDevices {
/** PipeWire converts freely, so any channel count works; keep it in the range we handle. */
private static int clampChannels(int channels) {
return Math.max(1, Math.min(MAX_CHANNELS, channels));
return Math.max(1, Math.min(VoiceFormat.MAX_CHANNELS, channels));
}
private static <T extends DataLine> T open(Class<T> lineClass, String deviceId, int preferredChannels)
@@ -180,7 +168,7 @@ public final class AudioDevices {
try {
T line = lineClass.cast(
(mixer != null) ? mixer.getLine(info) : AudioSystem.getLine(info));
openLine(line, fmt, FRAME_SIZE * 2 * fmt.getChannels() * BUFFER_FRAMES);
openLine(line, fmt, VoiceFormat.FRAME_SIZE * 2 * fmt.getChannels() * BUFFER_FRAMES);
return line;
} catch (Exception e) {
failure = e;

View File

@@ -1,26 +0,0 @@
package com.ts3client.audio.desktop;
/**
* An open speaker line, accepting 16-bit little-endian PCM at
* {@link AudioDevices#SAMPLE_RATE}.
*
* <p>Modelled on {@link javax.sound.sampled.SourceDataLine} so the playback pipeline does
* not care whether the audio goes to PipeWire or Java Sound.
*/
public interface AudioPlayback extends AutoCloseable {
/** Channel count actually negotiated with the device. */
int channels();
/** Begins playing whatever is written from now on. */
void start();
/** Blocks until the audio has been queued for playback. */
void write(byte[] buffer, int offset, int length);
/** Blocks until everything already written has been played out. */
void drain();
@Override
void close();
}

View File

@@ -1,9 +1,10 @@
package com.ts3client.audio.desktop;
import com.ts3client.audio.AudioBackend;
import com.ts3client.audio.VoiceInput;
import com.ts3client.audio.VoiceOutput;
import com.ts3client.audio.AudioIo;
import com.ts3client.audio.StreamingVoiceOutput;
import com.ts3client.audio.desktop.pipewire.PipeWire;
import com.ts3client.audio.opus.OpusCodec;
import com.ts3client.config.Settings;
import com.ts3client.sound.SoundPlayer;
@@ -13,19 +14,27 @@ import com.ts3client.sound.SoundPlayer;
*/
public final class DesktopAudioBackend implements AudioBackend {
private final OpusCodec opus = new NativeOpusCodec();
@Override
public VoiceInput createInput(Settings settings) {
return new DesktopVoiceInput(settings);
public AudioIo io() {
return AudioDevices.INSTANCE;
}
@Override
public VoiceOutput createOutput(Settings settings) {
return new DesktopVoiceOutput(settings.outputDevice);
public OpusCodec opus() {
return opus;
}
/** Every speaker gets a line of their own, so each shows up in the desktop's mixer. */
@Override
public StreamingVoiceOutput.Lines outputLines() {
return StreamingVoiceOutput.Lines.PER_SPEAKER;
}
@Override
public SoundPlayer createSoundPlayer(Settings settings) {
return new WavSoundPlayer(settings.outputDevice);
return new WavSoundPlayer(io(), settings.outputDevice);
}
@Override
@@ -33,9 +42,9 @@ public final class DesktopAudioBackend implements AudioBackend {
return codec() + ", " + audioSystem();
}
private static String codec() {
private String codec() {
try {
return "Opus " + Opus.getVersionString();
return "Opus " + opus.version();
} catch (Throwable t) {
return "Opus (native library unavailable)";
}

View File

@@ -1,529 +0,0 @@
package com.ts3client.audio.desktop;
import com.github.manevolent.ts3j.enums.CodecType;
import com.ts3client.audio.AudioFrameListener;
import com.ts3client.audio.InputLevel;
import com.ts3client.audio.OpusParameters;
import com.ts3client.audio.SpeechProbabilityDetector;
import com.ts3client.audio.VoiceInput;
import com.ts3client.audio.processing.AudioProcessor;
import com.ts3client.audio.processing.ns.SuppressionLevel;
import com.ts3client.audio.vad.RnnSpeechDetector;
import com.ts3client.config.Settings;
import java.util.concurrent.ConcurrentLinkedQueue;
import java.util.concurrent.atomic.AtomicBoolean;
import java.util.function.Consumer;
/**
* Desktop {@link VoiceInput}: captures the microphone, applies voice-activation
* or push-to-talk gating, and Opus-encodes 20&nbsp;ms frames.
*
* <p>The capture line comes from {@link AudioDevices}, which picks PipeWire or Java
* Sound; it is opened in stereo when the device offers it. Voice
* ({@code OPUS_VOICE}) is transmitted mono, from the downmix, so the pre-processing
* chain and VAD see a single channel; the music codec ({@code OPUS_MUSIC}) transmits
* the stereo capture as-is.
*
* <p>A dedicated capture thread fills a small packet queue while ts3j polls
* {@link #isReady()} / {@link #provide()} on its own timer, decoupling capture from
* network sending. When the gate closes and the queue drains {@link #isReady()}
* returns {@code false}, prompting ts3j to emit the terminating empty voice packet.
*/
public final class DesktopVoiceInput implements VoiceInput {
/**
* How long the gate stays open after the last active frame. TS3 counts 64 of its
* 10&nbsp;ms preprocessor frames before closing, so the tail of a word is never clipped.
*/
private static final int HANGOVER_MS = 640;
/**
* Audio replayed when the gate opens, so a word's onset is not swallowed by the frame
* that detected it. TS3 keeps up to three 10&nbsp;ms buffers ({@code vad_extrabuffersize}
* defaults to 2, plus one).
*/
private static final int PREROLL_MS = 30;
private static final int HANGOVER_FRAMES =
Math.max(1, HANGOVER_MS * AudioDevices.SAMPLE_RATE / 1000 / AudioDevices.FRAME_SIZE);
private static final int PREROLL_FRAMES =
Math.max(1, PREROLL_MS * AudioDevices.SAMPLE_RATE / 1000 / AudioDevices.FRAME_SIZE);
private final ConcurrentLinkedQueue<byte[]> queue = new ConcurrentLinkedQueue<>();
private final AtomicBoolean muted = new AtomicBoolean(false);
private final AtomicBoolean localMuted = new AtomicBoolean(false);
private final AtomicBoolean transmitting = new AtomicBoolean(false);
private final AtomicBoolean pttDown = new AtomicBoolean(false);
private final AtomicBoolean running = new AtomicBoolean(false);
private volatile Settings.InputMode mode;
private volatile Settings.VadMode vadMode;
private volatile double thresholdDb;
private volatile double speechThreshold;
private volatile boolean vadOverPtt;
private volatile double inputGain;
private volatile CodecType codec = CodecType.OPUS_VOICE;
private final SpeechProbabilityDetector speechDetector =
new RnnSpeechDetector(AudioDevices.SAMPLE_RATE);
private final AudioProcessor processor = new AudioProcessor();
private volatile Consumer<Double> levelListener; // input level, InputLevel scale
private volatile Consumer<Boolean> talkListener; // local talk-state changes
private volatile Runnable mutedTalkListener; // speech detected while muted
private volatile AudioFrameListener monitorListener; // local monitoring of what is sent
private boolean mutedTalking;
private final String deviceName;
private final Object encoderLock = new Object();
private volatile OpusParameters params;
private OpusParameters appliedParams;
private int encoderApplication = -1;
private volatile int encoderChannels = 1;
private volatile int captureChannels = 1;
private Thread captureThread;
private AudioCapture line;
private OpusEncoder encoder;
private int hangover;
private boolean lastTransmitting;
// Ring of recent frames captured while the gate was shut.
private final float[][] preroll = new float[PREROLL_FRAMES][];
private int prerollTail;
private int prerollCount;
public DesktopVoiceInput(Settings settings) {
this.deviceName = settings.inputDevice;
this.mode = settings.inputMode;
this.vadMode = settings.vadMode;
this.thresholdDb = settings.vadThresholdDb;
this.speechThreshold = settings.speechThreshold;
this.vadOverPtt = settings.vadOverPtt;
this.inputGain = settings.inputVolume;
this.params = OpusParameters.from(settings);
this.codec = codecFor(params);
processor.setNoiseSuppression(settings.denoise);
processor.setSuppressionLevel(SuppressionLevel.fromDenoiserLevel(settings.denoiserLevel));
processor.setTransientSuppression(settings.typingAttenuation);
processor.setGainControl(settings.agc);
}
// ---- live configuration (safe to call from the UI thread) ----
public void setMode(Settings.InputMode mode) {
this.mode = mode;
}
public void setVadMode(Settings.VadMode mode) {
this.vadMode = mode;
speechDetector.reset();
}
public void setThresholdDb(double db) {
this.thresholdDb = db;
}
public void setSpeechThreshold(double threshold) {
this.speechThreshold = threshold;
}
public void setVadOverPtt(boolean enabled) {
this.vadOverPtt = enabled;
}
public void setInputGain(double gain) {
this.inputGain = gain;
}
public void setNoiseSuppression(boolean enabled) {
processor.setNoiseSuppression(enabled);
}
public void setDenoiserLevel(int level) {
processor.setSuppressionLevel(SuppressionLevel.fromDenoiserLevel(level));
}
public void setTypingAttenuation(boolean enabled) {
processor.setTransientSuppression(enabled);
}
public void setAgc(boolean enabled) {
processor.setGainControl(enabled);
}
@Override
public void keyPressed() {
processor.keyPressed();
}
/**
* Stages new encoder settings; the capture thread picks them up on the next frame
* (or {@link #start()} applies them directly when idle).
*/
public void setOpusParameters(OpusParameters p) {
this.params = p;
synchronized (encoderLock) {
// While capturing, the codec flag flips together with the encoder swap so
// no packet is ever tagged with a codec it wasn't encoded for.
if (encoder == null) this.codec = codecFor(p);
}
}
private static int applicationFor(OpusParameters p) {
return p.music ? Opus.OPUS_APPLICATION_AUDIO : Opus.OPUS_APPLICATION_VOIP;
}
private static CodecType codecFor(OpusParameters p) {
return p.music ? CodecType.OPUS_MUSIC : CodecType.OPUS_VOICE;
}
/**
* Channels to transmit: TeamSpeak's {@code OPUS_MUSIC} stream is stereo,
* {@code OPUS_VOICE} is mono. A mono-only capture device caps this at one.
*/
private int channelsFor(OpusParameters p) {
return (p.music && captureChannels >= 2) ? 2 : 1;
}
/** Creates or reconfigures the encoder to match {@code p}. Call under {@link #encoderLock}. */
private void applyParams(OpusParameters p) {
int application = applicationFor(p);
int channels = channelsFor(p);
// Application (VOIP vs AUDIO) and channel count are fixed at creation, so
// switching voice<->music means building a new encoder.
if (encoder == null || application != encoderApplication || channels != encoderChannels) {
OpusEncoder replacement = new OpusEncoder(
AudioDevices.SAMPLE_RATE, AudioDevices.FRAME_SIZE, channels, application);
configureEncoder(replacement, p);
OpusEncoder previous = encoder;
encoder = replacement;
encoderApplication = application;
encoderChannels = channels;
if (previous != null) previous.close();
} else {
configureEncoder(encoder, p);
}
codec = codecFor(p);
appliedParams = p;
}
private static void configureEncoder(OpusEncoder enc, OpusParameters p) {
enc.setBitrate(p.bitrate);
enc.setComplexity(p.complexity);
enc.setVbr(p.vbr);
enc.setInbandFec(p.fec);
enc.setExpectedPacketLoss(p.expectedPacketLoss);
enc.setSignal(p.music ? Opus.OPUS_SIGNAL_MUSIC : Opus.OPUS_SIGNAL_VOICE);
}
public void setLevelListener(Consumer<Double> l) {
this.levelListener = l;
}
public void setTalkListener(Consumer<Boolean> l) {
this.talkListener = l;
}
@Override
public void setMonitorListener(AudioFrameListener l) {
this.monitorListener = l;
}
@Override
public void setMutedTalkListener(Runnable l) {
this.mutedTalkListener = l;
}
public void setPushToTalk(boolean down) {
this.pttDown.set(down);
}
public void setMuted(boolean m) {
this.muted.set(m);
}
@Override
public void setLocalMuted(boolean m) {
this.localMuted.set(m);
}
@Override
public boolean isLocalMuted() {
return localMuted.get();
}
// ---- lifecycle ----
public synchronized void start() {
if (running.get()) return;
try {
line = AudioDevices.openCapture(deviceName, AudioDevices.MAX_CHANNELS);
captureChannels = line.channels();
synchronized (encoderLock) {
applyParams(params);
}
} catch (Throwable t) {
cleanup();
throw new RuntimeException("Could not start microphone: " + t.getMessage(), t);
}
speechDetector.reset();
processor.reset();
running.set(true);
captureThread = new Thread(this::captureLoop, "ts3j-mic-capture");
captureThread.setDaemon(true);
captureThread.start();
}
public synchronized void stop() {
running.set(false);
if (captureThread != null) {
captureThread.interrupt();
captureThread = null;
}
cleanup();
queue.clear();
setTransmitting(false);
}
private void cleanup() {
if (line != null) {
line.close();
line = null;
}
synchronized (encoderLock) {
if (encoder != null) {
try {
encoder.close();
} catch (Exception ignored) {
}
encoder = null;
encoderApplication = -1;
appliedParams = null;
}
}
}
private void captureLoop() {
final int frameSamples = AudioDevices.FRAME_SIZE;
final int channels = captureChannels;
final byte[] buf = new byte[frameSamples * 2 * channels];
final float[] pcm = new float[frameSamples * channels]; // interleaved capture
// Analysis (level, VAD) always runs on a mono downmix; with a mono device that
// is the capture buffer itself, so nothing is copied.
final float[] mono = (channels == 1) ? pcm : new float[frameSamples];
// Held locally so a concurrent stop() closing the line cannot null it mid-loop.
final AudioCapture capture = line;
capture.start();
while (running.get()) {
// A short read only happens once the line is closing, or on interruption.
if (capture.read(buf, 0, buf.length) < buf.length) break;
// 16-bit LE -> float, with input gain
for (int i = 0; i < pcm.length; i++) {
int lo = buf[2 * i] & 0xFF;
int hi = buf[2 * i + 1];
short s = (short) ((hi << 8) | lo);
float f = (float) (s / 32768.0 * inputGain);
if (f > 1f) f = 1f;
else if (f < -1f) f = -1f;
pcm[i] = f;
}
if (channels > 1) {
for (int i = 0; i < frameSamples; i++) {
float sum = 0;
for (int c = 0; c < channels; c++) sum += pcm[i * channels + c];
mono[i] = sum / channels;
}
}
byte[] packet = null;
synchronized (encoderLock) {
OpusParameters p = params;
if (p != appliedParams && encoder != null) applyParams(p);
boolean stereo = encoderChannels == 2;
// As in TS3, the speech detector judges the raw microphone signal, while
// the level meter and volume gate see the processed one.
double probability = usesSpeechDetector()
? speechDetector.process(mono, frameSamples) : 0.0;
// Processing is a voice-chain stage on the mono path. A stereo (music)
// stream bypasses it and is transmitted as captured.
if (!stereo) processor.process(mono, frameSamples);
double db = InputLevel.toDb(mono, frameSamples);
Consumer<Double> ll = levelListener;
if (ll != null) ll.accept(db);
boolean wasOpen = transmitting.get();
boolean open = decideGate(db, probability);
setTransmitting(open);
if (open && !muted.get() && encoder != null) {
try {
// On the opening edge, send the buffered lead-in first so the
// word's onset isn't lost to the frame that detected it.
if (!wasOpen) {
flushPreroll(stereo);
}
float[] sent = stereo ? pcm : mono;
packet = encoder.encode(sent);
monitor(sent, stereo ? encoderChannels : 1);
} catch (Exception ignored) {
}
}
if (!open) {
rememberForPreroll(stereo ? pcm : mono);
}
}
if (packet != null && packet.length > 0) {
offer(packet);
}
}
}
private void offer(byte[] packet) {
queue.offer(packet);
// Guard against unbounded growth if the network stalls.
while (queue.size() > 10) queue.poll();
}
/** Keeps the most recent frames while the gate is shut, for {@link #flushPreroll}. */
private void rememberForPreroll(float[] frame) {
float[] slot = preroll[prerollTail];
if (slot == null || slot.length != frame.length) {
slot = new float[frame.length];
preroll[prerollTail] = slot;
}
System.arraycopy(frame, 0, slot, 0, frame.length);
prerollTail = (prerollTail + 1) % PREROLL_FRAMES;
if (prerollCount < PREROLL_FRAMES) prerollCount++;
}
/** Encodes and queues the buffered lead-in, oldest first. Call under {@link #encoderLock}. */
private void flushPreroll(boolean stereo) {
int expected = stereo ? AudioDevices.FRAME_SIZE * encoderChannels : AudioDevices.FRAME_SIZE;
for (int i = 0; i < prerollCount; i++) {
int idx = (prerollTail - prerollCount + i + PREROLL_FRAMES) % PREROLL_FRAMES;
float[] frame = preroll[idx];
// A channel-count change between capture and flush invalidates the buffer.
if (frame == null || frame.length != expected) continue;
try {
byte[] p = encoder.encode(frame);
if (p != null && p.length > 0) offer(p);
monitor(frame, stereo ? encoderChannels : 1);
} catch (Exception ignored) {
}
}
prerollCount = 0;
}
/** Whether the gate may consult the speech detector, so it must follow the signal. */
private boolean usesSpeechDetector() {
if (vadMode == Settings.VadMode.VOLUME_GATE) return false;
Settings.InputMode m = mode;
return m == Settings.InputMode.VOICE_ACTIVATION || (m == Settings.InputMode.PUSH_TO_TALK && vadOverPtt);
}
private boolean decideGate(double db, double probability) {
if (muted.get() || localMuted.get()) {
hangover = 0;
detectMutedSpeech(db);
return false;
}
mutedTalking = false;
switch (mode) {
case CONTINUOUS:
return true;
case PUSH_TO_TALK:
if (pttDown.get()) return true;
return vadOverPtt && voiceActivated(db, probability);
case VOICE_ACTIVATION:
default:
return voiceActivated(db, probability);
}
}
/**
* Applies the selected VAD mode with a hangover so trailing syllables aren't clipped.
*
* <p>The modes match the TS3 client's: the volume gate alone, the speech detector
* alone, or both together. Volume Gate skips the detector entirely, as TS3 does.
*/
private boolean voiceActivated(double db, double probability) {
boolean detected = switch (vadMode) {
case VOLUME_GATE -> db >= thresholdDb;
case AUTOMATIC -> probability >= speechThreshold;
case HYBRID -> db >= thresholdDb && probability >= speechThreshold;
};
if (detected) {
hangover = HANGOVER_FRAMES;
return true;
}
if (hangover > 0) {
hangover--;
return true;
}
return false;
}
/**
* Reports the start of a talk burst that the mute is swallowing. The volume gate
* alone decides here: the speech detector's state is kept for real transmission.
*/
private void detectMutedSpeech(double db) {
boolean talking = db >= thresholdDb;
if (talking && !mutedTalking) {
Runnable listener = mutedTalkListener;
if (listener != null) listener.run();
}
mutedTalking = talking;
}
/** Hands a transmitted frame to the monitor, if one is attached. */
private void monitor(float[] frame, int channels) {
AudioFrameListener l = monitorListener;
if (l != null) l.onFrame(frame, channels);
}
private void setTransmitting(boolean t) {
transmitting.set(t);
if (t != lastTransmitting) {
lastTransmitting = t;
Consumer<Boolean> tl = talkListener;
if (tl != null) tl.accept(t);
}
}
// ---- ts3j Microphone contract ----
@Override
public boolean isMuted() {
return muted.get();
}
@Override
public boolean isReady() {
// Keep the ts3j sender "active" while we're transmitting or still have
// buffered packets. When both are false ts3j sends the terminating packet.
return transmitting.get() || !queue.isEmpty();
}
@Override
public CodecType getCodec() {
return codec;
}
@Override
public byte[] provide() {
byte[] p = queue.poll();
return p != null ? p : new byte[0];
}
}

View File

@@ -1,404 +0,0 @@
package com.ts3client.audio.desktop;
import com.github.manevolent.ts3j.enums.CodecType;
import com.github.manevolent.ts3j.protocol.packet.PacketBody0Voice;
import com.github.manevolent.ts3j.protocol.packet.PacketBody1VoiceWhisper;
import com.ts3client.audio.VoiceOutput;
import java.util.HashMap;
import java.util.Map;
import java.util.Set;
import java.util.concurrent.ConcurrentHashMap;
import java.util.concurrent.ExecutorService;
import java.util.concurrent.Executors;
import java.util.concurrent.ScheduledFuture;
import java.util.concurrent.ScheduledExecutorService;
import java.util.concurrent.TimeUnit;
import java.util.concurrent.atomic.AtomicBoolean;
import java.util.function.BiConsumer;
/**
* Desktop {@link VoiceOutput}: decodes and plays incoming voice per speaker.
* Each client gets its own Opus decoder, playback line and worker thread, so
* simultaneous speakers are mixed by the audio server and one slow decode never
* blocks another (or the network thread). Lines come from {@link AudioDevices},
* so on a PipeWire desktop every speaker is a separate stream in the mixer.
*/
public final class DesktopVoiceOutput implements VoiceOutput {
/** Longest Opus frame (120 ms @ 48 kHz) a packet may decode to, per channel. */
private static final int MAX_FRAME = 5760;
/** TS3 voice packets normally carry one 20 ms Opus frame. */
private static final long FRAME_NANOS = TimeUnit.MILLISECONDS.toNanos(20);
/** Initial playout delay: enough room for common UDP jitter without feeling laggy. */
private static final long JITTER_DELAY_NANOS = FRAME_NANOS * 3;
/** Cap PLC during a lost end-of-talk marker or a hard network stall. */
private static final int MAX_CONSECUTIVE_PLC_FRAMES = 10;
/** Keep memory bounded when a talk burst runs far ahead of playback. */
private static final int MAX_PENDING_PACKETS = 32;
/**
* A speaker is only meant to stop when the empty voice packet marking the end of a
* talk burst arrives — but that packet is UDP too, and a lost one otherwise leaves
* the talking indicator stuck until the speaker's next burst. Opus frames are 20&nbsp;ms
* apart while someone is actually talking, so a gap several times that long, with no
* packet of either kind, is unambiguous: force the indicator off rather than trust
* the one packet that could go missing.
*/
private static final long TALK_TIMEOUT_NANOS = TimeUnit.MILLISECONDS.toNanos(400);
/** One speaker's decode + playback pipeline. */
private final class ClientStream {
final Object lock = new Object();
final int clientId;
final AudioPlayback line;
final int lineChannels;
final ExecutorService worker;
final Map<Integer, VoiceFrame> pending = new HashMap<>();
final AtomicBoolean tickQueued = new AtomicBoolean();
/** Decoder scratch and byte buffer, touched only by {@link #worker}. */
final float[] pcm = new float[MAX_FRAME * AudioDevices.MAX_CHANNELS];
byte[] out = new byte[0];
OpusDecoder decoder;
int decoderChannels;
ScheduledFuture<?> playout;
int expectedPacketId = -1;
int activeChannels = 1;
int consecutivePlcFrames;
volatile boolean talking;
volatile long lastPacketNanos;
ClientStream(int clientId) throws Exception {
this.clientId = clientId;
this.line = AudioDevices.openPlayback(outputDevice, AudioDevices.MAX_CHANNELS);
this.lineChannels = line.channels();
this.line.start();
this.worker = Executors.newSingleThreadExecutor(r -> {
Thread t = new Thread(r, "ts3j-play-" + clientId);
t.setDaemon(true);
return t;
});
}
/**
* Returns a decoder matching the stream's channel count, rebuilding it when a
* speaker switches between the mono {@code OPUS_VOICE} and stereo
* {@code OPUS_MUSIC} codecs.
*/
OpusDecoder decoderFor(int channels) {
if (decoder == null || decoderChannels != channels) {
if (decoder != null) decoder.close();
decoder = new OpusDecoder(AudioDevices.SAMPLE_RATE, MAX_FRAME, channels);
decoderChannels = channels;
}
return decoder;
}
void close() {
synchronized (lock) {
if (playout != null) playout.cancel(false);
pending.clear();
}
worker.shutdownNow();
line.close();
if (decoder != null) decoder.close();
}
}
private record VoiceFrame(int packetId, CodecType codec, byte[] data, boolean end) {
}
private final Map<Integer, ClientStream> streams = new ConcurrentHashMap<>();
private final Set<Integer> mutedClients = ConcurrentHashMap.newKeySet();
/** Per-client gain on top of the master volume; absent means 1.0. */
private final Map<Integer, Double> clientVolumes = new ConcurrentHashMap<>();
private volatile String outputDevice;
private volatile double masterVolume = 1.0;
private volatile boolean deafened = false;
/** Notified (clientId, talking) on the EDT-agnostic worker thread when a speaker starts/stops. */
private volatile BiConsumer<Integer, Boolean> talkListener;
/** Catches a talk burst whose end packet never arrived; see {@link #TALK_TIMEOUT_NANOS}. */
private final ScheduledExecutorService watchdog = Executors.newSingleThreadScheduledExecutor(r -> {
Thread t = new Thread(r, "ts3j-talk-watchdog");
t.setDaemon(true);
return t;
});
public DesktopVoiceOutput(String outputDevice) {
this.outputDevice = outputDevice;
watchdog.scheduleWithFixedDelay(this::checkTalkTimeouts, 50, 50, TimeUnit.MILLISECONDS);
}
private void checkTalkTimeouts() {
long now = System.nanoTime();
for (ClientStream s : streams.values()) {
if (s.talking && now - s.lastPacketNanos > TALK_TIMEOUT_NANOS) {
s.worker.submit(() -> stopStream(s));
}
}
}
public void setTalkListener(BiConsumer<Integer, Boolean> l) {
this.talkListener = l;
}
public void setMasterVolume(double v) {
this.masterVolume = Math.max(0, Math.min(2.0, v));
}
public void setDeafened(boolean d) {
this.deafened = d;
if (d) {
// Stop everyone talking immediately.
for (ClientStream s : streams.values()) {
s.worker.submit(() -> stopStream(s));
}
}
}
public boolean isDeafened() {
return deafened;
}
public void setClientMuted(int clientId, boolean muted) {
if (muted) {
mutedClients.add(clientId);
ClientStream s = streams.get(clientId);
if (s != null) s.worker.submit(() -> stopStream(s));
} else {
mutedClients.remove(clientId);
}
}
public boolean isClientMuted(int clientId) {
return mutedClients.contains(clientId);
}
public void setClientVolume(int clientId, double gain) {
if (gain == 1.0) clientVolumes.remove(clientId);
else clientVolumes.put(clientId, Math.max(0, gain));
}
public double getClientVolume(int clientId) {
return clientVolumes.getOrDefault(clientId, 1.0);
}
public void setOutputDevice(String device) {
this.outputDevice = device;
}
/** Entry point wired into {@code client.setVoiceHandler(...)}. */
public void handleVoice(PacketBody0Voice voice) {
route(voice.getClientId(), voice.getPacketId(), voice.getCodecType(), voice.getCodecData());
}
/** Entry point wired into {@code client.setWhisperHandler(...)}. */
public void handleWhisper(PacketBody1VoiceWhisper whisper) {
route(whisper.getClientId(), whisper.getPacketId(), whisper.getCodecType(), whisper.getCodecData());
}
/** TeamSpeak streams {@code OPUS_MUSIC} in stereo and everything else in mono. */
private static int channelsFor(CodecType codec) {
return codec == CodecType.OPUS_MUSIC ? 2 : 1;
}
private void route(int clientId, int packetId, CodecType codec, byte[] data) {
if (deafened) return;
if (mutedClients.contains(clientId)) return;
ClientStream stream = streams.get(clientId);
if (stream == null) {
try {
stream = new ClientStream(clientId);
ClientStream existing = streams.putIfAbsent(clientId, stream);
if (existing != null) {
stream.close();
stream = existing;
}
} catch (Exception e) {
return; // couldn't open a line; drop
}
}
final ClientStream target = stream;
target.lastPacketNanos = System.nanoTime();
VoiceFrame frame = new VoiceFrame(packetId & 0xFFFF, codec, data, data == null || data.length == 0);
synchronized (target.lock) {
if (target.expectedPacketId < 0 || target.playout == null || target.playout.isDone()) {
target.expectedPacketId = frame.packetId();
target.activeChannels = channelsFor(codec);
target.consecutivePlcFrames = 0;
target.playout = watchdog.scheduleAtFixedRate(
() -> queuePlayoutTick(target),
JITTER_DELAY_NANOS,
FRAME_NANOS,
TimeUnit.NANOSECONDS);
} else if (packetDistance(target.expectedPacketId, frame.packetId()) < 0) {
return; // too late for this burst; do not replay stale audio
}
target.pending.putIfAbsent(frame.packetId(), frame);
trimPending(target);
}
if (!frame.end()) markTalking(target, true);
}
private void queuePlayoutTick(ClientStream stream) {
if (!stream.tickQueued.compareAndSet(false, true)) return;
try {
stream.worker.submit(() -> {
try {
playNext(stream);
} finally {
stream.tickQueued.set(false);
}
});
} catch (RuntimeException e) {
if (stream.worker.isShutdown()) {
stream.tickQueued.set(false);
} else {
throw e;
}
}
}
private void playNext(ClientStream stream) {
VoiceFrame frame;
int channels;
boolean conceal;
synchronized (stream.lock) {
if (stream.expectedPacketId < 0) return;
frame = stream.pending.remove(stream.expectedPacketId);
if (frame != null && frame.end()) {
stopPlayoutLocked(stream);
return;
}
conceal = frame == null;
channels = frame != null ? channelsFor(frame.codec()) : stream.activeChannels;
stream.activeChannels = channels;
if (conceal) {
stream.consecutivePlcFrames++;
if (stream.consecutivePlcFrames > MAX_CONSECUTIVE_PLC_FRAMES) {
stopPlayoutLocked(stream);
return;
}
} else {
stream.consecutivePlcFrames = 0;
}
stream.expectedPacketId = (stream.expectedPacketId + 1) & 0xFFFF;
}
decodeAndPlay(stream, channels, conceal ? null : frame.data());
}
private void stopStream(ClientStream stream) {
synchronized (stream.lock) {
stopPlayoutLocked(stream);
}
}
private void stopPlayoutLocked(ClientStream stream) {
stream.expectedPacketId = -1;
if (stream.playout != null) stream.playout.cancel(false);
stream.playout = null;
stream.pending.clear();
finishStream(stream);
}
private void finishStream(ClientStream stream) {
if (stream.decoder != null) stream.decoder.reset();
markTalking(stream, false);
}
private static int packetDistance(int from, int to) {
int distance = (to - from) & 0xFFFF;
return distance >= 0x8000 ? distance - 0x10000 : distance;
}
private static void trimPending(ClientStream stream) {
while (stream.pending.size() > MAX_PENDING_PACKETS) {
Integer furthest = null;
int furthestDistance = Integer.MIN_VALUE;
for (Integer packetId : stream.pending.keySet()) {
int distance = packetDistance(stream.expectedPacketId, packetId);
if (distance > furthestDistance) {
furthestDistance = distance;
furthest = packetId;
}
}
if (furthest == null) return;
stream.pending.remove(furthest);
}
}
private void decodeAndPlay(ClientStream stream, int channels, byte[] data) {
try {
float[] pcm = stream.pcm;
int frames = stream.decoderFor(channels).decode(data, pcm);
double vol = masterVolume * clientVolumes.getOrDefault(stream.clientId, 1.0);
// Match the decoded stream to the line: duplicate mono across a stereo
// line, fold a stereo (music) stream down onto a mono-only line.
int lineChannels = stream.lineChannels;
int bytes = frames * lineChannels * 2;
if (stream.out.length < bytes) stream.out = new byte[bytes];
byte[] out = stream.out;
for (int i = 0, k = 0; i < frames; i++) {
for (int c = 0; c < lineChannels; c++, k += 2) {
double v;
if (channels == lineChannels) {
v = pcm[i * channels + c];
} else if (channels == 1) {
v = pcm[i];
} else {
double sum = 0;
for (int s = 0; s < channels; s++) sum += pcm[i * channels + s];
v = sum / channels;
}
v *= vol;
if (v > 1.0) v = 1.0;
else if (v < -1.0) v = -1.0;
short s = (short) Math.round(v * 32767.0);
out[k] = (byte) (s & 0xFF);
out[k + 1] = (byte) ((s >> 8) & 0xFF);
}
}
stream.line.write(out, 0, bytes);
} catch (Exception ignored) {
}
}
private void markTalking(ClientStream stream, boolean talking) {
if (stream.talking == talking) return;
stream.talking = talking;
BiConsumer<Integer, Boolean> l = talkListener;
if (l != null) l.accept(stream.clientId, talking);
}
/** Drop a speaker's pipeline entirely (e.g. they left the server). */
/** Forgets a client that left; the server may hand its id to somebody else. */
public void removeClient(int clientId) {
ClientStream s = streams.remove(clientId);
if (s != null) s.close();
mutedClients.remove(clientId);
clientVolumes.remove(clientId);
}
public void shutdown() {
watchdog.shutdownNow();
for (ClientStream s : streams.values()) {
s.close();
}
streams.clear();
}
}

View File

@@ -1,5 +1,7 @@
package com.ts3client.audio.desktop;
import com.ts3client.audio.AudioCapture;
import javax.sound.sampled.TargetDataLine;
/** {@link AudioCapture} backed by a Java Sound {@link TargetDataLine}. */

View File

@@ -1,5 +1,7 @@
package com.ts3client.audio.desktop;
import com.ts3client.audio.AudioPlayback;
import javax.sound.sampled.SourceDataLine;
/** {@link AudioPlayback} backed by a Java Sound {@link SourceDataLine}. */

View File

@@ -0,0 +1,25 @@
package com.ts3client.audio.desktop;
import com.ts3client.audio.opus.OpusCodec;
import com.ts3client.audio.opus.OpusDecoder;
import com.ts3client.audio.opus.OpusEncoder;
/** The system's (or the bundled) libopus, reached through the FFM API. */
public final class NativeOpusCodec implements OpusCodec {
@Override
public OpusEncoder createEncoder(int sampleRate, int frameSize, int channels, Application application) {
int app = application == Application.AUDIO ? Opus.OPUS_APPLICATION_AUDIO : Opus.OPUS_APPLICATION_VOIP;
return new NativeOpusEncoder(sampleRate, frameSize, channels, app);
}
@Override
public OpusDecoder createDecoder(int sampleRate, int maxFrameSize, int channels) {
return new NativeOpusDecoder(sampleRate, maxFrameSize, channels);
}
@Override
public String version() {
return Opus.getVersionString();
}
}

View File

@@ -1,5 +1,7 @@
package com.ts3client.audio.desktop;
import com.ts3client.audio.opus.OpusDecoder;
import java.lang.foreign.Arena;
import java.lang.foreign.MemorySegment;
import java.lang.foreign.ValueLayout;
@@ -11,7 +13,7 @@ import java.lang.foreign.ValueLayout;
* (mono for {@code OPUS_VOICE}, stereo for {@code OPUS_MUSIC}); Opus will up-/down-mix
* a mismatched stream, which costs the stereo image.
*/
public final class OpusDecoder implements AutoCloseable {
final class NativeOpusDecoder implements OpusDecoder {
private static final int MAX_PACKET_BYTES = 4096;
@@ -23,7 +25,7 @@ public final class OpusDecoder implements AutoCloseable {
private final int channels;
private boolean closed;
public OpusDecoder(int sampleRate, int frameSize, int channels) {
NativeOpusDecoder(int sampleRate, int frameSize, int channels) {
this.frameSize = frameSize;
this.channels = channels;
@@ -48,6 +50,7 @@ public final class OpusDecoder implements AutoCloseable {
* @param out output buffer, at least {@code frameSize * channels} long
* @return number of samples decoded per channel
*/
@Override
public int decode(byte[] packet, float[] out) {
if (closed) throw new IllegalStateException("decoder closed");
@@ -70,15 +73,12 @@ public final class OpusDecoder implements AutoCloseable {
return samples;
}
@Override
public void reset() {
if (closed) return;
Opus.decoderCtl(handle, Opus.OPUS_RESET_STATE, 0);
}
public int getChannels() {
return channels;
}
@Override
public void close() {
if (closed) return;

View File

@@ -1,5 +1,7 @@
package com.ts3client.audio.desktop;
import com.ts3client.audio.opus.OpusEncoder;
import java.lang.foreign.Arena;
import java.lang.foreign.MemorySegment;
import java.lang.foreign.ValueLayout;
@@ -14,7 +16,7 @@ import java.lang.foreign.ValueLayout;
* are freed by {@link #close()}; access is serialised by the instance lock, so the
* encoder may be driven from any thread.
*/
public final class OpusEncoder implements AutoCloseable {
final class NativeOpusEncoder implements OpusEncoder {
private static final int MAX_PACKET_BYTES = 4096;
@@ -27,7 +29,7 @@ public final class OpusEncoder implements AutoCloseable {
private final Object lock = new Object();
private boolean closed;
public OpusEncoder(int sampleRate, int frameSize, int channels, int application) {
NativeOpusEncoder(int sampleRate, int frameSize, int channels, int application) {
this.frameSize = frameSize;
this.channels = channels;
@@ -44,28 +46,34 @@ public final class OpusEncoder implements AutoCloseable {
this.packetBuffer = arena.allocate(MAX_PACKET_BYTES);
}
@Override
public void setBitrate(int bitsPerSecond) {
ctl(Opus.OPUS_SET_BITRATE_REQUEST, bitsPerSecond);
}
@Override
public void setComplexity(int complexity) {
ctl(Opus.OPUS_SET_COMPLEXITY_REQUEST, Math.max(0, Math.min(10, complexity)));
}
@Override
public void setVbr(boolean vbr) {
ctl(Opus.OPUS_SET_VBR_REQUEST, vbr ? 1 : 0);
}
@Override
public void setInbandFec(boolean fec) {
ctl(Opus.OPUS_SET_INBAND_FEC_REQUEST, fec ? 1 : 0);
}
@Override
public void setExpectedPacketLoss(int percent) {
ctl(Opus.OPUS_SET_PACKET_LOSS_PERC_REQUEST, Math.max(0, Math.min(100, percent)));
}
public void setSignal(int signal) {
ctl(Opus.OPUS_SET_SIGNAL_REQUEST, signal);
@Override
public void setSignal(Signal signal) {
ctl(Opus.OPUS_SET_SIGNAL_REQUEST, signal == Signal.MUSIC ? Opus.OPUS_SIGNAL_MUSIC : Opus.OPUS_SIGNAL_VOICE);
}
private void ctl(int request, int value) {
@@ -85,6 +93,7 @@ public final class OpusEncoder implements AutoCloseable {
*
* @return a newly allocated byte array holding the encoded packet
*/
@Override
public byte[] encode(float[] pcm) {
int expected = frameSize * channels;
if (pcm.length != expected) {

View File

@@ -1,5 +1,8 @@
package com.ts3client.audio.desktop;
import com.ts3client.audio.AudioIo;
import com.ts3client.audio.AudioPlayback;
import com.ts3client.audio.VoiceFormat;
import com.ts3client.sound.SoundPlayer;
import javax.sound.sampled.AudioFormat;
@@ -28,7 +31,7 @@ public final class WavSoundPlayer implements SoundPlayer {
* triggered, so it is kept around for a spell of quiet rather than per sound.
*/
private static final long IDLE_KEEP_OPEN_NANOS = 30_000_000_000L;
private static final long FRAME_NANOS = AudioDevices.FRAME_SIZE * 1_000_000_000L / AudioDevices.SAMPLE_RATE;
private static final long FRAME_NANOS = VoiceFormat.FRAME_SIZE * 1_000_000_000L / VoiceFormat.SAMPLE_RATE;
/**
* How far ahead of real time the mixer renders. The line would happily take a
* few hundred milliseconds at once, but anything written is fixed: a sound that
@@ -61,11 +64,13 @@ public final class WavSoundPlayer implements SoundPlayer {
private final List<Voice> voices = new ArrayList<>();
private final Object lock = new Object();
private final AudioIo io;
private volatile String outputDevice;
private volatile boolean running = true;
private Thread mixer;
public WavSoundPlayer(String outputDevice) {
public WavSoundPlayer(AudioIo io, String outputDevice) {
this.io = io;
this.outputDevice = outputDevice;
}
@@ -121,11 +126,11 @@ public final class WavSoundPlayer implements SoundPlayer {
private void mixLoop() {
AudioPlayback line = null;
try {
float[] mix = new float[AudioDevices.FRAME_SIZE];
float[] mix = new float[VoiceFormat.FRAME_SIZE];
long due = 0; // when the next frame is due to leave the speaker
while (awaitVoices()) {
if (line == null) {
line = AudioDevices.openPlayback(outputDevice, AudioDevices.MAX_CHANNELS);
line = io.openPlayback(outputDevice, VoiceFormat.MAX_CHANNELS);
line.start();
}
long now = System.nanoTime();
@@ -236,7 +241,7 @@ public final class WavSoundPlayer implements SoundPlayer {
}
mono[i] = sum / channels;
}
return resample(mono, source.getSampleRate(), AudioDevices.SAMPLE_RATE);
return resample(mono, source.getSampleRate(), VoiceFormat.SAMPLE_RATE);
}
}
}

View File

@@ -1,7 +1,7 @@
package com.ts3client.audio.desktop.pipewire;
import com.ts3client.audio.desktop.AudioCapture;
import com.ts3client.audio.desktop.AudioPlayback;
import com.ts3client.audio.AudioCapture;
import com.ts3client.audio.AudioPlayback;
import java.util.List;

View File

@@ -1,6 +1,6 @@
package com.ts3client.audio.desktop.pipewire;
import com.ts3client.audio.desktop.AudioCapture;
import com.ts3client.audio.AudioCapture;
import java.lang.foreign.MemorySegment;

View File

@@ -1,6 +1,6 @@
package com.ts3client.audio.desktop.pipewire;
import com.ts3client.audio.desktop.AudioPlayback;
import com.ts3client.audio.AudioPlayback;
import java.lang.foreign.MemorySegment;