Process the microphone like TS3, and tell it about key presses

The capture path now runs the ported WebRTC chain in place of the
home-made noise, typing and gain stages, which are removed.

- As in TS3, the speech detector judges the raw microphone signal, while
  the level meter and volume gate see the processed one.
- Noise removal takes TS3's four levels (6, 12, 18 or 21 dB); a stored
  0..1 level falls back to TS3's default of 12 dB.
- Every key press from the global input hook reaches the connected
  microphones and the microphone test, so typing attenuation engages
  while the user types.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-24 20:47:04 +00:00
parent de5f7da236
commit 1d61f458ae
12 changed files with 102 additions and 588 deletions

View File

@@ -1,12 +1,13 @@
package com.ts3client.audio.desktop;
import com.github.manevolent.ts3j.enums.CodecType;
import com.ts3client.audio.AudioEnhancer;
import com.ts3client.audio.AudioFrameListener;
import com.ts3client.audio.InputLevel;
import com.ts3client.audio.OpusParameters;
import com.ts3client.audio.SpeechProbabilityDetector;
import com.ts3client.audio.VoiceInput;
import com.ts3client.audio.processing.AudioProcessor;
import com.ts3client.audio.processing.ns.SuppressionLevel;
import com.ts3client.audio.vad.RnnSpeechDetector;
import com.ts3client.config.Settings;
@@ -66,7 +67,7 @@ public final class DesktopVoiceInput implements VoiceInput {
private final SpeechProbabilityDetector speechDetector =
new RnnSpeechDetector(AudioDevices.SAMPLE_RATE);
private final AudioEnhancer enhancer = new AudioEnhancer(AudioDevices.SAMPLE_RATE);
private final AudioProcessor processor = new AudioProcessor();
private volatile Consumer<Double> levelListener; // input level, InputLevel scale
private volatile Consumer<Boolean> talkListener; // local talk-state changes
@@ -105,10 +106,10 @@ public final class DesktopVoiceInput implements VoiceInput {
this.inputGain = settings.inputVolume;
this.params = OpusParameters.from(settings);
this.codec = codecFor(params);
enhancer.setNoiseSuppression(settings.denoise);
enhancer.setDenoiserLevel(settings.denoiserLevel);
enhancer.setTypingAttenuation(settings.typingAttenuation);
enhancer.setAgc(settings.agc);
processor.setNoiseSuppression(settings.denoise);
processor.setSuppressionLevel(SuppressionLevel.fromDenoiserLevel(settings.denoiserLevel));
processor.setTransientSuppression(settings.typingAttenuation);
processor.setGainControl(settings.agc);
}
// ---- live configuration (safe to call from the UI thread) ----
@@ -139,19 +140,24 @@ public final class DesktopVoiceInput implements VoiceInput {
}
public void setNoiseSuppression(boolean enabled) {
enhancer.setNoiseSuppression(enabled);
processor.setNoiseSuppression(enabled);
}
public void setDenoiserLevel(double level) {
enhancer.setDenoiserLevel(level);
public void setDenoiserLevel(int level) {
processor.setSuppressionLevel(SuppressionLevel.fromDenoiserLevel(level));
}
public void setTypingAttenuation(boolean enabled) {
enhancer.setTypingAttenuation(enabled);
processor.setTransientSuppression(enabled);
}
public void setAgc(boolean enabled) {
enhancer.setAgc(enabled);
processor.setGainControl(enabled);
}
@Override
public void keyPressed() {
processor.keyPressed();
}
/**
@@ -265,7 +271,7 @@ public final class DesktopVoiceInput implements VoiceInput {
throw new RuntimeException("Could not start microphone: " + t.getMessage(), t);
}
speechDetector.reset();
enhancer.reset();
processor.reset();
running.set(true);
captureThread = new Thread(this::captureLoop, "ts3j-mic-capture");
captureThread.setDaemon(true);
@@ -342,17 +348,21 @@ public final class DesktopVoiceInput implements VoiceInput {
if (p != appliedParams && encoder != null) applyParams(p);
boolean stereo = encoderChannels == 2;
// Denoise / typing attenuation are voice-chain stages: they feed the
// level meter, VAD and encoder alike on the mono path. A stereo (music)
// stream bypasses them and is transmitted as captured.
if (!stereo) enhancer.process(mono, frameSamples);
// As in TS3, the speech detector judges the raw microphone signal, while
// the level meter and volume gate see the processed one.
double probability = usesSpeechDetector()
? speechDetector.process(mono, frameSamples) : 0.0;
// Processing is a voice-chain stage on the mono path. A stereo (music)
// stream bypasses it and is transmitted as captured.
if (!stereo) processor.process(mono, frameSamples);
double db = InputLevel.toDb(mono, frameSamples);
Consumer<Double> ll = levelListener;
if (ll != null) ll.accept(db);
boolean wasOpen = transmitting.get();
boolean open = decideGate(db, mono);
boolean open = decideGate(db, probability);
setTransmitting(open);
if (open && !muted.get() && encoder != null) {
@@ -415,7 +425,14 @@ public final class DesktopVoiceInput implements VoiceInput {
prerollCount = 0;
}
private boolean decideGate(double db, float[] pcm) {
/** Whether the gate may consult the speech detector, so it must follow the signal. */
private boolean usesSpeechDetector() {
if (vadMode == Settings.VadMode.VOLUME_GATE) return false;
Settings.InputMode m = mode;
return m == Settings.InputMode.VOICE_ACTIVATION || (m == Settings.InputMode.PUSH_TO_TALK && vadOverPtt);
}
private boolean decideGate(double db, double probability) {
if (muted.get() || localMuted.get()) {
hangover = 0;
detectMutedSpeech(db);
@@ -427,10 +444,10 @@ public final class DesktopVoiceInput implements VoiceInput {
return true;
case PUSH_TO_TALK:
if (pttDown.get()) return true;
return vadOverPtt && voiceActivated(db, pcm);
return vadOverPtt && voiceActivated(db, probability);
case VOICE_ACTIVATION:
default:
return voiceActivated(db, pcm);
return voiceActivated(db, probability);
}
}
@@ -440,21 +457,12 @@ public final class DesktopVoiceInput implements VoiceInput {
* <p>The modes match the TS3 client's: the volume gate alone, the speech detector
* alone, or both together. Volume Gate skips the detector entirely, as TS3 does.
*/
private boolean voiceActivated(double db, float[] pcm) {
boolean detected;
switch (vadMode) {
case VOLUME_GATE:
detected = db >= thresholdDb;
break;
case AUTOMATIC:
detected = speechDetector.process(pcm, AudioDevices.FRAME_SIZE) >= speechThreshold;
break;
case HYBRID:
default:
double probability = speechDetector.process(pcm, AudioDevices.FRAME_SIZE);
detected = db >= thresholdDb && probability >= speechThreshold;
break;
}
private boolean voiceActivated(double db, double probability) {
boolean detected = switch (vadMode) {
case VOLUME_GATE -> db >= thresholdDb;
case AUTOMATIC -> probability >= speechThreshold;
case HYBRID -> db >= thresholdDb && probability >= speechThreshold;
};
if (detected) {
hangover = HANGOVER_FRAMES;
return true;