From 1d61f458aeacd02cad5b71ac88174ea3f3aca767 Mon Sep 17 00:00:00 2001 From: ericek111 Date: Thu, 24 Sep 2026 20:47:04 +0000 Subject: [PATCH] Process the microphone like TS3, and tell it about key presses The capture path now runs the ported WebRTC chain in place of the home-made noise, typing and gain stages, which are removed. - As in TS3, the speech detector judges the raw microphone signal, while the level meter and volume gate see the processed one. - Noise removal takes TS3's four levels (6, 12, 18 or 21 dB); a stored 0..1 level falls back to TS3's default of 12 dB. - Every key press from the global input hook reaches the connected microphones and the microphone test, so typing attenuation engages while the user types. Co-Authored-By: Claude Opus 5.5 --- .../com/ts3client/audio/AudioEnhancer.java | 238 ------------------ .../ts3client/audio/AutomaticGainControl.java | 66 ----- .../com/ts3client/audio/HighPassFilter.java | 43 ---- .../com/ts3client/audio/NoiseSuppressor.java | 112 --------- .../com/ts3client/audio/TypingAttenuator.java | 78 ------ .../java/com/ts3client/audio/VoiceInput.java | 11 +- .../java/com/ts3client/config/Settings.java | 12 +- .../audio/desktop/DesktopVoiceInput.java | 78 +++--- .../java/com/ts3client/ui/DevicesPanel.java | 14 +- .../java/com/ts3client/ui/HotkeyService.java | 21 +- .../main/java/com/ts3client/ui/MainFrame.java | 10 + .../java/com/ts3client/ui/SettingsDialog.java | 7 + 12 files changed, 102 insertions(+), 588 deletions(-) delete mode 100644 ts3-client/core/src/main/java/com/ts3client/audio/AudioEnhancer.java delete mode 100644 ts3-client/core/src/main/java/com/ts3client/audio/AutomaticGainControl.java delete mode 100644 ts3-client/core/src/main/java/com/ts3client/audio/HighPassFilter.java delete mode 100644 ts3-client/core/src/main/java/com/ts3client/audio/NoiseSuppressor.java delete mode 100644 ts3-client/core/src/main/java/com/ts3client/audio/TypingAttenuator.java diff --git a/ts3-client/core/src/main/java/com/ts3client/audio/AudioEnhancer.java b/ts3-client/core/src/main/java/com/ts3client/audio/AudioEnhancer.java deleted file mode 100644 index 8f5bedb..0000000 --- a/ts3-client/core/src/main/java/com/ts3client/audio/AudioEnhancer.java +++ /dev/null @@ -1,238 +0,0 @@ -package com.ts3client.audio; - -import java.util.Arrays; - -/** - * Microphone pre-processing chain applied before voice activation and Opus encoding, - * mirroring the capture-side denoise/typing filters of the TeamSpeak 3 client. - * - *

The stages run in the same order as the TeamSpeak client's WebRTC capture chain: - * a {@link HighPassFilter} (always-on rumble/DC removal), then a streaming short-time - * Fourier transform (square-root Hann window, 50% overlap-add) carrying a - * {@link NoiseSuppressor} ("Remove background noise") and a {@link TypingAttenuator} - * ("Typing attenuation") — sharing one FFT/IFFT per hop — and finally an - * {@link AutomaticGainControl} ("AGC"). Echo cancellation (WebRTC AEC3) is omitted as - * it requires the loudspeaker reference signal. - * - *

Input frames of any length are decoupled from the STFT hop by internal ring - * buffers; when noise/typing suppression is active the output is delayed by one hop - * (~5 ms). The high-pass filter and AGC are zero-latency. When every stage is - * disabled the chain is fully bypassed and audio passes through untouched. - * - *

Pure DSP with no platform dependencies, so any frontend/backend can reuse it. - * Not thread-safe: drive it from a single capture thread; the enable/level setters - * are cheap volatiles safe to call from the UI thread. - */ -public final class AudioEnhancer { - - private static final int FFT_SIZE = 512; // power of two -> 10.7 ms @ 48 kHz - private static final int HOP = FFT_SIZE / 2; // 50% overlap - private static final int BINS = FFT_SIZE / 2 + 1; - - private final double[] window = new double[FFT_SIZE]; - private final double[] re = new double[FFT_SIZE]; - private final double[] im = new double[FFT_SIZE]; - private final double[] power = new double[BINS]; - private final double[] gain = new double[BINS]; - - private final double[] frame = new double[FFT_SIZE]; // sliding analysis frame - private final double[] ola = new double[FFT_SIZE]; // overlap-add accumulator - - private final FloatRing input = new FloatRing(FFT_SIZE * 4); - private final FloatRing output = new FloatRing(FFT_SIZE * 4); - private final float[] hopIn = new float[HOP]; - - private final NoiseSuppressor noiseSuppressor = new NoiseSuppressor(BINS); - private final TypingAttenuator typingAttenuator; - private final HighPassFilter highPass; - private final AutomaticGainControl agc; - - private volatile boolean noiseEnabled; - private volatile boolean typingEnabled; - private volatile boolean agcEnabled; - private boolean active; // any stage on: HPF + AGC state is live - private boolean stftRunning; // noise/typing on: STFT rings are live - - public AudioEnhancer(int sampleRate) { - for (int i = 0; i < FFT_SIZE; i++) { - // sqrt(Hann): analysis*synthesis = Hann, which is COLA at 50% overlap. - window[i] = Math.sqrt(0.5 * (1 - Math.cos(2 * Math.PI * i / FFT_SIZE))); - } - this.typingAttenuator = new TypingAttenuator(BINS, sampleRate, FFT_SIZE); - this.highPass = new HighPassFilter(sampleRate); - this.agc = new AutomaticGainControl(sampleRate); - } - - public void setNoiseSuppression(boolean enabled) { - this.noiseEnabled = enabled; - } - - public void setDenoiserLevel(double level) { - noiseSuppressor.setLevel(level); - } - - public void setTypingAttenuation(boolean enabled) { - this.typingEnabled = enabled; - } - - public void setAgc(boolean enabled) { - this.agcEnabled = enabled; - } - - /** Clears all filter state; call when (re)starting capture. */ - public void reset() { - resetStft(); - highPass.reset(); - agc.reset(); - active = false; - stftRunning = false; - } - - private void resetStft() { - input.clear(); - output.clear(); - Arrays.fill(frame, 0); - Arrays.fill(ola, 0); - noiseSuppressor.reset(); - typingAttenuator.reset(); - } - - /** - * Enhances one frame of mono PCM in place. {@code buf[0..len)} is overwritten with - * the processed (one-hop-delayed when noise/typing suppression is on) signal. - * Returns immediately if every stage is disabled. - */ - public void process(float[] buf, int len) { - boolean stft = noiseEnabled || typingEnabled; - boolean anyStage = stft || agcEnabled; - if (!anyStage) { - if (active) reset(); // drop stale filter/delay state on full disable - return; - } - if (!active) { - reset(); - active = true; - } - - // 1) High-pass filter (always-on part of the active chain). - highPass.process(buf, len); - - // 2) STFT noise + typing suppression (only when either is enabled). - if (stft) { - if (!stftRunning) { - resetStft(); - stftRunning = true; - } - runStft(buf, len); - } else if (stftRunning) { - stftRunning = false; - } - - // 3) Automatic gain control (last, on the cleaned signal). - if (agcEnabled) { - agc.process(buf, len); - } - } - - private void runStft(float[] buf, int len) { - input.write(buf, len); - while (input.available() >= HOP) { - System.arraycopy(frame, HOP, frame, 0, FFT_SIZE - HOP); - input.read(hopIn, HOP); - for (int i = 0; i < HOP; i++) { - frame[FFT_SIZE - HOP + i] = hopIn[i]; - } - processBlock(); - } - - // During the initial one-hop priming the output ring is short; pad with zeros. - int ready = output.available(); - if (ready < len) { - for (int i = 0; i < len - ready; i++) buf[i] = 0f; - output.read(buf, len - ready, ready); - } else { - output.read(buf, 0, len); - } - } - - private void processBlock() { - for (int i = 0; i < FFT_SIZE; i++) { - re[i] = frame[i] * window[i]; - im[i] = 0; - } - Fft.forward(re, im); - - for (int k = 0; k < BINS; k++) { - power[k] = re[k] * re[k] + im[k] * im[k]; - gain[k] = 1.0; - } - if (noiseEnabled) noiseSuppressor.apply(power, gain); - if (typingEnabled) typingAttenuator.apply(power, gain); - - // Apply the real-valued gain to each bin and its conjugate mirror. - for (int k = 0; k < BINS; k++) { - double g = gain[k]; - re[k] *= g; - im[k] *= g; - if (k > 0 && k < FFT_SIZE - k) { - int m = FFT_SIZE - k; - re[m] *= g; - im[m] *= g; - } - } - Fft.inverse(re, im); - - for (int i = 0; i < FFT_SIZE; i++) { - ola[i] += re[i] * window[i]; - } - output.write(ola, HOP); - System.arraycopy(ola, HOP, ola, 0, FFT_SIZE - HOP); - Arrays.fill(ola, FFT_SIZE - HOP, FFT_SIZE, 0); - } - - /** Minimal single-producer/single-consumer float ring buffer. */ - private static final class FloatRing { - private final float[] buf; - private int head, tail, size; - - FloatRing(int capacity) { - this.buf = new float[capacity]; - } - - int available() { - return size; - } - - void clear() { - head = tail = size = 0; - } - - void write(float[] src, int len) { - for (int i = 0; i < len; i++) { - buf[tail] = src[i]; - tail = (tail + 1) % buf.length; - } - size += len; - } - - void write(double[] src, int len) { - for (int i = 0; i < len; i++) { - buf[tail] = (float) src[i]; - tail = (tail + 1) % buf.length; - } - size += len; - } - - void read(float[] dst, int len) { - read(dst, 0, len); - } - - void read(float[] dst, int offset, int len) { - for (int i = 0; i < len; i++) { - dst[offset + i] = buf[head]; - head = (head + 1) % buf.length; - } - size -= len; - } - } -} diff --git a/ts3-client/core/src/main/java/com/ts3client/audio/AutomaticGainControl.java b/ts3-client/core/src/main/java/com/ts3client/audio/AutomaticGainControl.java deleted file mode 100644 index 4cad850..0000000 --- a/ts3-client/core/src/main/java/com/ts3client/audio/AutomaticGainControl.java +++ /dev/null @@ -1,66 +0,0 @@ -package com.ts3client.audio; - -/** - * Automatic gain control — WebRTC APM's {@code gain_controller} (AGC2 adaptive - * digital) / Speex {@code AGC} stage. It normalises voice loudness toward a target - * level so quiet microphones are boosted and loud ones tamed, keeping perceived volume - * consistent across speakers. - * - *

Placed last in the capture chain (after noise suppression), it tracks the frame - * level and moves an applied gain toward {@code target / level}: it attenuates quickly - * to head off clipping and boosts slowly to avoid pumping. A noise gate freezes the - * gain while the input is near silence, so background noise between words is never - * amplified; the per-sample gain ramp avoids zipper artefacts and a final clamp guards - * against overshoot. - */ -final class AutomaticGainControl { - - private static final double TARGET_RMS = 0.12; // ~ -18.4 dBFS - private static final double MAX_GAIN = dbToGain(30); // up to +30 dB boost - private static final double MIN_GAIN = dbToGain(-20); // down to -20 dB - private static final double NOISE_GATE_RMS = dbToGain(-55); // freeze below this level - - private final double attackCoeff; // gain decreasing (signal too loud): fast - private final double releaseCoeff; // gain increasing (too quiet): slow - - private double gain = 1.0; - - AutomaticGainControl(int sampleRate) { - this.attackCoeff = 1 - Math.exp(-1.0 / (0.005 * sampleRate)); // ~5 ms - this.releaseCoeff = 1 - Math.exp(-1.0 / (0.300 * sampleRate)); // ~300 ms - } - - void reset() { - gain = 1.0; - } - - /** Applies gain normalisation to one mono frame in place. */ - void process(float[] buf, int len) { - double sumSq = 0; - for (int i = 0; i < len; i++) { - sumSq += (double) buf[i] * buf[i]; - } - double rms = Math.sqrt(sumSq / len); - - double desired = gain; - if (rms >= NOISE_GATE_RMS) { - desired = TARGET_RMS / rms; - if (desired > MAX_GAIN) desired = MAX_GAIN; - else if (desired < MIN_GAIN) desired = MIN_GAIN; - } - // Boost slowly, attenuate quickly. - double coeff = desired < gain ? attackCoeff : releaseCoeff; - - for (int i = 0; i < len; i++) { - gain += (desired - gain) * coeff; - double y = buf[i] * gain; - if (y > 1.0) y = 1.0; - else if (y < -1.0) y = -1.0; - buf[i] = (float) y; - } - } - - private static double dbToGain(double db) { - return Math.pow(10.0, db / 20.0); - } -} diff --git a/ts3-client/core/src/main/java/com/ts3client/audio/HighPassFilter.java b/ts3-client/core/src/main/java/com/ts3client/audio/HighPassFilter.java deleted file mode 100644 index d43f891..0000000 --- a/ts3-client/core/src/main/java/com/ts3client/audio/HighPassFilter.java +++ /dev/null @@ -1,43 +0,0 @@ -package com.ts3client.audio; - -/** - * Second-order Butterworth high-pass filter (RBJ biquad, transposed direct form II). - * Mirrors WebRTC APM's {@code high_pass_filter} stage — an always-on part of the - * capture chain that removes DC offset, mains hum and low-frequency rumble below the - * speech band before noise suppression sees the signal. - */ -final class HighPassFilter { - - private static final double CUTOFF_HZ = 80.0; // WebRTC APM high-pass cutoff - - private final double b0, b1, b2, a1, a2; - private double z1, z2; - - HighPassFilter(int sampleRate) { - double w0 = 2 * Math.PI * CUTOFF_HZ / sampleRate; - double cos = Math.cos(w0); - double alpha = Math.sin(w0) / Math.sqrt(2.0); // Q = 1/sqrt(2) (Butterworth) - double a0 = 1 + alpha; - this.b0 = (1 + cos) / 2 / a0; - this.b1 = -(1 + cos) / a0; - this.b2 = (1 + cos) / 2 / a0; - this.a1 = -2 * cos / a0; - this.a2 = (1 - alpha) / a0; - } - - void reset() { - z1 = 0; - z2 = 0; - } - - /** Filters one mono frame in place. */ - void process(float[] buf, int len) { - for (int i = 0; i < len; i++) { - double x = buf[i]; - double y = b0 * x + z1; - z1 = b1 * x - a1 * y + z2; - z2 = b2 * x - a2 * y; - buf[i] = (float) y; - } - } -} diff --git a/ts3-client/core/src/main/java/com/ts3client/audio/NoiseSuppressor.java b/ts3-client/core/src/main/java/com/ts3client/audio/NoiseSuppressor.java deleted file mode 100644 index d6986e1..0000000 --- a/ts3-client/core/src/main/java/com/ts3client/audio/NoiseSuppressor.java +++ /dev/null @@ -1,112 +0,0 @@ -package com.ts3client.audio; - -import java.util.Arrays; - -/** - * Single-channel spectral noise suppressor — the "Remove background noise" - * (denoise) stage. The TeamSpeak 3 client filters steady background noise with - * WebRTC's {@code noise_suppression} module (and a Speex denoiser fallback); this is - * a self-contained equivalent that operates on the STFT bins produced by - * {@link AudioEnhancer}. - * - *

The noise floor is tracked per bin by continuous minimum statistics (Doblinger's - * recursive minimum tracker): the estimate follows the valleys of the smoothed power - * spectrum, so modulated speech — which dips between syllables — is - * preserved while only near-stationary background energy is learned as noise. From - * that floor a Wiener gain is formed with a decision-directed a priori SNR - * (Ephraim–Malah smoothing, which keeps musical noise low). A configurable - * aggressiveness ({@code denoiser_level}, 0–1) sets both the over-subtraction - * factor and the gain floor, i.e. how deeply steady noise is cut. - */ -final class NoiseSuppressor { - - /** Power-spectrum smoothing feeding the minimum tracker. */ - private static final double POWER_SMOOTH = 0.7; - /** Doblinger minimum-tracker constants. */ - private static final double MIN_GAMMA = 0.998; - private static final double MIN_BETA = 0.96; - /** Over-estimation applied to the tracked minimum to get the noise power. */ - private static final double NOISE_OVEREST = 1.5; - /** Decision-directed smoothing of the a priori SNR (higher = less musical noise). */ - private static final double DD_ALPHA = 0.98; - /** Floor on the a priori SNR (~ -25 dB) to bound the deepest Wiener gain. */ - private static final double XI_MIN = 0.003; - - private final int bins; - private final double[] smoothed; - private final double[] prevSmoothed; - private final double[] minTrack; - private final double[] priorClean; // previous enhanced power, for the DD estimate - - private double overSubtraction = 1.5; - private double gainFloor = dbToGain(-18); - private boolean initialised; - - NoiseSuppressor(int bins) { - this.bins = bins; - this.smoothed = new double[bins]; - this.prevSmoothed = new double[bins]; - this.minTrack = new double[bins]; - this.priorClean = new double[bins]; - } - - /** - * Sets aggressiveness in [0,1]. 0 is a light touch (~6 dB max cut), 1 is - * heavy (~30 dB) with stronger over-subtraction. - */ - void setLevel(double level) { - double l = Math.max(0, Math.min(1, level)); - this.gainFloor = dbToGain(-(6 + 24 * l)); - this.overSubtraction = 1.0 + 1.5 * l; - } - - void reset() { - initialised = false; - Arrays.fill(smoothed, 0); - Arrays.fill(prevSmoothed, 0); - Arrays.fill(minTrack, 0); - Arrays.fill(priorClean, 0); - } - - /** Multiplies the running per-bin gain by this stage's Wiener gain. */ - void apply(double[] power, double[] gain) { - if (!initialised) { - for (int k = 0; k < bins; k++) { - smoothed[k] = prevSmoothed[k] = minTrack[k] = power[k]; - priorClean[k] = power[k]; - } - initialised = true; - } - for (int k = 0; k < bins; k++) { - double p = power[k] + 1e-12; - - double s = POWER_SMOOTH * smoothed[k] + (1 - POWER_SMOOTH) * p; - // Doblinger continuous minimum tracking of the smoothed power. - double mt; - if (minTrack[k] < s) { - mt = MIN_GAMMA * minTrack[k] - + ((1 - MIN_GAMMA) / (1 - MIN_BETA)) * (s - MIN_BETA * smoothed[k]); - } else { - mt = s; - } - minTrack[k] = mt; - smoothed[k] = s; - double noiseK = NOISE_OVEREST * mt + 1e-12; - - double gamma = p / (noiseK * overSubtraction); // a posteriori SNR - double xi = DD_ALPHA * (priorClean[k] / noiseK) - + (1 - DD_ALPHA) * Math.max(gamma - 1, 0); // a priori SNR - if (xi < XI_MIN) xi = XI_MIN; - - double g = xi / (1 + xi); // Wiener gain - if (g < gainFloor) g = gainFloor; - - priorClean[k] = g * g * p; - gain[k] *= g; - } - } - - private static double dbToGain(double db) { - return Math.pow(10.0, db / 20.0); - } -} diff --git a/ts3-client/core/src/main/java/com/ts3client/audio/TypingAttenuator.java b/ts3-client/core/src/main/java/com/ts3client/audio/TypingAttenuator.java deleted file mode 100644 index 70d7963..0000000 --- a/ts3-client/core/src/main/java/com/ts3client/audio/TypingAttenuator.java +++ /dev/null @@ -1,78 +0,0 @@ -package com.ts3client.audio; - -/** - * Transient (keystroke) suppressor — the "Typing attenuation" stage, which per - * the TeamSpeak 3 client "tries to detect and reduce the sounds made by typing" - * (WebRTC's {@code transient_suppression} module). Key clicks are short, impulsive, - * broadband bursts with a strong high-frequency component, unlike voiced speech which - * is sustained and low-frequency dominant. - * - *

Each STFT block is scored for a keystroke signature: a sudden jump in total - * power (both against the previous block and a slow running floor) together with an - * elevated high-frequency energy ratio. Matching blocks are ducked broadband with an - * immediate attack and a short release. A hold cap ensures only genuinely brief - * events are cut — a sustained sound such as a fricative outlasts the cap and is - * released, so speech is preserved. - */ -final class TypingAttenuator { - - private static final double HF_HZ = 4000.0; // high-frequency band start - private static final double ONSET_FACTOR = 2.5; // total power vs slow floor - private static final double FLUX_FACTOR = 3.0; // total power vs previous block - private static final double HF_RATIO = 0.30; // fraction of energy above HF_HZ - private static final double SUPPRESS = 0.12; // ducking gain on a detected click (~ -18 dB) - private static final double RELEASE = 0.25; // recovery fraction per block after a click - private static final double FLOOR_SMOOTH = 0.98; // slow power-floor tracking - private static final int MAX_HOLD = 4; // max consecutive ducked blocks (~clicks only) - - private final int bins; - private final int hfBin; - - private double slowPower; - private double prevPower; - private double envGain = 1.0; - private int heldBlocks; - - TypingAttenuator(int bins, int sampleRate, int fftSize) { - this.bins = bins; - this.hfBin = (int) Math.round(HF_HZ * fftSize / sampleRate); - } - - void reset() { - slowPower = 0; - prevPower = 0; - envGain = 1.0; - heldBlocks = 0; - } - - /** Multiplies the running per-bin gain by the current broadband ducking gain. */ - void apply(double[] power, double[] gain) { - double total = 0, high = 0; - for (int k = 0; k < bins; k++) { - total += power[k]; - if (k >= hfBin) high += power[k]; - } - double hfRatio = high / (total + 1e-12); - - boolean signature = total > slowPower * ONSET_FACTOR - && total > prevPower * FLUX_FACTOR - && hfRatio > HF_RATIO; - - if (signature && heldBlocks < MAX_HOLD) { - envGain = SUPPRESS; // fast attack: duck immediately - heldBlocks++; - } else { - envGain += (1 - envGain) * RELEASE; - if (!signature) { - heldBlocks = 0; - // Only let the floor track when we're not inside a transient. - slowPower = slowPower == 0 ? total : FLOOR_SMOOTH * slowPower + (1 - FLOOR_SMOOTH) * total; - } - } - prevPower = total; - - for (int k = 0; k < bins; k++) { - gain[k] *= envGain; - } - } -} diff --git a/ts3-client/core/src/main/java/com/ts3client/audio/VoiceInput.java b/ts3-client/core/src/main/java/com/ts3client/audio/VoiceInput.java index 3566de7..aae9830 100644 --- a/ts3-client/core/src/main/java/com/ts3client/audio/VoiceInput.java +++ b/ts3-client/core/src/main/java/com/ts3client/audio/VoiceInput.java @@ -42,15 +42,18 @@ public interface VoiceInput extends Microphone { void setInputGain(double gain); - /** Enables removal of steady background noise (spectral denoise). */ + /** Enables removal of steady background noise (WebRTC noise suppression). */ void setNoiseSuppression(boolean enabled); - /** Background-noise removal aggressiveness, 0 (light) .. 1 (heavy). */ - void setDenoiserLevel(double level); + /** Noise suppression strength as TS3's {@code denoiser_level}: 0 (6 dB) .. 3 (21 dB). */ + void setDenoiserLevel(int level); - /** Enables detection and attenuation of keyboard typing sounds. */ + /** Enables attenuation of keyboard clicks while the user is typing. */ void setTypingAttenuation(boolean enabled); + /** Reports a key press anywhere on the system; typing attenuation only acts while typing. */ + void keyPressed(); + /** Enables automatic gain control (normalise microphone loudness). */ void setAgc(boolean enabled); diff --git a/ts3-client/core/src/main/java/com/ts3client/config/Settings.java b/ts3-client/core/src/main/java/com/ts3client/config/Settings.java index 6399d26..891a563 100644 --- a/ts3-client/core/src/main/java/com/ts3client/config/Settings.java +++ b/ts3-client/core/src/main/java/com/ts3client/config/Settings.java @@ -130,11 +130,11 @@ public final class Settings { public double masterVolume = 1.0; /** Microphone input gain multiplier applied before VAD/encode. */ public double inputVolume = 1.0; - /** Remove steady background noise from the microphone (spectral denoise). */ + /** Remove steady background noise from the microphone (WebRTC noise suppression). */ public boolean denoise = true; - /** Background-noise removal aggressiveness, 0 (light) .. 1 (heavy). */ - public double denoiserLevel = 0.5; - /** Detect and attenuate keyboard typing sounds in the microphone. */ + /** TS3's {@code denoiser_level}: 0..3 for 6, 12, 18 or 21 dB of suppression. */ + public int denoiserLevel = 1; + /** Attenuate keyboard clicks while typing. */ public boolean typingAttenuation = true; /** Automatic gain control: normalise microphone loudness to a target level. */ public boolean agc = true; @@ -258,7 +258,7 @@ public final class Settings { masterVolume = parseD(props.getProperty("masterVolume"), masterVolume); inputVolume = parseD(props.getProperty("inputVolume"), inputVolume); denoise = parseB(props.getProperty("denoise"), denoise); - denoiserLevel = parseD(props.getProperty("denoiserLevel"), denoiserLevel); + denoiserLevel = parseI(props.getProperty("denoiserLevel"), denoiserLevel); typingAttenuation = parseB(props.getProperty("typingAttenuation"), typingAttenuation); mutedTalkWarning = parseB(props.getProperty("mutedTalkWarning"), mutedTalkWarning); agc = parseB(props.getProperty("agc"), agc); @@ -308,7 +308,7 @@ public final class Settings { props.setProperty("masterVolume", Double.toString(masterVolume)); props.setProperty("inputVolume", Double.toString(inputVolume)); props.setProperty("denoise", Boolean.toString(denoise)); - props.setProperty("denoiserLevel", Double.toString(denoiserLevel)); + props.setProperty("denoiserLevel", Integer.toString(denoiserLevel)); props.setProperty("typingAttenuation", Boolean.toString(typingAttenuation)); props.setProperty("mutedTalkWarning", Boolean.toString(mutedTalkWarning)); props.setProperty("agc", Boolean.toString(agc)); diff --git a/ts3-client/desktop/src/main/java/com/ts3client/audio/desktop/DesktopVoiceInput.java b/ts3-client/desktop/src/main/java/com/ts3client/audio/desktop/DesktopVoiceInput.java index 96eacfc..ce5330d 100644 --- a/ts3-client/desktop/src/main/java/com/ts3client/audio/desktop/DesktopVoiceInput.java +++ b/ts3-client/desktop/src/main/java/com/ts3client/audio/desktop/DesktopVoiceInput.java @@ -1,12 +1,13 @@ package com.ts3client.audio.desktop; import com.github.manevolent.ts3j.enums.CodecType; -import com.ts3client.audio.AudioEnhancer; import com.ts3client.audio.AudioFrameListener; import com.ts3client.audio.InputLevel; import com.ts3client.audio.OpusParameters; import com.ts3client.audio.SpeechProbabilityDetector; import com.ts3client.audio.VoiceInput; +import com.ts3client.audio.processing.AudioProcessor; +import com.ts3client.audio.processing.ns.SuppressionLevel; import com.ts3client.audio.vad.RnnSpeechDetector; import com.ts3client.config.Settings; @@ -66,7 +67,7 @@ public final class DesktopVoiceInput implements VoiceInput { private final SpeechProbabilityDetector speechDetector = new RnnSpeechDetector(AudioDevices.SAMPLE_RATE); - private final AudioEnhancer enhancer = new AudioEnhancer(AudioDevices.SAMPLE_RATE); + private final AudioProcessor processor = new AudioProcessor(); private volatile Consumer levelListener; // input level, InputLevel scale private volatile Consumer talkListener; // local talk-state changes @@ -105,10 +106,10 @@ public final class DesktopVoiceInput implements VoiceInput { this.inputGain = settings.inputVolume; this.params = OpusParameters.from(settings); this.codec = codecFor(params); - enhancer.setNoiseSuppression(settings.denoise); - enhancer.setDenoiserLevel(settings.denoiserLevel); - enhancer.setTypingAttenuation(settings.typingAttenuation); - enhancer.setAgc(settings.agc); + processor.setNoiseSuppression(settings.denoise); + processor.setSuppressionLevel(SuppressionLevel.fromDenoiserLevel(settings.denoiserLevel)); + processor.setTransientSuppression(settings.typingAttenuation); + processor.setGainControl(settings.agc); } // ---- live configuration (safe to call from the UI thread) ---- @@ -139,19 +140,24 @@ public final class DesktopVoiceInput implements VoiceInput { } public void setNoiseSuppression(boolean enabled) { - enhancer.setNoiseSuppression(enabled); + processor.setNoiseSuppression(enabled); } - public void setDenoiserLevel(double level) { - enhancer.setDenoiserLevel(level); + public void setDenoiserLevel(int level) { + processor.setSuppressionLevel(SuppressionLevel.fromDenoiserLevel(level)); } public void setTypingAttenuation(boolean enabled) { - enhancer.setTypingAttenuation(enabled); + processor.setTransientSuppression(enabled); } public void setAgc(boolean enabled) { - enhancer.setAgc(enabled); + processor.setGainControl(enabled); + } + + @Override + public void keyPressed() { + processor.keyPressed(); } /** @@ -265,7 +271,7 @@ public final class DesktopVoiceInput implements VoiceInput { throw new RuntimeException("Could not start microphone: " + t.getMessage(), t); } speechDetector.reset(); - enhancer.reset(); + processor.reset(); running.set(true); captureThread = new Thread(this::captureLoop, "ts3j-mic-capture"); captureThread.setDaemon(true); @@ -342,17 +348,21 @@ public final class DesktopVoiceInput implements VoiceInput { if (p != appliedParams && encoder != null) applyParams(p); boolean stereo = encoderChannels == 2; - // Denoise / typing attenuation are voice-chain stages: they feed the - // level meter, VAD and encoder alike on the mono path. A stereo (music) - // stream bypasses them and is transmitted as captured. - if (!stereo) enhancer.process(mono, frameSamples); + // As in TS3, the speech detector judges the raw microphone signal, while + // the level meter and volume gate see the processed one. + double probability = usesSpeechDetector() + ? speechDetector.process(mono, frameSamples) : 0.0; + + // Processing is a voice-chain stage on the mono path. A stereo (music) + // stream bypasses it and is transmitted as captured. + if (!stereo) processor.process(mono, frameSamples); double db = InputLevel.toDb(mono, frameSamples); Consumer ll = levelListener; if (ll != null) ll.accept(db); boolean wasOpen = transmitting.get(); - boolean open = decideGate(db, mono); + boolean open = decideGate(db, probability); setTransmitting(open); if (open && !muted.get() && encoder != null) { @@ -415,7 +425,14 @@ public final class DesktopVoiceInput implements VoiceInput { prerollCount = 0; } - private boolean decideGate(double db, float[] pcm) { + /** Whether the gate may consult the speech detector, so it must follow the signal. */ + private boolean usesSpeechDetector() { + if (vadMode == Settings.VadMode.VOLUME_GATE) return false; + Settings.InputMode m = mode; + return m == Settings.InputMode.VOICE_ACTIVATION || (m == Settings.InputMode.PUSH_TO_TALK && vadOverPtt); + } + + private boolean decideGate(double db, double probability) { if (muted.get() || localMuted.get()) { hangover = 0; detectMutedSpeech(db); @@ -427,10 +444,10 @@ public final class DesktopVoiceInput implements VoiceInput { return true; case PUSH_TO_TALK: if (pttDown.get()) return true; - return vadOverPtt && voiceActivated(db, pcm); + return vadOverPtt && voiceActivated(db, probability); case VOICE_ACTIVATION: default: - return voiceActivated(db, pcm); + return voiceActivated(db, probability); } } @@ -440,21 +457,12 @@ public final class DesktopVoiceInput implements VoiceInput { *

The modes match the TS3 client's: the volume gate alone, the speech detector * alone, or both together. Volume Gate skips the detector entirely, as TS3 does. */ - private boolean voiceActivated(double db, float[] pcm) { - boolean detected; - switch (vadMode) { - case VOLUME_GATE: - detected = db >= thresholdDb; - break; - case AUTOMATIC: - detected = speechDetector.process(pcm, AudioDevices.FRAME_SIZE) >= speechThreshold; - break; - case HYBRID: - default: - double probability = speechDetector.process(pcm, AudioDevices.FRAME_SIZE); - detected = db >= thresholdDb && probability >= speechThreshold; - break; - } + private boolean voiceActivated(double db, double probability) { + boolean detected = switch (vadMode) { + case VOLUME_GATE -> db >= thresholdDb; + case AUTOMATIC -> probability >= speechThreshold; + case HYBRID -> db >= thresholdDb && probability >= speechThreshold; + }; if (detected) { hangover = HANGOVER_FRAMES; return true; diff --git a/ts3-client/swing/src/main/java/com/ts3client/ui/DevicesPanel.java b/ts3-client/swing/src/main/java/com/ts3client/ui/DevicesPanel.java index c72f803..f920c79 100644 --- a/ts3-client/swing/src/main/java/com/ts3client/ui/DevicesPanel.java +++ b/ts3-client/swing/src/main/java/com/ts3client/ui/DevicesPanel.java @@ -84,9 +84,13 @@ final class DevicesPanel extends FormPanel { denoiseCheck = new JCheckBox("Remove background noise", settings.denoise); denoiseCheck.setToolTipText("Attempt to filter out background noises."); - denoiseLevel = new JSlider(0, 100, (int) Math.round(settings.denoiserLevel * 100)); + // TS3's four steps, 6/12/18/21 dB of suppression. + denoiseLevel = new JSlider(0, 3, Math.max(0, Math.min(3, settings.denoiserLevel))); + denoiseLevel.setMajorTickSpacing(1); + denoiseLevel.setPaintTicks(true); + denoiseLevel.setSnapToTicks(true); limitWidth(denoiseLevel, SLIDER_WIDTH); - denoiseLevel.setToolTipText("Higher = more aggressive noise removal."); + denoiseLevel.setToolTipText("How much steady noise to remove: 6, 12, 18 or 21 dB."); typingCheck = new JCheckBox("Typing attenuation", settings.typingAttenuation); typingCheck.setToolTipText("Typing attenuation tries to detect and " + "reduce the sounds made by typing."); @@ -113,7 +117,7 @@ final class DevicesPanel extends FormPanel { denoiseLevel.setEnabled(denoiseCheck.isSelected()); applyLive.accept(m -> { m.setNoiseSuppression(denoiseCheck.isSelected()); - m.setDenoiserLevel(denoiseLevel.getValue() / 100.0); + m.setDenoiserLevel(denoiseLevel.getValue()); m.setTypingAttenuation(typingCheck.isSelected()); m.setAgc(agcCheck.isSelected()); }); @@ -122,7 +126,7 @@ final class DevicesPanel extends FormPanel { typingCheck.addActionListener(e -> syncNoise.run()); agcCheck.addActionListener(e -> syncNoise.run()); denoiseLevel.addChangeListener(e -> - applyLive.accept(m -> m.setDenoiserLevel(denoiseLevel.getValue() / 100.0))); + applyLive.accept(m -> m.setDenoiserLevel(denoiseLevel.getValue()))); syncNoise.run(); mutedWarningCheck = new JCheckBox("Beep when talking while muted", settings.mutedTalkWarning); @@ -148,7 +152,7 @@ final class DevicesPanel extends FormPanel { target.inputVolume = inputGain.getValue() / 100.0; target.outputVolume = outputVol.getValue() / 100.0; target.denoise = denoiseCheck.isSelected(); - target.denoiserLevel = denoiseLevel.getValue() / 100.0; + target.denoiserLevel = denoiseLevel.getValue(); target.typingAttenuation = typingCheck.isSelected(); target.agc = agcCheck.isSelected(); target.mutedTalkWarning = mutedWarningCheck.isSelected(); diff --git a/ts3-client/swing/src/main/java/com/ts3client/ui/HotkeyService.java b/ts3-client/swing/src/main/java/com/ts3client/ui/HotkeyService.java index 10ae8e6..c2d9680 100644 --- a/ts3-client/swing/src/main/java/com/ts3client/ui/HotkeyService.java +++ b/ts3-client/swing/src/main/java/com/ts3client/ui/HotkeyService.java @@ -10,6 +10,7 @@ import com.ts3client.hotkey.Hotkeys; import com.ts3client.hotkey.desktop.DesktopInputHooks; import java.util.List; +import java.util.concurrent.CopyOnWriteArrayList; /** * Owns the hotkey machinery for the UI: the stored bindings, the matching engine and @@ -22,11 +23,29 @@ final class HotkeyService { private final HotkeyEngine engine; private final GlobalInputHook hook; private final HotkeyArguments arguments; + private final List keyPressListeners = new CopyOnWriteArrayList<>(); HotkeyService(HotkeyEngine.Handler handler, HotkeyArguments arguments) { this.arguments = arguments; engine = new HotkeyEngine(hotkeys, handler); - hook = DesktopInputHooks.start(engine); + hook = DesktopInputHooks.start((key, pressed) -> { + engine.onInput(key, pressed); + if (pressed && key.device() == HotkeyKey.Device.KEYBOARD) { + for (Runnable l : keyPressListeners) l.run(); + } + }); + } + + /** + * Runs {@code l} on the input hook's thread for every key going down anywhere on the + * system, which is what typing attenuation listens for. Keep it cheap. + */ + void addKeyPressListener(Runnable l) { + keyPressListeners.add(l); + } + + void removeKeyPressListener(Runnable l) { + keyPressListeners.remove(l); } /** The values this action's parameter can take right now, for the action tree. */ diff --git a/ts3-client/swing/src/main/java/com/ts3client/ui/MainFrame.java b/ts3-client/swing/src/main/java/com/ts3client/ui/MainFrame.java index b26350b..49785fd 100644 --- a/ts3-client/swing/src/main/java/com/ts3client/ui/MainFrame.java +++ b/ts3-client/swing/src/main/java/com/ts3client/ui/MainFrame.java @@ -95,6 +95,7 @@ public final class MainFrame extends JFrame implements ServerTabPane.Listener { this.soundPlayer = audio.createSoundPlayer(settings); this.sounds.setPlayer(soundPlayer); this.hotkeys = new HotkeyService(new HotkeyActions(this), this::hotkeyArgumentChoices); + hotkeys.addKeyPressListener(() -> SwingUtilities.invokeLater(this::keyPressed)); contacts.addListener(() -> SwingUtilities.invokeLater(this::contactsChanged)); setIconImages(Icons.appImages()); @@ -482,6 +483,15 @@ public final class MainFrame extends JFrame implements ServerTabPane.Listener { tab.connection().getMicrophone().setPushToTalk(talking); } + /** Tells every connected microphone about a key press, as TS3 tells all its connections. */ + private void keyPressed() { + for (ServerTab tab : tabs) { + if (tab.isConnected() && tab.connection().getMicrophone() != null) { + tab.connection().getMicrophone().keyPressed(); + } + } + } + void moveMicrophoneToSelectedTab() { if (selected != null && selected.isConnected()) setMicTab(selected); updateToolbar(); diff --git a/ts3-client/swing/src/main/java/com/ts3client/ui/SettingsDialog.java b/ts3-client/swing/src/main/java/com/ts3client/ui/SettingsDialog.java index 0039713..5370c06 100644 --- a/ts3-client/swing/src/main/java/com/ts3client/ui/SettingsDialog.java +++ b/ts3-client/swing/src/main/java/com/ts3client/ui/SettingsDialog.java @@ -10,6 +10,7 @@ import com.ts3client.sound.SoundNotifier; import javax.swing.JButton; import javax.swing.JDialog; import javax.swing.JPanel; +import javax.swing.SwingUtilities; import javax.swing.JTabbedPane; import java.awt.BorderLayout; import java.awt.Dimension; @@ -83,11 +84,17 @@ public final class SettingsDialog extends JDialog { getContentPane().add(tabs, BorderLayout.CENTER); getContentPane().add(buttons, BorderLayout.SOUTH); + // The microphone test hears typing too, so typing attenuation can be tried out here. + Runnable testKeyPress = () -> SwingUtilities.invokeLater( + () -> voiceActivationPanel.configureTest(VoiceInput::keyPressed)); + hotkeys.addKeyPressListener(testKeyPress); + Dialogs.closeOnEscape(this, this::cancel); setDefaultCloseOperation(DISPOSE_ON_CLOSE); addWindowListener(new java.awt.event.WindowAdapter() { @Override public void windowClosed(java.awt.event.WindowEvent e) { + hotkeys.removeKeyPressListener(testKeyPress); voiceActivationPanel.stopTest(); } });