Process the microphone like TS3, and tell it about key presses

The capture path now runs the ported WebRTC chain in place of the
home-made noise, typing and gain stages, which are removed.

- As in TS3, the speech detector judges the raw microphone signal, while
  the level meter and volume gate see the processed one.
- Noise removal takes TS3's four levels (6, 12, 18 or 21 dB); a stored
  0..1 level falls back to TS3's default of 12 dB.
- Every key press from the global input hook reaches the connected
  microphones and the microphone test, so typing attenuation engages
  while the user types.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-24 20:47:04 +00:00
parent de5f7da236
commit 1d61f458ae
12 changed files with 102 additions and 588 deletions

View File

@@ -1,238 +0,0 @@
package com.ts3client.audio;
import java.util.Arrays;
/**
* Microphone pre-processing chain applied before voice activation and Opus encoding,
* mirroring the capture-side denoise/typing filters of the TeamSpeak&nbsp;3 client.
*
* <p>The stages run in the same order as the TeamSpeak client's WebRTC capture chain:
* a {@link HighPassFilter} (always-on rumble/DC removal), then a streaming short-time
* Fourier transform (square-root Hann window, 50% overlap-add) carrying a
* {@link NoiseSuppressor} ("Remove background noise") and a {@link TypingAttenuator}
* ("Typing attenuation") &mdash; sharing one FFT/IFFT per hop &mdash; and finally an
* {@link AutomaticGainControl} ("AGC"). Echo cancellation (WebRTC AEC3) is omitted as
* it requires the loudspeaker reference signal.
*
* <p>Input frames of any length are decoupled from the STFT hop by internal ring
* buffers; when noise/typing suppression is active the output is delayed by one hop
* (~5&nbsp;ms). The high-pass filter and AGC are zero-latency. When every stage is
* disabled the chain is fully bypassed and audio passes through untouched.
*
* <p>Pure DSP with no platform dependencies, so any frontend/backend can reuse it.
* Not thread-safe: drive it from a single capture thread; the enable/level setters
* are cheap volatiles safe to call from the UI thread.
*/
public final class AudioEnhancer {
private static final int FFT_SIZE = 512; // power of two -> 10.7 ms @ 48 kHz
private static final int HOP = FFT_SIZE / 2; // 50% overlap
private static final int BINS = FFT_SIZE / 2 + 1;
private final double[] window = new double[FFT_SIZE];
private final double[] re = new double[FFT_SIZE];
private final double[] im = new double[FFT_SIZE];
private final double[] power = new double[BINS];
private final double[] gain = new double[BINS];
private final double[] frame = new double[FFT_SIZE]; // sliding analysis frame
private final double[] ola = new double[FFT_SIZE]; // overlap-add accumulator
private final FloatRing input = new FloatRing(FFT_SIZE * 4);
private final FloatRing output = new FloatRing(FFT_SIZE * 4);
private final float[] hopIn = new float[HOP];
private final NoiseSuppressor noiseSuppressor = new NoiseSuppressor(BINS);
private final TypingAttenuator typingAttenuator;
private final HighPassFilter highPass;
private final AutomaticGainControl agc;
private volatile boolean noiseEnabled;
private volatile boolean typingEnabled;
private volatile boolean agcEnabled;
private boolean active; // any stage on: HPF + AGC state is live
private boolean stftRunning; // noise/typing on: STFT rings are live
public AudioEnhancer(int sampleRate) {
for (int i = 0; i < FFT_SIZE; i++) {
// sqrt(Hann): analysis*synthesis = Hann, which is COLA at 50% overlap.
window[i] = Math.sqrt(0.5 * (1 - Math.cos(2 * Math.PI * i / FFT_SIZE)));
}
this.typingAttenuator = new TypingAttenuator(BINS, sampleRate, FFT_SIZE);
this.highPass = new HighPassFilter(sampleRate);
this.agc = new AutomaticGainControl(sampleRate);
}
public void setNoiseSuppression(boolean enabled) {
this.noiseEnabled = enabled;
}
public void setDenoiserLevel(double level) {
noiseSuppressor.setLevel(level);
}
public void setTypingAttenuation(boolean enabled) {
this.typingEnabled = enabled;
}
public void setAgc(boolean enabled) {
this.agcEnabled = enabled;
}
/** Clears all filter state; call when (re)starting capture. */
public void reset() {
resetStft();
highPass.reset();
agc.reset();
active = false;
stftRunning = false;
}
private void resetStft() {
input.clear();
output.clear();
Arrays.fill(frame, 0);
Arrays.fill(ola, 0);
noiseSuppressor.reset();
typingAttenuator.reset();
}
/**
* Enhances one frame of mono PCM in place. {@code buf[0..len)} is overwritten with
* the processed (one-hop-delayed when noise/typing suppression is on) signal.
* Returns immediately if every stage is disabled.
*/
public void process(float[] buf, int len) {
boolean stft = noiseEnabled || typingEnabled;
boolean anyStage = stft || agcEnabled;
if (!anyStage) {
if (active) reset(); // drop stale filter/delay state on full disable
return;
}
if (!active) {
reset();
active = true;
}
// 1) High-pass filter (always-on part of the active chain).
highPass.process(buf, len);
// 2) STFT noise + typing suppression (only when either is enabled).
if (stft) {
if (!stftRunning) {
resetStft();
stftRunning = true;
}
runStft(buf, len);
} else if (stftRunning) {
stftRunning = false;
}
// 3) Automatic gain control (last, on the cleaned signal).
if (agcEnabled) {
agc.process(buf, len);
}
}
private void runStft(float[] buf, int len) {
input.write(buf, len);
while (input.available() >= HOP) {
System.arraycopy(frame, HOP, frame, 0, FFT_SIZE - HOP);
input.read(hopIn, HOP);
for (int i = 0; i < HOP; i++) {
frame[FFT_SIZE - HOP + i] = hopIn[i];
}
processBlock();
}
// During the initial one-hop priming the output ring is short; pad with zeros.
int ready = output.available();
if (ready < len) {
for (int i = 0; i < len - ready; i++) buf[i] = 0f;
output.read(buf, len - ready, ready);
} else {
output.read(buf, 0, len);
}
}
private void processBlock() {
for (int i = 0; i < FFT_SIZE; i++) {
re[i] = frame[i] * window[i];
im[i] = 0;
}
Fft.forward(re, im);
for (int k = 0; k < BINS; k++) {
power[k] = re[k] * re[k] + im[k] * im[k];
gain[k] = 1.0;
}
if (noiseEnabled) noiseSuppressor.apply(power, gain);
if (typingEnabled) typingAttenuator.apply(power, gain);
// Apply the real-valued gain to each bin and its conjugate mirror.
for (int k = 0; k < BINS; k++) {
double g = gain[k];
re[k] *= g;
im[k] *= g;
if (k > 0 && k < FFT_SIZE - k) {
int m = FFT_SIZE - k;
re[m] *= g;
im[m] *= g;
}
}
Fft.inverse(re, im);
for (int i = 0; i < FFT_SIZE; i++) {
ola[i] += re[i] * window[i];
}
output.write(ola, HOP);
System.arraycopy(ola, HOP, ola, 0, FFT_SIZE - HOP);
Arrays.fill(ola, FFT_SIZE - HOP, FFT_SIZE, 0);
}
/** Minimal single-producer/single-consumer float ring buffer. */
private static final class FloatRing {
private final float[] buf;
private int head, tail, size;
FloatRing(int capacity) {
this.buf = new float[capacity];
}
int available() {
return size;
}
void clear() {
head = tail = size = 0;
}
void write(float[] src, int len) {
for (int i = 0; i < len; i++) {
buf[tail] = src[i];
tail = (tail + 1) % buf.length;
}
size += len;
}
void write(double[] src, int len) {
for (int i = 0; i < len; i++) {
buf[tail] = (float) src[i];
tail = (tail + 1) % buf.length;
}
size += len;
}
void read(float[] dst, int len) {
read(dst, 0, len);
}
void read(float[] dst, int offset, int len) {
for (int i = 0; i < len; i++) {
dst[offset + i] = buf[head];
head = (head + 1) % buf.length;
}
size -= len;
}
}
}

View File

@@ -1,66 +0,0 @@
package com.ts3client.audio;
/**
* Automatic gain control &mdash; WebRTC APM's {@code gain_controller} (AGC2 adaptive
* digital) / Speex {@code AGC} stage. It normalises voice loudness toward a target
* level so quiet microphones are boosted and loud ones tamed, keeping perceived volume
* consistent across speakers.
*
* <p>Placed last in the capture chain (after noise suppression), it tracks the frame
* level and moves an applied gain toward {@code target / level}: it attenuates quickly
* to head off clipping and boosts slowly to avoid pumping. A noise gate freezes the
* gain while the input is near silence, so background noise between words is never
* amplified; the per-sample gain ramp avoids zipper artefacts and a final clamp guards
* against overshoot.
*/
final class AutomaticGainControl {
private static final double TARGET_RMS = 0.12; // ~ -18.4 dBFS
private static final double MAX_GAIN = dbToGain(30); // up to +30 dB boost
private static final double MIN_GAIN = dbToGain(-20); // down to -20 dB
private static final double NOISE_GATE_RMS = dbToGain(-55); // freeze below this level
private final double attackCoeff; // gain decreasing (signal too loud): fast
private final double releaseCoeff; // gain increasing (too quiet): slow
private double gain = 1.0;
AutomaticGainControl(int sampleRate) {
this.attackCoeff = 1 - Math.exp(-1.0 / (0.005 * sampleRate)); // ~5 ms
this.releaseCoeff = 1 - Math.exp(-1.0 / (0.300 * sampleRate)); // ~300 ms
}
void reset() {
gain = 1.0;
}
/** Applies gain normalisation to one mono frame in place. */
void process(float[] buf, int len) {
double sumSq = 0;
for (int i = 0; i < len; i++) {
sumSq += (double) buf[i] * buf[i];
}
double rms = Math.sqrt(sumSq / len);
double desired = gain;
if (rms >= NOISE_GATE_RMS) {
desired = TARGET_RMS / rms;
if (desired > MAX_GAIN) desired = MAX_GAIN;
else if (desired < MIN_GAIN) desired = MIN_GAIN;
}
// Boost slowly, attenuate quickly.
double coeff = desired < gain ? attackCoeff : releaseCoeff;
for (int i = 0; i < len; i++) {
gain += (desired - gain) * coeff;
double y = buf[i] * gain;
if (y > 1.0) y = 1.0;
else if (y < -1.0) y = -1.0;
buf[i] = (float) y;
}
}
private static double dbToGain(double db) {
return Math.pow(10.0, db / 20.0);
}
}

View File

@@ -1,43 +0,0 @@
package com.ts3client.audio;
/**
* Second-order Butterworth high-pass filter (RBJ biquad, transposed direct form II).
* Mirrors WebRTC APM's {@code high_pass_filter} stage &mdash; an always-on part of the
* capture chain that removes DC offset, mains hum and low-frequency rumble below the
* speech band before noise suppression sees the signal.
*/
final class HighPassFilter {
private static final double CUTOFF_HZ = 80.0; // WebRTC APM high-pass cutoff
private final double b0, b1, b2, a1, a2;
private double z1, z2;
HighPassFilter(int sampleRate) {
double w0 = 2 * Math.PI * CUTOFF_HZ / sampleRate;
double cos = Math.cos(w0);
double alpha = Math.sin(w0) / Math.sqrt(2.0); // Q = 1/sqrt(2) (Butterworth)
double a0 = 1 + alpha;
this.b0 = (1 + cos) / 2 / a0;
this.b1 = -(1 + cos) / a0;
this.b2 = (1 + cos) / 2 / a0;
this.a1 = -2 * cos / a0;
this.a2 = (1 - alpha) / a0;
}
void reset() {
z1 = 0;
z2 = 0;
}
/** Filters one mono frame in place. */
void process(float[] buf, int len) {
for (int i = 0; i < len; i++) {
double x = buf[i];
double y = b0 * x + z1;
z1 = b1 * x - a1 * y + z2;
z2 = b2 * x - a2 * y;
buf[i] = (float) y;
}
}
}

View File

@@ -1,112 +0,0 @@
package com.ts3client.audio;
import java.util.Arrays;
/**
* Single-channel spectral noise suppressor &mdash; the "Remove background noise"
* (denoise) stage. The TeamSpeak&nbsp;3 client filters steady background noise with
* WebRTC's {@code noise_suppression} module (and a Speex denoiser fallback); this is
* a self-contained equivalent that operates on the STFT bins produced by
* {@link AudioEnhancer}.
*
* <p>The noise floor is tracked per bin by continuous minimum statistics (Doblinger's
* recursive minimum tracker): the estimate follows the valleys of the smoothed power
* spectrum, so modulated speech &mdash; which dips between syllables &mdash; is
* preserved while only near-stationary background energy is learned as noise. From
* that floor a Wiener gain is formed with a decision-directed a&nbsp;priori SNR
* (Ephraim&ndash;Malah smoothing, which keeps musical noise low). A configurable
* aggressiveness ({@code denoiser_level}, 0&ndash;1) sets both the over-subtraction
* factor and the gain floor, i.e. how deeply steady noise is cut.
*/
final class NoiseSuppressor {
/** Power-spectrum smoothing feeding the minimum tracker. */
private static final double POWER_SMOOTH = 0.7;
/** Doblinger minimum-tracker constants. */
private static final double MIN_GAMMA = 0.998;
private static final double MIN_BETA = 0.96;
/** Over-estimation applied to the tracked minimum to get the noise power. */
private static final double NOISE_OVEREST = 1.5;
/** Decision-directed smoothing of the a priori SNR (higher = less musical noise). */
private static final double DD_ALPHA = 0.98;
/** Floor on the a priori SNR (~ -25 dB) to bound the deepest Wiener gain. */
private static final double XI_MIN = 0.003;
private final int bins;
private final double[] smoothed;
private final double[] prevSmoothed;
private final double[] minTrack;
private final double[] priorClean; // previous enhanced power, for the DD estimate
private double overSubtraction = 1.5;
private double gainFloor = dbToGain(-18);
private boolean initialised;
NoiseSuppressor(int bins) {
this.bins = bins;
this.smoothed = new double[bins];
this.prevSmoothed = new double[bins];
this.minTrack = new double[bins];
this.priorClean = new double[bins];
}
/**
* Sets aggressiveness in [0,1]. 0 is a light touch (~6&nbsp;dB max cut), 1 is
* heavy (~30&nbsp;dB) with stronger over-subtraction.
*/
void setLevel(double level) {
double l = Math.max(0, Math.min(1, level));
this.gainFloor = dbToGain(-(6 + 24 * l));
this.overSubtraction = 1.0 + 1.5 * l;
}
void reset() {
initialised = false;
Arrays.fill(smoothed, 0);
Arrays.fill(prevSmoothed, 0);
Arrays.fill(minTrack, 0);
Arrays.fill(priorClean, 0);
}
/** Multiplies the running per-bin gain by this stage's Wiener gain. */
void apply(double[] power, double[] gain) {
if (!initialised) {
for (int k = 0; k < bins; k++) {
smoothed[k] = prevSmoothed[k] = minTrack[k] = power[k];
priorClean[k] = power[k];
}
initialised = true;
}
for (int k = 0; k < bins; k++) {
double p = power[k] + 1e-12;
double s = POWER_SMOOTH * smoothed[k] + (1 - POWER_SMOOTH) * p;
// Doblinger continuous minimum tracking of the smoothed power.
double mt;
if (minTrack[k] < s) {
mt = MIN_GAMMA * minTrack[k]
+ ((1 - MIN_GAMMA) / (1 - MIN_BETA)) * (s - MIN_BETA * smoothed[k]);
} else {
mt = s;
}
minTrack[k] = mt;
smoothed[k] = s;
double noiseK = NOISE_OVEREST * mt + 1e-12;
double gamma = p / (noiseK * overSubtraction); // a posteriori SNR
double xi = DD_ALPHA * (priorClean[k] / noiseK)
+ (1 - DD_ALPHA) * Math.max(gamma - 1, 0); // a priori SNR
if (xi < XI_MIN) xi = XI_MIN;
double g = xi / (1 + xi); // Wiener gain
if (g < gainFloor) g = gainFloor;
priorClean[k] = g * g * p;
gain[k] *= g;
}
}
private static double dbToGain(double db) {
return Math.pow(10.0, db / 20.0);
}
}

View File

@@ -1,78 +0,0 @@
package com.ts3client.audio;
/**
* Transient (keystroke) suppressor &mdash; the "Typing attenuation" stage, which per
* the TeamSpeak&nbsp;3 client "tries to detect and reduce the sounds made by typing"
* (WebRTC's {@code transient_suppression} module). Key clicks are short, impulsive,
* broadband bursts with a strong high-frequency component, unlike voiced speech which
* is sustained and low-frequency dominant.
*
* <p>Each STFT block is scored for a keystroke signature: a sudden jump in total
* power (both against the previous block and a slow running floor) together with an
* elevated high-frequency energy ratio. Matching blocks are ducked broadband with an
* immediate attack and a short release. A hold cap ensures only genuinely brief
* events are cut &mdash; a sustained sound such as a fricative outlasts the cap and is
* released, so speech is preserved.
*/
final class TypingAttenuator {
private static final double HF_HZ = 4000.0; // high-frequency band start
private static final double ONSET_FACTOR = 2.5; // total power vs slow floor
private static final double FLUX_FACTOR = 3.0; // total power vs previous block
private static final double HF_RATIO = 0.30; // fraction of energy above HF_HZ
private static final double SUPPRESS = 0.12; // ducking gain on a detected click (~ -18 dB)
private static final double RELEASE = 0.25; // recovery fraction per block after a click
private static final double FLOOR_SMOOTH = 0.98; // slow power-floor tracking
private static final int MAX_HOLD = 4; // max consecutive ducked blocks (~clicks only)
private final int bins;
private final int hfBin;
private double slowPower;
private double prevPower;
private double envGain = 1.0;
private int heldBlocks;
TypingAttenuator(int bins, int sampleRate, int fftSize) {
this.bins = bins;
this.hfBin = (int) Math.round(HF_HZ * fftSize / sampleRate);
}
void reset() {
slowPower = 0;
prevPower = 0;
envGain = 1.0;
heldBlocks = 0;
}
/** Multiplies the running per-bin gain by the current broadband ducking gain. */
void apply(double[] power, double[] gain) {
double total = 0, high = 0;
for (int k = 0; k < bins; k++) {
total += power[k];
if (k >= hfBin) high += power[k];
}
double hfRatio = high / (total + 1e-12);
boolean signature = total > slowPower * ONSET_FACTOR
&& total > prevPower * FLUX_FACTOR
&& hfRatio > HF_RATIO;
if (signature && heldBlocks < MAX_HOLD) {
envGain = SUPPRESS; // fast attack: duck immediately
heldBlocks++;
} else {
envGain += (1 - envGain) * RELEASE;
if (!signature) {
heldBlocks = 0;
// Only let the floor track when we're not inside a transient.
slowPower = slowPower == 0 ? total : FLOOR_SMOOTH * slowPower + (1 - FLOOR_SMOOTH) * total;
}
}
prevPower = total;
for (int k = 0; k < bins; k++) {
gain[k] *= envGain;
}
}
}

View File

@@ -42,15 +42,18 @@ public interface VoiceInput extends Microphone {
void setInputGain(double gain);
/** Enables removal of steady background noise (spectral denoise). */
/** Enables removal of steady background noise (WebRTC noise suppression). */
void setNoiseSuppression(boolean enabled);
/** Background-noise removal aggressiveness, 0 (light) .. 1 (heavy). */
void setDenoiserLevel(double level);
/** Noise suppression strength as TS3's {@code denoiser_level}: 0 (6 dB) .. 3 (21 dB). */
void setDenoiserLevel(int level);
/** Enables detection and attenuation of keyboard typing sounds. */
/** Enables attenuation of keyboard clicks while the user is typing. */
void setTypingAttenuation(boolean enabled);
/** Reports a key press anywhere on the system; typing attenuation only acts while typing. */
void keyPressed();
/** Enables automatic gain control (normalise microphone loudness). */
void setAgc(boolean enabled);

View File

@@ -130,11 +130,11 @@ public final class Settings {
public double masterVolume = 1.0;
/** Microphone input gain multiplier applied before VAD/encode. */
public double inputVolume = 1.0;
/** Remove steady background noise from the microphone (spectral denoise). */
/** Remove steady background noise from the microphone (WebRTC noise suppression). */
public boolean denoise = true;
/** Background-noise removal aggressiveness, 0 (light) .. 1 (heavy). */
public double denoiserLevel = 0.5;
/** Detect and attenuate keyboard typing sounds in the microphone. */
/** TS3's {@code denoiser_level}: 0..3 for 6, 12, 18 or 21 dB of suppression. */
public int denoiserLevel = 1;
/** Attenuate keyboard clicks while typing. */
public boolean typingAttenuation = true;
/** Automatic gain control: normalise microphone loudness to a target level. */
public boolean agc = true;
@@ -258,7 +258,7 @@ public final class Settings {
masterVolume = parseD(props.getProperty("masterVolume"), masterVolume);
inputVolume = parseD(props.getProperty("inputVolume"), inputVolume);
denoise = parseB(props.getProperty("denoise"), denoise);
denoiserLevel = parseD(props.getProperty("denoiserLevel"), denoiserLevel);
denoiserLevel = parseI(props.getProperty("denoiserLevel"), denoiserLevel);
typingAttenuation = parseB(props.getProperty("typingAttenuation"), typingAttenuation);
mutedTalkWarning = parseB(props.getProperty("mutedTalkWarning"), mutedTalkWarning);
agc = parseB(props.getProperty("agc"), agc);
@@ -308,7 +308,7 @@ public final class Settings {
props.setProperty("masterVolume", Double.toString(masterVolume));
props.setProperty("inputVolume", Double.toString(inputVolume));
props.setProperty("denoise", Boolean.toString(denoise));
props.setProperty("denoiserLevel", Double.toString(denoiserLevel));
props.setProperty("denoiserLevel", Integer.toString(denoiserLevel));
props.setProperty("typingAttenuation", Boolean.toString(typingAttenuation));
props.setProperty("mutedTalkWarning", Boolean.toString(mutedTalkWarning));
props.setProperty("agc", Boolean.toString(agc));