Port TS3's WebRTC capture processing to Java

TS3 pre-processes the microphone with WebRTC's audio processing module.
AudioProcessor ports the stages it enables, in APM's order, per 10 ms at
48 kHz:

- The high-pass filter APM forces on with noise suppression, then the
  three-band split and WebRTC's noise suppressor at TS3's four levels.
- The transient suppressor behind "Typing attenuation", which only acts
  while it is told about key presses.
- AGC2 adaptive digital, WebRTC's successor to the AGC1 that TS3 runs.

The tests compare against 16-bit outputs of upstream APM builds
(webrtc-audio-processing 2.1, and 1.3 for the transient suppressor).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-24 20:46:46 +00:00
parent ee0f15fb3c
commit de5f7da236
34 changed files with 2770 additions and 3 deletions

View File

@@ -0,0 +1,174 @@
package com.ts3client.audio.processing;
import com.ts3client.audio.processing.ns.SuppressionLevel;
import org.junit.jupiter.api.Test;
import java.io.IOException;
import java.io.InputStream;
import java.nio.ByteBuffer;
import java.nio.ByteOrder;
import java.util.Random;
import java.util.Set;
import static org.junit.jupiter.api.Assertions.assertArrayEquals;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
/**
* Checks the chain against WebRTC itself. The {@code golden-*.s16} resources are
* {@link #testSignal()} run through upstream APM (webrtc-audio-processing 2.1 for noise
* suppression, 1.3 for the transient suppressor, which 2.x no longer ships), configured as
* TS3 configures it, and stored as 16-bit PCM.
*/
class AudioProcessorTest {
private static final int RATE = AudioProcessor.SAMPLE_RATE;
private static final int FRAME = AudioProcessor.FRAME_SIZE;
/** 10 ms frames in which the test signal has a key click, reported as key presses. */
static final int[] KEY_FRAMES = {66, 72, 79, 85, 90, 97};
private static final Set<Integer> KEYS = Set.of(66, 72, 79, 85, 90, 97);
/**
* One second of microphone-like input: quiet pink noise throughout, a voiced harmonic
* burst from 0.3 to 0.6 s, and key clicks after it.
*/
static float[] testSignal() {
Random random = new Random(11);
float[] x = new float[RATE];
double b0 = 0, b1 = 0, b2 = 0;
for (int i = 0; i < x.length; i++) {
double w = random.nextGaussian();
b0 = 0.99765 * b0 + w * 0.0990460;
b1 = 0.96300 * b1 + w * 0.2965164;
b2 = 0.57000 * b2 + w * 1.0526913;
double noise = (b0 + b1 + b2 + w * 0.1848) * 0.0012;
double t = (double) i / RATE;
double voice = 0;
if (t >= 0.3 && t < 0.6) {
double envelope = Math.sin(Math.PI * (t - 0.3) / 0.3);
for (int h = 1; h <= 20; h++) {
voice += Math.sin(2 * Math.PI * 140 * h * t + h) / h;
}
voice *= 0.08 * envelope;
}
x[i] = (float) (noise + voice);
}
for (int frame : KEY_FRAMES) {
int start = frame * FRAME + 100;
for (int k = 0; k < 480; k++) {
x[start + k] += (float) (random.nextGaussian() * 0.1 * Math.exp(-k / 96.0));
}
}
return x;
}
private static float[] run(AudioProcessor processor, float[] input, Set<Integer> keyFrames) {
float[] x = input.clone();
float[] frame = new float[FRAME];
for (int f = 0; f * FRAME + FRAME <= x.length; f++) {
if (keyFrames.contains(f)) processor.keyPressed();
System.arraycopy(x, f * FRAME, frame, 0, FRAME);
processor.process(frame, FRAME);
System.arraycopy(frame, 0, x, f * FRAME, FRAME);
}
return x;
}
private static void assertMatchesGolden(String name, float[] actual) throws IOException {
short[] golden = loadGolden(name);
assertEquals(golden.length, actual.length);
double maxDiff = 0;
double sumSquares = 0;
for (int i = 0; i < actual.length; i++) {
double d = actual[i] * 32768.0 - golden[i];
maxDiff = Math.max(maxDiff, Math.abs(d));
sumSquares += d * d;
}
double rms = Math.sqrt(sumSquares / actual.length);
// The reference is quantised to 16 bits, and float rounding differs slightly from
// upstream's: anything within a few LSB is the same signal.
assertTrue(maxDiff <= 4, name + ": max deviation " + maxDiff + " LSB");
assertTrue(rms <= 0.6, name + ": RMS deviation " + rms + " LSB");
}
private static short[] loadGolden(String name) throws IOException {
try (InputStream in = AudioProcessorTest.class.getResourceAsStream(name)) {
assertNotNull(in, name);
ByteBuffer bytes = ByteBuffer.wrap(in.readAllBytes()).order(ByteOrder.LITTLE_ENDIAN);
short[] out = new short[bytes.remaining() / 2];
bytes.asShortBuffer().get(out);
return out;
}
}
@Test
void everythingOffPassesAudioThroughUntouched() {
float[] x = testSignal();
assertArrayEquals(x, run(new AudioProcessor(), x, KEYS));
}
@Test
void noiseSuppressionMatchesWebRtcAtEveryLevel() throws IOException {
for (int level = 0; level <= 3; level++) {
AudioProcessor p = new AudioProcessor();
p.setNoiseSuppression(true);
p.setSuppressionLevel(SuppressionLevel.fromDenoiserLevel(level));
assertMatchesGolden("golden-ns" + level + ".s16", run(p, testSignal(), Set.of()));
}
}
@Test
void typingAttenuationMatchesWebRtc() throws IOException {
AudioProcessor p = new AudioProcessor();
p.setNoiseSuppression(true);
p.setSuppressionLevel(SuppressionLevel.DB_12);
p.setTransientSuppression(true);
assertMatchesGolden("golden-ns1-typing.s16", run(p, testSignal(), KEYS));
}
@Test
void typingAttenuationOnlyDelaysWithoutKeyPresses() {
float[] x = testSignal();
AudioProcessor p = new AudioProcessor();
p.setTransientSuppression(true);
float[] y = run(p, x, Set.of());
int delay = 1024 - FRAME;
for (int i = delay; i < x.length; i++) {
assertEquals(x[i - delay], y[i], 0.0f);
}
}
@Test
void gainControlRaisesQuietSpeechAndLeavesSilenceAlone() {
AudioProcessor p = new AudioProcessor();
p.setGainControl(true);
float[] silence = run(p, new float[RATE], Set.of());
for (float s : silence) assertEquals(0.0f, s, 0.0f);
// 3 s of quiet voiced "words" at about -40 dBFS over a -70 dBFS noise floor. The
// pauses matter: without them AGC2 would take the voice for the noise floor.
Random random = new Random(5);
float[] quiet = new float[RATE * 3];
for (int i = 0; i < quiet.length; i++) {
double t = (double) i / RATE;
double v = 0;
if (t % 0.5 < 0.3) {
for (int h = 1; h <= 20; h++) v += Math.sin(2 * Math.PI * 130 * h * t + h) / h;
v *= 0.004 * Math.sin(Math.PI * (t % 0.5) / 0.3);
}
quiet[i] = (float) (v + random.nextGaussian() * 0.0003);
}
float[] out = run(p, quiet, Set.of());
assertTrue(rms(out, RATE * 2, RATE * 3) > 4 * rms(quiet, RATE * 2, RATE * 3),
"quiet speech should be raised by more than 12 dB");
}
private static double rms(float[] x, int from, int to) {
double sum = 0;
for (int i = from; i < to; i++) sum += x[i] * x[i];
return Math.sqrt(sum / (to - from));
}
}