Port TS3's WebRTC capture processing to Java
TS3 pre-processes the microphone with WebRTC's audio processing module. AudioProcessor ports the stages it enables, in APM's order, per 10 ms at 48 kHz: - The high-pass filter APM forces on with noise suppression, then the three-band split and WebRTC's noise suppressor at TS3's four levels. - The transient suppressor behind "Typing attenuation", which only acts while it is told about key presses. - AGC2 adaptive digital, WebRTC's successor to the AGC1 that TS3 runs. The tests compare against 16-bit outputs of upstream APM builds (webrtc-audio-processing 2.1, and 1.3 for the transient suppressor). Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,174 @@
|
||||
package com.ts3client.audio.processing;
|
||||
|
||||
import com.ts3client.audio.processing.ns.SuppressionLevel;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.nio.ByteBuffer;
|
||||
import java.nio.ByteOrder;
|
||||
import java.util.Random;
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertArrayEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* Checks the chain against WebRTC itself. The {@code golden-*.s16} resources are
|
||||
* {@link #testSignal()} run through upstream APM (webrtc-audio-processing 2.1 for noise
|
||||
* suppression, 1.3 for the transient suppressor, which 2.x no longer ships), configured as
|
||||
* TS3 configures it, and stored as 16-bit PCM.
|
||||
*/
|
||||
class AudioProcessorTest {
|
||||
|
||||
private static final int RATE = AudioProcessor.SAMPLE_RATE;
|
||||
private static final int FRAME = AudioProcessor.FRAME_SIZE;
|
||||
|
||||
/** 10 ms frames in which the test signal has a key click, reported as key presses. */
|
||||
static final int[] KEY_FRAMES = {66, 72, 79, 85, 90, 97};
|
||||
private static final Set<Integer> KEYS = Set.of(66, 72, 79, 85, 90, 97);
|
||||
|
||||
/**
|
||||
* One second of microphone-like input: quiet pink noise throughout, a voiced harmonic
|
||||
* burst from 0.3 to 0.6 s, and key clicks after it.
|
||||
*/
|
||||
static float[] testSignal() {
|
||||
Random random = new Random(11);
|
||||
float[] x = new float[RATE];
|
||||
double b0 = 0, b1 = 0, b2 = 0;
|
||||
for (int i = 0; i < x.length; i++) {
|
||||
double w = random.nextGaussian();
|
||||
b0 = 0.99765 * b0 + w * 0.0990460;
|
||||
b1 = 0.96300 * b1 + w * 0.2965164;
|
||||
b2 = 0.57000 * b2 + w * 1.0526913;
|
||||
double noise = (b0 + b1 + b2 + w * 0.1848) * 0.0012;
|
||||
|
||||
double t = (double) i / RATE;
|
||||
double voice = 0;
|
||||
if (t >= 0.3 && t < 0.6) {
|
||||
double envelope = Math.sin(Math.PI * (t - 0.3) / 0.3);
|
||||
for (int h = 1; h <= 20; h++) {
|
||||
voice += Math.sin(2 * Math.PI * 140 * h * t + h) / h;
|
||||
}
|
||||
voice *= 0.08 * envelope;
|
||||
}
|
||||
x[i] = (float) (noise + voice);
|
||||
}
|
||||
for (int frame : KEY_FRAMES) {
|
||||
int start = frame * FRAME + 100;
|
||||
for (int k = 0; k < 480; k++) {
|
||||
x[start + k] += (float) (random.nextGaussian() * 0.1 * Math.exp(-k / 96.0));
|
||||
}
|
||||
}
|
||||
return x;
|
||||
}
|
||||
|
||||
private static float[] run(AudioProcessor processor, float[] input, Set<Integer> keyFrames) {
|
||||
float[] x = input.clone();
|
||||
float[] frame = new float[FRAME];
|
||||
for (int f = 0; f * FRAME + FRAME <= x.length; f++) {
|
||||
if (keyFrames.contains(f)) processor.keyPressed();
|
||||
System.arraycopy(x, f * FRAME, frame, 0, FRAME);
|
||||
processor.process(frame, FRAME);
|
||||
System.arraycopy(frame, 0, x, f * FRAME, FRAME);
|
||||
}
|
||||
return x;
|
||||
}
|
||||
|
||||
private static void assertMatchesGolden(String name, float[] actual) throws IOException {
|
||||
short[] golden = loadGolden(name);
|
||||
assertEquals(golden.length, actual.length);
|
||||
double maxDiff = 0;
|
||||
double sumSquares = 0;
|
||||
for (int i = 0; i < actual.length; i++) {
|
||||
double d = actual[i] * 32768.0 - golden[i];
|
||||
maxDiff = Math.max(maxDiff, Math.abs(d));
|
||||
sumSquares += d * d;
|
||||
}
|
||||
double rms = Math.sqrt(sumSquares / actual.length);
|
||||
// The reference is quantised to 16 bits, and float rounding differs slightly from
|
||||
// upstream's: anything within a few LSB is the same signal.
|
||||
assertTrue(maxDiff <= 4, name + ": max deviation " + maxDiff + " LSB");
|
||||
assertTrue(rms <= 0.6, name + ": RMS deviation " + rms + " LSB");
|
||||
}
|
||||
|
||||
private static short[] loadGolden(String name) throws IOException {
|
||||
try (InputStream in = AudioProcessorTest.class.getResourceAsStream(name)) {
|
||||
assertNotNull(in, name);
|
||||
ByteBuffer bytes = ByteBuffer.wrap(in.readAllBytes()).order(ByteOrder.LITTLE_ENDIAN);
|
||||
short[] out = new short[bytes.remaining() / 2];
|
||||
bytes.asShortBuffer().get(out);
|
||||
return out;
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void everythingOffPassesAudioThroughUntouched() {
|
||||
float[] x = testSignal();
|
||||
assertArrayEquals(x, run(new AudioProcessor(), x, KEYS));
|
||||
}
|
||||
|
||||
@Test
|
||||
void noiseSuppressionMatchesWebRtcAtEveryLevel() throws IOException {
|
||||
for (int level = 0; level <= 3; level++) {
|
||||
AudioProcessor p = new AudioProcessor();
|
||||
p.setNoiseSuppression(true);
|
||||
p.setSuppressionLevel(SuppressionLevel.fromDenoiserLevel(level));
|
||||
assertMatchesGolden("golden-ns" + level + ".s16", run(p, testSignal(), Set.of()));
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void typingAttenuationMatchesWebRtc() throws IOException {
|
||||
AudioProcessor p = new AudioProcessor();
|
||||
p.setNoiseSuppression(true);
|
||||
p.setSuppressionLevel(SuppressionLevel.DB_12);
|
||||
p.setTransientSuppression(true);
|
||||
assertMatchesGolden("golden-ns1-typing.s16", run(p, testSignal(), KEYS));
|
||||
}
|
||||
|
||||
@Test
|
||||
void typingAttenuationOnlyDelaysWithoutKeyPresses() {
|
||||
float[] x = testSignal();
|
||||
AudioProcessor p = new AudioProcessor();
|
||||
p.setTransientSuppression(true);
|
||||
float[] y = run(p, x, Set.of());
|
||||
int delay = 1024 - FRAME;
|
||||
for (int i = delay; i < x.length; i++) {
|
||||
assertEquals(x[i - delay], y[i], 0.0f);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void gainControlRaisesQuietSpeechAndLeavesSilenceAlone() {
|
||||
AudioProcessor p = new AudioProcessor();
|
||||
p.setGainControl(true);
|
||||
float[] silence = run(p, new float[RATE], Set.of());
|
||||
for (float s : silence) assertEquals(0.0f, s, 0.0f);
|
||||
|
||||
// 3 s of quiet voiced "words" at about -40 dBFS over a -70 dBFS noise floor. The
|
||||
// pauses matter: without them AGC2 would take the voice for the noise floor.
|
||||
Random random = new Random(5);
|
||||
float[] quiet = new float[RATE * 3];
|
||||
for (int i = 0; i < quiet.length; i++) {
|
||||
double t = (double) i / RATE;
|
||||
double v = 0;
|
||||
if (t % 0.5 < 0.3) {
|
||||
for (int h = 1; h <= 20; h++) v += Math.sin(2 * Math.PI * 130 * h * t + h) / h;
|
||||
v *= 0.004 * Math.sin(Math.PI * (t % 0.5) / 0.3);
|
||||
}
|
||||
quiet[i] = (float) (v + random.nextGaussian() * 0.0003);
|
||||
}
|
||||
float[] out = run(p, quiet, Set.of());
|
||||
assertTrue(rms(out, RATE * 2, RATE * 3) > 4 * rms(quiet, RATE * 2, RATE * 3),
|
||||
"quiet speech should be raised by more than 12 dB");
|
||||
}
|
||||
|
||||
private static double rms(float[] x, int from, int to) {
|
||||
double sum = 0;
|
||||
for (int i = from; i < to; i++) sum += x[i] * x[i];
|
||||
return Math.sqrt(sum / (to - from));
|
||||
}
|
||||
}
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Reference in New Issue
Block a user