Let AGC raise the noise floor further to bring up a distant voice

AGC2 won't amplify the background noise above -50 dBFS, which also caps
the gain for a voice barely above that noise: a distant talker ended up
5 dB quieter than a near one with noise suppression on, 12 dB with it
off. A "Boost quiet speech" setting (0-20 dB, default 0) raises the cap,
live, on desktop and Android.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-25 22:08:50 +00:00
parent 491cbcb259
commit 73a15a8cfc
10 changed files with 87 additions and 2 deletions

View File

@@ -113,6 +113,7 @@ public final class CaptureVoiceInput implements VoiceInput {
processor.setSuppressionLevel(SuppressionLevel.fromDenoiserLevel(settings.denoiserLevel));
processor.setTransientSuppression(settings.typingAttenuation);
processor.setGainControl(settings.agc);
processor.setGainBoost(settings.agcBoostDb);
}
// ---- live configuration (safe to call from the UI thread) ----
@@ -169,6 +170,10 @@ public final class CaptureVoiceInput implements VoiceInput {
processor.setGainControl(enabled);
}
public void setAgcBoost(int db) {
processor.setGainBoost(db);
}
@Override
public void keyPressed() {
processor.keyPressed();

View File

@@ -60,6 +60,9 @@ public interface VoiceInput extends Microphone {
/** Enables automatic gain control (normalise microphone loudness). */
void setAgc(boolean enabled);
/** Lets AGC raise the background noise this many dB more to bring up a quiet voice. */
void setAgcBoost(int db);
/** Applies Opus encoder parameters, taking effect immediately if capturing. */
void setOpusParameters(OpusParameters parameters);
@@ -94,6 +97,7 @@ public interface VoiceInput extends Microphone {
setDenoiserLevel(settings.denoiserLevel);
setTypingAttenuation(settings.typingAttenuation);
setAgc(settings.agc);
setAgcBoost(settings.agcBoostDb);
setOpusParameters(OpusParameters.from(settings));
}
}

View File

@@ -18,7 +18,9 @@ import java.util.concurrent.atomic.AtomicBoolean;
* whenever noise suppression runs, so a 100&nbsp;Hz high-pass comes with it;</li>
* <li>transient suppression ("Typing attenuation"), told about key presses;</li>
* <li>automatic gain control. TS3 runs WebRTC's legacy AGC1 (adaptive digital, target
* -9&nbsp;dBFS, 20&nbsp;dB compression); this runs its successor, AGC2 adaptive digital.</li>
* -9&nbsp;dBFS, 20&nbsp;dB compression); this runs its successor, AGC2 adaptive digital. AGC2 won't
* amplify the noise floor above -50&nbsp;dBFS, which also holds back a distant voice
* barely above the noise; {@link #setGainBoost} trades more noise for more of it.</li>
* </ol>
*
* <p>One deliberate difference: APM only splits into bands for band-wise stages, so with noise
@@ -39,6 +41,7 @@ public final class AudioProcessor {
private volatile SuppressionLevel suppressionLevel = SuppressionLevel.DB_12;
private volatile boolean transientSuppression;
private volatile boolean gainControl;
private volatile float gainBoostDb;
private final AtomicBoolean keyPressed = new AtomicBoolean();
private final ThreeBandFilterBank filterBank = new ThreeBandFilterBank();
@@ -66,6 +69,11 @@ public final class AudioProcessor {
this.gainControl = enabled;
}
/** Lets the gain control raise the noise floor this many dB higher to reach quiet speech. */
public void setGainBoost(float db) {
this.gainBoostDb = db;
}
/** Reports a key press anywhere on the system; the transient suppressor only acts while typing. */
public void keyPressed() {
keyPressed.set(true);
@@ -146,5 +154,8 @@ public final class AudioProcessor {
} else if (gainController == null) {
gainController = new GainController2(AdaptiveDigitalConfig.DEFAULT);
}
if (gainController != null) {
gainController.setMaxOutputNoiseLevelDbfs(AdaptiveDigitalConfig.DEFAULT.maxOutputNoiseLevelDbfs() + gainBoostDb);
}
}
}

View File

@@ -21,6 +21,7 @@ final class AdaptiveDigitalGainController {
private final AdaptiveDigitalConfig config;
private final int adjacentSpeechFramesThreshold;
private final float maxGainChangeDbPer10ms;
private float maxOutputNoiseLevelDbfs;
private int framesToGainIncreaseAllowed;
private float lastGainDb;
@@ -31,6 +32,11 @@ final class AdaptiveDigitalGainController {
this.maxGainChangeDbPer10ms = config.maxGainChangeDbPerSecond() * FRAME_DURATION_MS / 1000.0f;
this.framesToGainIncreaseAllowed = adjacentSpeechFramesThreshold;
this.lastGainDb = config.initialGainDb();
this.maxOutputNoiseLevelDbfs = config.maxOutputNoiseLevelDbfs();
}
void setMaxOutputNoiseLevelDbfs(float dbfs) {
maxOutputNoiseLevelDbfs = dbfs;
}
void process(FrameInfo info, float[] frame, int samples) {
@@ -80,7 +86,7 @@ final class AdaptiveDigitalGainController {
}
private float limitGainByNoise(float targetGainDb, float inputNoiseLevelDbfs) {
final float maxAllowedGainDb = config.maxOutputNoiseLevelDbfs() - inputNoiseLevelDbfs;
final float maxAllowedGainDb = maxOutputNoiseLevelDbfs - inputNoiseLevelDbfs;
return Math.min(targetGainDb, Math.max(maxAllowedGainDb, 0.0f));
}

View File

@@ -16,6 +16,12 @@ import com.ts3client.audio.vad.RnnSpeechDetector;
* WebRTC: it measures the speech level only on frames its RNN VAD calls speech, and caps
* the gain so the measured noise floor stays below -50&nbsp;dBFS.
*
* <p>It is also slower to adapt: the gain rises at most 6&nbsp;dB/s, and only after 12
* frames in a row that its VAD rates at 0.95 or more, holding still through pauses. A
* distant voice in a reverberant room (half its frames that confident) took about 9&nbsp;s
* of talking to go from the initial 15&nbsp;dB to the 26&nbsp;dB it settled at. How fast
* AGC1 follows the same voice hasn't been measured.
*
* <p>Takes 10&nbsp;ms mono frames at 48&nbsp;kHz on the 16-bit scale, in place.
*/
public final class GainController2 {
@@ -39,6 +45,14 @@ public final class GainController2 {
this.adaptiveDigitalController = new AdaptiveDigitalGainController(config, ADJACENT_SPEECH_FRAMES_THRESHOLD);
}
/**
* Moves the cap on the amplified noise floor away from the configured
* {@code maxOutputNoiseLevelDbfs}, keeping the rest of the controller's state.
*/
public void setMaxOutputNoiseLevelDbfs(float dbfs) {
adaptiveDigitalController.setMaxOutputNoiseLevelDbfs(dbfs);
}
public void process(float[] frame) {
final float speechProbability = analyzeVad(frame);

View File

@@ -138,6 +138,9 @@ public final class Settings {
public boolean typingAttenuation = true;
/** Automatic gain control: normalise microphone loudness to a target level. */
public boolean agc = true;
public static final int MAX_AGC_BOOST_DB = 20;
/** How many dB louder AGC may make the background noise to bring up a quiet or distant voice. */
public int agcBoostDb = 0;
/** Beep when speech is picked up while the microphone is muted. */
public boolean mutedTalkWarning = true;
@@ -257,6 +260,7 @@ public final class Settings {
typingAttenuation = parseB(props.getProperty("typingAttenuation"), typingAttenuation);
mutedTalkWarning = parseB(props.getProperty("mutedTalkWarning"), mutedTalkWarning);
agc = parseB(props.getProperty("agc"), agc);
agcBoostDb = Math.max(0, Math.min(MAX_AGC_BOOST_DB, parseI(props.getProperty("agcBoostDb"), agcBoostDb)));
soundPack = props.getProperty("soundPack", soundPack);
soundVolume = parseD(props.getProperty("soundVolume"), soundVolume);
soundPackDir = props.getProperty("soundPackDir", soundPackDir);
@@ -309,6 +313,7 @@ public final class Settings {
props.setProperty("typingAttenuation", Boolean.toString(typingAttenuation));
props.setProperty("mutedTalkWarning", Boolean.toString(mutedTalkWarning));
props.setProperty("agc", Boolean.toString(agc));
props.setProperty("agcBoostDb", Integer.toString(agcBoostDb));
props.setProperty("soundPack", soundPack);
props.setProperty("soundVolume", Double.toString(soundVolume));
props.setProperty("soundPackDir", soundPackDir);

View File

@@ -166,6 +166,26 @@ class AudioProcessorTest {
"quiet speech should be raised by more than 12 dB");
}
@Test
void gainBoostLetsTheNoiseFloorRiseFurther() {
// -60 dBFS of noise alone: AGC2 starts at 15 dB of gain and, with no speech to
// raise it, can only lower it, down to what keeps the noise under its cap.
Random random = new Random(3);
float[] noise = new float[RATE * 3];
for (int i = 0; i < noise.length; i++) noise[i] = (float) (random.nextGaussian() * 0.001);
AudioProcessor plain = new AudioProcessor();
plain.setGainControl(true);
AudioProcessor boosted = new AudioProcessor();
boosted.setGainControl(true);
boosted.setGainBoost(20);
double plainDb = 20 * Math.log10(rms(run(plain, noise, Set.of()), RATE * 2, RATE * 3));
double boostedDb = 20 * Math.log10(rms(run(boosted, noise, Set.of()), RATE * 2, RATE * 3));
assertEquals(-50, plainDb, 1.5, "the noise is held at AGC2's -50 dBFS cap");
assertEquals(-45, boostedDb, 1.5, "with the cap 20 dB higher, the initial 15 dB of gain stays");
}
private static double rms(float[] x, int from, int to) {
double sum = 0;
for (int i = from; i < to; i++) sum += x[i] * x[i];