/* * test_noise_suppression — the RNNoise backend behind ApmProcessor actually denoises. * * This is the behavior exit-criterion for the noise-suppression feature (docs/voice.md §10-11): * a real DSP backend, not the old inert passthrough. ApmProcessor::create() returns the RNNoise * processor when the core is built with VOICECAT_HAS_NS (the dev/release presets). We feed it * mono 48 kHz white noise in 20 ms (960-sample) frames — exercising the internal 480-sample * chunking — and assert the output noise floor collapses while values stay finite/in-range. * * Registered only under VOICECAT_USE_VCPKG_DEPS, where VOICECAT_HAS_NS is defined, so a large * reduction is expected; a passthrough build would (correctly) fail this test. */ #include #include #include #include #include "audio/apm_processor.h" namespace vca = voicecat::audio; static int g_failures = 0; #define CHECK(cond) \ do { \ if (!(cond)) { \ std::printf("FAIL [%s:%d]: %s\n", __FILE__, __LINE__, #cond); \ ++g_failures; \ } \ } while (0) static constexpr int kFrameSamples = 960; // 20 ms @ 48 kHz (two RNNoise 480-sample frames) int main() { auto ns = vca::ApmProcessor::create(); CHECK(ns != nullptr); if (!ns) return 1; // Deterministic white noise (xorshift) at ~int16/10 amplitude, processed frame by frame. uint32_t rng = 0x12345678u; auto next_noise = [&]() -> int16_t { rng ^= rng << 13; rng ^= rng >> 17; rng ^= rng << 5; // map to roughly [-3000, 3000] return static_cast((static_cast(rng % 6001)) - 3000); }; const int kFrames = 200; const int kWarmup = 60; // let RNNoise's recurrent state settle before measuring double in_sumsq = 0.0, out_sumsq = 0.0; long measured = 0; std::vector frame(kFrameSamples); for (int f = 0; f < kFrames; ++f) { double frame_in_sq = 0.0; for (int i = 0; i < kFrameSamples; ++i) { frame[i] = next_noise(); frame_in_sq += static_cast(frame[i]) * frame[i]; } bool gate = ns->process_capture(frame.data(), kFrameSamples, 48000); CHECK(gate); // NS never gates — always passes the frame on if (f >= kWarmup) { in_sumsq += frame_in_sq; for (int i = 0; i < kFrameSamples; ++i) { // Output must stay finite and within int16 range (clamping correctness). CHECK(frame[i] >= -32768 && frame[i] <= 32767); out_sumsq += static_cast(frame[i]) * frame[i]; } measured += kFrameSamples; } } CHECK(measured > 0); double in_rms = std::sqrt(in_sumsq / measured); double out_rms = std::sqrt(out_sumsq / measured); double reduction = (in_rms > 0.0) ? (1.0 - out_rms / in_rms) : 0.0; std::printf("noise-only: in_rms=%.1f out_rms=%.1f reduction=%.1f%%\n", in_rms, out_rms, 100.0 * reduction); // RNNoise drops pure noise by ~99%; require a large, unambiguous reduction so a passthrough // (no real backend) is caught. The threshold is deliberately conservative vs. the ~99% seen. CHECK(reduction > 0.80); // A 48-kHz guard miss must pass audio through untouched (our clock is always 48 kHz, but the // backstop matters): feed a non-48k sample-rate and confirm the buffer is unchanged. std::vector probe(kFrameSamples); for (int i = 0; i < kFrameSamples; ++i) probe[i] = next_noise(); std::vector probe_copy = probe; ns->process_capture(probe.data(), kFrameSamples, 16000); CHECK(probe == probe_copy); if (g_failures == 0) { std::printf("noise_suppression: OK\n"); return 0; } std::printf("noise_suppression: %d failure(s)\n", g_failures); return 1; }