/* * test_jitter_depth — verifies the bounded-depth playout in AudioEngine::on_playback. * * Regression guard for the "latency keeps drifting backward, fixed only by rejoining" bug. The * sender omits VAD/PTT/DTX silence from its timestamps (a compressed timeline), while the * receiver's playout clock free-runs in real time. The old logic re-synced the playout clock to * the *oldest* buffered frame and could only ever *add* standing latency (a reordered/late frame * snapped the clock backward), with nothing to trim it — so latency ratcheted up across talkspurt * gaps. The fix keeps the clock a bounded `target` behind the *newest* arrival and frame-skips to * catch up, so depth stays bounded no matter the trigger. * * This drives many talkspurt/silence cycles with a compressed timeline plus a reordered straggler * each cycle (which previously snapped the clock backward), and asserts the buffered depth * (newest_ts - playout_ts) stays bounded while audio keeps playing. White-box via mix_for_test * (no audio hardware needed), same pattern as test_plc_cap. */ #include #include #include #include #if defined(VOICECAT_HAS_AUDIO) && defined(VOICECAT_HAS_OPUS) #include "audio/audio_engine.h" #include "codec/opus_codec.h" static int g_failures = 0; #define CHECK(cond) \ do { \ if (!(cond)) { \ std::printf("FAIL [%s:%d]: %s\n", __FILE__, __LINE__, #cond); \ ++g_failures; \ } \ } while (0) static double rms(const int16_t* pcm, int n) { double sum = 0.0; for (int i = 0; i < n; ++i) sum += static_cast(pcm[i]) * pcm[i]; return std::sqrt(sum / n); } int main() { voicecat::audio::AudioEngine engine; voicecat::audio::AudioParams p; p.sample_rate = 48000; p.capture_channels = 1; p.playback_channels = 2; p.frame_ms = 20; CHECK(engine.start(p)); // no capture_cb — headless safe voicecat::codec::OpusParams op; op.sample_rate = 48000; op.frame_ms = 20; op.stereo = false; const int frame_samples = voicecat::codec::opus_frame_samples(op); // 960 // A loud sine, encoded once, reused for every pushed frame. voicecat::codec::OpusEncoder enc; CHECK(enc.init(op)); std::vector sine(static_cast(frame_samples)); for (int i = 0; i < frame_samples; ++i) { float t = static_cast(i) / 48000.0f; sine[i] = static_cast(std::sin(2.0f * 3.14159265f * 440.0f * t) * 20000.0f); } uint8_t opus_buf[1500]; const int opus_len = enc.encode(sine.data(), frame_samples, opus_buf, sizeof(opus_buf)); CHECK(opus_len > 0); const uint32_t ssrc = 1; engine.init_recv_stream(ssrc, op, /*user_id=*/0, /*stream_id=*/0); const uint32_t pb_frames = 480; // 10 ms hardware period const int out_n = static_cast(pb_frames) * 2; // stereo interleaved std::vector out(static_cast(out_n), 0); auto mix_n = [&](int n) { for (int i = 0; i < n; ++i) engine.mix_for_test(out.data(), pb_frames); }; auto push = [&](uint32_t ts, bool marker) { voicecat::audio::JitterBuffer::Frame f; f.seq = 0; f.timestamp = ts; f.fec_present = false; f.marker = marker; f.payload.assign(opus_buf, opus_buf + opus_len); engine.push_recv_frame(ssrc, std::move(f)); }; uint32_t ts = 1000; // arbitrary non-zero start int32_t max_depth = 0; double last_voice_rms = 0.0; // Seed the stream (first frame is a talkspurt marker, like a real resume). push(ts, /*marker=*/true); ts += static_cast(frame_samples); mix_n(1); // Drive the producer FASTER than the consumer: push one 960-sample frame per step but drain // only 480 samples (one pb_frames callback) — i.e. arrivals outrun playout by ~480 samples a // step, exactly the clock-drift / bursty-arrival condition that made latency ratchet up. Also // inject a reordered straggler periodically (the old backward-snap trigger). The bounded-depth // catch-up must keep the standing latency from growing without limit. Pre-fix (no catch-up, // snap-to-oldest) the depth would climb to ~hundreds of frames here. const int kSteps = 400; const uint32_t kStraggler = 48000u * 250u / 1000u; // 250 ms behind the leading edge for (int s = 0; s < kSteps; ++s) { push(ts, /*marker=*/false); ts += static_cast(frame_samples); if (s % 25 == 12) push(ts - kStraggler, /*marker=*/false); // reordered straggler mix_n(1); // drain only 480 of the 960 produced — producer outruns consumer int32_t d = engine.stream_playout_depth_samples(ssrc); if (d > max_depth) max_depth = d; last_voice_rms = std::max(last_voice_rms, rms(out.data(), out_n)); } std::printf("jitter_depth: max_depth=%d samples (%.0f ms) voice_rms=%.1f\n", max_depth, static_cast(max_depth) * 1000.0 / 48000.0, last_voice_rms); // Bounded: with catch-up the standing latency stays near the adaptive target, well under // 200 ms even though arrivals outran playout for 400 steps (~4 s of pushed audio). CHECK(max_depth > 0); // playout ran / depth observed CHECK(max_depth < static_cast(48000 * 200 / 1000)); // bounded (was unbounded pre-fix) CHECK(last_voice_rms > 1.0); // audio keeps playing engine.remove_stream(ssrc); engine.stop(); enc.destroy(); if (g_failures == 0) { std::printf("jitter_depth: all checks passed\n"); return 0; } std::printf("jitter_depth: %d failure(s)\n", g_failures); return 1; } #else int main() { std::printf("jitter_depth: SKIP (VOICECAT_HAS_AUDIO or VOICECAT_HAS_OPUS not defined)\n"); return 0; } #endif