2026-06-22 20:02:20 +02:00
|
|
|
/*
|
|
|
|
|
* test_jitter_depth — verifies the bounded-depth playout in AudioEngine::on_playback.
|
|
|
|
|
*
|
|
|
|
|
* Regression guard for the "latency keeps drifting backward, fixed only by rejoining" bug. The
|
|
|
|
|
* sender omits VAD/PTT/DTX silence from its timestamps (a compressed timeline), while the
|
|
|
|
|
* receiver's playout clock free-runs in real time. The old logic re-synced the playout clock to
|
|
|
|
|
* the *oldest* buffered frame and could only ever *add* standing latency (a reordered/late frame
|
|
|
|
|
* snapped the clock backward), with nothing to trim it — so latency ratcheted up across talkspurt
|
|
|
|
|
* gaps. The fix keeps the clock a bounded `target` behind the *newest* arrival and frame-skips to
|
|
|
|
|
* catch up, so depth stays bounded no matter the trigger.
|
|
|
|
|
*
|
|
|
|
|
* This drives many talkspurt/silence cycles with a compressed timeline plus a reordered straggler
|
|
|
|
|
* each cycle (which previously snapped the clock backward), and asserts the buffered depth
|
|
|
|
|
* (newest_ts - playout_ts) stays bounded while audio keeps playing. White-box via mix_for_test
|
|
|
|
|
* (no audio hardware needed), same pattern as test_plc_cap.
|
|
|
|
|
*/
|
|
|
|
|
#include <cmath>
|
|
|
|
|
#include <cstdint>
|
|
|
|
|
#include <cstdio>
|
|
|
|
|
#include <vector>
|
|
|
|
|
|
|
|
|
|
#if defined(VOICECAT_HAS_AUDIO) && defined(VOICECAT_HAS_OPUS)
|
|
|
|
|
|
|
|
|
|
#include "audio/audio_engine.h"
|
|
|
|
|
#include "codec/opus_codec.h"
|
|
|
|
|
|
|
|
|
|
static int g_failures = 0;
|
|
|
|
|
#define CHECK(cond) \
|
|
|
|
|
do { \
|
|
|
|
|
if (!(cond)) { \
|
|
|
|
|
std::printf("FAIL [%s:%d]: %s\n", __FILE__, __LINE__, #cond); \
|
|
|
|
|
++g_failures; \
|
|
|
|
|
} \
|
|
|
|
|
} while (0)
|
|
|
|
|
|
|
|
|
|
static double rms(const int16_t* pcm, int n) {
|
|
|
|
|
double sum = 0.0;
|
|
|
|
|
for (int i = 0; i < n; ++i) sum += static_cast<double>(pcm[i]) * pcm[i];
|
|
|
|
|
return std::sqrt(sum / n);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
int main() {
|
|
|
|
|
voicecat::audio::AudioEngine engine;
|
|
|
|
|
voicecat::audio::AudioParams p;
|
|
|
|
|
p.sample_rate = 48000;
|
|
|
|
|
p.capture_channels = 1;
|
|
|
|
|
p.playback_channels = 2;
|
|
|
|
|
p.frame_ms = 20;
|
|
|
|
|
CHECK(engine.start(p)); // no capture_cb — headless safe
|
|
|
|
|
|
|
|
|
|
voicecat::codec::OpusParams op;
|
|
|
|
|
op.sample_rate = 48000;
|
|
|
|
|
op.frame_ms = 20;
|
|
|
|
|
op.stereo = false;
|
|
|
|
|
const int frame_samples = voicecat::codec::opus_frame_samples(op); // 960
|
|
|
|
|
|
|
|
|
|
// A loud sine, encoded once, reused for every pushed frame.
|
|
|
|
|
voicecat::codec::OpusEncoder enc;
|
|
|
|
|
CHECK(enc.init(op));
|
|
|
|
|
std::vector<int16_t> sine(static_cast<size_t>(frame_samples));
|
|
|
|
|
for (int i = 0; i < frame_samples; ++i) {
|
|
|
|
|
float t = static_cast<float>(i) / 48000.0f;
|
|
|
|
|
sine[i] = static_cast<int16_t>(std::sin(2.0f * 3.14159265f * 440.0f * t) * 20000.0f);
|
|
|
|
|
}
|
|
|
|
|
uint8_t opus_buf[1500];
|
|
|
|
|
const int opus_len = enc.encode(sine.data(), frame_samples, opus_buf, sizeof(opus_buf));
|
|
|
|
|
CHECK(opus_len > 0);
|
|
|
|
|
|
|
|
|
|
const uint32_t ssrc = 1;
|
2026-06-23 21:11:03 +02:00
|
|
|
engine.init_recv_stream(ssrc, op, /*user_id=*/0, /*stream_id=*/0, /*is_voice=*/false);
|
2026-06-22 20:02:20 +02:00
|
|
|
|
|
|
|
|
const uint32_t pb_frames = 480; // 10 ms hardware period
|
|
|
|
|
const int out_n = static_cast<int>(pb_frames) * 2; // stereo interleaved
|
|
|
|
|
std::vector<int16_t> out(static_cast<size_t>(out_n), 0);
|
|
|
|
|
|
|
|
|
|
auto mix_n = [&](int n) {
|
|
|
|
|
for (int i = 0; i < n; ++i) engine.mix_for_test(out.data(), pb_frames);
|
|
|
|
|
};
|
|
|
|
|
auto push = [&](uint32_t ts, bool marker) {
|
|
|
|
|
voicecat::audio::JitterBuffer::Frame f;
|
|
|
|
|
f.seq = 0;
|
|
|
|
|
f.timestamp = ts;
|
|
|
|
|
f.fec_present = false;
|
|
|
|
|
f.marker = marker;
|
|
|
|
|
f.payload.assign(opus_buf, opus_buf + opus_len);
|
|
|
|
|
engine.push_recv_frame(ssrc, std::move(f));
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
uint32_t ts = 1000; // arbitrary non-zero start
|
|
|
|
|
int32_t max_depth = 0;
|
|
|
|
|
double last_voice_rms = 0.0;
|
|
|
|
|
|
|
|
|
|
// Seed the stream (first frame is a talkspurt marker, like a real resume).
|
|
|
|
|
push(ts, /*marker=*/true);
|
|
|
|
|
ts += static_cast<uint32_t>(frame_samples);
|
|
|
|
|
mix_n(1);
|
|
|
|
|
|
|
|
|
|
// Drive the producer FASTER than the consumer: push one 960-sample frame per step but drain
|
|
|
|
|
// only 480 samples (one pb_frames callback) — i.e. arrivals outrun playout by ~480 samples a
|
|
|
|
|
// step, exactly the clock-drift / bursty-arrival condition that made latency ratchet up. Also
|
|
|
|
|
// inject a reordered straggler periodically (the old backward-snap trigger). The bounded-depth
|
|
|
|
|
// catch-up must keep the standing latency from growing without limit. Pre-fix (no catch-up,
|
|
|
|
|
// snap-to-oldest) the depth would climb to ~hundreds of frames here.
|
|
|
|
|
const int kSteps = 400;
|
|
|
|
|
const uint32_t kStraggler = 48000u * 250u / 1000u; // 250 ms behind the leading edge
|
|
|
|
|
for (int s = 0; s < kSteps; ++s) {
|
|
|
|
|
push(ts, /*marker=*/false);
|
|
|
|
|
ts += static_cast<uint32_t>(frame_samples);
|
|
|
|
|
if (s % 25 == 12) push(ts - kStraggler, /*marker=*/false); // reordered straggler
|
|
|
|
|
mix_n(1); // drain only 480 of the 960 produced — producer outruns consumer
|
|
|
|
|
int32_t d = engine.stream_playout_depth_samples(ssrc);
|
|
|
|
|
if (d > max_depth) max_depth = d;
|
|
|
|
|
last_voice_rms = std::max(last_voice_rms, rms(out.data(), out_n));
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
std::printf("jitter_depth: max_depth=%d samples (%.0f ms) voice_rms=%.1f\n", max_depth,
|
|
|
|
|
static_cast<double>(max_depth) * 1000.0 / 48000.0, last_voice_rms);
|
|
|
|
|
|
|
|
|
|
// Bounded: with catch-up the standing latency stays near the adaptive target, well under
|
|
|
|
|
// 200 ms even though arrivals outran playout for 400 steps (~4 s of pushed audio).
|
|
|
|
|
CHECK(max_depth > 0); // playout ran / depth observed
|
|
|
|
|
CHECK(max_depth < static_cast<int32_t>(48000 * 200 / 1000)); // bounded (was unbounded pre-fix)
|
|
|
|
|
CHECK(last_voice_rms > 1.0); // audio keeps playing
|
|
|
|
|
|
|
|
|
|
engine.remove_stream(ssrc);
|
|
|
|
|
engine.stop();
|
|
|
|
|
enc.destroy();
|
|
|
|
|
|
|
|
|
|
if (g_failures == 0) {
|
|
|
|
|
std::printf("jitter_depth: all checks passed\n");
|
|
|
|
|
return 0;
|
|
|
|
|
}
|
|
|
|
|
std::printf("jitter_depth: %d failure(s)\n", g_failures);
|
|
|
|
|
return 1;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#else
|
|
|
|
|
|
|
|
|
|
int main() {
|
|
|
|
|
std::printf("jitter_depth: SKIP (VOICECAT_HAS_AUDIO or VOICECAT_HAS_OPUS not defined)\n");
|
|
|
|
|
return 0;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#endif
|