fix(net): broadcast LEFT on disconnect, add keepalive/reaper, cap PLC hiss
Three reported bugs traced to one root cause plus two missing designed features:
1. Stale users + eternal PLC hiss (root cause): ConnSession::close() silently
erased dropped users without broadcasting UserEvent::LEFT, so peers never
learned the user left and their audio engines never called remove_stream —
Opus PLC synthesized comfort noise forever. Fix: broadcast_left() helper
+ close() broadcasts LEFT before erasing.
2. PLC cap (defense-in-depth): on_playback now caps pure PLC at ~2s, then
emits digital silence so a stale stream can never hiss forever even if
remove_stream is skipped. Resets automatically on fresh packets.
3. No timeout / no ping: client never sent Ping, server had no last_seen /
reaper, so half-open connections (NAT timeout, wifi loss, sleep) left
ghost users forever. Fix: client Ping every 15s with RTT measurement,
ConnSession::last_seen bumped on every inbound TCP/UDP frame, steady_timer
reaper sweeps every 15s and drops sessions older than 45s (configurable
via server::Config).
4. UDP KEEPALIVE: client sends plaintext kFrameKeepalive every 5s; server
bumps last_seen + echoes back. Keeps NAT bindings alive and lets media
activity defer the reaper independently of TCP.
5. Graceful client disconnect: vc_disconnect() sends Disconnect{code=0} via
a flag-based io-thread exit (no double-close race); server handles
client-sent Disconnect with immediate close() + LEFT broadcast.
3 new tests: disconnect_left, plc_cap, reaper_timeout. 21/21 ctest green.
Docs: protocol.md §6/§7, voice.md §6, architecture.md §5, PROGRESS.md.
2026-06-18 01:18:33 +02:00
|
|
|
/*
|
|
|
|
|
* test_plc_cap — verifies the PLC cap in AudioEngine::on_playback.
|
|
|
|
|
*
|
|
|
|
|
* After ~2s of pure packet-loss concealment (no real packets decoded), the mixer stops
|
|
|
|
|
* calling opus_decode(nullptr,0,...) and emits digital silence instead. This bounds the
|
|
|
|
|
* Opus comfort-noise hiss so a stale stream left in the mixer can never hiss forever —
|
|
|
|
|
* defense-in-depth for the server's UserEvent::LEFT broadcast (the primary fix that
|
|
|
|
|
* triggers remove_stream on disconnect). Also verifies that a fresh real packet resets
|
|
|
|
|
* the PLC streak and audio resumes.
|
|
|
|
|
*
|
|
|
|
|
* White-box: drives AudioEngine::mix_for_test directly (no audio hardware needed).
|
|
|
|
|
*/
|
|
|
|
|
#include <cmath>
|
|
|
|
|
#include <cstdio>
|
|
|
|
|
#include <cstring>
|
|
|
|
|
#include <vector>
|
|
|
|
|
|
|
|
|
|
#if defined(VOICECAT_HAS_AUDIO) && defined(VOICECAT_HAS_OPUS)
|
|
|
|
|
|
|
|
|
|
#include "audio/audio_engine.h"
|
|
|
|
|
#include "codec/opus_codec.h"
|
|
|
|
|
|
|
|
|
|
static int g_failures = 0;
|
|
|
|
|
#define CHECK(cond) \
|
|
|
|
|
do { \
|
|
|
|
|
if (!(cond)) { \
|
|
|
|
|
std::printf("FAIL [%s:%d]: %s\n", __FILE__, __LINE__, #cond); \
|
|
|
|
|
++g_failures; \
|
|
|
|
|
} \
|
|
|
|
|
} while (0)
|
|
|
|
|
|
|
|
|
|
static double rms(const int16_t* pcm, int n) {
|
|
|
|
|
double sum = 0.0;
|
|
|
|
|
for (int i = 0; i < n; ++i) sum += static_cast<double>(pcm[i]) * pcm[i];
|
|
|
|
|
return std::sqrt(sum / n);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
static int64_t abs_energy(const int16_t* pcm, int n) {
|
|
|
|
|
int64_t e = 0;
|
|
|
|
|
for (int i = 0; i < n; ++i) e += static_cast<int64_t>(std::abs(static_cast<int>(pcm[i])));
|
|
|
|
|
return e;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
int main() {
|
|
|
|
|
voicecat::audio::AudioEngine engine;
|
|
|
|
|
voicecat::audio::AudioParams p;
|
|
|
|
|
p.sample_rate = 48000;
|
|
|
|
|
p.capture_channels = 1;
|
|
|
|
|
p.playback_channels = 2;
|
|
|
|
|
p.frame_ms = 20;
|
|
|
|
|
CHECK(engine.start(p)); // no capture_cb — headless safe (devices may fail to init; ok)
|
|
|
|
|
|
|
|
|
|
voicecat::codec::OpusParams op;
|
|
|
|
|
op.sample_rate = 48000;
|
|
|
|
|
op.frame_ms = 20;
|
|
|
|
|
op.stereo = false;
|
|
|
|
|
int frame_samples = voicecat::codec::opus_frame_samples(op); // 960
|
|
|
|
|
|
|
|
|
|
// Encode a loud sine wave to seed the decoder's PLC state.
|
|
|
|
|
voicecat::codec::OpusEncoder enc;
|
|
|
|
|
CHECK(enc.init(op));
|
|
|
|
|
std::vector<int16_t> sine(static_cast<size_t>(frame_samples));
|
|
|
|
|
for (int i = 0; i < frame_samples; ++i) {
|
|
|
|
|
float t = static_cast<float>(i) / 48000.0f;
|
|
|
|
|
sine[i] = static_cast<int16_t>(std::sin(2.0f * 3.14159265f * 440.0f * t) * 20000.0f);
|
|
|
|
|
}
|
|
|
|
|
uint8_t opus_buf[1500];
|
|
|
|
|
int opus_len = enc.encode(sine.data(), frame_samples, opus_buf, sizeof(opus_buf));
|
|
|
|
|
CHECK(opus_len > 0);
|
|
|
|
|
|
|
|
|
|
const uint32_t ssrc = 1;
|
feat: external PCM feed/tap API (vc_stream_feed_pcm + vc_set_pcm_sink)
Promotes vc_test_inject_capture (mono-only, TEST-ONLY) to a public,
stereo-capable production API and adds a symmetric PCM tap on the
receive side. Enables ReplayKit (iOS), ScreenCaptureKit (macOS), bots,
soundboards, and custom clients — all without a hardware audio device.
Core C++:
- voicecat.h: new vc_stream_feed_pcm, vc_pcm_sink_cb typedef,
vc_set_pcm_sink; vc_test_inject_capture kept as deprecated alias
- audio_engine: stereo-aware inject_capture (channels param + ring
reset on channel-count change); atomic pcm_sink_ fired per decoded
frame in on_playback; RemoteStream carries user_id/stream_id for
RT-safe sink metadata; init_recv_stream takes user_id+stream_id
- client.cpp: stream_feed_pcm / set_pcm_sink implementations;
sync_remote_streams passes user_id/stream_id to init_recv_stream
- voicecat.cpp: trampolines + channels=1/2 validation
Tests: test_external_pcm (headless, 3 sub-tests: mono round-trip,
stereo feed L≠R, sink metadata+disable). ctest 23/23.
Swift: feedPcm / setPcmSink in VoiceCatClient.swift + 4 XCTest
smoke tests (ExternalPcmTests.swift).
C#: StreamFeedPcm / SetPcmSink in VoiceCatClient.cs + NativeMethods.cs
(vc_stream_feed_pcm unsafe P/Invoke, VcPcmSinkCallback delegate,
vc_set_pcm_sink via nint) + 4 xUnit smoke tests (ExternalPcmTests.cs).
Docs: architecture.md §4 new subsection, voice.md §9 updated
(macOS/iOS now reference vc_stream_feed_pcm), protocol.md §8 explicit
no-protocol-change note, roadmap.md M5 entry.
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-20 17:52:09 +02:00
|
|
|
engine.init_recv_stream(ssrc, op, /*user_id=*/0, /*stream_id=*/0);
|
fix(net): broadcast LEFT on disconnect, add keepalive/reaper, cap PLC hiss
Three reported bugs traced to one root cause plus two missing designed features:
1. Stale users + eternal PLC hiss (root cause): ConnSession::close() silently
erased dropped users without broadcasting UserEvent::LEFT, so peers never
learned the user left and their audio engines never called remove_stream —
Opus PLC synthesized comfort noise forever. Fix: broadcast_left() helper
+ close() broadcasts LEFT before erasing.
2. PLC cap (defense-in-depth): on_playback now caps pure PLC at ~2s, then
emits digital silence so a stale stream can never hiss forever even if
remove_stream is skipped. Resets automatically on fresh packets.
3. No timeout / no ping: client never sent Ping, server had no last_seen /
reaper, so half-open connections (NAT timeout, wifi loss, sleep) left
ghost users forever. Fix: client Ping every 15s with RTT measurement,
ConnSession::last_seen bumped on every inbound TCP/UDP frame, steady_timer
reaper sweeps every 15s and drops sessions older than 45s (configurable
via server::Config).
4. UDP KEEPALIVE: client sends plaintext kFrameKeepalive every 5s; server
bumps last_seen + echoes back. Keeps NAT bindings alive and lets media
activity defer the reaper independently of TCP.
5. Graceful client disconnect: vc_disconnect() sends Disconnect{code=0} via
a flag-based io-thread exit (no double-close race); server handles
client-sent Disconnect with immediate close() + LEFT broadcast.
3 new tests: disconnect_left, plc_cap, reaper_timeout. 21/21 ctest green.
Docs: protocol.md §6/§7, voice.md §6, architecture.md §5, PROGRESS.md.
2026-06-18 01:18:33 +02:00
|
|
|
|
|
|
|
|
// Push one real frame to seed the decoder.
|
|
|
|
|
voicecat::audio::JitterBuffer::Frame f;
|
|
|
|
|
f.seq = 0;
|
|
|
|
|
f.timestamp = 0;
|
|
|
|
|
f.fec_present = false;
|
|
|
|
|
f.payload.assign(opus_buf, opus_buf + opus_len);
|
|
|
|
|
engine.push_recv_frame(ssrc, std::move(f));
|
|
|
|
|
|
|
|
|
|
// mix_for_test period — 480 frames @ 48kHz = 10ms (typical WASAPI shared period).
|
|
|
|
|
const uint32_t pb_frames = 480;
|
|
|
|
|
const int out_n = static_cast<int>(pb_frames) * 2; // stereo interleaved
|
|
|
|
|
std::vector<int16_t> out(static_cast<size_t>(out_n), 0);
|
|
|
|
|
|
|
|
|
|
// 1) Decode the real frame (first mix call) — seeds PLC state.
|
|
|
|
|
engine.mix_for_test(out.data(), pb_frames);
|
|
|
|
|
|
|
|
|
|
// 2) Drive ~50ms of pure PLC — should produce comfort noise (non-zero).
|
|
|
|
|
double early_rms = 0.0;
|
|
|
|
|
for (int i = 0; i < 5; ++i) {
|
|
|
|
|
engine.mix_for_test(out.data(), pb_frames);
|
|
|
|
|
early_rms = std::max(early_rms, rms(out.data(), out_n));
|
|
|
|
|
}
|
|
|
|
|
CHECK(early_rms > 1.0); // PLC of a loud sine is audible, not digital silence
|
|
|
|
|
|
|
|
|
|
// 3) Drive well past the 2s PLC cap (250 callbacks = 2.5s of output).
|
|
|
|
|
// After the cap, on_playback emits silence (memset 0) instead of PLC noise.
|
|
|
|
|
for (int i = 0; i < 250; ++i) {
|
|
|
|
|
engine.mix_for_test(out.data(), pb_frames);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// 4) The output must now be digital silence (all zeros), not comfort noise.
|
|
|
|
|
// Drain a couple more callbacks to flush any ring residue, then assert.
|
|
|
|
|
int64_t energy = 0;
|
|
|
|
|
for (int i = 0; i < 3; ++i) {
|
|
|
|
|
engine.mix_for_test(out.data(), pb_frames);
|
|
|
|
|
energy = std::max(energy, abs_energy(out.data(), out_n));
|
|
|
|
|
}
|
|
|
|
|
CHECK(energy == 0); // capped PLC = silence
|
|
|
|
|
|
|
|
|
|
// 5) Resumption: push a fresh real frame — PLC streak resets, audio returns.
|
|
|
|
|
voicecat::audio::JitterBuffer::Frame f2;
|
|
|
|
|
f2.seq = 1;
|
|
|
|
|
f2.timestamp = 200000; // far ahead — playout-clock re-sync snaps to it
|
|
|
|
|
f2.fec_present = false;
|
|
|
|
|
f2.payload.assign(opus_buf, opus_buf + opus_len);
|
|
|
|
|
engine.push_recv_frame(ssrc, std::move(f2));
|
|
|
|
|
|
|
|
|
|
double resume_rms = 0.0;
|
|
|
|
|
for (int i = 0; i < 5; ++i) { // a few calls to flush silence residue + decode real
|
|
|
|
|
engine.mix_for_test(out.data(), pb_frames);
|
|
|
|
|
resume_rms = std::max(resume_rms, rms(out.data(), out_n));
|
|
|
|
|
}
|
|
|
|
|
CHECK(resume_rms > 1.0); // real audio is back
|
|
|
|
|
|
|
|
|
|
engine.remove_stream(ssrc);
|
|
|
|
|
engine.stop();
|
|
|
|
|
enc.destroy();
|
|
|
|
|
|
|
|
|
|
if (g_failures == 0) {
|
|
|
|
|
std::printf("plc_cap: all checks passed (early_rms=%.1f resume_rms=%.1f)\n",
|
|
|
|
|
early_rms, resume_rms);
|
|
|
|
|
return 0;
|
|
|
|
|
}
|
|
|
|
|
std::printf("plc_cap: %d failure(s)\n", g_failures);
|
|
|
|
|
return 1;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#else
|
|
|
|
|
|
|
|
|
|
int main() {
|
|
|
|
|
std::printf("plc_cap: SKIP (VOICECAT_HAS_AUDIO or VOICECAT_HAS_OPUS not defined)\n");
|
|
|
|
|
return 0;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
#endif
|