/* * audio/apm_processor.h — Send-side audio processing module (AEC/NS/AGC/VAD). * * Design: docs/voice.md §11. Uses webrtc-audio-processing when VOICECAT_HAS_APM is defined; * falls back to a no-op passthrough (VAD always open, PCM unmodified) otherwise. */ #ifndef VOICECAT_AUDIO_APM_PROCESSOR_H #define VOICECAT_AUDIO_APM_PROCESSOR_H #include #include namespace voicecat::audio { class ApmProcessor { public: virtual ~ApmProcessor() = default; // Feed the most recent playback reference (for AEC). Call before process_capture(). virtual void process_render(const int16_t* pcm, int samples, int sample_rate) = 0; // Process one capture frame in-place (AEC, NS, AGC). // Returns true if VAD detects speech (or always true in passthrough mode). // Returns false → caller should skip encode/send (silence gate). virtual bool process_capture(int16_t* pcm, int samples, int sample_rate) = 0; // Update the VAD RMS threshold in-place (used by EnergyVadProcessor; no-op in passthrough). // Safe to call from any thread — EnergyVadProcessor stores it atomically. virtual void set_threshold(float) {} // Factory: returns a real APM if VOICECAT_HAS_APM is defined, else a passthrough. Used for // recv-side per-stream noise reduction (docs/voice.md §10) — gating doesn't apply there, so // this stays a passthrough until a real APM/NS backend exists (still inert; see // PROGRESS.md). Do not use this for the send-side VAD gate — see create_vad() below. static std::unique_ptr create(); // Factory for the send-side input gate (docs/voice.md §11): a lightweight, dependency-free // energy/RMS VAD with configurable threshold + hang-time. webrtc-audio-processing (the // originally-planned APM) has no working Windows/MSVC build upstream (GCC-only Meson build, // unfinished MinGW support, hard abseil-cpp dependency — see PROGRESS.md), so this is the // real v1 implementation behind the same ApmProcessor interface, not a passthrough. No AEC // — process_render() is a no-op here; that's a real limitation versus the originally-planned // APM, not just a deferred VAD. // rms_threshold: normalized 0.0-1.0 RMS-of-int16-range; default ~0.025. // hang_time_ms: how long the gate stays open after the last loud frame; default 300 ms // (matches AudioEngine's kTalkHangoverMs so "talking" and "gate open" agree). static std::unique_ptr create_vad(float rms_threshold = 0.025f, int64_t hang_time_ms = 300); }; } // namespace voicecat::audio #endif // VOICECAT_AUDIO_APM_PROCESSOR_H