From dd811a0bb891127e58afb7b10eac727e37910695 Mon Sep 17 00:00:00 2001 From: Talon Date: Sun, 20 Sep 2026 16:25:30 +0200 Subject: [PATCH] Fix audio clock drift and adaptive jitter buffering --- PROGRESS.md | 3 +- .../dotnet/VoiceCat.Mac/MacAudioBackend.cs | 5 +- .../apple/dotnet/VoiceCat.Mac/MacSettings.cs | 8 +- .../VoiceCat.Mac/MainWindowController.cs | 3 + .../VoiceCat.Mac/SettingsWindowController.cs | 10 +- clients/apple/dotnet/VoiceCat.iOS/AppModel.cs | 2 + .../dotnet/VoiceCat.iOS/IosAudioEngine.cs | 3 +- .../apple/dotnet/VoiceCat.iOS/IosSettings.cs | 7 +- .../dotnet/VoiceCat.iOS/SettingsController.cs | 11 ++- .../VoiceCat.App/Audio/WasapiAudioBackend.cs | 6 +- .../VoiceCat.App/Forms/AudioSettingsForm.cs | 27 +++++- .../windows/VoiceCat.App/Forms/MainForm.cs | 5 +- .../VoiceCat.App/Models/VoiceSettings.cs | 7 +- .../windows/VoiceCat.Managed/AudioControls.cs | 10 ++ .../VoiceCat.Managed/VoiceCatClient.cs | 6 +- docs/api-dotnet.md | 6 +- docs/voice.md | 26 +++-- .../src/VoiceCat.Audio/AdaptivePcmBuffer.cs | 97 +++++++++++++++++++ .../src/VoiceCat.Audio/AudioDeviceBackend.cs | 6 +- dotnet/src/VoiceCat.Audio/AudioEngine.cs | 15 ++- dotnet/src/VoiceCat.Audio/LocalStream.cs | 18 +++- dotnet/src/VoiceCat.Audio/ReceiveStream.cs | 95 ++++++++++++++---- .../tests/VoiceCat.Tests/AudioEngineTests.cs | 86 ++++++++++++++++ 23 files changed, 401 insertions(+), 61 deletions(-) create mode 100644 dotnet/src/VoiceCat.Audio/AdaptivePcmBuffer.cs diff --git a/PROGRESS.md b/PROGRESS.md index e2f8a57..66be97a 100644 --- a/PROGRESS.md +++ b/PROGRESS.md @@ -20,7 +20,8 @@ The supported source-of-truth layout is: ## Release gates -- Run real multi-person calls on Windows, macOS, and physical iOS hardware. +- Run real multi-person calls on Windows, macOS, and physical iOS hardware, including adaptive + 20/40/60 ms buffering, duration-aware DRED/FEC, and mismatched input/output endpoints. - Complete NVDA and VoiceOver navigation/announcement passes. - Exercise iOS background/lock, interruption, Bluetooth, route-change, ReplayKit, and iOS 27 ScreenCaptureKit paths on devices. diff --git a/clients/apple/dotnet/VoiceCat.Mac/MacAudioBackend.cs b/clients/apple/dotnet/VoiceCat.Mac/MacAudioBackend.cs index f7a2365..ba97148 100644 --- a/clients/apple/dotnet/VoiceCat.Mac/MacAudioBackend.cs +++ b/clients/apple/dotnet/VoiceCat.Mac/MacAudioBackend.cs @@ -139,7 +139,7 @@ internal sealed class MacAudioBackend : IAudioDeviceBackend private sealed class Playback : IAudioPlayback { - private readonly PcmRing pcm = new(32_768); + private readonly AdaptivePcmBuffer pcm = new(2); private readonly short[] renderScratch = new short[8_192]; private readonly AVAudioEngine engine = new(); private readonly AVAudioFormat format; @@ -187,6 +187,7 @@ internal sealed class MacAudioBackend : IAudioDeviceBackend } public void Write(ReadOnlySpan stereoPcm) => pcm.TryWrite(stereoPcm); + public int BufferMilliseconds { get => pcm.BufferMilliseconds; set => pcm.BufferMilliseconds = value; } private unsafe int Render(IntPtr isSilence, IntPtr _, uint frameCount, IntPtr outputData) { @@ -199,8 +200,6 @@ internal sealed class MacAudioBackend : IAudioDeviceBackend int frames = checked((int)frameCount); int requested = checked(frames * 2); if (bufferCount != 2 || requested > renderScratch.Length) return Fail("Core Audio returned an invalid planar stereo playback buffer."); - Span discard = stackalloc short[1_920]; - while (pcm.Count > 11_520) pcm.Read(discard[..Math.Min(discard.Length, pcm.Count - 11_520)]); Span source = renderScratch.AsSpan(0, requested); int read = pcm.Read(source); source[read..].Clear(); diff --git a/clients/apple/dotnet/VoiceCat.Mac/MacSettings.cs b/clients/apple/dotnet/VoiceCat.Mac/MacSettings.cs index a255ee6..b34f8cc 100644 --- a/clients/apple/dotnet/VoiceCat.Mac/MacSettings.cs +++ b/clients/apple/dotnet/VoiceCat.Mac/MacSettings.cs @@ -19,6 +19,7 @@ internal sealed class MacSettings internal bool AuxiliaryEnabled { get; set; } internal string? AuxiliaryDeviceId { get; set; } internal float AuxiliaryGain { get; set; } = 1f; + internal int AudioBufferMilliseconds { get; set; } = 40; internal bool EventSounds { get; set; } = true; internal bool SpokenEvents { get; set; } internal float EventVolume { get; set; } = 1f; @@ -28,9 +29,9 @@ internal sealed class MacSettings internal void Load() { NSString[] keys = [(NSString)"voice.inputMode", (NSString)"voice.vadThreshold", (NSString)"voice.inputGain", (NSString)"voice.outputGain", - (NSString)"voice.pttKeyCode", (NSString)"voice.auxGain", (NSString)"feedback.sounds", (NSString)"feedback.volume"]; + (NSString)"voice.pttKeyCode", (NSString)"voice.auxGain", (NSString)"voice.audioBufferMs", (NSString)"feedback.sounds", (NSString)"feedback.volume"]; NSObject[] defaults = [NSNumber.FromInt32((int)AudioInputMode.VoiceActivation), NSNumber.FromFloat(0.05f), - NSNumber.FromFloat(1f), NSNumber.FromFloat(0.8f), NSNumber.FromInt32(0x60), NSNumber.FromFloat(1f), + NSNumber.FromFloat(1f), NSNumber.FromFloat(0.8f), NSNumber.FromInt32(0x60), NSNumber.FromFloat(1f), NSNumber.FromInt32(40), NSNumber.FromBoolean(true), NSNumber.FromFloat(1f)]; values.RegisterDefaults(new NSDictionary(keys, defaults)); InputMode = Enum.IsDefined(typeof(AudioInputMode), (int)values.IntForKey("voice.inputMode")) @@ -46,6 +47,7 @@ internal sealed class MacSettings AuxiliaryEnabled = values.BoolForKey("voice.auxEnabled"); AuxiliaryDeviceId = EmptyToNull(values.StringForKey("voice.auxDeviceUID")); AuxiliaryGain = Math.Clamp(values.FloatForKey("voice.auxGain"), 0f, 4f); + int audioBuffer = checked((int)values.IntForKey("voice.audioBufferMs")); AudioBufferMilliseconds = audioBuffer is 20 or 40 or 60 ? audioBuffer : 40; EventSounds = values.BoolForKey("feedback.sounds"); SpokenEvents = values.BoolForKey("feedback.speech"); EventVolume = Math.Clamp(values.FloatForKey("feedback.volume"), 0f, 1f); @@ -59,7 +61,7 @@ internal sealed class MacSettings values.SetFloat(InputGain, "voice.inputGain"); values.SetFloat(OutputGain, "voice.outputGain"); values.SetBool(InputNoiseReduction, "voice.inputNoiseReduction"); values.SetBool(StereoMicrophone, "voice.stereoMic"); values.SetInt(PushToTalkKeyCode, "voice.pttKeyCode"); Set("voice.inputDevice", InputDeviceId); Set("voice.outputDevice", OutputDeviceId); - values.SetBool(AuxiliaryEnabled, "voice.auxEnabled"); Set("voice.auxDeviceUID", AuxiliaryDeviceId); values.SetFloat(AuxiliaryGain, "voice.auxGain"); + values.SetBool(AuxiliaryEnabled, "voice.auxEnabled"); Set("voice.auxDeviceUID", AuxiliaryDeviceId); values.SetFloat(AuxiliaryGain, "voice.auxGain"); values.SetInt(AudioBufferMilliseconds, "voice.audioBufferMs"); values.SetBool(EventSounds, "feedback.sounds"); values.SetBool(SpokenEvents, "feedback.speech"); values.SetFloat(EventVolume, "feedback.volume"); values.SetBool(SelfTalkSounds, "feedback.selfTalk"); values.SetBool(PushToTalkSound, "feedback.ptt"); values.Synchronize(); } diff --git a/clients/apple/dotnet/VoiceCat.Mac/MainWindowController.cs b/clients/apple/dotnet/VoiceCat.Mac/MainWindowController.cs index 3aab98e..8b4477b 100644 --- a/clients/apple/dotnet/VoiceCat.Mac/MainWindowController.cs +++ b/clients/apple/dotnet/VoiceCat.Mac/MainWindowController.cs @@ -210,6 +210,8 @@ internal sealed class MainWindowController : NSWindowController if (settings.InputMode != AudioInputMode.PushToTalk) { client.Audio.PushToTalk = false; pushToTalkEngaged = false; } client.Audio.InputGain = settings.InputGain; client.Audio.OutputGain = settings.OutputGain; client.Audio.InputNoiseReduction = settings.InputNoiseReduction; + client.Audio.DeviceBufferMilliseconds = settings.AudioBufferMilliseconds; + if (playback is not null) playback.BufferMilliseconds = settings.AudioBufferMilliseconds; if (auxiliaryStreamId != 0) client.Audio.SetLocalGain(auxiliaryStreamId, settings.AuxiliaryGain); } @@ -367,6 +369,7 @@ internal sealed class MainWindowController : NSWindowController activeOutputDevice = settings.OutputDeviceId ?? SelectedDevice(outputDevice); activeStereo = settings.StereoMicrophone; playback = audioBackend.OpenPlayback(activeOutputDevice); + playback.BufferMilliseconds = settings.AudioBufferMilliseconds; client.Audio.MixedPcm += PlayMixedPcm; StreamInfo stream = await client.StartStreamAsync(StreamKind.StreamMic, "Microphone", settings.StereoMicrophone ? 2 : 1); microphoneStreamId = stream.StreamId; diff --git a/clients/apple/dotnet/VoiceCat.Mac/SettingsWindowController.cs b/clients/apple/dotnet/VoiceCat.Mac/SettingsWindowController.cs index fe3c75d..83fd948 100644 --- a/clients/apple/dotnet/VoiceCat.Mac/SettingsWindowController.cs +++ b/clients/apple/dotnet/VoiceCat.Mac/SettingsWindowController.cs @@ -26,15 +26,18 @@ internal sealed class SettingsWindowController : NSWindowController private readonly NSButton pttSound = NSButton.CreateCheckbox("Push-to-talk cue", () => { }); private readonly NSButton pttKey = new(new CGRect(220, 12, 200, 32)) { Title = "Change PTT key…" }; private readonly NSSlider eventVolume = new(new CGRect(160, 505, 260, 24)) { MinValue = 0, MaxValue = 100 }; + private readonly NSPopUpButton audioBuffer = new(new CGRect(160, 545, 260, 28), false); internal SettingsWindowController(MainWindowController owner, MacSettings settings, MacAudioBackend audio) : base( - new NSWindow(new CGRect(0, 0, 460, 620), NSWindowStyle.Titled | NSWindowStyle.Closable, + new NSWindow(new CGRect(0, 0, 460, 660), NSWindowStyle.Titled | NSWindowStyle.Closable, NSBackingStore.Buffered, false)) { this.owner = owner; this.settings = settings; this.audio = audio; Window!.Title = "VoiceCat Settings"; Window.Center(); NSView view = Window.ContentView!; mode.AddItems(["Voice activation", "Push to talk", "Always on"]); + audioBuffer.AddItems(["Low latency (20 ms)", "Balanced (40 ms)", "Stable (60 ms)"]); + Add(view, "Audio buffering", audioBuffer, 555); Add(view, "Sound volume", eventVolume, 515); Add(view, "Input mode", mode, 465); Add(view, "VAD sensitivity", vad, 425); Add(view, "Microphone volume", inputGain, 385); noiseReduction.Frame = new CGRect(160, 340, 260, 24); stereo.Frame = new CGRect(160, 315, 260, 24); @@ -46,7 +49,7 @@ internal sealed class SettingsWindowController : NSWindowController selfTalk.Frame = new CGRect(265, 45, 180, 24); pttSound.Frame = new CGRect(25, 18, 180, 24); view.AddSubview(sounds); view.AddSubview(speech); view.AddSubview(selfTalk); view.AddSubview(pttSound); view.AddSubview(pttKey); foreach (NSControl control in new NSControl[] { mode, vad, inputGain, noiseReduction, stereo, input, output, - outputGain, auxiliary, auxiliaryDevice, auxiliaryGain, sounds, speech, selfTalk, pttSound, eventVolume }) control.Activated += Changed; + outputGain, auxiliary, auxiliaryDevice, auxiliaryGain, sounds, speech, selfTalk, pttSound, eventVolume, audioBuffer }) control.Activated += Changed; pttKey.Activated += CapturePushToTalkKey; Populate(input, true, settings.InputDeviceId); Populate(output, false, settings.OutputDeviceId); Populate(auxiliaryDevice, true, settings.AuxiliaryDeviceId); LoadValues(); SetAccessibility(); @@ -76,6 +79,7 @@ internal sealed class SettingsWindowController : NSWindowController eventVolume.DoubleValue = settings.EventVolume * 100; selfTalk.State = State(settings.SelfTalkSounds); pttSound.State = State(settings.PushToTalkSound); pttKey.Title = $"PTT key code: {settings.PushToTalkKeyCode}"; + audioBuffer.SelectItem(settings.AudioBufferMilliseconds switch { 20 => 0, 60 => 2, _ => 1 }); UpdateEnabled(); } @@ -102,6 +106,7 @@ internal sealed class SettingsWindowController : NSWindowController settings.AuxiliaryGain = (float)auxiliaryGain.DoubleValue / 100; settings.EventSounds = On(sounds); settings.EventVolume = (float)eventVolume.DoubleValue / 100; settings.SpokenEvents = On(speech); settings.SelfTalkSounds = On(selfTalk); settings.PushToTalkSound = On(pttSound); + settings.AudioBufferMilliseconds = audioBuffer.IndexOfSelectedItem switch { 0 => 20, 2 => 60, _ => 40 }; settings.Save(); UpdateEnabled(); await owner.ApplySettingsAsync(); } @@ -113,6 +118,7 @@ internal sealed class SettingsWindowController : NSWindowController ((INSAccessibility)output).AccessibilityLabel = "Output device"; ((INSAccessibility)outputGain).AccessibilityLabel = "Output volume"; ((INSAccessibility)auxiliaryDevice).AccessibilityLabel = "Auxiliary input device"; ((INSAccessibility)auxiliaryGain).AccessibilityLabel = "Auxiliary input volume"; ((INSAccessibility)eventVolume).AccessibilityLabel = "Event sound volume"; + ((INSAccessibility)audioBuffer).AccessibilityLabel = "Audio buffering"; } private static bool On(NSButton button) => button.State == NSCellStateValue.On; private static NSCellStateValue State(bool value) => value ? NSCellStateValue.On : NSCellStateValue.Off; diff --git a/clients/apple/dotnet/VoiceCat.iOS/AppModel.cs b/clients/apple/dotnet/VoiceCat.iOS/AppModel.cs index fe76eb5..53ec62f 100644 --- a/clients/apple/dotnet/VoiceCat.iOS/AppModel.cs +++ b/clients/apple/dotnet/VoiceCat.iOS/AppModel.cs @@ -266,6 +266,8 @@ internal sealed class AppModel owner.Audio.InputMode = settings.InputMode; owner.Audio.VadThreshold = settings.VadThreshold; owner.Audio.InputGain = settings.InputGain; owner.Audio.OutputGain = settings.OutputGain; owner.Audio.InputNoiseReduction = settings.InputNoiseReduction; + owner.Audio.DeviceBufferMilliseconds = settings.AudioBufferMilliseconds; + IosAudioEngine.Shared.BufferMilliseconds = settings.AudioBufferMilliseconds; } private void PollAudio() diff --git a/clients/apple/dotnet/VoiceCat.iOS/IosAudioEngine.cs b/clients/apple/dotnet/VoiceCat.iOS/IosAudioEngine.cs index fbfbcaf..33bcf91 100644 --- a/clients/apple/dotnet/VoiceCat.iOS/IosAudioEngine.cs +++ b/clients/apple/dotnet/VoiceCat.iOS/IosAudioEngine.cs @@ -8,7 +8,7 @@ namespace VoiceCat.iOS; internal sealed class IosAudioEngine { internal static IosAudioEngine Shared { get; } = new(); - private readonly PcmRing playbackRing = new(131_072); + private readonly AdaptivePcmBuffer playbackRing = new(2, capacityFrames: 65_536); private readonly short[] renderScratch = new short[16_384]; private AVAudioEngine? engine; private AVAudioSourceNode? source; @@ -28,6 +28,7 @@ internal sealed class IosAudioEngine private bool inputProvided; private bool tapInstalled; internal bool IsConnected { get; private set; } + internal int BufferMilliseconds { get => playbackRing.BufferMilliseconds; set => playbackRing.BufferMilliseconds = value; } private IosAudioEngine() { microphoneWorker = PumpMicrophoneAsync(microphoneStop.Token); } diff --git a/clients/apple/dotnet/VoiceCat.iOS/IosSettings.cs b/clients/apple/dotnet/VoiceCat.iOS/IosSettings.cs index c4656ae..7479459 100644 --- a/clients/apple/dotnet/VoiceCat.iOS/IosSettings.cs +++ b/clients/apple/dotnet/VoiceCat.iOS/IosSettings.cs @@ -17,13 +17,14 @@ internal sealed class IosSettings internal float EventVolume { get; set; } = 1f; internal bool SelfTalkSounds { get; set; } internal bool PushToTalkSound { get; set; } + internal int AudioBufferMilliseconds { get; set; } = 40; internal void Load() { NSString[] keys = [(NSString)"voice.inputMode", (NSString)"voice.vadThreshold", (NSString)"voice.inputGain", - (NSString)"voice.outputGain", (NSString)"feedback.sounds", (NSString)"feedback.volume"]; + (NSString)"voice.outputGain", (NSString)"voice.audioBufferMs", (NSString)"feedback.sounds", (NSString)"feedback.volume"]; NSObject[] defaults = [NSNumber.FromInt32((int)AudioInputMode.VoiceActivation), NSNumber.FromFloat(0.025f), - NSNumber.FromFloat(1f), NSNumber.FromFloat(1f), NSNumber.FromBoolean(true), NSNumber.FromFloat(1f)]; + NSNumber.FromFloat(1f), NSNumber.FromFloat(1f), NSNumber.FromInt32(40), NSNumber.FromBoolean(true), NSNumber.FromFloat(1f)]; values.RegisterDefaults(new NSDictionary(keys, defaults)); int inputMode = checked((int)values.IntForKey("voice.inputMode")); InputMode = Enum.IsDefined(typeof(AudioInputMode), inputMode) ? (AudioInputMode)inputMode : AudioInputMode.VoiceActivation; @@ -31,6 +32,7 @@ internal sealed class IosSettings InputGain = Math.Clamp(values.FloatForKey("voice.inputGain"), 0f, 4f); OutputGain = Math.Clamp(values.FloatForKey("voice.outputGain"), 0f, 1f); InputNoiseReduction = values.BoolForKey("voice.inputNoiseReduction"); + int audioBuffer = checked((int)values.IntForKey("voice.audioBufferMs")); AudioBufferMilliseconds = audioBuffer is 20 or 40 or 60 ? audioBuffer : 40; EventSounds = values.BoolForKey("feedback.sounds"); SpokenEvents = values.BoolForKey("feedback.speech"); EventVolume = Math.Clamp(values.FloatForKey("feedback.volume"), 0f, 1f); @@ -42,6 +44,7 @@ internal sealed class IosSettings { values.SetInt((int)InputMode, "voice.inputMode"); values.SetFloat(VadThreshold, "voice.vadThreshold"); values.SetFloat(InputGain, "voice.inputGain"); values.SetFloat(OutputGain, "voice.outputGain"); + values.SetInt(AudioBufferMilliseconds, "voice.audioBufferMs"); values.SetBool(InputNoiseReduction, "voice.inputNoiseReduction"); values.SetBool(EventSounds, "feedback.sounds"); values.SetBool(SpokenEvents, "feedback.speech"); values.SetFloat(EventVolume, "feedback.volume"); values.SetBool(SelfTalkSounds, "feedback.selfTalk"); values.SetBool(PushToTalkSound, "feedback.ptt"); values.Synchronize(); diff --git a/clients/apple/dotnet/VoiceCat.iOS/SettingsController.cs b/clients/apple/dotnet/VoiceCat.iOS/SettingsController.cs index 3095526..8159e97 100644 --- a/clients/apple/dotnet/VoiceCat.iOS/SettingsController.cs +++ b/clients/apple/dotnet/VoiceCat.iOS/SettingsController.cs @@ -14,23 +14,23 @@ internal sealed class SettingsController : UITableViewController public override void ViewDidLoad() { base.ViewDidLoad(); TableView.RegisterClassForCellReuse(typeof(UITableViewCell), "setting"); } private void Reload() => TableView.ReloadData(); public override nint NumberOfSections(UITableView tableView) => 5; - public override nint RowsInSection(UITableView tableView, nint section) => section switch { 0 => 4, 1 => 4, 2 => 5, 3 => model.Client?.Permissions is { } p && (p.IsAdmin || p.CanAdminAccounts) ? 1 : 0, _ => 2 }; + public override nint RowsInSection(UITableView tableView, nint section) => section switch { 0 => 5, 1 => 4, 2 => 5, 3 => model.Client?.Permissions is { } p && (p.IsAdmin || p.CanAdminAccounts) ? 1 : 0, _ => 2 }; public override string? TitleForHeader(UITableView tableView, nint section) => section switch { 0 => "Audio", 1 => "Voice", 2 => "Notifications", 3 => "Administration", _ => "Server" }; public override UITableViewCell GetCell(UITableView tableView, NSIndexPath path) { UITableViewCell cell = tableView.DequeueReusableCell("setting", path); cell.AccessoryView = null; cell.Accessory = UITableViewCellAccessory.None; string title = path.Section switch { - 0 => path.Row switch { 0 => "Audio preset", 1 => "Speaker output", 2 => "Advanced audio", _ => model.ScreenSharing ? "Stop sharing screen audio" : "Share screen audio" }, + 0 => path.Row switch { 0 => "Audio preset", 1 => "Speaker output", 2 => "Advanced audio", 3 => "Audio buffering", _ => model.ScreenSharing ? "Stop sharing screen audio" : "Share screen audio" }, 1 => path.Row switch { 0 => "Input mode", 1 => $"VAD threshold: {model.Settings.VadThreshold:0.000}", 2 => $"Microphone volume: {model.Settings.InputGain:P0}", _ => "Microphone noise reduction" }, 2 => path.Row switch { 0 => "Event sounds", 1 => $"Sound volume: {model.Settings.EventVolume:P0}", 2 => "Speak events", 3 => "Own voice activity sounds", _ => "Push-to-talk cue" }, 3 => "Manage accounts", _ => path.Row == 0 ? "Disconnect" : "VoiceCat 0.0.1" }; - var content = cell.DefaultContentConfiguration; content.Text = title; content.SecondaryText = path.Section == 0 && path.Row == 0 ? IosAudioRouter.Shared.Preset.ToString() : path.Section == 1 && path.Row == 0 ? model.Settings.InputMode.ToString() : path.Section == 0 && path.Row == 3 && model.ScreenSharing ? "Sharing" : null; cell.ContentConfiguration = content; cell.AccessibilityLabel = title; + var content = cell.DefaultContentConfiguration; content.Text = title; content.SecondaryText = path.Section == 0 && path.Row == 0 ? IosAudioRouter.Shared.Preset.ToString() : path.Section == 1 && path.Row == 0 ? model.Settings.InputMode.ToString() : path.Section == 0 && path.Row == 3 ? $"{model.Settings.AudioBufferMilliseconds} ms" : path.Section == 0 && path.Row == 4 && model.ScreenSharing ? "Sharing" : null; cell.ContentConfiguration = content; cell.AccessibilityLabel = title + (content.SecondaryText is null ? "" : ", " + content.SecondaryText); if (path.Section == 0 && path.Row == 1) cell.AccessoryView = Toggle(IosAudioRouter.Shared.ForceSpeaker, "Speaker output", (_, _) => IosAudioRouter.Shared.SetForceSpeaker(((UISwitch)cell.AccessoryView!).On)); else if (path.Section == 1 && path.Row == 3) cell.AccessoryView = Toggle(model.Settings.InputNoiseReduction, title, (_, _) => { model.Settings.InputNoiseReduction = ((UISwitch)cell.AccessoryView!).On; model.ApplyVoiceSettings(); }); else if (path.Section == 2 && path.Row is 0 or 2 or 3 or 4) { bool value = path.Row switch { 0 => model.Settings.EventSounds, 2 => model.Settings.SpokenEvents, 3 => model.Settings.SelfTalkSounds, _ => model.Settings.PushToTalkSound }; int row = path.Row; cell.AccessoryView = Toggle(value, title, (_, _) => { bool on = ((UISwitch)cell.AccessoryView!).On; if (row == 0) model.Settings.EventSounds = on; else if (row == 2) model.Settings.SpokenEvents = on; else if (row == 3) model.Settings.SelfTalkSounds = on; else model.Settings.PushToTalkSound = on; model.Save(); }); } - else if (path.Section == 0 && path.Row == 2 || path.Section == 3) cell.Accessory = UITableViewCellAccessory.DisclosureIndicator; + else if (path.Section == 0 && path.Row is 2 or 3 || path.Section == 3) cell.Accessory = UITableViewCellAccessory.DisclosureIndicator; return cell; } public override async void RowSelected(UITableView tableView, NSIndexPath path) @@ -39,7 +39,8 @@ internal sealed class SettingsController : UITableViewController { if (path.Section == 0 && path.Row == 0) Choice("Audio preset", Enum.GetValues().Select(value => value.ToString()).ToArray(), index => IosAudioRouter.Shared.SelectPreset(Enum.GetValues()[index])); else if (path.Section == 0 && path.Row == 2) NavigationController?.PushViewController(new AdvancedAudioController(), true); - else if (path.Section == 0 && path.Row == 3) ShowScreenSharing(); + else if (path.Section == 0 && path.Row == 3) Choice("Audio buffering", ["Low latency (20 ms)", "Balanced (40 ms)", "Stable (60 ms)"], index => { model.Settings.AudioBufferMilliseconds = index switch { 0 => 20, 2 => 60, _ => 40 }; model.ApplyVoiceSettings(); }); + else if (path.Section == 0 && path.Row == 4) ShowScreenSharing(); else if (path.Section == 1 && path.Row == 0) Choice("Input mode", ["Voice activation", "Push to talk", "Always on"], index => { model.Settings.InputMode = (AudioInputMode)index; model.ApplyVoiceSettings(); }); else if (path.Section == 1 && path.Row == 1) Slider("Voice activation threshold", 0.001f, 0.1f, model.Settings.VadThreshold, value => { model.Settings.VadThreshold = value; model.ApplyVoiceSettings(); }); else if (path.Section == 1 && path.Row == 2) Slider("Microphone volume", 0, 4, model.Settings.InputGain, value => { model.Settings.InputGain = value; model.ApplyVoiceSettings(); }); diff --git a/clients/windows/VoiceCat.App/Audio/WasapiAudioBackend.cs b/clients/windows/VoiceCat.App/Audio/WasapiAudioBackend.cs index 714deb0..1db9668 100644 --- a/clients/windows/VoiceCat.App/Audio/WasapiAudioBackend.cs +++ b/clients/windows/VoiceCat.App/Audio/WasapiAudioBackend.cs @@ -34,7 +34,7 @@ public sealed class WasapiAudioBackend : IAudioDeviceBackend private sealed class Playback : IAudioPlayback { private readonly string? deviceId; - private readonly PcmRing pcm = new(32768); + private readonly AdaptivePcmBuffer pcm = new(2); private readonly ManualResetEventSlim initialized = new(false); private readonly Thread thread; private volatile bool running = true; @@ -47,6 +47,7 @@ public sealed class WasapiAudioBackend : IAudioDeviceBackend thread.Start(); if (!initialized.Wait(5000) || !ready) { Dispose(); throw new InvalidOperationException("WASAPI playback could not start."); } } + public int BufferMilliseconds { get => pcm.BufferMilliseconds; set => pcm.BufferMilliseconds = value; } public void Write(ReadOnlySpan stereoPcm) => pcm.TryWrite(stereoPcm); private unsafe void Work() { @@ -68,7 +69,6 @@ public sealed class WasapiAudioBackend : IAudioDeviceBackend render = (IAudioRenderClient)output; if (client.Start() < 0) return; ready = true; initialized.Set(); - Span discard = stackalloc short[1920]; while (running) { bufferReady.WaitOne(20); // Scheduling wait is outside the buffer-fill cycle. @@ -77,8 +77,6 @@ public sealed class WasapiAudioBackend : IAudioDeviceBackend if (frames == 0) continue; if (render.GetBuffer(frames, out nint buffer) < 0) break; var destination = new Span((void*)buffer, checked((int)frames * 2)); - // Keep queued playback bounded to ~120 ms, then fill underflow with silence. - while (pcm.Count > 11520) pcm.Read(discard); int count = pcm.Read(destination); destination[count..].Clear(); if (render.ReleaseBuffer(frames, 0) < 0) break; } diff --git a/clients/windows/VoiceCat.App/Forms/AudioSettingsForm.cs b/clients/windows/VoiceCat.App/Forms/AudioSettingsForm.cs index bfefb5e..9ad35e4 100644 --- a/clients/windows/VoiceCat.App/Forms/AudioSettingsForm.cs +++ b/clients/windows/VoiceCat.App/Forms/AudioSettingsForm.cs @@ -34,6 +34,7 @@ public sealed class AudioSettingsForm : Form private readonly bool _origAuxEnabled; private readonly string? _origAuxDeviceId; private readonly int _origAuxGain; + private readonly int _origAudioBufferMilliseconds; private readonly ComboBox _cboDevice; private readonly Button _btnRefresh; @@ -54,6 +55,7 @@ public sealed class AudioSettingsForm : Form private readonly Button _btnAuxRefresh; private readonly Label _lblAuxGain; private readonly TrackBar _trkAuxGain; + private readonly ComboBox _cboBuffer; private Keys _pttKey; @@ -80,6 +82,7 @@ public sealed class AudioSettingsForm : Form _origAuxEnabled = settings.AuxEnabled; _origAuxDeviceId = settings.AuxDeviceId; _origAuxGain = settings.AuxGain; + _origAudioBufferMilliseconds = settings.AudioBufferMilliseconds; Text = "Audio settings"; FormBorderStyle = FormBorderStyle.FixedDialog; @@ -87,7 +90,7 @@ public sealed class AudioSettingsForm : Form MinimizeBox = false; StartPosition = FormStartPosition.CenterParent; AutoScaleMode = AutoScaleMode.Font; - ClientSize = new Size(420, 639); + ClientSize = new Size(420, 690); // ── Device row ──────────────────────────────────────────────────────── var lblDevice = new Label @@ -319,19 +322,34 @@ public sealed class AudioSettingsForm : Form "Volume of the aux input stream. 100 is unity gain; range 0–400 percent."; _trkAuxGain.Scroll += TrkAuxGain_Scroll; + var lblBuffer = new Label { Text = "Audio &buffering:", Location = new Point(12, 590), AutoSize = true }; + _cboBuffer = new ComboBox + { + Location = new Point(130, 586), Width = 190, DropDownStyle = ComboBoxStyle.DropDownList, + AccessibleName = "Audio buffering", + AccessibleDescription = "Local capture and playback buffering. Lower values reduce latency; higher values tolerate more scheduling jitter." + }; + _cboBuffer.Items.AddRange(["Low latency (20 ms)", "Balanced (40 ms)", "Stable (60 ms)"]); + _cboBuffer.SelectedIndex = settings.AudioBufferMilliseconds switch { 20 => 0, 60 => 2, _ => 1 }; + _cboBuffer.SelectedIndexChanged += (_, _) => + { + _settings.AudioBufferMilliseconds = _cboBuffer.SelectedIndex switch { 0 => 20, 2 => 60, _ => 40 }; + _client.SetAudioBufferMilliseconds(_settings.AudioBufferMilliseconds); + }; + // ── OK / Cancel ─────────────────────────────────────────────────────── var btnOk = new Button { Text = "&OK", DialogResult = DialogResult.OK, - Location = new Point(228, 600), + Location = new Point(228, 642), Size = new Size(80, 27), }; var btnCancel = new Button { Text = "&Cancel", DialogResult = DialogResult.Cancel, - Location = new Point(316, 600), + Location = new Point(316, 642), Size = new Size(80, 27), }; @@ -350,6 +368,7 @@ public sealed class AudioSettingsForm : Form _radioPtt, _lblPttKey, _btnChangePtt, _chkSystemWidePtt, _radioAlwaysOn, lblGain, _trkGain, _chkNoiseReduction, _chkStereoMic, _chkAux, _lblAuxDevice, _cboAuxDevice, _btnAuxRefresh, _lblAuxGain, _trkAuxGain, + lblBuffer, _cboBuffer, btnOk, btnCancel, ]); @@ -534,6 +553,8 @@ public sealed class AudioSettingsForm : Form _settings.StereoMic = _origStereoMic; _settings.PttKey = (int)_origPttKey; _settings.SystemWidePtt = _origSystemWidePtt; + _settings.AudioBufferMilliseconds = _origAudioBufferMilliseconds; + _client.SetAudioBufferMilliseconds(_origAudioBufferMilliseconds); if (_micStreamId != 0) { diff --git a/clients/windows/VoiceCat.App/Forms/MainForm.cs b/clients/windows/VoiceCat.App/Forms/MainForm.cs index 8bdfc95..46d8480 100644 --- a/clients/windows/VoiceCat.App/Forms/MainForm.cs +++ b/clients/windows/VoiceCat.App/Forms/MainForm.cs @@ -109,8 +109,11 @@ public partial class MainForm : Form BootstrapFromServer(); } - private void ApplyPersistedVoiceSettings() => + private void ApplyPersistedVoiceSettings() + { _pttKey = (Keys)_voiceSettings.PttKey; + _client.SetAudioBufferMilliseconds(_voiceSettings.AudioBufferMilliseconds); + } // ── Startup ────────────────────────────────────────────────────────────── diff --git a/clients/windows/VoiceCat.App/Models/VoiceSettings.cs b/clients/windows/VoiceCat.App/Models/VoiceSettings.cs index 199311a..5a7dc17 100644 --- a/clients/windows/VoiceCat.App/Models/VoiceSettings.cs +++ b/clients/windows/VoiceCat.App/Models/VoiceSettings.cs @@ -59,6 +59,9 @@ public sealed class VoiceSettings /// to the captured PCM before feeding (the core's input gain is mic-only and global). public int AuxGain { get; set; } = 100; + /// Local capture/playback safety buffer in milliseconds: 20, 40, or 60. + public int AudioBufferMilliseconds { get; set; } = 40; + private static readonly JsonSerializerOptions JsonOptions = new() { WriteIndented = true }; private static string AppDataDir => Path.Combine( @@ -72,7 +75,9 @@ public sealed class VoiceSettings { if (!File.Exists(FilePath)) return new VoiceSettings(); string json = File.ReadAllText(FilePath); - return JsonSerializer.Deserialize(json) ?? new VoiceSettings(); + VoiceSettings result = JsonSerializer.Deserialize(json) ?? new VoiceSettings(); + if (result.AudioBufferMilliseconds is not (20 or 40 or 60)) result.AudioBufferMilliseconds = 40; + return result; } catch { diff --git a/clients/windows/VoiceCat.Managed/AudioControls.cs b/clients/windows/VoiceCat.Managed/AudioControls.cs index ad67d4f..6c107f5 100644 --- a/clients/windows/VoiceCat.Managed/AudioControls.cs +++ b/clients/windows/VoiceCat.Managed/AudioControls.cs @@ -55,6 +55,16 @@ public sealed partial class VoiceCatClient public VcResult SetOutputVolume(float gain) { if (!float.IsFinite(gain) || gain is < 0 or > 4) return VcResult.InvalidArg; core.Audio.OutputGain = gain; return VcResult.Ok; } public VcResult SetInputGain(float gain) { if (!float.IsFinite(gain) || gain is < 0 or > 4) return VcResult.InvalidArg; core.Audio.InputGain = gain; return VcResult.Ok; } public VcResult SetInputNoiseReduction(bool enabled) { core.Audio.InputNoiseReduction = enabled; return VcResult.Ok; } + public VcResult SetAudioBufferMilliseconds(int milliseconds) + { + try + { + core.Audio.DeviceBufferMilliseconds = milliseconds; + if (Volatile.Read(ref playback) is { } output) output.BufferMilliseconds = milliseconds; + return VcResult.Ok; + } + catch (ArgumentOutOfRangeException) { return VcResult.InvalidArg; } + } public VcResult SetRemoteStream(uint userId, uint streamId, float gain, bool muted, bool nr) { try { core.Audio.SetRemotePlayback(userId, streamId, gain, muted, nr); return VcResult.Ok; } catch { return VcResult.InvalidArg; } } public (VcResult Result, RemoteStreamState? State) GetRemoteStream(uint userId, uint streamId) diff --git a/clients/windows/VoiceCat.Managed/VoiceCatClient.cs b/clients/windows/VoiceCat.Managed/VoiceCatClient.cs index 6d896f0..c36f2e6 100644 --- a/clients/windows/VoiceCat.Managed/VoiceCatClient.cs +++ b/clients/windows/VoiceCat.Managed/VoiceCatClient.cs @@ -159,7 +159,11 @@ public sealed partial class VoiceCatClient : IDisposable { try { - if (backend is not null && playback is null) playback = backend.OpenPlayback(); + if (backend is not null && playback is null) + { + playback = backend.OpenPlayback(); + playback.BufferMilliseconds = core.Audio.DeviceBufferMilliseconds; + } var result = core.SubscribeVoiceAsync().GetAwaiter().GetResult(); if (!result.Ok) { Interlocked.Exchange(ref playback, null)?.Dispose(); } return result.Ok ? VcResult.Ok : VcResult.Audio; diff --git a/docs/api-dotnet.md b/docs/api-dotnet.md index fbce67e..f881b98 100644 --- a/docs/api-dotnet.md +++ b/docs/api-dotnet.md @@ -177,8 +177,10 @@ Noise reduction does not gate speech. `TimeProvider` timestamps; tests inject a clock. Threshold changes are atomic; all processing state otherwise has one owner. Codec/DSP processing methods allocate no managed memory after initialization, verified across 1,000 combined cycles. They run -on a managed worker, never the native real-time device callback. Native device rings, -jitter, mixer, and audio scheduling remain later work. +on a managed worker, never the native real-time device callback. Managed capture/playback +rings compensate independent clock drift around a configurable 20/40/60 ms target. Receive +jitter is duration-aware, reserves one codec frame for DRED/FEC look-ahead, and decodes each +channel packet at its actual duration before the fixed 20 ms mixer stage. ## Initial managed server diff --git a/docs/voice.md b/docs/voice.md index cb1502a..9cb744a 100644 --- a/docs/voice.md +++ b/docs/voice.md @@ -117,6 +117,9 @@ Guidance baked into defaults / docs: (accumulating two 960-frames for a 40 ms channel, splitting each into two 480-frames for a 10 ms channel, etc.). This keeps the hardware/`vc_stream_feed_pcm` contract a single 48 kHz clock regardless of the channel's window — see `vc_client::on_capture_frame`. + The device/mixer quantum is not the Opus packet duration: managed send streams reframe it + into the channel's 5/10/20/40/60 ms packets, and receive streams decode at that duration + before slicing decoded PCM back into 20 ms mixer blocks. - **Mode/bitrate:** speech channels → `MONO`, `VOIP`, 24–32 kbps, DTX on, FEC on. Music/screen-audio channels → `STEREO`, `AUDIO`, 96–128 kbps, DTX off, FEC optional. - **`application`:** `VOIP` for talk, `AUDIO` for music/screen-share, `LOWDELAY` for @@ -152,26 +155,30 @@ straight through to PLC. ## 5. Jitter buffer Each receiver keeps an **adaptive jitter buffer per ssrc** with **bounded-depth playout** -(`core/src/audio/audio_engine.cpp` — `JitterBuffer` + `AudioEngine::on_playback`). +(`dotnet/src/VoiceCat.Audio/ReceiveStream.cs`). - Frames are inserted by `timestamp`; playback reads in order at the device callback rate. - **The playout clock is always bounded against the stream's *leading edge* (newest buffered - frame), never re-synced to the oldest.** The clock free-runs at the playback hardware rate, + frame), never re-synced to the oldest.** The managed playout clock is isolated from hardware + drift by the adaptive PCM ring, while the sender omits VAD/PTT/DTX silence from its timestamps, so the two diverge across gaps and late joins. Two corrections keep latency bounded: - **(Re)seed to the leading edge** on first frame, on a talkspurt `marker`, or when the clock - has run past the newest frame (starved after silence). No artificial prebuffer — latency - starts as low as possible; buffered frames still play oldest-first. + has run past the newest frame (starved after silence). Playout retains the adaptive target; + when DRED/FEC is enabled that includes one codec frame of recovery look-ahead. - **Frame-skip catch-up:** when the backlog grows past `target + hysteresis` (clock drift, bursty arrival, reordering), fast-forward the clock to leave `target` buffered and drop the now-stale frames. This is the downward force that prevents latency from ratcheting upward. -- `target` is the adaptive jitter estimate (EWMA of inter-arrival gap vs. the per-frame gap), - floored; silence gaps and reordered stragglers are rejected as outliers so they don't inflate - it. The late-drop window tracks `target` (floored/capped at 500 ms). +- `target` is an EWMA of arrival-gap variation measured in 48 kHz sample time. DRED or FEC + reserves one complete channel Opus frame of look-ahead, variation can raise the target to + 120 ms, and packet history is bounded to 500 ms. Limits are durations rather than packet + counts, so 5 ms and 60 ms channels receive the same policy. - Late frames past the playout point are dropped; gaps are filled by DRED (if the next frame arrived) or PLC. - The `marker` flag (start of talkspurt) — set by the sender on the first frame after a transmission gap — lets the buffer reseed cleanly after silence/DTX without accumulating drift. +- One late device callback does not end a talkspurt. Capture resets only after 200 ms of + continuous starvation, avoiding a marker/rebuffer cascade from an isolated scheduling miss. - Diagnostics per stream: `packets_lost`, `duplicates`, `underruns`, `target_depth_ms`. ``` @@ -180,6 +187,11 @@ Each receiver keeps an **adaptive jitter buffer per ssrc** with **bounded-depth jitter estimate feeds depth ``` +Capture and playback use an allocation-free adaptive PCM ring between the managed 20 ms clock +and the hardware clock. Linear-interpolation correction, limited to ±0.5%, holds the ring near +its target instead of periodically dropping a block or rendering silence as device clocks drift. +The local **Audio buffering** preset is 20, 40 (default), or 60 ms and is persisted per client. + ## 6. UDP keepalive & NAT - A `KEEPALIVE` (type 2) frame flows both directions on the media channel every ~5 s to diff --git a/dotnet/src/VoiceCat.Audio/AdaptivePcmBuffer.cs b/dotnet/src/VoiceCat.Audio/AdaptivePcmBuffer.cs new file mode 100644 index 0000000..eef0acd --- /dev/null +++ b/dotnet/src/VoiceCat.Audio/AdaptivePcmBuffer.cs @@ -0,0 +1,97 @@ +namespace VoiceCat.Audio; + +// SPSC PCM handoff with a small occupancy-controlled sample-rate correction. Producers and +// consumers remain non-waiting; the correction keeps independent hardware and managed clocks +// from periodically reaching the hard underflow/overflow edges of a conventional ring. +public sealed class AdaptivePcmBuffer +{ + private const int SampleRate = 48_000; + private const double MaximumCorrection = 0.005; + private readonly short[] samples; + private readonly int channels, frameMask; + private int readFrame, writtenFrame, producer, targetFrames; + private double phase; + private bool primed; + + public AdaptivePcmBuffer(int channels, int bufferMilliseconds = 40, int capacityFrames = 16_384) + { + if (channels is not (1 or 2)) throw new ArgumentOutOfRangeException(nameof(channels)); + if (capacityFrames < 2 || (capacityFrames & (capacityFrames - 1)) != 0) throw new ArgumentOutOfRangeException(nameof(capacityFrames)); + this.channels = channels; samples = new short[checked(capacityFrames * channels)]; frameMask = capacityFrames - 1; + BufferMilliseconds = bufferMilliseconds; + } + + public int Channels => channels; + public int CountFrames => unchecked(Volatile.Read(ref writtenFrame) - Volatile.Read(ref readFrame)); + public int BufferMilliseconds + { + get => Volatile.Read(ref targetFrames) * 1000 / SampleRate; + set + { + if (value is not (20 or 40 or 60)) throw new ArgumentOutOfRangeException(nameof(value)); + Volatile.Write(ref targetFrames, value * SampleRate / 1000); + } + } + + public bool TryWrite(ReadOnlySpan source) + { + if (source.Length == 0 || source.Length % channels != 0 || Interlocked.CompareExchange(ref producer, 1, 0) != 0) return false; + try + { + int frames = source.Length / channels, index = writtenFrame; + if (frames > frameMask + 1 - unchecked(index - Volatile.Read(ref readFrame))) return false; + for (int frame = 0; frame < frames; frame++) + { + int target = ((index + frame) & frameMask) * channels; + for (int channel = 0; channel < channels; channel++) samples[target + channel] = source[frame * channels + channel]; + } + Volatile.Write(ref writtenFrame, unchecked(index + frames)); return true; + } + finally { Volatile.Write(ref producer, 0); } + } + + // Returns interleaved samples written. A zero return means the caller should treat the + // already-cleared destination as silence. Once primed, short scheduling stalls re-prime + // instead of repeatedly clicking at the ring edge. + public int Read(Span destination) + { + if (destination.Length % channels != 0) throw new ArgumentException("PCM must contain complete interleaved frames.", nameof(destination)); + int requestedFrames = destination.Length / channels; + if (requestedFrames == 0) return 0; + int available = CountFrames, target = Volatile.Read(ref targetFrames); + if (!primed) + { + if (available < target) { destination.Clear(); return 0; } + primed = true; phase = 0; + } + if (available <= 0) { primed = false; phase = 0; destination.Clear(); return 0; } + + double correction = Math.Clamp((available - target) / (SampleRate * 2.0), -MaximumCorrection, MaximumCorrection); + double step = 1.0 + correction; + int produced = 0, read = readFrame; + for (int frame = 0; frame < requestedFrames; frame++) + { + int baseOffset = (read & frameMask) * channels; + int nextOffset = ((read + 1) & frameMask) * channels; + int remaining = unchecked(Volatile.Read(ref writtenFrame) - read); + if (remaining <= 0) break; + double fraction = phase; + for (int channel = 0; channel < channels; channel++) + { + int first = samples[baseOffset + channel]; + int second = remaining > 1 ? samples[nextOffset + channel] : first; + destination[produced++] = (short)Math.Clamp((int)Math.Round(first + (second - first) * fraction), short.MinValue, short.MaxValue); + } + phase += step; + int advance = (int)phase; + if (advance > remaining) advance = remaining; + read = unchecked(read + advance); phase -= advance; + } + Volatile.Write(ref readFrame, read); + if (produced < destination.Length) + { + destination[produced..].Clear(); primed = false; phase = 0; + } + return produced; + } +} diff --git a/dotnet/src/VoiceCat.Audio/AudioDeviceBackend.cs b/dotnet/src/VoiceCat.Audio/AudioDeviceBackend.cs index 6ae9787..705487b 100644 --- a/dotnet/src/VoiceCat.Audio/AudioDeviceBackend.cs +++ b/dotnet/src/VoiceCat.Audio/AudioDeviceBackend.cs @@ -3,7 +3,11 @@ namespace VoiceCat.Audio; public sealed record AudioDeviceInfo(string Id, string Name, bool IsDefault); public delegate void CapturePcmHandler(ReadOnlySpan pcm, int channels); public interface IAudioCapture : IDisposable { } -public interface IAudioPlayback : IDisposable { void Write(ReadOnlySpan stereoPcm); } +public interface IAudioPlayback : IDisposable +{ + int BufferMilliseconds { get; set; } + void Write(ReadOnlySpan stereoPcm); +} public interface IAudioDeviceBackend { IReadOnlyList Enumerate(bool input); diff --git a/dotnet/src/VoiceCat.Audio/AudioEngine.cs b/dotnet/src/VoiceCat.Audio/AudioEngine.cs index d549f52..4e2cc93 100644 --- a/dotnet/src/VoiceCat.Audio/AudioEngine.cs +++ b/dotnet/src/VoiceCat.Audio/AudioEngine.cs @@ -23,6 +23,17 @@ public sealed class AudioEngine : IDisposable public volatile float InputGain = 1, OutputGain = 1, VadThreshold = 0.02f; public volatile bool InputNoiseReduction, MicMuted, Deafened, PushToTalk; public volatile AudioInputMode InputMode = AudioInputMode.VoiceActivation; + private int deviceBufferMilliseconds = 40; + public int DeviceBufferMilliseconds + { + get => Volatile.Read(ref deviceBufferMilliseconds); + set + { + if (value is not (20 or 40 or 60)) throw new ArgumentOutOfRangeException(nameof(value)); + Volatile.Write(ref deviceBufferMilliseconds, value); + foreach (LocalStream stream in Volatile.Read(ref routes).Local) stream.BufferMilliseconds = value; + } + } public Exception? Failure { get; private set; } public AudioEngine(EncodedVoiceSender sender, bool startWorker = true) @@ -42,7 +53,7 @@ public sealed class AudioEngine : IDisposable lock (gate) { ObjectDisposedException.ThrowIf(disposed != 0, this); - var stream = new LocalStream(info, captureChannels); + var stream = new LocalStream(info, captureChannels, DeviceBufferMilliseconds); Routes previous = routes; var locals = previous.Local.Where(s => s.Info.StreamId != info.StreamId).Append(stream).ToArray(); Publish(locals, previous.Remote); @@ -58,7 +69,7 @@ public sealed class AudioEngine : IDisposable lock (gate) { var current = routes.Local.FirstOrDefault(s => s.Info.StreamId == streamId) ?? throw new ArgumentException("Stream not found."); - if (current.CaptureChannels != channels) Publish(routes.Local.Select(s => s == current ? new LocalStream(s.Info, channels) : s).ToArray(), routes.Remote); + if (current.CaptureChannels != channels) Publish(routes.Local.Select(s => s == current ? new LocalStream(s.Info, channels, DeviceBufferMilliseconds) : s).ToArray(), routes.Remote); } } diff --git a/dotnet/src/VoiceCat.Audio/LocalStream.cs b/dotnet/src/VoiceCat.Audio/LocalStream.cs index 82bb647..7d8683f 100644 --- a/dotnet/src/VoiceCat.Audio/LocalStream.cs +++ b/dotnet/src/VoiceCat.Audio/LocalStream.cs @@ -15,7 +15,7 @@ internal sealed class LocalStream : IDisposable { internal readonly StreamInfo Info; internal readonly int CaptureChannels; - internal readonly PcmRing Input = new(16384); + internal readonly AdaptivePcmBuffer Input; internal volatile float Level; internal volatile bool Talking; internal volatile float Gain = 1; @@ -27,6 +27,7 @@ internal sealed class LocalStream : IDisposable private readonly short[] converted = new short[16384]; private int feeding; private int buffered; + private int starvedSamples; private uint timestamp; private bool wasTransmitting, marker; @@ -45,9 +46,12 @@ internal sealed class LocalStream : IDisposable finally { Volatile.Write(ref feeding, 0); } } - internal LocalStream(StreamInfo stream, int captureChannels) + internal int BufferMilliseconds { get => Input.BufferMilliseconds; set => Input.BufferMilliseconds = value; } + + internal LocalStream(StreamInfo stream, int captureChannels, int bufferMilliseconds = 40) { Info = stream.Clone(); CaptureChannels = captureChannels; + Input = new(captureChannels, bufferMilliseconds); encoder = new(new() { Channels = stream.Audio.Mode == ChannelMode.ModeStereo ? 2 : 1, @@ -64,10 +68,14 @@ internal sealed class LocalStream : IDisposable internal void Process(AudioEngine engine, EncodedVoiceSender sender) { var input = capture.AsSpan(0, 960 * CaptureChannels); - if (Input.Count < input.Length) { Level = 0; Talking = false; buffered = 0; wasTransmitting = false; return; } - while (Input.Count > input.Length * 6) Input.Read(input); + if (Input.Read(input) != input.Length) + { + Level = 0; Talking = false; starvedSamples += 960; + if (starvedSamples >= 9600) { buffered = 0; wasTransmitting = false; } + return; + } + starvedSamples = 0; if (buffered == 0) timestamp = engine.SampleClock; - Input.Read(input); bool mic = Info.Kind == StreamKind.StreamMic; if (mic && engine.InputNoiseReduction) { diff --git a/dotnet/src/VoiceCat.Audio/ReceiveStream.cs b/dotnet/src/VoiceCat.Audio/ReceiveStream.cs index 2371f3a..17b79ce 100644 --- a/dotnet/src/VoiceCat.Audio/ReceiveStream.cs +++ b/dotnet/src/VoiceCat.Audio/ReceiveStream.cs @@ -20,28 +20,36 @@ internal sealed class ReceiveStream : IDisposable private readonly byte[][] packets = Enumerable.Range(0, 64).Select(_ => new byte[1275]).ToArray(); private readonly VoiceFrameHeader[] headers = new VoiceFrameHeader[64]; private readonly int[] lengths = new int[64]; + private readonly long[] arrivals = new long[64]; private int read, written; - private readonly byte[][] jitter = Enumerable.Range(0, 6).Select(_ => new byte[1275]).ToArray(); - private readonly uint[] timestamps = new uint[6]; - private readonly int[] sizes = new int[6]; + private readonly byte[][] jitter; + private readonly uint[] timestamps; + private readonly int[] sizes; private int count, available, offset, missing, waiting; private uint expected; private bool started, hasTimestamp; private bool hasMarker; private uint lastMarker; private readonly int channels, frameSamples, maximumDepth; + private long lastArrival; + private uint lastArrivalTimestamp; + private double jitterSamples; + private readonly TimeProvider clock; internal int Depth => count; + internal int TargetDepthSamples => TargetSamples(); internal int ConcealedFrames { get; private set; } internal int DredFrames { get; private set; } internal int FecFrames { get; private set; } - internal ReceiveStream(uint userId, StreamInfo info) + internal ReceiveStream(uint userId, StreamInfo info, TimeProvider? clock = null) { if (info.Audio.FrameMs is not (5 or 10 or 20 or 40 or 60) || !Enum.IsDefined(info.Audio.Mode)) throw new ArgumentException("Unsupported remote audio configuration.", nameof(info)); - UserId = userId; Info = info.Clone(); + UserId = userId; Info = info.Clone(); this.clock = clock ?? TimeProvider.System; channels = info.Audio.Mode == ChannelMode.ModeStereo ? 2 : 1; frameSamples = checked((int)info.Audio.FrameMs * 48); - maximumDepth = Math.Clamp(120 / (int)info.Audio.FrameMs, 2, 6); + maximumDepth = Math.Clamp(500 / (int)info.Audio.FrameMs + 2, 8, 104); + jitter = Enumerable.Range(0, maximumDepth).Select(_ => new byte[1275]).ToArray(); + timestamps = new uint[maximumDepth]; sizes = new int[maximumDepth]; decoder = new(48000, channels); try { left = new(); } catch { decoder.Dispose(); throw; } try { right = new(); } catch { left.Dispose(); decoder.Dispose(); throw; } @@ -53,7 +61,7 @@ internal sealed class ReceiveStream : IDisposable { int index = written; if (payload.Length is < 1 or > 1275 || unchecked(index - Volatile.Read(ref read)) >= 64) return false; - int slot = index & 63; payload.CopyTo(packets[slot]); headers[slot] = header; lengths[slot] = payload.Length; + int slot = index & 63; payload.CopyTo(packets[slot]); headers[slot] = header; lengths[slot] = payload.Length; arrivals[slot] = clock.GetTimestamp(); Volatile.Write(ref written, unchecked(index + 1)); return true; } @@ -67,8 +75,8 @@ internal sealed class ReceiveStream : IDisposable if ((headers[source].Flags & VoiceFrameFlags.Marker) != 0 && (!hasMarker || unchecked((int)(timestamp - lastMarker)) > 0)) { hasMarker = true; lastMarker = timestamp; - sizes.AsSpan().Clear(); count = available = offset = missing = waiting = 0; - expected = timestamp; started = false; delta = 0; + DropBefore(timestamp); available = offset = missing = waiting = 0; + expected = timestamp; started = false; delta = 0; lastArrival = 0; jitterSamples = 0; } bool duplicate = false; for (int i = 0; i < sizes.Length; i++) if (sizes[i] != 0 && timestamps[i] == timestamp) duplicate = true; @@ -79,14 +87,61 @@ internal sealed class ReceiveStream : IDisposable int oldest = Oldest(); sizes[oldest] = 0; count--; } int target = Array.IndexOf(sizes, 0); - timestamps[target] = timestamp; sizes[target] = lengths[source]; packets[source].AsSpan(0, lengths[source]).CopyTo(jitter[target]); count++; - int oldestRemaining = Oldest(); - if (count >= maximumDepth && unchecked((int)(timestamps[oldestRemaining] - expected)) > 0) expected = timestamps[oldestRemaining]; + timestamps[target] = timestamp; sizes[target] = lengths[source]; + packets[source].AsSpan(0, lengths[source]).CopyTo(jitter[target]); count++; + ObserveArrival(timestamp, arrivals[source]); } Volatile.Write(ref read, unchecked(read + 1)); } } + private void ObserveArrival(uint timestamp, long arrival) + { + if (lastArrival != 0) + { + int timestampDelta = unchecked((int)(timestamp - lastArrivalTimestamp)); + if (timestampDelta <= 0) return; + if (timestampDelta > 0 && timestampDelta <= frameSamples * 10) + { + double arrivalDelta = clock.GetElapsedTime(lastArrival, arrival).TotalSeconds * 48_000; + double deviation = Math.Abs(arrivalDelta - timestampDelta); + jitterSamples += (deviation - jitterSamples) / 16.0; + } + } + lastArrival = arrival; lastArrivalTimestamp = timestamp; + } + + private int TargetSamples() + { + int recovery = Info.Audio.Dred || Info.Audio.Fec ? frameSamples : 0; + int variation = checked((int)Math.Ceiling(4 * jitterSamples / frameSamples)) * frameSamples; + return Math.Min(5760, recovery + variation); + } + + private int Newest() + { + int newest = -1; + for (int i = 0; i < sizes.Length; i++) if (sizes[i] != 0 && (newest < 0 || unchecked((int)(timestamps[i] - timestamps[newest])) > 0)) newest = i; + return newest; + } + + private void DropBefore(uint timestamp) + { + for (int i = 0; i < sizes.Length; i++) + if (sizes[i] != 0 && unchecked((int)(timestamps[i] - timestamp)) < 0) { sizes[i] = 0; count--; } + } + + private void CatchUp() + { + if (!started || available != 0 || count == 0) return; + int newest = Newest(); int target = TargetSamples(); + int lead = unchecked((int)(timestamps[newest] - expected)); + if (lead <= target + frameSamples) return; + int keepBehind = target / frameSamples * frameSamples; + expected = unchecked(timestamps[newest] - (uint)keepBehind); + DropBefore(expected); missing = 0; + } + private int Oldest() { int oldest = -1; @@ -108,11 +163,12 @@ internal sealed class ReceiveStream : IDisposable else { int next = Oldest(); missing++; - if (next >= 0 && unchecked((int)(timestamps[next] - expected)) == frameSamples) + int recoveryOffset = next < 0 ? 0 : unchecked((int)(timestamps[next] - expected)); + if (next >= 0 && recoveryOffset > 0 && recoveryOffset % frameSamples == 0) { var packet = jitter[next].AsSpan(0, sizes[next]); - if (dred?.TryRecover(decoder, packet, pcm, frameSamples) == true) { decoded = true; DredFrames++; } - else if (Info.Audio.Fec && decoder.TryDecode(packet, pcm, frameSamples, out int recovered, true) && recovered == frameSamples) { decoded = true; FecFrames++; } + if (dred?.TryRecover(decoder, packet, pcm, frameSamples, recoveryOffset) == true) { decoded = true; DredFrames++; } + else if (recoveryOffset == frameSamples && Info.Audio.Fec && decoder.TryDecode(packet, pcm, frameSamples, out int recovered, true) && recovered == frameSamples) { decoded = true; FecFrames++; } } } int maximumConcealment = Math.Max(1, 200 / (int)Info.Audio.FrameMs); @@ -127,11 +183,16 @@ internal sealed class ReceiveStream : IDisposable internal void Mix(Span output, bool deafened, PcmStreamHandler? sink) { Drain(); + CatchUp(); if (!started) { waiting++; - if (!hasTimestamp || count < Math.Min(3, maximumDepth) && waiting < 3) return; - expected = timestamps[Oldest()]; started = true; + if (!hasTimestamp || count == 0) return; + int oldest = Oldest(), newest = Newest(); + if (unchecked((int)(timestamps[newest] - timestamps[oldest])) < TargetSamples()) return; + int keepBehind = TargetSamples() / frameSamples * frameSamples; + expected = unchecked(timestamps[newest] - (uint)keepBehind); + DropBefore(expected); started = true; waiting = 0; } int copied = 0; while (copied < 960) diff --git a/dotnet/tests/VoiceCat.Tests/AudioEngineTests.cs b/dotnet/tests/VoiceCat.Tests/AudioEngineTests.cs index 3c0565d..567a588 100644 --- a/dotnet/tests/VoiceCat.Tests/AudioEngineTests.cs +++ b/dotnet/tests/VoiceCat.Tests/AudioEngineTests.cs @@ -7,6 +7,84 @@ namespace VoiceCat.Tests; public class AudioEngineTests { + [Theory] + [InlineData(-1000)] + [InlineData(1000)] + public void AdaptivePcmBufferAbsorbsIndependentClockDrift(int partsPerMillion) + { + var buffer = new AdaptivePcmBuffer(1, 40); short[] input = new short[962], output = new short[960]; + input.AsSpan().Fill(1234); Assert.True(buffer.TryWrite(input.AsSpan(0, 960))); + Assert.True(buffer.TryWrite(input.AsSpan(0, 960))); + double produced = 0; + for (int cycle = 0; cycle < 10_000; cycle++) + { + produced += 960 * (1 + partsPerMillion / 1_000_000.0); + int frames = (int)produced; produced -= frames; + Assert.True(buffer.TryWrite(input.AsSpan(0, frames))); + Assert.Equal(output.Length, buffer.Read(output)); + } + Assert.InRange(buffer.CountFrames, 480, 3840); + long before = GC.GetAllocatedBytesForCurrentThread(); + for (int i = 0; i < 100; i++) { buffer.TryWrite(input.AsSpan(0, 960)); buffer.Read(output); } + Assert.Equal(0, GC.GetAllocatedBytesForCurrentThread() - before); + } + + [Fact] + public void OneCaptureMissDoesNotRestartTalkspurtButSustainedStarvationDoes() + { + var sent = new List<(uint Timestamp, VoiceFrameFlags Flags)>(); + using var engine = new AudioEngine((_, timestamp, _, flags) => { sent.Add((timestamp, flags)); return true; }, false) + { InputMode = AudioInputMode.AlwaysOn, DeviceBufferMilliseconds = 20 }; + engine.AddLocalStream(Stream()); short[] tone = Tone(); + engine.FeedPcm(1, tone, 1); engine.ProcessCycle(); + engine.ProcessCycle(); + engine.FeedPcm(1, tone, 1); engine.ProcessCycle(); + Assert.Equal(2, sent.Count); Assert.True((sent[0].Flags & VoiceFrameFlags.Marker) != 0); Assert.Equal(VoiceFrameFlags.None, sent[1].Flags & VoiceFrameFlags.Marker); + for (int i = 0; i < 10; i++) engine.ProcessCycle(); + engine.FeedPcm(1, tone, 1); engine.ProcessCycle(); + Assert.True((sent[^1].Flags & VoiceFrameFlags.Marker) != 0); + } + + [Theory] + [InlineData(5)] [InlineData(10)] [InlineData(20)] [InlineData(40)] [InlineData(60)] + public void RecoveryLookaheadTracksChannelFrameDuration(int frameMilliseconds) + { + using var stream = new ReceiveStream(2, Stream(frameMilliseconds)); + Assert.Equal(frameMilliseconds * 48, stream.TargetDepthSamples); + } + + [Fact] + public void JitterTargetAdaptsInSampleTimeAndRemainsCapped() + { + var clock = new ManualAudioClock(); StreamInfo info = Stream(); info.Audio.Fec = false; + using var stream = new ReceiveStream(2, info, clock); using var encoder = new OpusEncoder(new() { Bitrate = 32000 }); + byte[] packet = new byte[1275]; int length = encoder.Encode(Tone(), packet); + for (uint i = 0; i < 40; i++) + { + clock.Advance(i % 2 == 0 ? 5 : 35); + stream.Enqueue(new(MediaFrameType.Voice, 0, 0, 42, i, i * 960), packet.AsSpan(0, length)); + } + int[] output = new int[1920]; stream.Mix(output, false, null); + Assert.InRange(stream.TargetDepthSamples, 960, 5760); + } + + [Fact] + public void DredUsesTimestampOffsetForConsecutiveMissingShortFrames() + { + using var probe = new OpusEncoder(); + if (!probe.SupportsDeepRedundancy) return; + StreamInfo info = Stream(10, dred: true); using var stream = new ReceiveStream(2, info); + using var encoder = new OpusEncoder(new() { FrameDurationMilliseconds = 10, DeepRedundancy = true, ExpectedPacketLossPercent = 30, Bitrate = 64000 }); + byte[] packet = new byte[1275]; short[] tone = new short[480]; int[] output = new int[1920]; + for (uint i = 0; i < 50; i++) + { + CodecTests.FillTone(tone, 480, 1, 48000, (int)i); int length = encoder.Encode(tone, packet); + if (i is not (25 or 26)) stream.Enqueue(new(MediaFrameType.Voice, 0, 0, 42, i, i * 480), packet.AsSpan(0, length)); + if (i % 2 == 1) { output.AsSpan().Clear(); stream.Mix(output, false, null); } + } + Assert.True(stream.DredFrames >= 2); + } + [Theory] [InlineData(true)] [InlineData(false)] @@ -114,4 +192,12 @@ public class AudioEngineTests for (int i = 0; i < 15; i++) { send.FeedPcm(1, stereo, 2); send.ProcessCycle(); receive.ProcessCycle(); } Assert.True(energy > 100000); } + + private sealed class ManualAudioClock : TimeProvider + { + private long milliseconds; + public override long TimestampFrequency => 1000; + public override long GetTimestamp() => milliseconds; + internal void Advance(int value) => milliseconds += value; + } }