feat(clients): event sound effects + optional text-to-speech

Add audible cues and optional spoken announcements for session events
(join/leave, channel + PM sent/recv, login, logout/connection-lost,
mic on/off, voice-activity, PTT) across all three clients, driven off
the shared C ABI vc_event stream so the mapping stays consistent.

TTS is off by default; when enabled it announces events and reads
message/PM bodies aloud. Master toggles + a sound-volume slider; the
per-utterance voice-activity and PTT cues default off. WAVs ship from
assets/sounds/.

Windows (built + verified): new VoiceCat.App/Notifications/ layer
(FeedbackSettings -> %AppData%\VoiceCat\feedback.json, SoundPlayerPool
via System.Media.SoundPlayer, SpeechAnnouncer via Prismatoid 0.3.0,
EventFeedback dispatcher); MainForm hooks; NotificationSettingsForm
under Settings > Notifications; csproj adds the Prismatoid PackageRef
and copies the WAVs into sounds\.

macOS + iOS (written, not yet built -- needs a Mac): shared
VoiceCatCore/Feedback/ (SoundEvent, EventFeedback = AVAudioPlayer pool
+ native AVSpeechSynthesizer, FeedbackSettings over UserDefaults); WAVs
bundled via Package.swift resources (.process). Hooks in SessionState/
AppState (iOS) and MainWindowController (macOS); settings UI in
SettingsView (iOS) and SettingsWindowController (macOS).

No core/server code touched; ctest --preset dev unaffected.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-06-22 15:20:31 +02:00
co-authored by Claude Opus 4.8
parent 725bd8e925
commit 50416c33a2
45 changed files with 812 additions and 19 deletions
+5 -1
View File
@@ -40,7 +40,11 @@ let package = Package(
.target(
name: "VoiceCatCore",
dependencies: ["VoiceCatCoreXCF"],
path: "Sources/VoiceCatCore"
path: "Sources/VoiceCatCore",
// Event-cue WAVs (shared with the Windows client) bundled into the package's
// resource bundle; EventFeedback loads them via Bundle.module. Copied from
// assets/sounds/ into Sources/VoiceCatCore/Sounds/.
resources: [.process("Sounds")]
),
// Smoke tests against a real voicecat-server mirrors clients/windows/
// VoiceCat.Interop.Tests/VoiceCatClientSmokeTests.cs. Requires the `dev` CMake preset
@@ -0,0 +1,105 @@
// EventFeedback shared audible + spoken feedback for session events, used by both the macOS
// (AppKit) and iOS (SwiftUI) clients. Mirrors the Windows client's EventFeedback policy
// (clients/windows/.../Notifications/EventFeedback.cs): the platform event handlers decide WHAT
// to play (they own the model/nickname/channel context); this type owns the "should I, and how"
// policy plus the AVFoundation playback/synthesis.
//
// Sound effects use AVAudioPlayer; spoken announcements use the OS-native AVSpeechSynthesizer.
//
// NOTE (iOS): on the voice path the app runs a play-and-record AVAudioSession (VPIO). Playing
// these cues / speech over that session can interact with the live call (ducking, route, or the
// mute switch). The session category should allow mixing verify on device. This is the most
// likely place for platform bugs to surface.
import Foundation
import AVFoundation
/// User preferences for event sounds and spoken feedback, backed by UserDefaults so the macOS
/// and iOS settings screens and this player share one source of truth.
public struct FeedbackSettings: Sendable {
public var sounds: Bool
public var speech: Bool
public var volume: Float
public var selfTalkSounds: Bool
public var pttSound: Bool
static let keySounds = "feedback.sounds"
static let keySpeech = "feedback.speech"
static let keyVolume = "feedback.volume"
static let keySelfTalk = "feedback.selfTalk"
static let keyPtt = "feedback.ptt"
/// Default values registered with UserDefaults (so "unset" reads as the intended default
/// rather than false/0).
static let defaults: [String: Any] = [
keySounds: true,
keySpeech: false,
keyVolume: 1.0,
keySelfTalk: false,
keyPtt: false,
]
/// The current settings, read live from UserDefaults.
public static var current: FeedbackSettings {
let d = UserDefaults.standard
d.register(defaults: defaults) // idempotent ensures "unset" reads as the intended default
return FeedbackSettings(
sounds: d.bool(forKey: keySounds),
speech: d.bool(forKey: keySpeech),
volume: d.float(forKey: keyVolume),
selfTalkSounds: d.bool(forKey: keySelfTalk),
pttSound: d.bool(forKey: keyPtt))
}
}
@MainActor
public final class EventFeedback {
public static let shared = EventFeedback()
private var players: [SoundEvent: AVAudioPlayer] = [:]
private let synthesizer = AVSpeechSynthesizer()
private init() {
UserDefaults.standard.register(defaults: FeedbackSettings.defaults)
}
// MARK: - Sounds
/// Play an event cue, honouring the user's settings. The two opt-in categories (your own
/// voice-activity, and the PTT cue) are gated by their own flags.
public func play(_ event: SoundEvent) {
let s = FeedbackSettings.current
guard s.sounds, s.volume > 0 else { return }
if (event == .vaStart || event == .vaStop), !s.selfTalkSounds { return }
if event == .ptt, !s.pttSound { return }
guard let player = player(for: event) else { return }
player.volume = s.volume
player.currentTime = 0
player.play()
}
/// Lazily load and cache an AVAudioPlayer for the event's bundled WAV. Returns nil (silent)
/// if the resource is missing or fails to load.
private func player(for event: SoundEvent) -> AVAudioPlayer? {
if let cached = players[event] { return cached }
guard let url = Bundle.module.url(forResource: event.resourceName, withExtension: "wav"),
let player = try? AVAudioPlayer(contentsOf: url) else {
return nil
}
player.prepareToPlay()
players[event] = player
return player
}
// MARK: - Speech
/// Speak `text` when spoken feedback is enabled. Utterances queue (do not interrupt prior
/// speech) so a burst of events is read in order.
public func speak(_ text: String) {
guard FeedbackSettings.current.speech else { return }
let trimmed = text.trimmingCharacters(in: .whitespacesAndNewlines)
guard !trimmed.isEmpty else { return }
synthesizer.speak(AVSpeechUtterance(string: trimmed))
}
}
@@ -0,0 +1,43 @@
// SoundEvent the cross-platform set of audible event cues. Each maps to a WAV bundled as an
// SPM resource (Sources/VoiceCatCore/Sounds/, copied from assets/sounds/). The same logical set
// is mirrored in the Windows client (clients/windows/.../Notifications/SoundEvent.cs) so feedback
// stays consistent across platforms.
import Foundation
public enum SoundEvent: CaseIterable, Sendable {
case channelJoin // another user joined my channel
case channelLeave // another user left my channel
case channelRecv // channel text message from someone else
case channelSent // channel text message I sent
case pmRecv // private message received
case pmSent // private message I sent
case login // connected / authenticated
case logout // clean disconnect
case connectionLost // unexpected disconnect
case voiceOn // my microphone stream started
case voiceOff // my microphone stream stopped
case vaStart // my voice-activity began (off by default)
case vaStop // my voice-activity ended (off by default)
case ptt // push-to-talk engaged (off by default)
/// Resource name (without extension) as bundled in Sources/VoiceCatCore/Sounds/.
var resourceName: String {
switch self {
case .channelJoin: return "channel_join"
case .channelLeave: return "channel_leave"
case .channelRecv: return "channel_recv"
case .channelSent: return "channel_sent"
case .pmRecv: return "pm_recv"
case .pmSent: return "pm_sent"
case .login: return "login"
case .logout: return "logout"
case .connectionLost: return "connection_lost"
case .voiceOn: return "voice_on"
case .voiceOff: return "voice_off"
case .vaStart: return "va_start"
case .vaStop: return "va_stop"
case .ptt: return "ptt"
}
}
}
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -158,6 +158,8 @@ final class AppState {
connectStatus = ""
showPasswordPrompt = false
self.session = newSession
EventFeedback.shared.play(.login)
EventFeedback.shared.speak("Connected")
// Activate the audio session now, while connected NOT lazily when the first
// remote stream arrives. The core opens its miniaudio playback device the moment
// a remote stream starts and only THEN emits .streamStarted; if we waited for
@@ -92,7 +92,22 @@ final class SessionState {
case .channelList:
refreshChannels()
syncSelfChannel()
case .userJoined, .userLeft:
case .userJoined:
// ev.text = nickname, ev.channelId = the channel they joined (per voicecat.h).
if ev.userId != selfUserId && ev.channelId == currentChannelId {
EventFeedback.shared.play(.channelJoin)
EventFeedback.shared.speak("\(ev.text ?? "Someone") joined")
}
refreshUsers()
syncSelfChannel()
case .userLeft:
// Capture the leaving user's prior nickname/channel before refreshUsers() drops them.
if ev.userId != selfUserId,
let gone = users.first(where: { $0.id == ev.userId }),
gone.channelId == currentChannelId {
EventFeedback.shared.play(.channelLeave)
EventFeedback.shared.speak("\(gone.nickname) left")
}
refreshUsers()
syncSelfChannel()
case .userUpdated:
@@ -103,13 +118,27 @@ final class SessionState {
}
case .textMessage:
let sender = users.first(where: { $0.id == ev.userId })?.nickname ?? "Unknown"
let body = ev.text ?? ""
let isSelf = ev.userId == selfUserId
let isPrivate = ev.textScope == .private
messages.append(ChatMessage(
timestamp: Date(timeIntervalSince1970: Double(ev.timestampUnixMs) / 1000),
senderName: sender,
text: ev.text ?? "",
text: body,
scope: ev.textScope))
EventFeedback.shared.play(isPrivate
? (isSelf ? .pmSent : .pmRecv)
: (isSelf ? .channelSent : .channelRecv))
if !isSelf {
EventFeedback.shared.speak(isPrivate
? "Private message from \(sender): \(body)"
: "\(sender): \(body)")
}
case .talkState:
let talking = ev.u32a != 0
if ev.userId == selfUserId {
EventFeedback.shared.play(talking ? .vaStart : .vaStop)
}
let who = users.first(where: { $0.id == ev.userId })?.nickname ?? "user \(ev.userId)"
addActivity(talking ? "\(who) started talking" : "\(who) stopped talking")
case .streamStarted:
@@ -155,6 +184,10 @@ final class SessionState {
}
case .accountList:
accounts = client.listAccounts()
case .disconnected:
// Audible cue only session teardown is driven elsewhere (AppState / UI).
EventFeedback.shared.play(ev.result == .ok ? .logout : .connectionLost)
EventFeedback.shared.speak(ev.result == .ok ? "Disconnected" : "Connection lost")
default:
break
}
@@ -247,6 +280,7 @@ final class SessionState {
if result == .ok {
voiceState.micActive = true
voiceState.localStreamId = streamId
EventFeedback.shared.play(.voiceOn)
// Publish the active mic stream ID so IOSAudioRouter can reset the core's capture
// channel count when the user switches monostereo (selectCaptureChannels /
// applyPreset). Without this, switching stereomono leaves the LocalStream's
@@ -291,6 +325,7 @@ final class SessionState {
client.stopStream(voiceState.localStreamId)
voiceState.localStreamId = 0
AudioSessionManager.shared.activeMicStreamId = nil
EventFeedback.shared.play(.voiceOff)
}
if wasVPIO {
client.setExternalPlayback(false)
@@ -367,8 +402,12 @@ final class SessionState {
voiceState.vadThreshold = threshold
}
private var pttEngaged = false
func setPushToTalk(_ active: Bool) {
client.setPushToTalk(active)
// Play the PTT cue only on the press transition (the gesture fires repeatedly while held).
if active && !pttEngaged { EventFeedback.shared.play(.ptt) }
pttEngaged = active
}
// MARK: - Text
@@ -8,6 +8,14 @@ struct SettingsView: View {
@StateObject private var router = IOSAudioRouter.shared
@State private var showAdvanced = false
// Notification feedback prefs keys shared with VoiceCatCore's FeedbackSettings, so the
// EventFeedback player reads the same values these toggles write.
@AppStorage("feedback.sounds") private var soundsEnabled = true
@AppStorage("feedback.speech") private var speechEnabled = false
@AppStorage("feedback.volume") private var soundsVolume = 1.0
@AppStorage("feedback.selfTalk") private var selfTalkEnabled = false
@AppStorage("feedback.ptt") private var pttSoundEnabled = false
var body: some View {
NavigationStack {
Form {
@@ -217,6 +225,27 @@ struct SettingsView: View {
}
}
// MARK: - Notifications
Section("Notifications") {
Toggle("Event sounds", isOn: $soundsEnabled)
.accessibilityLabel("Play event sounds")
if soundsEnabled {
VStack(alignment: .leading, spacing: 4) {
Text("Sound volume")
.font(.caption)
Slider(value: $soundsVolume, in: 0...1)
.accessibilityLabel("Sound volume")
}
}
Toggle("Speak events (text-to-speech)", isOn: $speechEnabled)
.accessibilityLabel("Speak events")
.accessibilityHint("Announces joins and leaves and reads message text aloud.")
Toggle("Your own voice-activity sounds", isOn: $selfTalkEnabled)
.accessibilityLabel("Voice activity sounds")
Toggle("Push-to-talk cue", isOn: $pttSoundEnabled)
.accessibilityLabel("Push to talk cue")
}
// MARK: - Admin
if session.permissions.canAdminAccounts || session.permissions.isAdmin {
Section("Administration") {
@@ -35,6 +35,7 @@ final class MainWindowController: NSWindowController, NSWindowDelegate {
private var screenAudioSelection: ScreenAudioSelection = .default
internal var pttKeyCode: UInt16 = 0x60 // F8
private var pttMonitor: Any?
private var pttEngaged = false // guards the PTT cue against key-repeat
private var serverMuted = false
private var serverDeafened = false
private var channelTree: [ChannelNode] = []
@@ -343,7 +344,11 @@ final class MainWindowController: NSWindowController, NSWindowDelegate {
pttMonitor = NSEvent.addLocalMonitorForEvents(matching: [.keyDown, .keyUp]) { [weak self] event in
guard let self, self.selectedInputMode == .pushToTalk,
self.micStreamId != 0, event.keyCode == self.pttKeyCode else { return event }
self.client.setPushToTalk(event.type == .keyDown)
let down = event.type == .keyDown
self.client.setPushToTalk(down)
// Cue only on the press transition key-down auto-repeats while held.
if down && !self.pttEngaged { EventFeedback.shared.play(.ptt) }
self.pttEngaged = down
return nil
}
@@ -385,6 +390,8 @@ final class MainWindowController: NSWindowController, NSWindowDelegate {
buildMessagesMenu()
buildAdminMenu()
addActivity("Connected to server as \(nickname)")
EventFeedback.shared.play(.login)
EventFeedback.shared.speak("Connected")
}
// MARK: - Event handling
@@ -415,6 +422,8 @@ final class MainWindowController: NSWindowController, NSWindowDelegate {
refreshChannelTree(); refreshUserList()
if event.channelId == currentChannelId && event.userId != selfUserId {
addActivity("\(u.nickname) joined the channel")
EventFeedback.shared.play(.channelJoin)
EventFeedback.shared.speak("\(u.nickname) joined")
}
case .userLeft:
@@ -423,7 +432,11 @@ final class MainWindowController: NSWindowController, NSWindowDelegate {
users.removeValue(forKey: event.userId)
talkingUsers.remove(event.userId)
refreshChannelTree(); refreshUserList()
if wasHere { addActivity("\(nick) left the channel") }
if wasHere {
addActivity("\(nick) left the channel")
EventFeedback.shared.play(.channelLeave)
EventFeedback.shared.speak("\(nick) left")
}
if let pmWin = pmWindows[event.userId] {
pmWin.appendActivity("\(nick) disconnected from server")
}
@@ -473,6 +486,9 @@ final class MainWindowController: NSWindowController, NSWindowDelegate {
let talking = event.u32a == 1
if talking { talkingUsers.insert(event.userId) } else { talkingUsers.remove(event.userId) }
refreshUserList()
if event.userId == selfUserId {
EventFeedback.shared.play(talking ? .vaStart : .vaStop)
}
if talking && event.userId != selfUserId,
let u = users[event.userId], u.channelId == currentChannelId {
addActivity("\(u.nickname) started talking")
@@ -521,21 +537,28 @@ final class MainWindowController: NSWindowController, NSWindowDelegate {
time = DateFormatter.localizedString(from: Date(), dateStyle: .none, timeStyle: .short)
}
let sender = nickname(for: event.userId)
let isSelf = event.userId == selfUserId
let body = event.text ?? ""
if event.textScope == .private {
// Route PMs to per-conversation windows. For our own outgoing PM, ev.channelId
// carries the recipient user ID; for incoming, ev.userId is the sender.
let otherId = event.userId == selfUserId ? event.channelId : event.userId
let otherId = isSelf ? event.channelId : event.userId
let win = getOrOpenPmWindow(otherId)
win.appendMessage(time: time, isSelf: event.userId == selfUserId, sender: sender,
text: event.text ?? "")
if event.userId != selfUserId {
win.appendMessage(time: time, isSelf: isSelf, sender: sender, text: body)
EventFeedback.shared.play(isSelf ? .pmSent : .pmRecv)
if !isSelf {
addActivity("Private message from \(sender)")
EventFeedback.shared.speak("Private message from \(sender): \(body)")
}
} else {
let line = "[\(time)] \(sender): \(event.text ?? "")\n"
let line = "[\(time)] \(sender): \(body)\n"
logTextView.textStorage?.append(NSAttributedString(string: line))
logTextView.scrollToEndOfDocument(nil)
EventFeedback.shared.play(isSelf ? .channelSent : .channelRecv)
if !isSelf {
EventFeedback.shared.speak("\(sender): \(body)")
}
}
}
@@ -607,6 +630,8 @@ final class MainWindowController: NSWindowController, NSWindowDelegate {
let msg = event.text.map { "Disconnected: \($0)" } ?? "Disconnected from server."
statusLabel.stringValue = msg
addActivity(msg)
EventFeedback.shared.play(event.result == .ok ? .logout : .connectionLost)
EventFeedback.shared.speak(event.result == .ok ? "Disconnected" : "Connection lost")
channelTree = []; channelOutlineView.reloadData()
displayedUsers = []; userTableView.reloadData()
users.removeAll(); talkingUsers.removeAll()
@@ -689,6 +714,7 @@ final class MainWindowController: NSWindowController, NSWindowDelegate {
}
setVoiceJoinedState(true)
addActivity("Joined voice — microphone active")
EventFeedback.shared.play(.voiceOn)
NSAccessibility.post(element: logTextView, notification: .announcementRequested,
userInfo: [.announcement: "Joined voice", .priority: NSAccessibilityPriorityLevel.medium])
} else {
@@ -701,6 +727,7 @@ final class MainWindowController: NSWindowController, NSWindowDelegate {
settingsWindowController?.resetLevel()
setVoiceJoinedState(false)
addActivity("Left voice")
EventFeedback.shared.play(.voiceOff)
}
}
@@ -50,6 +50,21 @@ final class SettingsWindowController: NSWindowController, NSWindowDelegate {
// Cached VAD slider position so we can restore it when the window reopens.
private var vadSliderValue: Double = 50
// Notification feedback controls. Read/write UserDefaults with the same keys VoiceCatCore's
// FeedbackSettings reads, so EventFeedback honours these immediately.
private let soundsCheckbox = NSButton(checkboxWithTitle: "Event sounds", target: nil, action: nil)
private let soundsVolumeSlider: NSSlider = {
let s = NSSlider(value: 1, minValue: 0, maxValue: 1, target: nil, action: nil)
s.numberOfTickMarks = 0
return s
}()
private let speechCheckbox = NSButton(checkboxWithTitle: "Speak events (text-to-speech)",
target: nil, action: nil)
private let selfTalkCheckbox = NSButton(checkboxWithTitle: "Your own voice-activity sounds",
target: nil, action: nil)
private let pttSoundCheckbox = NSButton(checkboxWithTitle: "Push-to-talk cue",
target: nil, action: nil)
// MARK: - Init
init(client: VoiceCatClient, mainController: MainWindowController) {
@@ -57,13 +72,13 @@ final class SettingsWindowController: NSWindowController, NSWindowDelegate {
self.mainController = mainController
let window = NSWindow(
contentRect: NSRect(x: 0, y: 0, width: 380, height: 260),
contentRect: NSRect(x: 0, y: 0, width: 380, height: 420),
styleMask: [.titled, .closable, .miniaturizable],
backing: .buffered,
defer: false
)
window.title = "Audio Settings"
window.minSize = NSSize(width: 340, height: 220)
window.title = "Settings"
window.minSize = NSSize(width: 340, height: 380)
window.center()
super.init(window: window)
window.delegate = self
@@ -140,7 +155,31 @@ final class SettingsWindowController: NSWindowController, NSWindowDelegate {
levelRow.orientation = .horizontal
levelRow.spacing = 8
let stack = NSStackView(views: [inputModeRow, vadRow, pttRow, deviceRow, levelRow])
// Notifications
let notificationsHeader = NSTextField(labelWithString: "Notifications")
notificationsHeader.font = .boldSystemFont(ofSize: NSFont.systemFontSize)
for box in [soundsCheckbox, speechCheckbox, selfTalkCheckbox, pttSoundCheckbox] {
box.target = self
box.action = #selector(notificationSettingChanged)
}
soundsCheckbox.setAccessibilityLabel("Play event sounds")
speechCheckbox.setAccessibilityLabel("Speak events")
selfTalkCheckbox.setAccessibilityLabel("Your own voice-activity sounds")
pttSoundCheckbox.setAccessibilityLabel("Push-to-talk cue")
let volumeLabel = NSTextField(labelWithString: "Sound volume:")
volumeLabel.setAccessibilityLabel("Sound volume")
soundsVolumeSlider.target = self
soundsVolumeSlider.action = #selector(notificationSettingChanged)
soundsVolumeSlider.setAccessibilityLabel("Sound volume")
let volumeRow = NSStackView(views: [volumeLabel, soundsVolumeSlider])
volumeRow.orientation = .horizontal
volumeRow.spacing = 8
let stack = NSStackView(views: [inputModeRow, vadRow, pttRow, deviceRow, levelRow,
notificationsHeader, soundsCheckbox, volumeRow,
speechCheckbox, selfTalkCheckbox, pttSoundCheckbox])
stack.orientation = .vertical
stack.spacing = 12
stack.alignment = .leading
@@ -157,7 +196,30 @@ final class SettingsWindowController: NSWindowController, NSWindowDelegate {
vadSlider.widthAnchor.constraint(greaterThanOrEqualToConstant: 200),
levelMeter.widthAnchor.constraint(equalToConstant: 200),
devicePicker.widthAnchor.constraint(greaterThanOrEqualToConstant: 180),
soundsVolumeSlider.widthAnchor.constraint(greaterThanOrEqualToConstant: 200),
])
syncNotificationControls()
}
/// Load the notification checkbox/slider states from UserDefaults. Touches EventFeedback.shared
/// first so its default values are registered before we read them.
private func syncNotificationControls() {
let s = FeedbackSettings.current
soundsCheckbox.state = s.sounds ? .on : .off
speechCheckbox.state = s.speech ? .on : .off
selfTalkCheckbox.state = s.selfTalkSounds ? .on : .off
pttSoundCheckbox.state = s.pttSound ? .on : .off
soundsVolumeSlider.doubleValue = Double(s.volume)
}
@objc private func notificationSettingChanged() {
let d = UserDefaults.standard
d.set(soundsCheckbox.state == .on, forKey: "feedback.sounds")
d.set(speechCheckbox.state == .on, forKey: "feedback.speech")
d.set(selfTalkCheckbox.state == .on, forKey: "feedback.selfTalk")
d.set(pttSoundCheckbox.state == .on, forKey: "feedback.ptt")
d.set(soundsVolumeSlider.doubleValue, forKey: "feedback.volume")
}
// MARK: - Sync from MainWindowController