Files
voice-cat/server/src/conn_session.h

135 lines
5.8 KiB
C
Raw Normal View History

2026-06-15 23:48:44 +02:00
/*
* server/conn_session.h Per-client connection state machine.
*
* State: WaitingHello WaitingAuth Authenticated Disconnecting
*
* Design: ConnSession is a pure state machine. It receives frames via on_frame()
* (called from TcpServerConn's strand) and sends via a send_fn set after construction.
* The server creates the TcpServerConn first (with callbacks referencing the session),
* then calls set_tcp() to give the session its send capability.
*/
#ifndef VOICECAT_SERVER_CONN_SESSION_H
#define VOICECAT_SERVER_CONN_SESSION_H
#ifdef VOICECAT_HAS_NET
#include <array>
#include <atomic>
#include <cstdint>
#include <functional>
#include <memory>
#include <string>
#include <vector>
#define ASIO_STANDALONE 1
#include <asio.hpp>
#include "crypto/crypto.h"
2026-06-15 23:48:44 +02:00
#include "proto/voicecat.pb.h"
namespace voicecat { class WorkerPool; }
2026-06-15 23:48:44 +02:00
namespace voicecat::server {
class Database;
class SessionRegistry;
class ConnSession : public std::enable_shared_from_this<ConnSession> {
public:
enum class State { WaitingHello, WaitingAuth, Authenticated, Disconnecting };
using SendFn = std::function<void(std::vector<uint8_t>)>;
2026-06-15 23:48:44 +02:00
using CloseFn = std::function<void()>;
ConnSession(std::shared_ptr<Database> db,
std::shared_ptr<SessionRegistry> registry,
std::shared_ptr<voicecat::WorkerPool> workers,
const std::array<uint8_t, 32>& server_fp,
bool allow_guests,
uint16_t udp_media_port = 0);
2026-06-15 23:48:44 +02:00
void set_io(SendFn send_fn, CloseFn close_fn);
void set_session_id(uint64_t id) { session_id_ = id; }
void begin();
void on_frame(std::vector<uint8_t> frame);
void on_disconnect();
void send_envelope(const voicecat::v1::Envelope& env);
void close();
// ── M2: media key injection (called from on_tls_ready) ───────────────────
void set_media_crypto(std::unique_ptr<voicecat::crypto::SodiumMediaCrypto> send,
std::unique_ptr<voicecat::crypto::SodiumMediaCrypto> recv);
// ── M2: UDP endpoint (set by MediaRelay on UdpBinding) ────────────────────
void set_udp_endpoint(asio::ip::udp::endpoint ep);
asio::ip::udp::endpoint udp_endpoint() const;
bool has_udp_endpoint() const { return has_udp_ep_.load(); }
// ── M2: media crypto access (for SFU relay) ──────────────────────────────
voicecat::crypto::SodiumMediaCrypto* send_crypto();
voicecat::crypto::SodiumMediaCrypto* recv_crypto();
// ── M2: UDP token (for binding) ───────────────────────────────────────────
const std::array<uint8_t, 16>& udp_token() const { return udp_token_; }
// ── Accessors ──────────────────────────────────────────────────────────────
2026-06-15 23:48:44 +02:00
State state() const { return state_.load(); }
uint64_t session_id() const { return session_id_; }
uint32_t user_id() const { return user_id_; }
private:
void handle_client_hello(uint64_t req_id, const voicecat::v1::ClientHello& msg);
void handle_auth_request(uint64_t req_id, const voicecat::v1::AuthRequest& msg);
void handle_join_channel(uint64_t req_id, const voicecat::v1::JoinChannelRequest& msg);
void handle_text_message(const voicecat::v1::TextMessage& msg);
void handle_ping(const voicecat::v1::Ping& msg);
void handle_udp_binding(uint64_t req_id, const voicecat::v1::UdpBinding& msg);
void handle_stream_announce(uint64_t req_id, const voicecat::v1::StreamAnnounce& msg);
void handle_stream_stop(const voicecat::v1::StreamStop& msg);
2026-06-15 23:48:44 +02:00
void finish_guest_auth(const voicecat::v1::GuestAuth& guest, uint64_t req_id);
void finish_password_auth(const std::string& username, const std::string& password,
uint64_t req_id);
void send_auth_result_ok(uint64_t req_id, const voicecat::v1::User& user,
const voicecat::v1::Permissions* perms = nullptr);
2026-06-15 23:48:44 +02:00
void send_state_snapshot();
void broadcast_user_joined(const voicecat::v1::User& user);
void send_disconnect_and_close(uint32_t code, const std::string& reason);
std::shared_ptr<Database> db_;
std::shared_ptr<SessionRegistry> registry_;
2026-06-15 23:48:44 +02:00
std::shared_ptr<voicecat::WorkerPool> workers_;
std::array<uint8_t, 32> server_fp_;
bool allow_guests_;
uint16_t udp_media_port_;
SendFn send_fn_;
CloseFn close_fn_;
std::atomic<State> state_{State::WaitingHello};
uint64_t session_id_{0};
std::atomic<uint32_t> user_id_{0};
std::atomic<bool> closed_{false};
// M2 UDP / media
std::array<uint8_t, 16> udp_token_{};
mutable std::mutex udp_ep_mu_;
asio::ip::udp::endpoint udp_ep_;
std::atomic<bool> has_udp_ep_{false};
mutable std::mutex crypto_mu_;
std::unique_ptr<voicecat::crypto::SodiumMediaCrypto> send_crypto_;
std::unique_ptr<voicecat::crypto::SodiumMediaCrypto> recv_crypto_;
feat(M3): multi-stream & per-channel tuning Implements docs/roadmap.md M3: multiple concurrent streams per user (MIC + SCREEN_AUDIO + AUX_DEVICE), independent per-stream receiver gain/mute/noise- reduction, talk indicators, and enforced per-channel Opus configurability (mono/stereo, bitrate, frame size, FEC/DTX, application). Bugs fixed along the way (found while implementing, not pre-existing scope): - Server hard-coded stream_id=1 for every announce, so a second stream from the same user silently overwrote the first in SessionRegistry::set_user_stream. Now a per-session counter (ConnSession::next_stream_id_); handle_stream_stop validates against announced_stream_ids_ before clearing. - Client dropped mode/dtx/complexity/application from effective_audio even for the single M2 stream -- only sample_rate/bitrate_bps/frame_ms/fec were ever applied to OpusParams. Fixed on both the send (handle_stream_announce_result) and receive (sync_remote_streams) paths via a shared opus_params_from_audio_config() helper. - OpusEncoder always used OPUS_APPLICATION_VOIP; added OpusParams::application and wired it through. - on_playback's per-stream decode passed the wrong frame_size to opus_decode (total samples instead of samples-per-channel), which would have overflowed the decode buffer for any stereo stream. - teardown_voice() raced when called concurrently from run_io()'s own cleanup and from disconnect() on a different thread -- both could see udp_thread_/talk_timer_thread_ as joinable() at once and race to join() the same std::thread (intermittent std::system_error under ctest). Fixed with a teardown_mu_ guard instead of carrying the flake forward. New: - Per-channel AudioConfig: SessionRegistry now seeds Lobby (mono/24kbps/VOIP/ FEC+DTX) and a new "Music Room" channel (stereo/128kbps/AUDIO/no DTX); handle_stream_announce enforces the channel's config, clamping (not overriding) bitrate_bps to its ceiling. - core/src/core/client.h/.cpp: local-stream state is now a std::unordered_map<int, LocalStream> keyed by vc_stream_kind, with request_id-correlated announce/result handling (request_id already round-tripped on the wire; just wasn't read before). on_capture_frame is kind-aware and upmixes mono capture to stereo when a stream's config calls for it. set_self_mute's mic_muted now only gates the MIC kind. NS is wired through set_remote_stream. New run_talk_timer() thread emits VC_EVENT_TALK_STATE from both remote and local edge detection. - core/src/audio/audio_engine.h/.cpp: kind-keyed injection taps (inject_capture), stereo-to-mono downmix at the decode/mix boundary, RemoteStream gains recv_ns (lazy ApmProcessor) + noise_reduction_enabled and last_voice_ms/talking; new set_stream_noise_reduction() and poll_talk_transitions(). - core/src/session/session.h/.cpp: Stream now carries the full AudioConfig, not just sample_rate/frame_ms. - New additive C ABI (core/include/voicecat.h): vc_audio_config + vc_get_stream_audio_config (effective Opus config for any stream you own or a peer's); vc_test_inject_capture (test-only synthetic PCM injection, clearly marked, mirrors AudioEngine::inject_capture). - tests/test_m3_multistream.cpp: the M3 exit criterion through the real ABI (mirrors test_voice_client_abi.cpp's approach, not raw sockets) -- two concurrent local streams, independent gain/mute/NS control, per-channel config divergence via vc_get_stream_audio_config, talk indicators. Explicitly out of scope for this pass (tracked in PROGRESS.md, not silently dropped): VAD/PTT input gate + device enumeration; real WASAPI loopback capture for SCREEN_AUDIO (synthetic injection only); true stereo playback output (AudioEngine's mixer/output device stays mono -- Opus itself is fully stereo-correct on the wire). ctest --test-dir build/m1-dev: 11/11 green, verified across 3 consecutive full-suite runs plus 8 standalone runs of the new test. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-16 14:12:37 +02:00
// M2/M3: locally-announced streams. The server assigns the stream_id (unique per
// session), so a per-session counter + the set of currently-active ids is enough to
// support multiple concurrent streams (MIC + SCREEN_AUDIO + AUX_DEVICE) per user.
uint32_t next_stream_id_{1};
std::vector<uint32_t> announced_stream_ids_;
2026-06-15 23:48:44 +02:00
};
} // namespace voicecat::server
#endif // VOICECAT_HAS_NET
#endif // VOICECAT_SERVER_CONN_SESSION_H