//! Audio engine — owns the cpal input/output streams, the Opus //! encoder, and the tsclientlib `AudioHandler` for decode+mix. //! //! The engine is started after a protocol connection is established //! and stopped before disconnect. It does not retry on device //! change. use std::sync::atomic::{AtomicBool, AtomicU32, Ordering}; use std::sync::{Arc, Mutex}; use cpal::traits::{DeviceTrait, HostTrait, StreamTrait}; use cpal::{SampleFormat, SizedSample}; use tokio::sync::mpsc; use tracing::{debug, error, info, warn}; use audiopus::coder::Encoder as OpusEncoder; use audiopus::{Application as OpusApp, Channels as OpusChannels, SampleRate as OpusSampleRate}; use tsclientlib::audio::AudioHandler; use chanora_protocol::{ AudioData, CodecType, InboundVoice, OutAudio, OutPacket, }; use crate::AudioError; /// Stable Chanora-side identifier for AudioHandler bookkeeping. /// We only ever have one connection at a time (DEC-006), so this is /// trivially unique. #[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)] pub struct SessionAudioId(pub u64); /// Audio framing: 48 kHz mono, 20 ms = 960 samples per frame. const SAMPLE_RATE: u32 = 48_000; const FRAME_SAMPLES: usize = 48_000 / 50; // 960 const MAX_OPUS_FRAME: usize = 1275; /// Engine configuration. #[derive(Debug, Clone)] pub struct AudioEngineConfig { /// Input gain applied before encoding (1.0 = pass-through). pub mic_gain: f32, /// Initial PTT state. When false the encoder is bypassed and no /// outbound packets are produced. pub ptt_initial: bool, /// Audio-effect toggles. The struct is honoured by *naming* but /// the filters themselves are still no-op in Beta — see the /// crate-level docs and DEC-007/008/009/010. pub effects: crate::AudioEffects, /// A.5 mobile: prefer the OS-provided "voice communication" /// audio source on mobile platforms (Android /// `MediaRecorder.AudioSource.VOICE_COMMUNICATION`, iOS /// `AVAudioSession.Mode.voiceChat`). On Linux desktop this is /// ignored — the DSP chain stays a no-op and we use the /// default ALSA/PipeWire source. /// /// Beta status: the *config flag* is plumbed through every /// layer; the *Android-side preset switch* is documented but /// not yet wired through cpal, which currently uses the /// AAudio default input. RISK-AUDIO-MOBILE-001 tracks this gap. /// Setting `true` is a forward-compatible no-op for Beta and /// will become active once cpal exposes input-preset hooks (or /// when Chanora ships an Oboe-based fork). pub mobile_voice_preset: bool, } impl Default for AudioEngineConfig { fn default() -> Self { Self { mic_gain: 1.0, ptt_initial: false, effects: crate::AudioEffects::default(), mobile_voice_preset: true, } } } /// Running audio engine. Drop = stop. pub struct AudioEngine { ptt: Arc, frames_sent: Arc, frames_received: Arc, // Streams must be dropped to stop audio. Both are `!Send` because // cpal's Stream isn't Send on some backends; we keep them in an // Option wrapped by Mutex so stop() can move them out. _input_stream: Mutex>, _output_stream: Mutex>, // Hand the inbound-voice forwarder task a shutdown signal. shutdown_tx: Option>, /// True if the capture stream actually opened. If false (typical /// in headless environments with null sources, or where the user /// denied microphone permission), PTT becomes a no-op and /// `frames_sent` stays at 0. capture_active: bool, } // cpal::Stream is not Send. We keep the engine pinned to the thread // it was constructed on — `chanora_core` spawns it inside a // `tokio::task::spawn_blocking` so the streams stay on that worker. // This `unsafe impl Send` is necessary because the outer Arc // is stored in core's session and must move into a task. The streams // themselves are only mutated through the Mutex and are dropped on // the same thread that owns them. // // SAFETY: cpal's Stream is not Send because the underlying audio API // callback thread may not be transferable. We never invoke methods on // the streams from any thread but the owning one; we only ever *drop* // them, which cpal documents as safe from any thread for ALSA and // PipeWire (Linux backend used here). For Windows/macOS the contract // may differ; production Beta+ work must revisit per-platform. unsafe impl Send for AudioEngine {} unsafe impl Sync for AudioEngine {} impl AudioEngine { /// Start the engine: open capture + playback streams, spawn the /// inbound-voice forwarder, return a handle. pub fn start( cfg: AudioEngineConfig, voice_out_tx: mpsc::Sender, mut voice_in_rx: mpsc::Receiver, ) -> Result { let host = cpal::default_host(); let in_dev = host .default_input_device() .ok_or(AudioError::NoInputDevice)?; let out_dev = host .default_output_device() .ok_or(AudioError::NoOutputDevice)?; info!( target: "chanora_audio", in_device = %in_dev.name().unwrap_or_default(), out_device = %out_dev.name().unwrap_or_default(), "starting audio engine" ); // A.5 mobile-only preset acknowledgement. On Linux desktop // the flag is ignored; on Android we log it so a future cpal // / Oboe wiring can be observed in the diagnostic export. #[cfg(target_os = "android")] { if cfg.mobile_voice_preset { info!( target: "chanora_audio", "android: mobile_voice_preset requested (RISK-AUDIO-MOBILE-001 — flag plumbed, switch pending cpal upstream)" ); } if cfg.effects.aec || cfg.effects.noise_suppression { info!( target: "chanora_audio", aec = cfg.effects.aec, ns = cfg.effects.noise_suppression, "android: effects requested; awaiting OS-source switch to engage hardware AEC/NS" ); } } #[cfg(target_os = "ios")] { if cfg.mobile_voice_preset { info!( target: "chanora_audio", "ios: voice-chat session mode requested (binding pending — Chanora iOS audio is documented-only for Beta)" ); } } let ptt = Arc::new(AtomicBool::new(cfg.ptt_initial)); let frames_sent = Arc::new(AtomicU32::new(0)); let frames_received = Arc::new(AtomicU32::new(0)); // ---------- Capture ---------- // Capture is best-effort. If the platform default input // device refuses any supported config (typical for // headless null sources or for users who deny the mic // permission) we log and continue — playback alone is // still useful. PTT becomes a no-op in that case. let capture_result = try_open_capture( &in_dev, voice_out_tx, ptt.clone(), frames_sent.clone(), cfg.mic_gain, ); let (input_stream, capture_active) = match capture_result { Ok(s) => (Some(s), true), Err(e) => { warn!( target: "chanora_audio", error = %e, "capture stream unavailable; continuing with playback only" ); (None, false) } }; if let Some(s) = &input_stream { s.play() .map_err(|e| AudioError::Backend(format!("input play: {e}")))?; } // ---------- Playback ---------- let audio_handler: Arc>> = Arc::new(Mutex::new(AudioHandler::new())); let out_cfg = out_dev .default_output_config() .map_err(|e| AudioError::StreamConfig(format!("output default: {e}")))?; let out_format = out_cfg.sample_format(); // AudioHandler::fill_buffer expects 48 kHz stereo f32. let out_stream_cfg = cpal::StreamConfig { channels: 2, sample_rate: cpal::SampleRate(SAMPLE_RATE), buffer_size: cpal::BufferSize::Default, }; let output_stream = match out_format { SampleFormat::F32 => build_output_stream::( &out_dev, &out_stream_cfg, audio_handler.clone(), )?, SampleFormat::I16 => build_output_stream::( &out_dev, &out_stream_cfg, audio_handler.clone(), )?, SampleFormat::U16 => build_output_stream::( &out_dev, &out_stream_cfg, audio_handler.clone(), )?, other => { return Err(AudioError::StreamConfig(format!( "unsupported output format: {other:?}" ))) } }; output_stream .play() .map_err(|e| AudioError::Backend(format!("output play: {e}")))?; // ---------- Inbound forwarder ---------- let (shutdown_tx, mut shutdown_rx) = tokio::sync::oneshot::channel(); let handler_for_task = audio_handler.clone(); let frames_received_for_task = frames_received.clone(); tokio::spawn(async move { loop { tokio::select! { _ = &mut shutdown_rx => { debug!(target: "chanora_audio", "inbound forwarder shutting down"); break; } item = voice_in_rx.recv() => { match item { Some(v) => { let id = SessionAudioId(v.from_client); let mut h = handler_for_task.lock().unwrap(); if let Err(e) = h.handle_packet(id, v.packet) { debug!(target: "chanora_audio", error = %e, "decode failed"); } else { frames_received_for_task.fetch_add(1, Ordering::Relaxed); } } None => break, } } } } }); Ok(Self { ptt, frames_sent, frames_received, _input_stream: Mutex::new(input_stream), _output_stream: Mutex::new(Some(output_stream)), shutdown_tx: Some(shutdown_tx), capture_active, }) } /// Stop the engine. Idempotent. pub fn stop(&mut self) { if let Some(tx) = self.shutdown_tx.take() { let _ = tx.send(()); } // Drop the streams, which stops their callback threads. let _ = self._input_stream.lock().unwrap().take(); let _ = self._output_stream.lock().unwrap().take(); info!(target: "chanora_audio", "audio engine stopped"); } /// Set the push-to-talk active state. When false, captured audio /// is discarded before encoding. No-op if capture is inactive. pub fn set_ptt(&self, active: bool) { self.ptt.store(active, Ordering::Relaxed); } /// Current PTT state. pub fn ptt(&self) -> bool { self.ptt.load(Ordering::Relaxed) } /// True if the capture stream opened. When false, the engine /// runs in playback-only mode and PTT is a no-op. pub fn capture_active(&self) -> bool { self.capture_active } /// Number of Opus frames sent since the engine started. pub fn frames_sent(&self) -> u32 { self.frames_sent.load(Ordering::Relaxed) } /// Number of inbound voice packets received and decoded. pub fn frames_received(&self) -> u32 { self.frames_received.load(Ordering::Relaxed) } } impl Drop for AudioEngine { fn drop(&mut self) { self.stop(); } } // ---------- Capture pipeline ---------- fn try_open_capture( in_dev: &cpal::Device, voice_out_tx: mpsc::Sender, ptt: Arc, frames_sent: Arc, mic_gain: f32, ) -> Result { let in_cfg = in_dev .default_input_config() .map_err(|e| AudioError::StreamConfig(format!("input default: {e}")))?; let in_sample_rate = in_cfg.sample_rate().0; let in_channels = in_cfg.channels() as usize; let in_format = in_cfg.sample_format(); let in_stream_cfg: cpal::StreamConfig = in_cfg.into(); let opus_enc = OpusEncoder::new( OpusSampleRate::Hz48000, OpusChannels::Mono, OpusApp::Voip, ) .map_err(|e| AudioError::Opus(format!("encoder new: {e}")))?; let capture_state = Arc::new(Mutex::new(CaptureState::new( opus_enc, in_sample_rate, in_channels, mic_gain, voice_out_tx, ptt, frames_sent, ))); let stream = match in_format { SampleFormat::F32 => build_input_stream::(in_dev, &in_stream_cfg, capture_state)?, SampleFormat::I16 => build_input_stream::(in_dev, &in_stream_cfg, capture_state)?, SampleFormat::U16 => build_input_stream::(in_dev, &in_stream_cfg, capture_state)?, other => { return Err(AudioError::StreamConfig(format!( "unsupported input format: {other:?}" ))) } }; Ok(stream) } struct CaptureState { encoder: OpusEncoder, in_sample_rate: u32, in_channels: usize, mic_gain: f32, /// 48 kHz mono buffer accumulated to FRAME_SAMPLES before each encode. pcm_accum: Vec, /// Resampling state for non-48k sources (very simple linear resampler). resample_pos: f64, opus_out: [u8; MAX_OPUS_FRAME], voice_out_tx: mpsc::Sender, ptt: Arc, frames_sent: Arc, } impl CaptureState { fn new( encoder: OpusEncoder, in_sample_rate: u32, in_channels: usize, mic_gain: f32, voice_out_tx: mpsc::Sender, ptt: Arc, frames_sent: Arc, ) -> Self { Self { encoder, in_sample_rate, in_channels, mic_gain, pcm_accum: Vec::with_capacity(FRAME_SAMPLES * 2), resample_pos: 0.0, opus_out: [0u8; MAX_OPUS_FRAME], voice_out_tx, ptt, frames_sent, } } /// Consume an arbitrary-rate, multichannel cpal buffer; produce /// 48 kHz mono frames; encode and send on PTT. fn ingest(&mut self, buf: &[T]) { if !self.ptt.load(Ordering::Relaxed) { // Drain accumulator while muted so we don't pop on PTT release. self.pcm_accum.clear(); return; } // 1. Down-mix to mono + gain. let mono: Vec = buf .chunks(self.in_channels) .map(|frame| { let sum: f32 = frame.iter().map(|s| s.to_f32_sample()).sum(); (sum / frame.len() as f32) * self.mic_gain }) .collect(); // 2. Resample to 48 kHz if needed. if self.in_sample_rate == SAMPLE_RATE { self.pcm_accum.extend_from_slice(&mono); } else { self.resample_into_accum(&mono); } // 3. Encode any complete frames. while self.pcm_accum.len() >= FRAME_SAMPLES { let frame: Vec = self.pcm_accum.drain(..FRAME_SAMPLES).collect(); match self.encoder.encode_float(&frame, &mut self.opus_out[..]) { Ok(len) => { let packet = OutAudio::new(&AudioData::C2S { id: 0, codec: CodecType::OpusVoice, data: &self.opus_out[..len], }); match self.voice_out_tx.try_send(packet) { Ok(()) => { self.frames_sent.fetch_add(1, Ordering::Relaxed); } Err(mpsc::error::TrySendError::Full(_)) => { warn!(target: "chanora_audio", "voice_out queue full; dropping frame"); } Err(mpsc::error::TrySendError::Closed(_)) => { warn!(target: "chanora_audio", "voice_out closed; stopping send"); } } } Err(e) => { error!(target: "chanora_audio", error = %e, "opus encode failed"); } } } } /// Simple linear resampler for `in_sample_rate → 48000`. /// Production quality work belongs in a Beta+ DSP module. fn resample_into_accum(&mut self, mono: &[f32]) { let ratio = self.in_sample_rate as f64 / SAMPLE_RATE as f64; let mut pos = self.resample_pos; while pos < mono.len() as f64 { let i = pos as usize; let frac = pos - i as f64; let a = mono[i]; let b = if i + 1 < mono.len() { mono[i + 1] } else { a }; self.pcm_accum .push((a as f64 + frac * (b - a) as f64) as f32); pos += ratio; } // Keep the leftover sub-sample offset for the next buffer. self.resample_pos = pos - mono.len() as f64; } } /// Per-sample format conversion to f32 in the range [-1.0, 1.0]. trait ToF32 { fn to_f32_sample(self) -> f32; } impl ToF32 for f32 { fn to_f32_sample(self) -> f32 { self } } impl ToF32 for i16 { fn to_f32_sample(self) -> f32 { f32::from(self) / f32::from(i16::MAX) } } impl ToF32 for u16 { fn to_f32_sample(self) -> f32 { (f32::from(self) - f32::from(i16::MAX) - 1.0) / f32::from(i16::MAX) } } fn build_input_stream( device: &cpal::Device, config: &cpal::StreamConfig, state: Arc>, ) -> Result where T: SizedSample + ToF32 + Send + 'static, { let stream = device .build_input_stream( config, move |data: &[T], _| { let mut s = state.lock().unwrap(); s.ingest(data); }, move |e| { error!(target: "chanora_audio", error = %e, "input stream error"); }, None, ) .map_err(|e| AudioError::Backend(format!("build_input_stream: {e}")))?; Ok(stream) } // ---------- Playback pipeline ---------- fn build_output_stream( device: &cpal::Device, config: &cpal::StreamConfig, handler: Arc>>, ) -> Result where T: SizedSample + FromF32 + Send + 'static, { // Reusable f32 scratch buffer. cpal callbacks ask for a max // buffer size known at construction time; we allocate per-call // because reusing across calls would need an Arc> and // we already hold one for the handler. let stream = device .build_output_stream( config, move |out: &mut [T], _| { let mut scratch = vec![0.0f32; out.len()]; { let mut h = handler.lock().unwrap(); h.fill_buffer(&mut scratch); } for (dst, src) in out.iter_mut().zip(scratch.into_iter()) { *dst = T::from_f32_sample(src); } }, move |e| { error!(target: "chanora_audio", error = %e, "output stream error"); }, None, ) .map_err(|e| AudioError::Backend(format!("build_output_stream: {e}")))?; Ok(stream) } trait FromF32 { fn from_f32_sample(v: f32) -> Self; } impl FromF32 for f32 { fn from_f32_sample(v: f32) -> Self { v } } impl FromF32 for i16 { fn from_f32_sample(v: f32) -> Self { (v.clamp(-1.0, 1.0) * f32::from(i16::MAX)) as i16 } } impl FromF32 for u16 { fn from_f32_sample(v: f32) -> Self { let s = (v.clamp(-1.0, 1.0) * f32::from(i16::MAX)) as i32; (s + i32::from(i16::MAX) + 1) as u16 } }