Files
chanora/crates/chanora_audio/src/audio_processing.rs
T
Edison Jwa 82441f3d97 feat(voice): real-time mic input level metering at 30 Hz (#25)
* feat(voice): add real-time mic input level metering at 30 Hz

Expose input RMS from the audio engine through the bridge as a
dedicated Rust→Dart Stream<double>, replacing the binary on/off
indicator with a proportional dBFS level meter.

Rust side:
- chanora_audio: add set_input_dbfs/input_dbfs accessors to
  SharedAudioProcessingStats; restructure CaptureState::ingest()
  to compute dBFS from mono buffer before the PTT guard so the
  meter shows mic activity even when not transmitting.
- chanora_core: widen audio_stats() return to include f32 input
  level.
- chanora_bridge: add input_level: f32 to BridgeAudioStats and
  new input_level_stream(sink: StreamSink<f32>) that pushes at
  ~30 Hz via tokio interval task.
- Update frb_generated.rs serialization for the new field.

Flutter side:
- VoiceLevelMeter: accept optional double level (dBFS), map
  -60..0 dBFS to 0..1 fill fraction, animate with
  TweenAnimationBuilder for smooth transitions.
- voice_compact.dart: subscribe to inputLevelStream in the voice
  details sheet for 30 Hz meter updates, keeping 250 ms poll for
  TX/RX counters.
- voice_bar.dart: accept optional inputLevel from the stream.
- main.dart: subscribe to inputLevelStream, pass to VoiceBar.

* chore: sync Flutter build config and dependency updates

- Add Flutter migrator flags to gradle.properties (builtInKotlin, newDsl)
- Add FlutterGeneratedPluginSwiftPackage to iOS/macOS Xcode projects
- Update meta 1.17→1.18, test_api 0.7.10→0.7.11
- Rebuild chanora_bridge framework for macOS
- Update Podfile.lock for iOS and macOS

* fix(voice): correct meter animation, pre-gain dBFS, stream lifecycle, and protocol warnings

B1: Convert VoiceLevelMeter to StatefulWidget tracking previous fill
     as Tween begin so the meter animates smoothly instead of resetting
     to zero on every frame.

B2: Compute dBFS from pre-gain mono samples in CaptureState::ingest()
     so the level meter reflects raw mic input, matching mobile paths.

B4: End input_level_stream after 10 consecutive session errors instead
     of emitting -120 dBFS forever when the session is gone.

Also fixes all 13 clippy warnings in chanora_protocol: collapsed
nested if-let patterns, replaced .ok() + Some matching with Ok, used
? operator, and introduced EventChannels struct to reduce the four
helper functions below the 7-argument threshold.

* fix(voice): use MissedTickBehavior::Skip for level meter stream and align dBFS doc

Set MissedTickBehavior::Skip on the input_level_stream tokio interval
so slow audio_stats() calls skip missed ticks instead of bursting,
preventing CPU spikes on the UI meter thread.

Align VoiceLevelMeter class doc: the mapping floors at -60 dBFS
(via dbfsToFraction), not the full -120 range.
2026-06-05 20:57:16 +09:00

529 lines
19 KiB
Rust

//! P1 audio-processing configuration and statistics.
//!
//! iOS P1 ships the platform VoiceProcessingIO path by default. Rust
//! software AEC/NS/AGC backends are represented in the schema so the
//! bridge can reject unsafe combinations instead of silently enabling
//! double processing.
use std::sync::atomic::{AtomicBool, AtomicU32, AtomicU64, Ordering};
use crate::AudioError;
/// Physical/logical audio route class used for route-aware policy.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum AudioRoute {
/// Built-in speakerphone path.
Speaker,
/// Built-in receiver/earpiece path.
Earpiece,
/// Wired headset or USB headset.
WiredHeadset,
/// Bluetooth Hands-Free Profile duplex route.
BluetoothHfp,
/// Bluetooth A2DP output-only route.
BluetoothA2dp,
/// Route could not be classified yet.
Unknown,
}
impl AudioRoute {
/// Stable bridge/debug string.
pub fn as_str(self) -> &'static str {
match self {
Self::Speaker => "speaker",
Self::Earpiece => "earpiece",
Self::WiredHeadset => "wired_headset",
Self::BluetoothHfp => "bluetooth_hfp",
Self::BluetoothA2dp => "bluetooth_a2dp",
Self::Unknown => "unknown",
}
}
/// Parse from the Swift-side route class string (case-insensitive).
/// Unrecognised strings map to `Unknown`.
pub fn from_route_class(s: &str) -> Self {
match s {
"Speaker" | "speaker" => Self::Speaker,
"Earpiece" | "earpiece" => Self::Earpiece,
"WiredHeadset" | "wired_headset" => Self::WiredHeadset,
"BluetoothHfp" | "bluetooth_hfp" => Self::BluetoothHfp,
"BluetoothA2dp" | "bluetooth_a2dp" => Self::BluetoothA2dp,
_ => Self::Unknown,
}
}
}
/// iOS voice-processing mode.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum IosVoiceProcessingMode {
/// Shipping default: Apple VoiceProcessingIO owns AEC/NS/AGC.
PlatformVoiceProcessing,
/// Experimental raw capture-processing path.
SonoraExperimental,
}
/// Processing backend selected by policy/config.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum AudioBackend {
/// Platform voice processing, VPIO on iOS.
PlatformVoiceProcessing,
/// Rust-native Sonora backend.
Sonora,
/// WebRTC Audio Processing Module backend.
WebrtcApm,
/// No processing.
Noop,
}
impl AudioBackend {
/// Stable bridge/debug string.
pub fn as_str(self) -> &'static str {
match self {
Self::PlatformVoiceProcessing => "platform_voice_processing",
Self::Sonora => "sonora",
Self::WebrtcApm => "webrtc_apm",
Self::Noop => "noop",
}
}
}
/// VAD backend selected by policy/config.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum VadBackend {
/// Silero ONNX VAD. P1 schema default when model/runtime exist.
SileroOnnx,
/// WebRTC-style fallback VAD.
WebrtcVad,
/// Debug-only energy VAD.
EnergyDebug,
/// VAD disabled.
Disabled,
}
fn default_vad_backend() -> VadBackend {
VadBackend::SileroOnnx
}
impl VadBackend {
/// Stable bridge/debug string.
pub fn as_str(self) -> &'static str {
match self {
#[cfg(any(target_os = "ios", target_os = "macos"))]
Self::SileroOnnx => "apple_coreml",
#[cfg(not(any(target_os = "ios", target_os = "macos")))]
Self::SileroOnnx => "silero_vad_onnx",
Self::WebrtcVad => "webrtc_vad",
Self::EnergyDebug => "energy_debug",
Self::Disabled => "disabled",
}
}
}
/// Effect owner for AEC/NS/AGC policy fields.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum EffectOwner {
/// Owned by platform voice processing.
Platform,
/// Owned by Sonora.
Sonora,
/// Owned by WebRTC APM.
WebrtcApm,
/// Conservative route-managed setting.
Conservative,
/// Disabled.
Off,
}
/// Voice-processing configuration owned by the Rust audio engine.
#[derive(Debug, Clone, PartialEq)]
pub struct AudioProcessingConfig {
/// Current route class.
pub route: AudioRoute,
/// iOS processing mode.
pub ios_mode: IosVoiceProcessingMode,
/// Processing backend.
pub processing_backend: AudioBackend,
/// VAD backend.
pub vad_backend: VadBackend,
/// AEC owner.
pub aec: EffectOwner,
/// Noise suppression owner.
pub ns: EffectOwner,
/// AGC owner.
pub agc: EffectOwner,
/// High-pass filter enabled.
pub hpf_enabled: bool,
/// Limiter enabled.
pub limiter_enabled: bool,
/// Hangover after speech closes.
pub vad_hangover_ms: u32,
/// Pre-roll before open.
pub vad_pre_roll_ms: u32,
/// Minimum transmit duration after open.
pub vad_min_tx_ms: u32,
/// Debug WAV dumps enabled.
pub debug_wav_dump_enabled: bool,
}
impl Default for AudioProcessingConfig {
fn default() -> Self {
Self {
route: AudioRoute::Speaker,
ios_mode: IosVoiceProcessingMode::PlatformVoiceProcessing,
processing_backend: AudioBackend::PlatformVoiceProcessing,
vad_backend: default_vad_backend(),
aec: EffectOwner::Platform,
// iOS VPIO owns NS/AGC on the default shipping path. Software
// effects are opt-in through the experimental raw route only.
ns: EffectOwner::Platform,
agc: EffectOwner::Platform,
hpf_enabled: true,
limiter_enabled: true,
vad_hangover_ms: crate::voice_activity::VAD_HANGOVER_MS,
vad_pre_roll_ms: 160,
vad_min_tx_ms: crate::voice_activity::VAD_MIN_TX_MS,
debug_wav_dump_enabled: false,
}
}
}
impl AudioProcessingConfig {
/// Validate P1 iOS invariants before applying a config.
pub fn validate_for_ios(&self) -> Result<(), AudioError> {
if self.route == AudioRoute::BluetoothA2dp {
return Err(AudioError::InvalidAudioProcessingConfig(
"bluetooth_a2dp is output-only and cannot transmit duplex voice".to_string(),
));
}
if self.ios_mode == IosVoiceProcessingMode::PlatformVoiceProcessing
&& (self.processing_backend == AudioBackend::Sonora
|| self.processing_backend == AudioBackend::WebrtcApm
|| self.aec == EffectOwner::Sonora
|| self.aec == EffectOwner::WebrtcApm
|| self.ns == EffectOwner::Sonora
|| self.ns == EffectOwner::WebrtcApm
|| self.agc == EffectOwner::Sonora
|| self.agc == EffectOwner::WebrtcApm)
{
return Err(AudioError::InvalidAudioProcessingConfig(
"software audio processing cannot be enabled with iOS VoiceProcessingIO"
.to_string(),
));
}
if self.ios_mode == IosVoiceProcessingMode::SonoraExperimental
&& self.processing_backend != AudioBackend::WebrtcApm
{
return Err(AudioError::InvalidAudioProcessingConfig(
"ios raw processing mode requires the WebRTC APM backend".to_string(),
));
}
Ok(())
}
/// Demote a failed VAD backend to the WebRTC fallback.
///
/// Returns `true` when the config changed.
pub fn disable_failed_vad_backend(&mut self, failed_backend: VadBackend) -> bool {
if self.vad_backend == failed_backend && failed_backend != VadBackend::WebrtcVad {
self.vad_backend = VadBackend::WebrtcVad;
true
} else {
false
}
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn default_config_is_valid_for_ios_vpio() {
let config = AudioProcessingConfig::default();
assert!(config.validate_for_ios().is_ok());
assert_eq!(
config.processing_backend,
AudioBackend::PlatformVoiceProcessing
);
assert_eq!(config.aec, EffectOwner::Platform);
assert_eq!(config.ns, EffectOwner::Platform);
assert_eq!(config.agc, EffectOwner::Platform);
assert_eq!(config.vad_backend, VadBackend::SileroOnnx);
}
#[test]
fn platform_voice_processing_rejects_software_effects() {
let config = AudioProcessingConfig {
ns: EffectOwner::WebrtcApm,
..AudioProcessingConfig::default()
};
assert!(config.validate_for_ios().is_err());
}
#[test]
fn raw_processing_allows_full_webrtc_apm_chain() {
let config = AudioProcessingConfig {
ios_mode: IosVoiceProcessingMode::SonoraExperimental,
processing_backend: AudioBackend::WebrtcApm,
aec: EffectOwner::WebrtcApm,
ns: EffectOwner::WebrtcApm,
agc: EffectOwner::WebrtcApm,
..AudioProcessingConfig::default()
};
assert!(config.validate_for_ios().is_ok());
}
#[test]
fn raw_processing_rejects_non_webrtc_apm_backend() {
let config = AudioProcessingConfig {
ios_mode: IosVoiceProcessingMode::SonoraExperimental,
processing_backend: AudioBackend::PlatformVoiceProcessing,
aec: EffectOwner::WebrtcApm,
ns: EffectOwner::WebrtcApm,
agc: EffectOwner::WebrtcApm,
..AudioProcessingConfig::default()
};
assert!(config.validate_for_ios().is_err());
}
#[test]
fn disable_failed_vad_backend_demotes_to_webrtc() {
let mut config = AudioProcessingConfig {
vad_backend: VadBackend::SileroOnnx,
..AudioProcessingConfig::default()
};
assert!(config.disable_failed_vad_backend(VadBackend::SileroOnnx));
assert_eq!(config.vad_backend, VadBackend::WebrtcVad);
assert!(!config.disable_failed_vad_backend(VadBackend::SileroOnnx));
}
}
/// Runtime audio processing stats exposed to bridge/UI diagnostics.
#[derive(Debug, Clone)]
pub struct AudioProcessingStats {
/// Input dBFS before processing.
pub input_dbfs: f32,
/// Render dBFS before playout.
pub render_dbfs: f32,
/// Processed capture dBFS.
pub processed_dbfs: f32,
/// Latest VAD probability or fallback confidence.
pub vad_probability: f32,
/// VAD active state.
pub vad_active: bool,
/// Current resolved transmit state.
pub transmitting: bool,
/// VAD backend.
pub vad_backend: VadBackend,
/// Whether fallback VAD is active.
pub vad_fallback_active: bool,
/// Processing backend.
pub processing_backend: AudioBackend,
/// iOS mode.
pub ios_voice_processing_mode: IosVoiceProcessingMode,
/// Route class.
pub audio_route: AudioRoute,
/// Actual sample rate.
pub actual_sample_rate_hz: u32,
/// Actual IO buffer frame count.
pub actual_io_buffer_frames: u32,
/// Input overrun count.
pub input_overruns: u64,
/// Output underrun count.
pub output_underruns: u64,
/// Callback xrun count.
pub callback_xruns: u64,
/// Clipped sample count.
pub clipped_samples: u64,
/// Number of effectively silent processed capture frames.
pub zero_frames: u64,
/// Number of processed capture frames.
pub capture_frames: u64,
/// Number of input callbacks carrying 10 ms of audio.
pub callbacks_10ms: u64,
/// Number of input callbacks carrying 20 ms of audio.
pub callbacks_20ms: u64,
/// Number of input callbacks carrying any other size.
pub callbacks_other: u64,
/// Sonora enabled.
pub sonora_enabled: bool,
/// Platform voice processing enabled.
pub platform_voice_processing_enabled: bool,
}
/// Lock-free stats storage shared with callbacks.
pub struct SharedAudioProcessingStats {
input_dbfs: AtomicU32,
render_dbfs: AtomicU32,
processed_dbfs: AtomicU32,
vad_probability: AtomicU32,
vad_active: AtomicBool,
transmitting: AtomicBool,
vad_fallback_active: AtomicBool,
input_overruns: AtomicU64,
output_underruns: AtomicU64,
callback_xruns: AtomicU64,
clipped_samples: AtomicU64,
zero_frames: AtomicU64,
capture_frames: AtomicU64,
callbacks_10ms: AtomicU64,
callbacks_20ms: AtomicU64,
callbacks_other: AtomicU64,
actual_sample_rate_hz: AtomicU32,
actual_io_buffer_frames: AtomicU32,
}
impl Default for SharedAudioProcessingStats {
fn default() -> Self {
Self {
input_dbfs: AtomicU32::new((-120.0_f32).to_bits()),
render_dbfs: AtomicU32::new((-120.0_f32).to_bits()),
processed_dbfs: AtomicU32::new((-120.0_f32).to_bits()),
vad_probability: AtomicU32::new(0.0_f32.to_bits()),
vad_active: AtomicBool::new(false),
transmitting: AtomicBool::new(false),
vad_fallback_active: AtomicBool::new(false),
input_overruns: AtomicU64::new(0),
output_underruns: AtomicU64::new(0),
callback_xruns: AtomicU64::new(0),
clipped_samples: AtomicU64::new(0),
zero_frames: AtomicU64::new(0),
capture_frames: AtomicU64::new(0),
callbacks_10ms: AtomicU64::new(0),
callbacks_20ms: AtomicU64::new(0),
callbacks_other: AtomicU64::new(0),
actual_sample_rate_hz: AtomicU32::new(crate::frame::SAMPLE_RATE_HZ),
actual_io_buffer_frames: AtomicU32::new(crate::frame::FRAME_20MS_SAMPLES as u32),
}
}
}
impl SharedAudioProcessingStats {
/// Store the raw input dBFS level (desktop capture path).
/// Mobile platforms use [`Self::update_capture`] instead, which
/// also records VAD state; this lighter method is for the cpal
/// capture path that has no VAD pipeline.
pub fn set_input_dbfs(&self, dbfs: f32) {
self.input_dbfs.store(dbfs.to_bits(), Ordering::Relaxed);
}
/// Read the current input dBFS level.
pub fn input_dbfs(&self) -> f32 {
f32::from_bits(self.input_dbfs.load(Ordering::Relaxed))
}
/// Store capture levels and VAD state.
pub fn update_capture(
&self,
input_dbfs: f32,
processed_dbfs: f32,
probability: f32,
vad_active: bool,
transmitting: bool,
) {
self.input_dbfs
.store(input_dbfs.to_bits(), Ordering::Relaxed);
self.processed_dbfs
.store(processed_dbfs.to_bits(), Ordering::Relaxed);
self.vad_probability
.store(probability.clamp(0.0, 1.0).to_bits(), Ordering::Relaxed);
self.vad_active.store(vad_active, Ordering::Relaxed);
self.transmitting.store(transmitting, Ordering::Relaxed);
}
/// Store render level and callback buffer size.
pub fn update_render(&self, dbfs: f32, io_buffer_frames: u32) {
self.render_dbfs.store(dbfs.to_bits(), Ordering::Relaxed);
self.actual_io_buffer_frames
.store(io_buffer_frames, Ordering::Relaxed);
}
/// Record the actual device sample rate.
pub fn set_actual_sample_rate_hz(&self, sample_rate_hz: u32) {
self.actual_sample_rate_hz
.store(sample_rate_hz, Ordering::Relaxed);
}
/// Increment output underrun count.
pub fn increment_output_underrun(&self) {
self.output_underruns.fetch_add(1, Ordering::Relaxed);
}
/// Increment callback xrun count.
pub fn increment_callback_xrun(&self) {
self.callback_xruns.fetch_add(1, Ordering::Relaxed);
}
/// Add clipped samples.
pub fn add_clipped_samples(&self, count: u64) {
self.clipped_samples.fetch_add(count, Ordering::Relaxed);
}
/// Record one processed capture frame and whether it was effectively silent.
pub fn record_capture_frame(&self, zero_frame: bool) {
self.capture_frames.fetch_add(1, Ordering::Relaxed);
if zero_frame {
self.zero_frames.fetch_add(1, Ordering::Relaxed);
}
}
/// Bucket callback delivery sizes to diagnose timing jitter and packetization.
pub fn record_callback_frames(&self, frames: u64) {
let sample_rate_hz = self.actual_sample_rate_hz.load(Ordering::Relaxed).max(1);
let frames_10ms = (sample_rate_hz / 100) as u64;
let frames_20ms = (sample_rate_hz / 50) as u64;
match frames {
value if value == frames_10ms => {
self.callbacks_10ms.fetch_add(1, Ordering::Relaxed);
}
value if value == frames_20ms => {
self.callbacks_20ms.fetch_add(1, Ordering::Relaxed);
}
_ => {
self.callbacks_other.fetch_add(1, Ordering::Relaxed);
}
}
}
/// Store whether the selected VAD backend is currently using a fallback.
pub fn set_vad_fallback_active(&self, active: bool) {
self.vad_fallback_active.store(active, Ordering::Relaxed);
}
/// Build an owned stats snapshot with config-derived labels.
pub fn snapshot(&self, config: &AudioProcessingConfig) -> AudioProcessingStats {
AudioProcessingStats {
input_dbfs: f32::from_bits(self.input_dbfs.load(Ordering::Relaxed)),
render_dbfs: f32::from_bits(self.render_dbfs.load(Ordering::Relaxed)),
processed_dbfs: f32::from_bits(self.processed_dbfs.load(Ordering::Relaxed)),
vad_probability: f32::from_bits(self.vad_probability.load(Ordering::Relaxed)),
vad_active: self.vad_active.load(Ordering::Relaxed),
transmitting: self.transmitting.load(Ordering::Relaxed),
vad_backend: config.vad_backend,
vad_fallback_active: self.vad_fallback_active.load(Ordering::Relaxed),
processing_backend: config.processing_backend,
ios_voice_processing_mode: config.ios_mode,
audio_route: config.route,
actual_sample_rate_hz: self.actual_sample_rate_hz.load(Ordering::Relaxed),
actual_io_buffer_frames: self.actual_io_buffer_frames.load(Ordering::Relaxed),
input_overruns: self.input_overruns.load(Ordering::Relaxed),
output_underruns: self.output_underruns.load(Ordering::Relaxed),
callback_xruns: self.callback_xruns.load(Ordering::Relaxed),
clipped_samples: self.clipped_samples.load(Ordering::Relaxed),
zero_frames: self.zero_frames.load(Ordering::Relaxed),
capture_frames: self.capture_frames.load(Ordering::Relaxed),
callbacks_10ms: self.callbacks_10ms.load(Ordering::Relaxed),
callbacks_20ms: self.callbacks_20ms.load(Ordering::Relaxed),
callbacks_other: self.callbacks_other.load(Ordering::Relaxed),
sonora_enabled: config.processing_backend == AudioBackend::Sonora,
platform_voice_processing_enabled: config.processing_backend
== AudioBackend::PlatformVoiceProcessing,
}
}
}