| | |
| | | }; |
| | | |
| | | use anyhow::{Context, Result, anyhow}; |
| | | use audio::{AudioDiagnostics, PcmFrame, load_pre_recorded_frames}; |
| | | use audio::{AudioDiagnostics, load_pre_recorded_frames}; |
| | | use base64::{Engine as _, engine::general_purpose}; |
| | | use futures_util::StreamExt; |
| | | use libwebrtc::{ |
| | |
| | | use tokio::{sync::mpsc::UnboundedReceiver, task::JoinHandle}; |
| | | use tracing::{info, warn}; |
| | | |
| | | const TARGET_SAMPLE_RATE_HZ: u32 = 48_000; |
| | | const TARGET_NUM_CHANNELS: u16 = 1; |
| | | const USER_AUDIO_SAMPLE_RATE_HZ: u32 = 48_000; |
| | | const USER_AUDIO_NUM_CHANNELS: u16 = 1; |
| | | const DEFAULT_BOT_AUDIO_PROFILE: &str = "pcm-16k"; |
| | | const LIVEKIT_48K_SAMPLE_RATE_HZ: u32 = 48_000; |
| | | const PCM_16K_SAMPLE_RATE_HZ: u32 = 16_000; |
| | | const BOT_NUM_CHANNELS: u16 = 1; |
| | | const TRACK_NAME: &str = "bot-main-audio"; |
| | | |
| | | #[tokio::main(flavor = "multi_thread")] |
| | |
| | | &http, |
| | | config.greeting_audio_file.as_deref(), |
| | | config.greeting_audio_url.as_deref(), |
| | | TARGET_SAMPLE_RATE_HZ, |
| | | TARGET_NUM_CHANNELS, |
| | | config.bot_audio_profile.sample_rate_hz, |
| | | config.bot_audio_profile.num_channels, |
| | | config.audio_debug_dump_dir.as_deref(), |
| | | &config.call_id, |
| | | "greeting", |
| | |
| | | room_alias = %redact(&config.room_id), |
| | | participant_alias = %redact(&config.bot_participant_identity), |
| | | greeting_source = %config.greeting_source, |
| | | bot_audio_profile = %config.bot_audio_profile.profile, |
| | | bot_sample_rate_hz = config.bot_audio_profile.sample_rate_hz, |
| | | bot_num_channels = config.bot_audio_profile.num_channels, |
| | | "combrabo voice runtime helper connected" |
| | | ); |
| | | emit_activity( |
| | |
| | | &config.call_id, |
| | | &config.trace_id, |
| | | TRACK_NAME, |
| | | TARGET_SAMPLE_RATE_HZ, |
| | | u32::from(TARGET_NUM_CHANNELS), |
| | | config.bot_audio_profile.profile.clone(), |
| | | config.bot_audio_profile.sample_rate_hz, |
| | | u32::from(config.bot_audio_profile.num_channels), |
| | | config.user_participant_identity.clone(), |
| | | ) |
| | | .await?; |
| | |
| | | simple_vad_gate_until_greeting_done: bool, |
| | | simple_vad_post_greeting_delay_ms: u64, |
| | | simple_vad_config: SimpleVadConfig, |
| | | bot_audio_profile: BotAudioProfile, |
| | | } |
| | | |
| | | #[derive(Clone)] |
| | | struct BotAudioProfile { |
| | | profile: String, |
| | | sample_rate_hz: u32, |
| | | num_channels: u16, |
| | | } |
| | | |
| | | impl BotAudioProfile { |
| | | fn from_env() -> Result<Self> { |
| | | let profile = env::var("CV_BOT_AUDIO_PROFILE") |
| | | .unwrap_or_else(|_| DEFAULT_BOT_AUDIO_PROFILE.to_string()) |
| | | .trim() |
| | | .to_ascii_lowercase(); |
| | | match profile.as_str() { |
| | | "livekit-48k" | "48k" => Ok(Self { |
| | | profile: "livekit-48k".to_string(), |
| | | sample_rate_hz: LIVEKIT_48K_SAMPLE_RATE_HZ, |
| | | num_channels: BOT_NUM_CHANNELS, |
| | | }), |
| | | "pcm-16k" | "16k" => Ok(Self { |
| | | profile: "pcm-16k".to_string(), |
| | | sample_rate_hz: PCM_16K_SAMPLE_RATE_HZ, |
| | | num_channels: BOT_NUM_CHANNELS, |
| | | }), |
| | | "custom" => { |
| | | let sample_rate_hz = u32_env("CV_BOT_SAMPLE_RATE_HZ", LIVEKIT_48K_SAMPLE_RATE_HZ); |
| | | let num_channels = u16_env("CV_BOT_NUM_CHANNELS", BOT_NUM_CHANNELS); |
| | | if sample_rate_hz == 0 { |
| | | return Err(anyhow!("CV_BOT_SAMPLE_RATE_HZ must be positive")); |
| | | } |
| | | if num_channels == 0 { |
| | | return Err(anyhow!("CV_BOT_NUM_CHANNELS must be positive")); |
| | | } |
| | | Ok(Self { |
| | | profile, |
| | | sample_rate_hz, |
| | | num_channels, |
| | | }) |
| | | } |
| | | _ => Err(anyhow!( |
| | | "unsupported CV_BOT_AUDIO_PROFILE {}; expected livekit-48k, pcm-16k or custom", |
| | | profile |
| | | )), |
| | | } |
| | | } |
| | | } |
| | | |
| | | #[derive(Clone)] |
| | |
| | | simple_vad_gate_until_greeting_done: bool_env("CV_VAD_GATE_UNTIL_GREETING_DONE", true), |
| | | simple_vad_post_greeting_delay_ms: u64_env("CV_VAD_POST_GREETING_DELAY_MS", 800), |
| | | simple_vad_config: SimpleVadConfig::from_env(), |
| | | bot_audio_profile: BotAudioProfile::from_env()?, |
| | | }) |
| | | } |
| | | } |
| | |
| | | .unwrap_or(default_value) |
| | | } |
| | | |
| | | fn u16_env(key: &str, default_value: u16) -> u16 { |
| | | env::var(key) |
| | | .ok() |
| | | .and_then(|value| value.trim().parse::<u16>().ok()) |
| | | .unwrap_or(default_value) |
| | | } |
| | | |
| | | fn redact(value: &str) -> String { |
| | | if value.len() <= 8 { |
| | | return "redacted".to_string(); |
| | |
| | | "traceId": trace_id, |
| | | "turnId": turn_id, |
| | | "eventName": event_name, |
| | | "eventWallTimeMs": current_time_millis(), |
| | | "result": result, |
| | | "reasonCode": reason_code, |
| | | "retryable": retryable, |
| | |
| | | audio::write_pcm_wav( |
| | | &output_path, |
| | | &turn.samples, |
| | | TARGET_SAMPLE_RATE_HZ, |
| | | TARGET_NUM_CHANNELS, |
| | | USER_AUDIO_SAMPLE_RATE_HZ, |
| | | USER_AUDIO_NUM_CHANNELS, |
| | | )?; |
| | | let byte_size = fs::metadata(&output_path) |
| | | .context("failed to stat turn artifact")? |
| | |
| | | artifact_type: "local_file".to_string(), |
| | | path_ref: path_ref.to_string(), |
| | | format: "wav".to_string(), |
| | | sample_rate: TARGET_SAMPLE_RATE_HZ, |
| | | channels: u32::from(TARGET_NUM_CHANNELS), |
| | | sample_rate: USER_AUDIO_SAMPLE_RATE_HZ, |
| | | channels: u32::from(USER_AUDIO_NUM_CHANNELS), |
| | | duration_ms: turn.duration_ms, |
| | | byte_size, |
| | | }, |
| | |
| | | .trim() |
| | | .to_ascii_lowercase(); |
| | | let frames = if format == "pcm_s16le" { |
| | | pcm_s16le_payload_to_frames( |
| | | &payload, |
| | | audio_chunk.sample_rate.unwrap_or(TARGET_SAMPLE_RATE_HZ), |
| | | audio_chunk |
| | | .channels |
| | | .unwrap_or(u32::from(TARGET_NUM_CHANNELS)), |
| | | )? |
| | | let sample_rate = audio_chunk.sample_rate.unwrap_or(sink.sample_rate_hz); |
| | | let channels = audio_chunk.channels.unwrap_or(u32::from(sink.num_channels)); |
| | | if state.pcm_stream_decoder.is_none() { |
| | | state.pcm_stream_decoder = Some(audio::PcmS16leStreamDecoder::new( |
| | | sample_rate, |
| | | channels, |
| | | sink.sample_rate_hz, |
| | | sink.num_channels, |
| | | )?); |
| | | } |
| | | state.pcm_stream_network_chunk_count = |
| | | state.pcm_stream_network_chunk_count.saturating_add(1); |
| | | let stream_result = state |
| | | .pcm_stream_decoder |
| | | .as_mut() |
| | | .expect("pcm stream decoder initialized") |
| | | .push_bytes( |
| | | &payload, |
| | | sample_rate, |
| | | channels, |
| | | audio_chunk.last.unwrap_or(false), |
| | | )?; |
| | | if stream_result.dropped_tail_bytes > 0 { |
| | | warn!( |
| | | call_id = %call_id, |
| | | trace_id = %trace_id, |
| | | turn_id = %turn.turn_id, |
| | | chunk_seq = audio_chunk.chunk_seq.unwrap_or_default(), |
| | | dropped_tail_bytes = stream_result.dropped_tail_bytes, |
| | | "runtime helper stream_audio_pcm_unaligned_tail_dropped" |
| | | ); |
| | | } |
| | | if stream_result.frames.is_empty() { |
| | | warn!( |
| | | call_id = %call_id, |
| | | trace_id = %trace_id, |
| | | turn_id = %turn.turn_id, |
| | | chunk_seq = audio_chunk.chunk_seq.unwrap_or_default(), |
| | | buffered_source_bytes = stream_result.buffered_source_bytes, |
| | | buffered_source_samples = stream_result.buffered_source_samples, |
| | | network_chunk_count = state.pcm_stream_network_chunk_count, |
| | | "runtime helper stream_audio_pcm_waiting_for_20ms_frame" |
| | | ); |
| | | return Ok(0); |
| | | } |
| | | stream_result.frames |
| | | } else if matches!(format.as_str(), "mp3" | "mpeg" | "wav") { |
| | | state.encoded_audio_buffer.extend_from_slice(&payload); |
| | | match audio::decode_audio_bytes_to_frames( |
| | | &state.encoded_audio_buffer, |
| | | "stream_chunk", |
| | | TARGET_SAMPLE_RATE_HZ, |
| | | TARGET_NUM_CHANNELS, |
| | | sink.sample_rate_hz, |
| | | sink.num_channels, |
| | | bridge_config.audio_debug_dump_dir.as_deref(), |
| | | call_id, |
| | | &format!("stream-reply-{}", turn.turn_id), |
| | |
| | | error = %safe_error(&error.to_string()), |
| | | "runtime helper stream_audio_chunk_decode_waiting_for_more_data" |
| | | ); |
| | | return Ok(0); |
| | | } |
| | | Err(error) if state.first_audio_frame_written => { |
| | | warn!( |
| | | call_id = %call_id, |
| | | trace_id = %trace_id, |
| | | turn_id = %turn.turn_id, |
| | | chunk_seq = audio_chunk.chunk_seq.unwrap_or_default(), |
| | | format = %format, |
| | | buffered_bytes = state.encoded_audio_buffer.len(), |
| | | error = %safe_error(&error.to_string()), |
| | | "runtime helper stream_audio_final_chunk_decode_ignored" |
| | | ); |
| | | state.encoded_audio_buffer.clear(); |
| | | return Ok(0); |
| | | } |
| | | Err(error) => return Err(error).context("failed to decode final stream audio chunk"), |
| | |
| | | sleep_until(pacing_started_at + Duration::from_millis(((index + 1) as u64) * 20)).await; |
| | | } |
| | | if audio_chunk.last.unwrap_or(false) { |
| | | let mut debug_source_path = None; |
| | | let mut debug_pcm_wav_path = None; |
| | | let mut debug_pcm_wav_size_bytes = None; |
| | | if format == "pcm_s16le" { |
| | | if let (Some(debug_dump_dir), Some(decoder)) = ( |
| | | bridge_config.audio_debug_dump_dir.as_deref(), |
| | | state.pcm_stream_decoder.as_ref(), |
| | | ) { |
| | | match decoder.write_debug_dump( |
| | | debug_dump_dir, |
| | | call_id, |
| | | &format!("stream-reply-{}", turn.turn_id), |
| | | ) { |
| | | Ok(debug_dump) => { |
| | | debug_source_path = debug_dump.debug_source_path; |
| | | debug_pcm_wav_path = debug_dump.debug_pcm_wav_path; |
| | | debug_pcm_wav_size_bytes = debug_dump.debug_pcm_wav_size_bytes; |
| | | } |
| | | Err(error) => warn!( |
| | | call_id = %call_id, |
| | | trace_id = %trace_id, |
| | | turn_id = %turn.turn_id, |
| | | error = %safe_error(&error.to_string()), |
| | | "runtime helper stream_audio_pcm_debug_dump_failed" |
| | | ), |
| | | } |
| | | } |
| | | } |
| | | emit_activity( |
| | | call_id, |
| | | trace_id, |
| | |
| | | "replyPlaybackMode": state.reply_playback_mode.as_str(), |
| | | "format": format.as_str(), |
| | | "chunkSeq": audio_chunk.chunk_seq, |
| | | "networkChunkCount": state.pcm_stream_network_chunk_count, |
| | | "debugSourcePath": debug_source_path, |
| | | "debugPcmWavPath": debug_pcm_wav_path, |
| | | "debugPcmWavSizeBytes": debug_pcm_wav_size_bytes, |
| | | "sourceSampleRate": audio_chunk.sample_rate, |
| | | "sourceChannels": audio_chunk.channels, |
| | | "targetAudioProfile": sink.profile.as_str(), |
| | | "targetSampleRate": sink.sample_rate_hz, |
| | | "targetChannels": sink.num_channels, |
| | | "replyTotalAfterVadEndMs": turn_pipeline_started_at.elapsed().as_millis() as u64, |
| | | }), |
| | | ); |
| | | } |
| | | Ok(frames.len()) |
| | | } |
| | | |
| | | fn pcm_s16le_payload_to_frames( |
| | | payload: &[u8], |
| | | sample_rate: u32, |
| | | channels: u32, |
| | | ) -> Result<Vec<PcmFrame>> { |
| | | if sample_rate != TARGET_SAMPLE_RATE_HZ || channels != u32::from(TARGET_NUM_CHANNELS) { |
| | | return Err(anyhow!( |
| | | "unsupported pcm_s16le stream format: sample_rate={}, channels={}", |
| | | sample_rate, |
| | | channels |
| | | )); |
| | | } |
| | | if payload.len() % 2 != 0 { |
| | | return Err(anyhow!("pcm_s16le payload has odd byte length")); |
| | | } |
| | | let samples: Vec<i16> = payload |
| | | .chunks_exact(2) |
| | | .map(|chunk| i16::from_le_bytes([chunk[0], chunk[1]])) |
| | | .collect(); |
| | | if samples.is_empty() { |
| | | return Ok(Vec::new()); |
| | | } |
| | | let samples_per_channel = (sample_rate / 50).max(1); |
| | | let frame_sample_count = samples_per_channel as usize * channels as usize; |
| | | let mut frames = Vec::new(); |
| | | for chunk in samples.chunks(frame_sample_count) { |
| | | let chunk_samples_per_channel = (chunk.len() / channels as usize) as u32; |
| | | if chunk_samples_per_channel == 0 { |
| | | continue; |
| | | } |
| | | frames.push(PcmFrame::new( |
| | | chunk.to_vec(), |
| | | sample_rate, |
| | | channels, |
| | | chunk_samples_per_channel, |
| | | )); |
| | | } |
| | | Ok(frames) |
| | | } |
| | | |
| | | fn trim_ascii_whitespace(value: &[u8]) -> &[u8] { |
| | |
| | | artifact_type: "local_file".to_string(), |
| | | path_ref: path_ref.to_string(), |
| | | format: "wav".to_string(), |
| | | sample_rate: TARGET_SAMPLE_RATE_HZ, |
| | | channels: u32::from(TARGET_NUM_CHANNELS), |
| | | sample_rate: USER_AUDIO_SAMPLE_RATE_HZ, |
| | | channels: u32::from(USER_AUDIO_NUM_CHANNELS), |
| | | duration_ms: turn.duration_ms, |
| | | byte_size, |
| | | }, |
| | |
| | | http, |
| | | Some(&audio_path_string), |
| | | None, |
| | | TARGET_SAMPLE_RATE_HZ, |
| | | TARGET_NUM_CHANNELS, |
| | | sink.sample_rate_hz, |
| | | sink.num_channels, |
| | | bridge_config.audio_debug_dump_dir.as_deref(), |
| | | call_id, |
| | | &reply_debug_label, |
| | |
| | | audio_chunk_count: u64, |
| | | device_output_count: u64, |
| | | encoded_audio_buffer: Vec<u8>, |
| | | pcm_stream_decoder: Option<audio::PcmS16leStreamDecoder>, |
| | | pcm_stream_network_chunk_count: u64, |
| | | } |
| | | |
| | | impl Default for RuntimeTurnStreamState { |
| | |
| | | audio_chunk_count: 0, |
| | | device_output_count: 0, |
| | | encoded_audio_buffer: Vec::new(), |
| | | pcm_stream_decoder: None, |
| | | pcm_stream_network_chunk_count: 0, |
| | | } |
| | | } |
| | | } |
| | |
| | | tokio::spawn(async move { |
| | | let mut stream = NativeAudioStream::new( |
| | | track.rtc_track(), |
| | | TARGET_SAMPLE_RATE_HZ as i32, |
| | | i32::from(TARGET_NUM_CHANNELS), |
| | | USER_AUDIO_SAMPLE_RATE_HZ as i32, |
| | | i32::from(USER_AUDIO_NUM_CHANNELS), |
| | | ); |
| | | let started_at = Instant::now(); |
| | | let mut frame_count: u64 = 0; |
| | |
| | | rms_threshold: f64_env("CV_VAD_RMS_THRESHOLD", 0.012), |
| | | peak_threshold: f64_env("CV_VAD_PEAK_THRESHOLD", 0.08), |
| | | start_frames: u32_env("CV_VAD_START_FRAMES", 5).max(1), |
| | | end_silence_ms: u64_env("CV_VAD_END_SILENCE_MS", 700).max(100), |
| | | min_speech_ms: u64_env("CV_VAD_MIN_SPEECH_MS", 300).max(1), |
| | | end_silence_ms: u64_env("CV_VAD_END_SILENCE_MS", 400).max(100), |
| | | min_speech_ms: u64_env("CV_VAD_MIN_SPEECH_MS", 250).max(1), |
| | | max_turn_ms: u64_env("CV_VAD_MAX_TURN_MS", 10_000).max(1_000), |
| | | initial_ignore_ms: u64_env("CV_VAD_INITIAL_IGNORE_MS", 500), |
| | | } |
| | |
| | | return None; |
| | | } |
| | | |
| | | let voiced = rms >= self.config.rms_threshold || peak >= self.config.peak_threshold; |
| | | // Align with cb-sdk's energy-based segmentation: peak is diagnostic only, |
| | | // otherwise isolated spikes can keep a turn open until max_turn_ms. |
| | | let voiced = rms >= self.config.rms_threshold; |
| | | if !self.in_speech { |
| | | self.remember_pre_speech_frame(frame); |
| | | } |
| | |
| | | rtc_source: NativeAudioSource, |
| | | track: LocalAudioTrack, |
| | | device_output_destination_identity: Option<String>, |
| | | profile: String, |
| | | sample_rate_hz: u32, |
| | | num_channels: u16, |
| | | } |
| | | |
| | | impl BotAudioOutputSink { |
| | |
| | | call_id: &str, |
| | | trace_id: &str, |
| | | track_name: &str, |
| | | profile: String, |
| | | sample_rate: u32, |
| | | num_channels: u32, |
| | | device_output_destination_identity: Option<String>, |
| | |
| | | "failed to publish bot audio track in room {room_alias} for participant {participant_alias}: {error}" |
| | | ) |
| | | })?; |
| | | let num_channels_u16 = u16::try_from(num_channels) |
| | | .map_err(|_| anyhow!("unsupported bot audio channel count {num_channels}"))?; |
| | | |
| | | info!( |
| | | room_alias = %room_alias, |
| | | participant_alias = %participant_alias, |
| | | track_name = %track_name, |
| | | bot_audio_profile = %profile, |
| | | sample_rate, |
| | | num_channels, |
| | | "runtime helper published bot audio track" |
| | |
| | | None, |
| | | json!({ |
| | | "trackName": track_name, |
| | | "audioProfile": profile, |
| | | "sampleRate": sample_rate, |
| | | "numChannels": num_channels, |
| | | }), |
| | |
| | | rtc_source, |
| | | track, |
| | | device_output_destination_identity, |
| | | profile, |
| | | sample_rate_hz: sample_rate, |
| | | num_channels: num_channels_u16, |
| | | }) |
| | | } |
| | | |