cai
2026-07-09 57753e1b1e169369bdc83f3945ce5246c844b316
src/main.rs
@@ -12,7 +12,7 @@
};
use anyhow::{Context, Result, anyhow};
use audio::{AudioDiagnostics, PcmFrame, load_pre_recorded_frames};
use audio::{AudioDiagnostics, load_pre_recorded_frames};
use base64::{Engine as _, engine::general_purpose};
use futures_util::StreamExt;
use libwebrtc::{
@@ -34,8 +34,12 @@
use tokio::{sync::mpsc::UnboundedReceiver, task::JoinHandle};
use tracing::{info, warn};
const TARGET_SAMPLE_RATE_HZ: u32 = 48_000;
const TARGET_NUM_CHANNELS: u16 = 1;
const USER_AUDIO_SAMPLE_RATE_HZ: u32 = 48_000;
const USER_AUDIO_NUM_CHANNELS: u16 = 1;
const DEFAULT_BOT_AUDIO_PROFILE: &str = "pcm-16k";
const LIVEKIT_48K_SAMPLE_RATE_HZ: u32 = 48_000;
const PCM_16K_SAMPLE_RATE_HZ: u32 = 16_000;
const BOT_NUM_CHANNELS: u16 = 1;
const TRACK_NAME: &str = "bot-main-audio";
#[tokio::main(flavor = "multi_thread")]
@@ -58,8 +62,8 @@
        &http,
        config.greeting_audio_file.as_deref(),
        config.greeting_audio_url.as_deref(),
        TARGET_SAMPLE_RATE_HZ,
        TARGET_NUM_CHANNELS,
        config.bot_audio_profile.sample_rate_hz,
        config.bot_audio_profile.num_channels,
        config.audio_debug_dump_dir.as_deref(),
        &config.call_id,
        "greeting",
@@ -81,6 +85,9 @@
        room_alias = %redact(&config.room_id),
        participant_alias = %redact(&config.bot_participant_identity),
        greeting_source = %config.greeting_source,
        bot_audio_profile = %config.bot_audio_profile.profile,
        bot_sample_rate_hz = config.bot_audio_profile.sample_rate_hz,
        bot_num_channels = config.bot_audio_profile.num_channels,
        "combrabo voice runtime helper connected"
    );
    emit_activity(
@@ -129,8 +136,9 @@
        &config.call_id,
        &config.trace_id,
        TRACK_NAME,
        TARGET_SAMPLE_RATE_HZ,
        u32::from(TARGET_NUM_CHANNELS),
        config.bot_audio_profile.profile.clone(),
        config.bot_audio_profile.sample_rate_hz,
        u32::from(config.bot_audio_profile.num_channels),
        config.user_participant_identity.clone(),
    )
    .await?;
@@ -269,6 +277,54 @@
    simple_vad_gate_until_greeting_done: bool,
    simple_vad_post_greeting_delay_ms: u64,
    simple_vad_config: SimpleVadConfig,
    bot_audio_profile: BotAudioProfile,
}
#[derive(Clone)]
struct BotAudioProfile {
    profile: String,
    sample_rate_hz: u32,
    num_channels: u16,
}
impl BotAudioProfile {
    fn from_env() -> Result<Self> {
        let profile = env::var("CV_BOT_AUDIO_PROFILE")
            .unwrap_or_else(|_| DEFAULT_BOT_AUDIO_PROFILE.to_string())
            .trim()
            .to_ascii_lowercase();
        match profile.as_str() {
            "livekit-48k" | "48k" => Ok(Self {
                profile: "livekit-48k".to_string(),
                sample_rate_hz: LIVEKIT_48K_SAMPLE_RATE_HZ,
                num_channels: BOT_NUM_CHANNELS,
            }),
            "pcm-16k" | "16k" => Ok(Self {
                profile: "pcm-16k".to_string(),
                sample_rate_hz: PCM_16K_SAMPLE_RATE_HZ,
                num_channels: BOT_NUM_CHANNELS,
            }),
            "custom" => {
                let sample_rate_hz = u32_env("CV_BOT_SAMPLE_RATE_HZ", LIVEKIT_48K_SAMPLE_RATE_HZ);
                let num_channels = u16_env("CV_BOT_NUM_CHANNELS", BOT_NUM_CHANNELS);
                if sample_rate_hz == 0 {
                    return Err(anyhow!("CV_BOT_SAMPLE_RATE_HZ must be positive"));
                }
                if num_channels == 0 {
                    return Err(anyhow!("CV_BOT_NUM_CHANNELS must be positive"));
                }
                Ok(Self {
                    profile,
                    sample_rate_hz,
                    num_channels,
                })
            }
            _ => Err(anyhow!(
                "unsupported CV_BOT_AUDIO_PROFILE {}; expected livekit-48k, pcm-16k or custom",
                profile
            )),
        }
    }
}
#[derive(Clone)]
@@ -346,6 +402,7 @@
            simple_vad_gate_until_greeting_done: bool_env("CV_VAD_GATE_UNTIL_GREETING_DONE", true),
            simple_vad_post_greeting_delay_ms: u64_env("CV_VAD_POST_GREETING_DELAY_MS", 800),
            simple_vad_config: SimpleVadConfig::from_env(),
            bot_audio_profile: BotAudioProfile::from_env()?,
        })
    }
}
@@ -489,6 +546,13 @@
    env::var(key)
        .ok()
        .and_then(|value| value.trim().parse::<u32>().ok())
        .unwrap_or(default_value)
}
fn u16_env(key: &str, default_value: u16) -> u16 {
    env::var(key)
        .ok()
        .and_then(|value| value.trim().parse::<u16>().ok())
        .unwrap_or(default_value)
}
@@ -930,8 +994,8 @@
    audio::write_pcm_wav(
        &output_path,
        &turn.samples,
        TARGET_SAMPLE_RATE_HZ,
        TARGET_NUM_CHANNELS,
        USER_AUDIO_SAMPLE_RATE_HZ,
        USER_AUDIO_NUM_CHANNELS,
    )?;
    let byte_size = fs::metadata(&output_path)
        .context("failed to stat turn artifact")?
@@ -959,8 +1023,8 @@
            artifact_type: "local_file".to_string(),
            path_ref: path_ref.to_string(),
            format: "wav".to_string(),
            sample_rate: TARGET_SAMPLE_RATE_HZ,
            channels: u32::from(TARGET_NUM_CHANNELS),
            sample_rate: USER_AUDIO_SAMPLE_RATE_HZ,
            channels: u32::from(USER_AUDIO_NUM_CHANNELS),
            duration_ms: turn.duration_ms,
            byte_size,
        },
@@ -1343,20 +1407,59 @@
        .trim()
        .to_ascii_lowercase();
    let frames = if format == "pcm_s16le" {
        pcm_s16le_payload_to_frames(
            &payload,
            audio_chunk.sample_rate.unwrap_or(TARGET_SAMPLE_RATE_HZ),
            audio_chunk
                .channels
                .unwrap_or(u32::from(TARGET_NUM_CHANNELS)),
        )?
        let sample_rate = audio_chunk.sample_rate.unwrap_or(sink.sample_rate_hz);
        let channels = audio_chunk.channels.unwrap_or(u32::from(sink.num_channels));
        if state.pcm_stream_decoder.is_none() {
            state.pcm_stream_decoder = Some(audio::PcmS16leStreamDecoder::new(
                sample_rate,
                channels,
                sink.sample_rate_hz,
                sink.num_channels,
            )?);
        }
        state.pcm_stream_network_chunk_count =
            state.pcm_stream_network_chunk_count.saturating_add(1);
        let stream_result = state
            .pcm_stream_decoder
            .as_mut()
            .expect("pcm stream decoder initialized")
            .push_bytes(
                &payload,
                sample_rate,
                channels,
                audio_chunk.last.unwrap_or(false),
            )?;
        if stream_result.dropped_tail_bytes > 0 {
            warn!(
                call_id = %call_id,
                trace_id = %trace_id,
                turn_id = %turn.turn_id,
                chunk_seq = audio_chunk.chunk_seq.unwrap_or_default(),
                dropped_tail_bytes = stream_result.dropped_tail_bytes,
                "runtime helper stream_audio_pcm_unaligned_tail_dropped"
            );
        }
        if stream_result.frames.is_empty() {
            warn!(
                call_id = %call_id,
                trace_id = %trace_id,
                turn_id = %turn.turn_id,
                chunk_seq = audio_chunk.chunk_seq.unwrap_or_default(),
                buffered_source_bytes = stream_result.buffered_source_bytes,
                buffered_source_samples = stream_result.buffered_source_samples,
                network_chunk_count = state.pcm_stream_network_chunk_count,
                "runtime helper stream_audio_pcm_waiting_for_20ms_frame"
            );
            return Ok(0);
        }
        stream_result.frames
    } else if matches!(format.as_str(), "mp3" | "mpeg" | "wav") {
        state.encoded_audio_buffer.extend_from_slice(&payload);
        match audio::decode_audio_bytes_to_frames(
            &state.encoded_audio_buffer,
            "stream_chunk",
            TARGET_SAMPLE_RATE_HZ,
            TARGET_NUM_CHANNELS,
            sink.sample_rate_hz,
            sink.num_channels,
            bridge_config.audio_debug_dump_dir.as_deref(),
            call_id,
            &format!("stream-reply-{}", turn.turn_id),
@@ -1375,6 +1478,20 @@
                    error = %safe_error(&error.to_string()),
                    "runtime helper stream_audio_chunk_decode_waiting_for_more_data"
                );
                return Ok(0);
            }
            Err(error) if state.first_audio_frame_written => {
                warn!(
                    call_id = %call_id,
                    trace_id = %trace_id,
                    turn_id = %turn.turn_id,
                    chunk_seq = audio_chunk.chunk_seq.unwrap_or_default(),
                    format = %format,
                    buffered_bytes = state.encoded_audio_buffer.len(),
                    error = %safe_error(&error.to_string()),
                    "runtime helper stream_audio_final_chunk_decode_ignored"
                );
                state.encoded_audio_buffer.clear();
                return Ok(0);
            }
            Err(error) => return Err(error).context("failed to decode final stream audio chunk"),
@@ -1427,6 +1544,34 @@
        sleep_until(pacing_started_at + Duration::from_millis(((index + 1) as u64) * 20)).await;
    }
    if audio_chunk.last.unwrap_or(false) {
        let mut debug_source_path = None;
        let mut debug_pcm_wav_path = None;
        let mut debug_pcm_wav_size_bytes = None;
        if format == "pcm_s16le" {
            if let (Some(debug_dump_dir), Some(decoder)) = (
                bridge_config.audio_debug_dump_dir.as_deref(),
                state.pcm_stream_decoder.as_ref(),
            ) {
                match decoder.write_debug_dump(
                    debug_dump_dir,
                    call_id,
                    &format!("stream-reply-{}", turn.turn_id),
                ) {
                    Ok(debug_dump) => {
                        debug_source_path = debug_dump.debug_source_path;
                        debug_pcm_wav_path = debug_dump.debug_pcm_wav_path;
                        debug_pcm_wav_size_bytes = debug_dump.debug_pcm_wav_size_bytes;
                    }
                    Err(error) => warn!(
                        call_id = %call_id,
                        trace_id = %trace_id,
                        turn_id = %turn.turn_id,
                        error = %safe_error(&error.to_string()),
                        "runtime helper stream_audio_pcm_debug_dump_failed"
                    ),
                }
            }
        }
        emit_activity(
            call_id,
            trace_id,
@@ -1439,51 +1584,20 @@
                "replyPlaybackMode": state.reply_playback_mode.as_str(),
                "format": format.as_str(),
                "chunkSeq": audio_chunk.chunk_seq,
                "networkChunkCount": state.pcm_stream_network_chunk_count,
                "debugSourcePath": debug_source_path,
                "debugPcmWavPath": debug_pcm_wav_path,
                "debugPcmWavSizeBytes": debug_pcm_wav_size_bytes,
                "sourceSampleRate": audio_chunk.sample_rate,
                "sourceChannels": audio_chunk.channels,
                "targetAudioProfile": sink.profile.as_str(),
                "targetSampleRate": sink.sample_rate_hz,
                "targetChannels": sink.num_channels,
                "replyTotalAfterVadEndMs": turn_pipeline_started_at.elapsed().as_millis() as u64,
            }),
        );
    }
    Ok(frames.len())
}
fn pcm_s16le_payload_to_frames(
    payload: &[u8],
    sample_rate: u32,
    channels: u32,
) -> Result<Vec<PcmFrame>> {
    if sample_rate != TARGET_SAMPLE_RATE_HZ || channels != u32::from(TARGET_NUM_CHANNELS) {
        return Err(anyhow!(
            "unsupported pcm_s16le stream format: sample_rate={}, channels={}",
            sample_rate,
            channels
        ));
    }
    if payload.len() % 2 != 0 {
        return Err(anyhow!("pcm_s16le payload has odd byte length"));
    }
    let samples: Vec<i16> = payload
        .chunks_exact(2)
        .map(|chunk| i16::from_le_bytes([chunk[0], chunk[1]]))
        .collect();
    if samples.is_empty() {
        return Ok(Vec::new());
    }
    let samples_per_channel = (sample_rate / 50).max(1);
    let frame_sample_count = samples_per_channel as usize * channels as usize;
    let mut frames = Vec::new();
    for chunk in samples.chunks(frame_sample_count) {
        let chunk_samples_per_channel = (chunk.len() / channels as usize) as u32;
        if chunk_samples_per_channel == 0 {
            continue;
        }
        frames.push(PcmFrame::new(
            chunk.to_vec(),
            sample_rate,
            channels,
            chunk_samples_per_channel,
        ));
    }
    Ok(frames)
}
fn trim_ascii_whitespace(value: &[u8]) -> &[u8] {
@@ -1517,8 +1631,8 @@
            artifact_type: "local_file".to_string(),
            path_ref: path_ref.to_string(),
            format: "wav".to_string(),
            sample_rate: TARGET_SAMPLE_RATE_HZ,
            channels: u32::from(TARGET_NUM_CHANNELS),
            sample_rate: USER_AUDIO_SAMPLE_RATE_HZ,
            channels: u32::from(USER_AUDIO_NUM_CHANNELS),
            duration_ms: turn.duration_ms,
            byte_size,
        },
@@ -1681,8 +1795,8 @@
        http,
        Some(&audio_path_string),
        None,
        TARGET_SAMPLE_RATE_HZ,
        TARGET_NUM_CHANNELS,
        sink.sample_rate_hz,
        sink.num_channels,
        bridge_config.audio_debug_dump_dir.as_deref(),
        call_id,
        &reply_debug_label,
@@ -1861,6 +1975,8 @@
    audio_chunk_count: u64,
    device_output_count: u64,
    encoded_audio_buffer: Vec<u8>,
    pcm_stream_decoder: Option<audio::PcmS16leStreamDecoder>,
    pcm_stream_network_chunk_count: u64,
}
impl Default for RuntimeTurnStreamState {
@@ -1874,6 +1990,8 @@
            audio_chunk_count: 0,
            device_output_count: 0,
            encoded_audio_buffer: Vec::new(),
            pcm_stream_decoder: None,
            pcm_stream_network_chunk_count: 0,
        }
    }
}
@@ -2022,8 +2140,8 @@
    tokio::spawn(async move {
        let mut stream = NativeAudioStream::new(
            track.rtc_track(),
            TARGET_SAMPLE_RATE_HZ as i32,
            i32::from(TARGET_NUM_CHANNELS),
            USER_AUDIO_SAMPLE_RATE_HZ as i32,
            i32::from(USER_AUDIO_NUM_CHANNELS),
        );
        let started_at = Instant::now();
        let mut frame_count: u64 = 0;
@@ -2140,8 +2258,8 @@
            rms_threshold: f64_env("CV_VAD_RMS_THRESHOLD", 0.012),
            peak_threshold: f64_env("CV_VAD_PEAK_THRESHOLD", 0.08),
            start_frames: u32_env("CV_VAD_START_FRAMES", 5).max(1),
            end_silence_ms: u64_env("CV_VAD_END_SILENCE_MS", 700).max(100),
            min_speech_ms: u64_env("CV_VAD_MIN_SPEECH_MS", 300).max(1),
            end_silence_ms: u64_env("CV_VAD_END_SILENCE_MS", 400).max(100),
            min_speech_ms: u64_env("CV_VAD_MIN_SPEECH_MS", 250).max(1),
            max_turn_ms: u64_env("CV_VAD_MAX_TURN_MS", 10_000).max(1_000),
            initial_ignore_ms: u64_env("CV_VAD_INITIAL_IGNORE_MS", 500),
        }
@@ -2254,7 +2372,9 @@
            return None;
        }
        let voiced = rms >= self.config.rms_threshold || peak >= self.config.peak_threshold;
        // Align with cb-sdk's energy-based segmentation: peak is diagnostic only,
        // otherwise isolated spikes can keep a turn open until max_turn_ms.
        let voiced = rms >= self.config.rms_threshold;
        if !self.in_speech {
            self.remember_pre_speech_frame(frame);
        }
@@ -2510,6 +2630,9 @@
    rtc_source: NativeAudioSource,
    track: LocalAudioTrack,
    device_output_destination_identity: Option<String>,
    profile: String,
    sample_rate_hz: u32,
    num_channels: u16,
}
impl BotAudioOutputSink {
@@ -2520,6 +2643,7 @@
        call_id: &str,
        trace_id: &str,
        track_name: &str,
        profile: String,
        sample_rate: u32,
        num_channels: u32,
        device_output_destination_identity: Option<String>,
@@ -2548,11 +2672,14 @@
                    "failed to publish bot audio track in room {room_alias} for participant {participant_alias}: {error}"
                )
            })?;
        let num_channels_u16 = u16::try_from(num_channels)
            .map_err(|_| anyhow!("unsupported bot audio channel count {num_channels}"))?;
        info!(
            room_alias = %room_alias,
            participant_alias = %participant_alias,
            track_name = %track_name,
            bot_audio_profile = %profile,
            sample_rate,
            num_channels,
            "runtime helper published bot audio track"
@@ -2567,6 +2694,7 @@
            None,
            json!({
                "trackName": track_name,
                "audioProfile": profile,
                "sampleRate": sample_rate,
                "numChannels": num_channels,
            }),
@@ -2577,6 +2705,9 @@
            rtc_source,
            track,
            device_output_destination_identity,
            profile,
            sample_rate_hz: sample_rate,
            num_channels: num_channels_u16,
        })
    }