cai
2026-07-09 2bb392c03cdc76e27716f1d957cbb59016feaa4d
src/main.rs
@@ -34,8 +34,12 @@
use tokio::{sync::mpsc::UnboundedReceiver, task::JoinHandle};
use tracing::{info, warn};
const TARGET_SAMPLE_RATE_HZ: u32 = 48_000;
const TARGET_NUM_CHANNELS: u16 = 1;
const USER_AUDIO_SAMPLE_RATE_HZ: u32 = 48_000;
const USER_AUDIO_NUM_CHANNELS: u16 = 1;
const DEFAULT_BOT_AUDIO_PROFILE: &str = "pcm-16k";
const LIVEKIT_48K_SAMPLE_RATE_HZ: u32 = 48_000;
const PCM_16K_SAMPLE_RATE_HZ: u32 = 16_000;
const BOT_NUM_CHANNELS: u16 = 1;
const TRACK_NAME: &str = "bot-main-audio";
#[tokio::main(flavor = "multi_thread")]
@@ -58,8 +62,8 @@
        &http,
        config.greeting_audio_file.as_deref(),
        config.greeting_audio_url.as_deref(),
        TARGET_SAMPLE_RATE_HZ,
        TARGET_NUM_CHANNELS,
        config.bot_audio_profile.sample_rate_hz,
        config.bot_audio_profile.num_channels,
        config.audio_debug_dump_dir.as_deref(),
        &config.call_id,
        "greeting",
@@ -81,6 +85,9 @@
        room_alias = %redact(&config.room_id),
        participant_alias = %redact(&config.bot_participant_identity),
        greeting_source = %config.greeting_source,
        bot_audio_profile = %config.bot_audio_profile.profile,
        bot_sample_rate_hz = config.bot_audio_profile.sample_rate_hz,
        bot_num_channels = config.bot_audio_profile.num_channels,
        "combrabo voice runtime helper connected"
    );
    emit_activity(
@@ -129,8 +136,9 @@
        &config.call_id,
        &config.trace_id,
        TRACK_NAME,
        TARGET_SAMPLE_RATE_HZ,
        u32::from(TARGET_NUM_CHANNELS),
        config.bot_audio_profile.profile.clone(),
        config.bot_audio_profile.sample_rate_hz,
        u32::from(config.bot_audio_profile.num_channels),
        config.user_participant_identity.clone(),
    )
    .await?;
@@ -269,6 +277,54 @@
    simple_vad_gate_until_greeting_done: bool,
    simple_vad_post_greeting_delay_ms: u64,
    simple_vad_config: SimpleVadConfig,
    bot_audio_profile: BotAudioProfile,
}
#[derive(Clone)]
struct BotAudioProfile {
    profile: String,
    sample_rate_hz: u32,
    num_channels: u16,
}
impl BotAudioProfile {
    fn from_env() -> Result<Self> {
        let profile = env::var("CV_BOT_AUDIO_PROFILE")
            .unwrap_or_else(|_| DEFAULT_BOT_AUDIO_PROFILE.to_string())
            .trim()
            .to_ascii_lowercase();
        match profile.as_str() {
            "livekit-48k" | "48k" => Ok(Self {
                profile: "livekit-48k".to_string(),
                sample_rate_hz: LIVEKIT_48K_SAMPLE_RATE_HZ,
                num_channels: BOT_NUM_CHANNELS,
            }),
            "pcm-16k" | "16k" => Ok(Self {
                profile: "pcm-16k".to_string(),
                sample_rate_hz: PCM_16K_SAMPLE_RATE_HZ,
                num_channels: BOT_NUM_CHANNELS,
            }),
            "custom" => {
                let sample_rate_hz = u32_env("CV_BOT_SAMPLE_RATE_HZ", LIVEKIT_48K_SAMPLE_RATE_HZ);
                let num_channels = u16_env("CV_BOT_NUM_CHANNELS", BOT_NUM_CHANNELS);
                if sample_rate_hz == 0 {
                    return Err(anyhow!("CV_BOT_SAMPLE_RATE_HZ must be positive"));
                }
                if num_channels == 0 {
                    return Err(anyhow!("CV_BOT_NUM_CHANNELS must be positive"));
                }
                Ok(Self {
                    profile,
                    sample_rate_hz,
                    num_channels,
                })
            }
            _ => Err(anyhow!(
                "unsupported CV_BOT_AUDIO_PROFILE {}; expected livekit-48k, pcm-16k or custom",
                profile
            )),
        }
    }
}
#[derive(Clone)]
@@ -346,6 +402,7 @@
            simple_vad_gate_until_greeting_done: bool_env("CV_VAD_GATE_UNTIL_GREETING_DONE", true),
            simple_vad_post_greeting_delay_ms: u64_env("CV_VAD_POST_GREETING_DELAY_MS", 800),
            simple_vad_config: SimpleVadConfig::from_env(),
            bot_audio_profile: BotAudioProfile::from_env()?,
        })
    }
}
@@ -492,6 +549,13 @@
        .unwrap_or(default_value)
}
fn u16_env(key: &str, default_value: u16) -> u16 {
    env::var(key)
        .ok()
        .and_then(|value| value.trim().parse::<u16>().ok())
        .unwrap_or(default_value)
}
fn redact(value: &str) -> String {
    if value.len() <= 8 {
        return "redacted".to_string();
@@ -532,6 +596,7 @@
        "traceId": trace_id,
        "turnId": turn_id,
        "eventName": event_name,
        "eventWallTimeMs": current_time_millis(),
        "result": result,
        "reasonCode": reason_code,
        "retryable": retryable,
@@ -930,8 +995,8 @@
    audio::write_pcm_wav(
        &output_path,
        &turn.samples,
        TARGET_SAMPLE_RATE_HZ,
        TARGET_NUM_CHANNELS,
        USER_AUDIO_SAMPLE_RATE_HZ,
        USER_AUDIO_NUM_CHANNELS,
    )?;
    let byte_size = fs::metadata(&output_path)
        .context("failed to stat turn artifact")?
@@ -959,8 +1024,8 @@
            artifact_type: "local_file".to_string(),
            path_ref: path_ref.to_string(),
            format: "wav".to_string(),
            sample_rate: TARGET_SAMPLE_RATE_HZ,
            channels: u32::from(TARGET_NUM_CHANNELS),
            sample_rate: USER_AUDIO_SAMPLE_RATE_HZ,
            channels: u32::from(USER_AUDIO_NUM_CHANNELS),
            duration_ms: turn.duration_ms,
            byte_size,
        },
@@ -1343,16 +1408,14 @@
        .trim()
        .to_ascii_lowercase();
    let frames = if format == "pcm_s16le" {
        let sample_rate = audio_chunk.sample_rate.unwrap_or(TARGET_SAMPLE_RATE_HZ);
        let channels = audio_chunk
            .channels
            .unwrap_or(u32::from(TARGET_NUM_CHANNELS));
        let sample_rate = audio_chunk.sample_rate.unwrap_or(sink.sample_rate_hz);
        let channels = audio_chunk.channels.unwrap_or(u32::from(sink.num_channels));
        if state.pcm_stream_decoder.is_none() {
            state.pcm_stream_decoder = Some(audio::PcmS16leStreamDecoder::new(
                sample_rate,
                channels,
                TARGET_SAMPLE_RATE_HZ,
                TARGET_NUM_CHANNELS,
                sink.sample_rate_hz,
                sink.num_channels,
            )?);
        }
        state.pcm_stream_network_chunk_count =
@@ -1396,8 +1459,8 @@
        match audio::decode_audio_bytes_to_frames(
            &state.encoded_audio_buffer,
            "stream_chunk",
            TARGET_SAMPLE_RATE_HZ,
            TARGET_NUM_CHANNELS,
            sink.sample_rate_hz,
            sink.num_channels,
            bridge_config.audio_debug_dump_dir.as_deref(),
            call_id,
            &format!("stream-reply-{}", turn.turn_id),
@@ -1526,8 +1589,11 @@
                "debugSourcePath": debug_source_path,
                "debugPcmWavPath": debug_pcm_wav_path,
                "debugPcmWavSizeBytes": debug_pcm_wav_size_bytes,
                "sampleRate": audio_chunk.sample_rate,
                "channels": audio_chunk.channels,
                "sourceSampleRate": audio_chunk.sample_rate,
                "sourceChannels": audio_chunk.channels,
                "targetAudioProfile": sink.profile.as_str(),
                "targetSampleRate": sink.sample_rate_hz,
                "targetChannels": sink.num_channels,
                "replyTotalAfterVadEndMs": turn_pipeline_started_at.elapsed().as_millis() as u64,
            }),
        );
@@ -1566,8 +1632,8 @@
            artifact_type: "local_file".to_string(),
            path_ref: path_ref.to_string(),
            format: "wav".to_string(),
            sample_rate: TARGET_SAMPLE_RATE_HZ,
            channels: u32::from(TARGET_NUM_CHANNELS),
            sample_rate: USER_AUDIO_SAMPLE_RATE_HZ,
            channels: u32::from(USER_AUDIO_NUM_CHANNELS),
            duration_ms: turn.duration_ms,
            byte_size,
        },
@@ -1730,8 +1796,8 @@
        http,
        Some(&audio_path_string),
        None,
        TARGET_SAMPLE_RATE_HZ,
        TARGET_NUM_CHANNELS,
        sink.sample_rate_hz,
        sink.num_channels,
        bridge_config.audio_debug_dump_dir.as_deref(),
        call_id,
        &reply_debug_label,
@@ -2075,8 +2141,8 @@
    tokio::spawn(async move {
        let mut stream = NativeAudioStream::new(
            track.rtc_track(),
            TARGET_SAMPLE_RATE_HZ as i32,
            i32::from(TARGET_NUM_CHANNELS),
            USER_AUDIO_SAMPLE_RATE_HZ as i32,
            i32::from(USER_AUDIO_NUM_CHANNELS),
        );
        let started_at = Instant::now();
        let mut frame_count: u64 = 0;
@@ -2193,8 +2259,8 @@
            rms_threshold: f64_env("CV_VAD_RMS_THRESHOLD", 0.012),
            peak_threshold: f64_env("CV_VAD_PEAK_THRESHOLD", 0.08),
            start_frames: u32_env("CV_VAD_START_FRAMES", 5).max(1),
            end_silence_ms: u64_env("CV_VAD_END_SILENCE_MS", 700).max(100),
            min_speech_ms: u64_env("CV_VAD_MIN_SPEECH_MS", 300).max(1),
            end_silence_ms: u64_env("CV_VAD_END_SILENCE_MS", 400).max(100),
            min_speech_ms: u64_env("CV_VAD_MIN_SPEECH_MS", 250).max(1),
            max_turn_ms: u64_env("CV_VAD_MAX_TURN_MS", 10_000).max(1_000),
            initial_ignore_ms: u64_env("CV_VAD_INITIAL_IGNORE_MS", 500),
        }
@@ -2307,7 +2373,9 @@
            return None;
        }
        let voiced = rms >= self.config.rms_threshold || peak >= self.config.peak_threshold;
        // Align with cb-sdk's energy-based segmentation: peak is diagnostic only,
        // otherwise isolated spikes can keep a turn open until max_turn_ms.
        let voiced = rms >= self.config.rms_threshold;
        if !self.in_speech {
            self.remember_pre_speech_frame(frame);
        }
@@ -2563,6 +2631,9 @@
    rtc_source: NativeAudioSource,
    track: LocalAudioTrack,
    device_output_destination_identity: Option<String>,
    profile: String,
    sample_rate_hz: u32,
    num_channels: u16,
}
impl BotAudioOutputSink {
@@ -2573,6 +2644,7 @@
        call_id: &str,
        trace_id: &str,
        track_name: &str,
        profile: String,
        sample_rate: u32,
        num_channels: u32,
        device_output_destination_identity: Option<String>,
@@ -2601,11 +2673,14 @@
                    "failed to publish bot audio track in room {room_alias} for participant {participant_alias}: {error}"
                )
            })?;
        let num_channels_u16 = u16::try_from(num_channels)
            .map_err(|_| anyhow!("unsupported bot audio channel count {num_channels}"))?;
        info!(
            room_alias = %room_alias,
            participant_alias = %participant_alias,
            track_name = %track_name,
            bot_audio_profile = %profile,
            sample_rate,
            num_channels,
            "runtime helper published bot audio track"
@@ -2620,6 +2695,7 @@
            None,
            json!({
                "trackName": track_name,
                "audioProfile": profile,
                "sampleRate": sample_rate,
                "numChannels": num_channels,
            }),
@@ -2630,6 +2706,9 @@
            rtc_source,
            track,
            device_output_destination_identity,
            profile,
            sample_rate_hz: sample_rate,
            num_channels: num_channels_u16,
        })
    }