From 3c5a17954a45d279295059da078aefcf956568de Mon Sep 17 00:00:00 2001
From: cai <cai@nbcai.cc>
Date: Sat, 08 Aug 2026 19:07:38 +0800
Subject: [PATCH] chore: bind final observer test candidate
---
src/main.rs | 257 +++++++++++++++++++++++++++++++++++++++++++++-----
1 files changed, 228 insertions(+), 29 deletions(-)
diff --git a/src/main.rs b/src/main.rs
index d357e4a..4ab9a13 100644
--- a/src/main.rs
+++ b/src/main.rs
@@ -737,6 +737,10 @@
})
}
+fn is_bound_user_participant(identity: &str, expected: Option<&str>) -> bool {
+ expected.is_none_or(|value| identity == value)
+}
+
async fn observe_user_audio_events(
mut events: UnboundedReceiver<RoomEvent>,
call_id: String,
@@ -771,10 +775,10 @@
publication: _,
participant,
} => {
- if user_participant_identity
- .as_deref()
- .is_some_and(|expected| participant.identity().to_string() != expected)
- {
+ if !is_bound_user_participant(
+ &participant.identity().to_string(),
+ user_participant_identity.as_deref(),
+ ) {
warn!(call_id = %call_id, trace_id = %trace_id,
metadata_status = "wrong_participant",
"runtime helper ignored non-user audio participant");
@@ -3042,36 +3046,16 @@
let is_in_speech = vad.in_speech;
if !was_in_speech && is_in_speech {
- let turn_id = format!("turn-{:04}", vad.turn_index);
- match RealtimeAsrUpload::start_with_participant_attributes(
+ start_realtime_session_for_new_speech(
http.clone(),
turn_bridge_config.realtime_asr_config(),
&call_id,
&trace_id,
- &turn_id,
- &vad.speech_samples,
+ vad,
|| participant.attributes(),
- ) {
- Ok(upload) => {
- info!(
- call_id = %call_id,
- trace_id = %trace_id,
- turn_id = %turn_id,
- "runtime helper asr_realtime_session_started"
- );
- realtime_asr_upload = Some(upload);
- }
- Err(error) if turn_bridge_config.asr_realtime_enabled => {
- warn!(
- call_id = %call_id,
- trace_id = %trace_id,
- turn_id = %turn_id,
- error = %safe_error(&error.to_string()),
- "runtime helper asr_realtime_start_failed_fallback"
- );
- }
- Err(_) => {}
- }
+ &mut realtime_asr_upload,
+ turn_bridge_config.asr_realtime_enabled,
+ );
} else if was_in_speech {
let push_failed = realtime_asr_upload
.as_mut()
@@ -3189,6 +3173,40 @@
"runtime helper user_audio_stream_ended"
);
})
+}
+
+fn start_realtime_session_for_new_speech(
+ http: Client,
+ config: RealtimeAsrConfig,
+ call_id: &str,
+ trace_id: &str,
+ vad: &SimpleVad,
+ read_attributes: impl FnOnce() -> std::collections::HashMap<String, String>,
+ upload_slot: &mut Option<RealtimeAsrUpload>,
+ realtime_enabled: bool,
+) {
+ let turn_id = format!("turn-{:04}", vad.turn_index);
+ match RealtimeAsrUpload::start_with_participant_attributes(
+ http,
+ config,
+ call_id,
+ trace_id,
+ &turn_id,
+ &vad.speech_samples,
+ read_attributes,
+ ) {
+ Ok(upload) => {
+ info!(call_id = %call_id, trace_id = %trace_id, turn_id = %turn_id,
+ "runtime helper asr_realtime_session_started");
+ *upload_slot = Some(upload);
+ }
+ Err(error) if realtime_enabled => {
+ warn!(call_id = %call_id, trace_id = %trace_id, turn_id = %turn_id,
+ error = %safe_error(&error.to_string()),
+ "runtime helper asr_realtime_start_failed_fallback");
+ }
+ Err(_) => {}
+ }
}
struct DrainedUserAudioFrame {
@@ -3951,6 +3969,187 @@
use std::collections::HashSet;
#[test]
+ fn production_vad_session_boundary_reads_updated_attributes() {
+ let config = SimpleVadConfig {
+ rms_threshold: 0.001,
+ peak_threshold: 0.01,
+ start_frames: 2,
+ end_silence_ms: 100,
+ min_speech_ms: 1,
+ max_turn_ms: 1_000,
+ initial_ignore_ms: 0,
+ };
+ let mut vad = SimpleVad::new(config);
+ let samples = vec![1_000i16; 160];
+ let frame = AudioFrame {
+ data: samples.as_slice().into(),
+ sample_rate: 16_000,
+ num_channels: 1,
+ samples_per_channel: 160,
+ };
+ let mut attributes = std::collections::HashMap::from([
+ (
+ "inputSourceCategory".to_string(),
+ "controlled_fixture".to_string(),
+ ),
+ (
+ "clientFixtureSequence".to_string(),
+ "fixture-01".to_string(),
+ ),
+ ]);
+ let mut starts = Vec::new();
+ for (session_index, sequence) in [(1, "fixture-01"), (2, "fixture-02")] {
+ let was_in_speech = vad.in_speech;
+ vad.observe_frame(
+ "call-001",
+ "trace-001",
+ "participant",
+ "track",
+ session_index * 2 - 1,
+ 1_000 * session_index,
+ &frame,
+ );
+ vad.observe_frame(
+ "call-001",
+ "trace-001",
+ "participant",
+ "track",
+ session_index * 2,
+ 1_000 * session_index + 10,
+ &frame,
+ );
+ let is_in_speech = vad.in_speech;
+ assert!(!was_in_speech && is_in_speech);
+ attributes.insert("clientFixtureSequence".to_string(), sequence.to_string());
+ let metadata = AudioIngressMetadata::from_participant(&attributes)
+ .expect("valid participant attributes")
+ .expect("controlled fixture metadata");
+ let session_line = asr_realtime::session_start_line(
+ "call-001",
+ "trace-001",
+ &format!("turn-{session_index:04}"),
+ "nonce-001",
+ Some(&metadata),
+ )
+ .expect("session start line");
+ let session_json: serde_json::Value =
+ serde_json::from_slice(&session_line).expect("session start json");
+ assert_eq!(sequence, session_json["clientFixtureSequence"]);
+ starts.push(metadata.client_fixture_sequence);
+ vad.reset_current_turn();
+ }
+ assert_eq!(vec!["fixture-01", "fixture-02"], starts);
+ attributes.insert("inputSourceCategory".to_string(), "other".to_string());
+ assert!(AudioIngressMetadata::from_participant(&attributes).is_err());
+ assert!(
+ AudioIngressMetadata::from_participant(&std::collections::HashMap::new())
+ .expect("missing attributes is absent")
+ .is_none()
+ );
+ attributes.insert(
+ "clientFixtureSequence".to_string(),
+ "fixture-01".to_string(),
+ );
+ assert!(AudioIngressMetadata::from_participant(&attributes).is_err());
+ }
+
+ #[test]
+ fn production_observer_rejects_wrong_participant_before_vad_session() {
+ assert!(!is_bound_user_participant(
+ "participant-other",
+ Some("participant-user")
+ ));
+ assert!(is_bound_user_participant(
+ "participant-user",
+ Some("participant-user")
+ ));
+ assert!(is_bound_user_participant("participant-any", None));
+ }
+
+ #[tokio::test]
+ async fn production_observer_vad_to_session_entry_reads_each_updated_attribute() {
+ let mut vad = SimpleVad::new(SimpleVadConfig {
+ rms_threshold: 0.001,
+ peak_threshold: 0.01,
+ start_frames: 1,
+ end_silence_ms: 100,
+ min_speech_ms: 1,
+ max_turn_ms: 1_000,
+ initial_ignore_ms: 0,
+ });
+ let frame_data = vec![1_000i16; 160];
+ let frame = AudioFrame {
+ data: frame_data.as_slice().into(),
+ sample_rate: 16_000,
+ num_channels: 1,
+ samples_per_channel: 160,
+ };
+ let mut attrs = std::collections::HashMap::from([
+ (
+ "inputSourceCategory".to_string(),
+ "controlled_fixture".to_string(),
+ ),
+ (
+ "clientFixtureSequence".to_string(),
+ "fixture-01".to_string(),
+ ),
+ ]);
+ let mut upload = None;
+ let was = vad.in_speech;
+ vad.observe_frame(
+ "call-001",
+ "trace-001",
+ "participant",
+ "track",
+ 1,
+ 1_000,
+ &frame,
+ );
+ assert!(!was && vad.in_speech);
+ start_realtime_session_for_new_speech(
+ Client::new(),
+ RealtimeAsrConfig {
+ enabled: true,
+ url: Some("http://127.0.0.1:9".to_string()),
+ runtime_token: Some("test".to_string()),
+ runtime_session_nonce: Some("test".to_string()),
+ chunk_duration_ms: 200,
+ },
+ "call-001",
+ "trace-001",
+ &vad,
+ || attrs.clone(),
+ &mut upload,
+ true,
+ );
+ assert!(upload.is_some());
+ upload.take().unwrap().cancel("test").await;
+ vad.reset_current_turn();
+ attrs.insert(
+ "clientFixtureSequence".to_string(),
+ "fixture-02".to_string(),
+ );
+ let was = vad.in_speech;
+ vad.observe_frame(
+ "call-001",
+ "trace-001",
+ "participant",
+ "track",
+ 2,
+ 2_000,
+ &frame,
+ );
+ assert!(!was && vad.in_speech);
+ assert_eq!(
+ "fixture-02",
+ AudioIngressMetadata::from_participant(&attrs)
+ .expect("valid attributes")
+ .expect("bound")
+ .client_fixture_sequence
+ );
+ }
+
+ #[test]
fn reply_chunk_marker_state_emits_turn_first_once_and_later_segment_first_once() {
let mut state = ReplyChunkMarkerState::default();
--
Gitblit v1.9.3