@vellumai/assistant 0.11.3-staging.2 → 0.11.3-staging.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. package/package.json +1 -1
  2. package/src/__tests__/call-controller.test.ts +120 -0
  3. package/src/__tests__/config-loader-backfill.test.ts +19 -0
  4. package/src/__tests__/config-schema.test.ts +142 -5
  5. package/src/__tests__/events-dev-bypass-actor.test.ts +112 -1
  6. package/src/__tests__/media-stream-output.test.ts +175 -0
  7. package/src/__tests__/media-stream-stt-session.test.ts +67 -0
  8. package/src/calls/__tests__/tts-text-sanitizer.test.ts +13 -0
  9. package/src/calls/__tests__/voice-session-bridge.test.ts +81 -12
  10. package/src/calls/__tests__/voice-triage-escalate.test.ts +106 -0
  11. package/src/calls/call-controller.ts +30 -2
  12. package/src/calls/call-speech-output.ts +12 -4
  13. package/src/calls/call-transport.ts +18 -1
  14. package/src/calls/media-stream-output.ts +53 -5
  15. package/src/calls/media-stream-server.ts +12 -0
  16. package/src/calls/media-stream-stt-session.ts +34 -0
  17. package/src/calls/telephony-synthesis-language.ts +84 -0
  18. package/src/calls/tts-text-sanitizer.ts +12 -5
  19. package/src/calls/voice-session-bridge.ts +31 -3
  20. package/src/calls/voice-triage-escalate.ts +52 -5
  21. package/src/config/bundled-skills/phone-calls/references/CONFIG.md +15 -15
  22. package/src/config/loader.ts +5 -0
  23. package/src/config/schemas/__tests__/live-voice.test.ts +35 -0
  24. package/src/config/schemas/calls.ts +0 -4
  25. package/src/config/schemas/live-voice.ts +25 -0
  26. package/src/config/schemas/tts.ts +63 -0
  27. package/src/live-voice/__tests__/front-decision.test.ts +120 -0
  28. package/src/live-voice/__tests__/live-voice-events.test.ts +1 -1
  29. package/src/live-voice/__tests__/live-voice-integration.test.ts +5 -0
  30. package/src/live-voice/__tests__/live-voice-progress.test.ts +178 -11
  31. package/src/live-voice/__tests__/live-voice-stt.test.ts +304 -1
  32. package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +102 -2
  33. package/src/live-voice/__tests__/live-voice-tts.test.ts +148 -2
  34. package/src/live-voice/__tests__/live-voice-vad.test.ts +278 -1
  35. package/src/live-voice/__tests__/progress-phrases.test.ts +167 -0
  36. package/src/live-voice/front-decision.ts +50 -3
  37. package/src/live-voice/live-voice-session.ts +517 -45
  38. package/src/live-voice/live-voice-tts.ts +18 -2
  39. package/src/live-voice/progress-phrases.ts +105 -2
  40. package/src/providers/speech-to-text/deepgram-realtime.test.ts +283 -1
  41. package/src/providers/speech-to-text/deepgram-realtime.ts +117 -3
  42. package/src/providers/speech-to-text/provider-catalog.ts +38 -0
  43. package/src/runtime/__tests__/local-actor-identity-force-refresh.test.ts +104 -0
  44. package/src/runtime/assistant-event-hub.ts +23 -0
  45. package/src/runtime/local-actor-identity.ts +18 -5
  46. package/src/runtime/routes/__tests__/sse-actor-principal-heal.test.ts +165 -0
  47. package/src/runtime/routes/__tests__/surface-action-routes.test.ts +2 -0
  48. package/src/runtime/routes/events-routes.ts +17 -16
  49. package/src/runtime/routes/sse-actor-principal-heal.ts +113 -0
  50. package/src/stt/__tests__/language-metadata.test.ts +85 -0
  51. package/src/stt/__tests__/speech-energy.test.ts +79 -0
  52. package/src/stt/language-metadata.ts +65 -0
  53. package/src/stt/speech-energy.ts +115 -12
  54. package/src/stt/types.ts +16 -0
  55. package/src/tts/__tests__/provider-adapters.test.ts +147 -0
  56. package/src/tts/__tests__/speakable-segments.test.ts +470 -0
  57. package/src/tts/language-voices.ts +23 -0
  58. package/src/tts/providers/deepgram-provider.ts +3 -1
  59. package/src/tts/providers/elevenlabs-provider.ts +73 -1
  60. package/src/tts/providers/xai-provider.ts +28 -2
  61. package/src/tts/speakable-segments.ts +293 -23
  62. package/src/tts/synthesis-stream.ts +7 -0
  63. package/src/tts/types.ts +7 -0
  64. package/src/util/__tests__/language-subtag.test.ts +54 -0
  65. package/src/util/language-subtag.ts +43 -0
  66. package/src/util/unicode.ts +1 -1
@@ -25,7 +25,8 @@ import {
25
25
  classifyFrontDoorLeading,
26
26
  ESCALATE_VERDICT_TOKEN,
27
27
  ESCALATION_CONTINUATION_CONTENT,
28
- FALLBACK_ESCALATION_BRIDGE,
28
+ FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
29
+ fallbackEscalationBridgeFor,
29
30
  isEscalationBridgeComplete,
30
31
  MIN_SPOKEN_BRIDGE_CHARS,
31
32
  type VoiceRoutingLeg,
@@ -46,14 +47,24 @@ import { isInstalledStaticSkillLoad } from "../permissions/checker.js";
46
47
  import { ensureConversationExists } from "../persistence/conversation-crud.js";
47
48
  import {
48
49
  listProviderIds,
50
+ pinnedListeningLanguage,
49
51
  supportsBoundary,
50
52
  } from "../providers/speech-to-text/provider-catalog.js";
51
53
  import type { ResolveStreamingTranscriberOptions } from "../providers/speech-to-text/resolve.js";
52
54
  import { broadcastMessage } from "../runtime/assistant-event-hub.js";
53
55
  import { publishConversationListAndMetadataChanged } from "../runtime/sync/resource-sync-events.js";
54
- import { detectPcm16SpeechActivity } from "../stt/speech-energy.js";
56
+ import {
57
+ dominantLanguageTag,
58
+ voteDominantLanguage,
59
+ } from "../stt/language-metadata.js";
60
+ import {
61
+ DEFAULT_SPEECH_ENERGY_THRESHOLD,
62
+ pcm16MaxNormalizedCorrelation,
63
+ pcm16MeanAmplitude,
64
+ } from "../stt/speech-energy.js";
55
65
  import type {
56
66
  StreamingTranscriber,
67
+ SttProviderId,
57
68
  SttStreamServerErrorEvent,
58
69
  SttStreamServerEvent,
59
70
  } from "../stt/types.js";
@@ -62,6 +73,7 @@ import { liveVoiceEndScreen } from "../telemetry/live-voice-funnel.js";
62
73
  import { getToolOwner } from "../tools/registry.js";
63
74
  import { extractSpeakableSegments } from "../tts/speakable-segments.js";
64
75
  import { createAbortReason } from "../util/abort-reasons.js";
76
+ import { hasLocalizedEntry } from "../util/language-subtag.js";
65
77
  import { getLogger } from "../util/logger.js";
66
78
  import {
67
79
  activityLabelForTool,
@@ -101,7 +113,12 @@ import type {
101
113
  LiveVoiceTtsOptions,
102
114
  LiveVoiceTtsResult,
103
115
  } from "./live-voice-tts.js";
104
- import { pickProgressPhrase } from "./progress-phrases.js";
116
+ import {
117
+ APPROVAL_PENDING_PHRASE_BY_LANGUAGE,
118
+ approvalPendingPhraseFor,
119
+ pickProgressPhrase,
120
+ PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE,
121
+ } from "./progress-phrases.js";
105
122
  import {
106
123
  type LiveVoiceClientAttachImageFrame,
107
124
  type LiveVoiceClientFrame,
@@ -119,6 +136,13 @@ type LiveVoiceSessionState =
119
136
  | "failed"
120
137
  | "closed";
121
138
 
139
+ type VadEnergyClassification = "speech" | "silence" | "echo";
140
+
141
+ interface VadClassifiedChunk {
142
+ readonly chunk: Buffer;
143
+ readonly classification: VadEnergyClassification;
144
+ }
145
+
122
146
  // Cap on audio buffered while a server-VAD utterance waits for its
123
147
  // transcriber (PCM16 mono seconds; oldest chunks are dropped past the cap).
124
148
  const SERVER_VAD_PENDING_AUDIO_MAX_SECONDS = 10;
@@ -138,6 +162,25 @@ const FINALIZE_GRACE_MS = 1_000;
138
162
  // liveVoice.vad.bargeInMinSpeechMs schema default; 0 disables the guard for
139
163
  // instant barge-in.
140
164
  const DEFAULT_BARGE_IN_MIN_SPEECH_MS = 250;
165
+ // The playback echo gate learns microphone energy while assistant audio is
166
+ // expected at the speaker. Input must rise above the learned level by this
167
+ // margin to count as user speech.
168
+ const DEFAULT_ECHO_BARGE_IN_MARGIN = 1.5;
169
+ const DEFAULT_ECHO_EMA_HALF_LIFE_MS = 400;
170
+ // Before learning a microphone power baseline, compare a short input window
171
+ // with the PCM sent to the speaker. This keeps a user's first interruption
172
+ // from becoming its own echo threshold.
173
+ const ECHO_CORRELATION_PROBE_MS = 100;
174
+ const ECHO_CORRELATION_MIN_MS = 50;
175
+ const ECHO_CORRELATION_THRESHOLD = 0.65;
176
+ const ECHO_REFERENCE_MAX_MS = 10_000;
177
+ // Echo should reach the microphone near playback onset. If no signal arrives
178
+ // within this much input audio, the gate returns to the fixed base threshold.
179
+ // The same interval expires a learned reference after a real silent gap.
180
+ const ECHO_ONSET_ELIGIBILITY_MS = 300;
181
+ // Client buffering makes audible playback trail the server's send-time
182
+ // estimate. Keep the echo window open briefly past that estimate.
183
+ const DEFAULT_ECHO_DRAIN_SLACK_MS = 300;
141
184
  // Mirrors MediaTurnDetector's DEFAULT_SILENCE_THRESHOLD_MS: the session
142
185
  // tracks the effective trailing-silence threshold (the detector keeps its own
143
186
  // copy private) so the endpoint decider can report the pause length.
@@ -268,6 +311,17 @@ export interface LiveVoiceSessionOptions {
268
311
  * defaults to `DEFAULT_BARGE_IN_MIN_SPEECH_MS`.
269
312
  */
270
313
  bargeInMinSpeechMs?: number;
314
+ /**
315
+ * Multiplier over the learned playback echo level that input must exceed
316
+ * to count as speech while assistant audio is playing. Values at or below
317
+ * 1 disable adaptation for internal fixed-gate callers; workspace config
318
+ * requires a value greater than 1.
319
+ */
320
+ echoBargeInMargin?: number;
321
+ /** Half-life in milliseconds for the learned playback echo level. */
322
+ echoEmaHalfLifeMs?: number;
323
+ /** Extra time after the playback estimate during which echo is expected. */
324
+ echoDrainSlackMs?: number;
271
325
  /**
272
326
  * Overrides the bounded wait for the shared transcriber's finalize
273
327
  * flush in persistent mode (test hook). Defaults to `FINALIZE_GRACE_MS`.
@@ -369,6 +423,23 @@ interface UtteranceCycle {
369
423
  // text replays the boundary immediately — the hold was judged on stale
370
424
  // text, so waiting out the extension only adds silence.
371
425
  heldSpeculativeContent: string | null;
426
+ // Count per detected-language base subtag (see voteDominantLanguage)
427
+ // across this cycle's final transcript events. Resolves the turn's spoken
428
+ // language (see turnLanguageFor); empty when the provider tags nothing.
429
+ languageTally: Map<string, number>;
430
+ // Detected languages of the most recent partial that carried any, already
431
+ // normalized, dominance order. Speculative turns dispatch from partials
432
+ // before the first tagged final lands, so turnLanguageFor falls back to
433
+ // this when the final tally is still empty. Never cleared: the tally
434
+ // outranks it once finals arrive, and a revising partial without tags
435
+ // must not wipe an earlier partial's detection.
436
+ latestPartialLanguages: readonly string[] | null;
437
+ // The provider that actually transcribed this cycle, recorded when its
438
+ // transcriber is assigned and kept after teardown nulls `transcriber`.
439
+ // The resolver can silently dial managed vellum when a BYOK provider has
440
+ // no credential, so the language-pin gate in turnLanguageFor must follow
441
+ // this, not the configured provider.
442
+ dialedSttProvider: SttProviderId | null;
372
443
  turnId: string | null;
373
444
  userMessageId: string | null;
374
445
  userAudioChunks: Buffer[];
@@ -399,6 +470,11 @@ type UtteranceStartResult =
399
470
  // client in job-list order.
400
471
  interface TtsSegmentJob {
401
472
  readonly text: string;
473
+ // Per-segment language-hint override, preferred over the turn's language.
474
+ // Set on fixed phrases whose localized table lacks the turn's language:
475
+ // the English fallback text carries "en" so an enforcing provider never
476
+ // renders English words as ar/ko/ta. Undefined means the turn language.
477
+ readonly language: string | undefined;
402
478
  // The provider stream was started (the job holds an open-job slot).
403
479
  started: boolean;
404
480
  // Emission finished; the slot is free for the next queued segment.
@@ -483,6 +559,13 @@ interface ActiveAssistantTurn {
483
559
  token: symbol;
484
560
  turnId: string;
485
561
  utterance: UtteranceCycle;
562
+ // The caller's spoken language for this turn as a lowercase base subtag
563
+ // (see turnLanguageFor): the dominant STT-detected language, else a
564
+ // monolingual services.stt.language pin. Undefined when unknown, which
565
+ // disables every language-aware path (prompt note, TTS hint, localized
566
+ // fallbacks). Re-resolved when a speculative turn commits, since finals
567
+ // can land between dispatch and verdict.
568
+ language: string | undefined;
486
569
  abortController: AbortController;
487
570
  handle: VoiceTurnHandle | null;
488
571
  // When the turn launched, for narration's turnElapsedMs.
@@ -681,14 +764,8 @@ function createControlMarkerHoldback(
681
764
  // barge-in, the interruption merge note is appended to it (see
682
765
  // buildInterruptionMergeNote) so the model reconciles the interrupted request
683
766
  // with the new utterance.
684
- // Spoken once when a turn starts waiting on the user's decision. Fixed rather
685
- // than generated: this is a statement about the system's state, not about the
686
- // work, and it has to be true every time. Kept in the shape of the progress
687
- // phrases it displaces (short, neutral, no claim about tools).
688
- const APPROVAL_PENDING_PHRASE = "I need your okay for that one. Take a look.";
689
-
690
767
  const LIVE_VOICE_CONTROL_PROMPT_BASE =
691
- "You are speaking in a local live voice session. Keep replies brief and conversational. Speech is the main channel: say the answer, and do not narrate a surface instead of answering. You can also put something on screen when it genuinely helps (a form, a list to pick from, a progress card for long work); the call overlay minimizes by itself once you finish speaking, so the user sees it without doing anything. Never tell the user you cannot show them something. ";
768
+ "You are speaking in a local live voice session. Keep replies brief and conversational. Speech is the main channel: say the answer, and do not narrate a surface instead of answering. You can also put something on screen when it genuinely helps (a form, a list to pick from, a progress card for long work); the call overlay minimizes by itself once you finish speaking, so the user sees it without doing anything. Never tell the user you cannot show them something. Reply in the language the caller is speaking; if they switch languages, switch with them. ";
692
769
 
693
770
  // Appended for the legs that can actually put something on screen: the main
694
771
  // leg and the escalated leg. The front-door (fast) leg never receives it, for
@@ -777,6 +854,9 @@ function buildVoiceControlPrompt(
777
854
  LIVE_VOICE_CONTROL_PROMPT_BASE +
778
855
  (leg.frontDoor === true ? "" : LIVE_VOICE_SCREEN_REVEAL_TEACHING) +
779
856
  VOICE_NO_SETUP_FLOWS_RULE;
857
+ if (turn.language !== undefined) {
858
+ prompt = `${prompt}\n\nThe caller has been speaking the language with code "${turn.language}" this turn. Reply in that language unless they clearly switch to another.`;
859
+ }
780
860
  if (turn.interruptedRequest) {
781
861
  prompt = `${prompt}\n\n${buildInterruptionMergeNote(turn.interruptedRequest)}`;
782
862
  }
@@ -825,6 +905,9 @@ function createUtteranceCycle(): UtteranceCycle {
825
905
  latestPartialText: null,
826
906
  endpointExtensionCount: 0,
827
907
  heldSpeculativeContent: null,
908
+ languageTally: new Map(),
909
+ latestPartialLanguages: null,
910
+ dialedSttProvider: null,
828
911
  turnId: null,
829
912
  userMessageId: null,
830
913
  userAudioChunks: [],
@@ -991,9 +1074,29 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
991
1074
  private failureCode: LiveVoiceProtocolErrorCode | null = null;
992
1075
  // Non-null iff the start frame requested turnDetection "server_vad".
993
1076
  private readonly turnDetector: MediaTurnDetector | null;
994
- // Energy gate for server-VAD speech classification; undefined defers to
995
- // DEFAULT_SPEECH_ENERGY_THRESHOLD.
1077
+ // Base energy gate for server-VAD speech classification. During estimated
1078
+ // playback, classifyVadEnergy raises this above the learned echo level.
996
1079
  private readonly speechEnergyThreshold: number | undefined;
1080
+ private readonly echoBargeInMargin: number;
1081
+ private readonly echoEmaHalfLifeMs: number;
1082
+ private readonly echoDrainSlackMs: number;
1083
+ // Learned microphone energy attributable to assistant playback.
1084
+ private echoEnergyEma = 0;
1085
+ // Signal-bearing microphone audio held until it can be compared with the
1086
+ // assistant PCM. A nonmatch is replayed through VAD in original order.
1087
+ private echoProbeChunks: Buffer[] = [];
1088
+ // Recent raw assistant PCM from the current playback burst.
1089
+ private echoReferenceAudio = Buffer.alloc(0);
1090
+ private echoWindowTotalAudioMs = 0;
1091
+ // Consecutive sub-base input expires a reference that can no longer
1092
+ // describe audible playback.
1093
+ private echoSubBaseRunMs = 0;
1094
+ // Once onset eligibility lapses, later user speech cannot seed a new echo
1095
+ // reference in the same playback window.
1096
+ private echoOnsetLapsed = false;
1097
+ // A live speech run that predates playback belongs to the user and bypasses
1098
+ // echo warm-up until that run genuinely resets.
1099
+ private echoWindowGuardCarryover = false;
997
1100
  // Mutable so a mid-session `update_config` frame can retune "interrupt
998
1101
  // sensitivity" live (see applyConfigUpdate).
999
1102
  private bargeInMinSpeechMs: number;
@@ -1166,6 +1269,12 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1166
1269
  context.startFrame.bargeInMinSpeechMs ??
1167
1270
  options.bargeInMinSpeechMs ??
1168
1271
  DEFAULT_BARGE_IN_MIN_SPEECH_MS;
1272
+ this.echoBargeInMargin =
1273
+ options.echoBargeInMargin ?? DEFAULT_ECHO_BARGE_IN_MARGIN;
1274
+ this.echoEmaHalfLifeMs =
1275
+ options.echoEmaHalfLifeMs ?? DEFAULT_ECHO_EMA_HALF_LIFE_MS;
1276
+ this.echoDrainSlackMs =
1277
+ options.echoDrainSlackMs ?? DEFAULT_ECHO_DRAIN_SLACK_MS;
1169
1278
  this.finalizeGraceMs = options.finalizeGraceMs ?? FINALIZE_GRACE_MS;
1170
1279
  this.frontDecider = options.frontDecider ?? null;
1171
1280
  this.frontModelConfig = LiveVoiceFrontModelConfigSchema.parse(
@@ -1387,6 +1496,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1387
1496
  // Persistent re-arm: the shared stream is already open, so the cycle
1388
1497
  // goes straight to streaming with no resolve/start round-trip.
1389
1498
  utterance.transcriber = shared;
1499
+ utterance.dialedSttProvider = shared.providerId;
1390
1500
  return await this.activateUtterance(utterance, replayTurnEnd);
1391
1501
  }
1392
1502
  // The shared stream is pinned to the old language, so retire it and
@@ -1419,6 +1529,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1419
1529
  }
1420
1530
 
1421
1531
  utterance.transcriber = transcriber;
1532
+ utterance.dialedSttProvider = transcriber.providerId;
1422
1533
  if (
1423
1534
  this.turnDetector &&
1424
1535
  typeof transcriber.finalizeUtterance === "function"
@@ -1639,12 +1750,26 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1639
1750
  return;
1640
1751
  }
1641
1752
 
1642
- const hasSpeech = detectPcm16SpeechActivity(
1643
- chunk,
1644
- this.speechEnergyThreshold,
1645
- );
1753
+ for (const classified of this.classifyVadEnergy(chunk)) {
1754
+ await this.handleClassifiedVadAudio(detector, classified);
1755
+ }
1756
+ }
1757
+
1758
+ private async handleClassifiedVadAudio(
1759
+ detector: MediaTurnDetector,
1760
+ classified: VadClassifiedChunk,
1761
+ ): Promise<void> {
1762
+ const { chunk, classification: energyClassification } = classified;
1763
+ const hasSpeech = energyClassification === "speech";
1646
1764
  detector.onMediaChunk(hasSpeech);
1647
- this.trackBargeInGuard(hasSpeech, chunk);
1765
+ this.trackBargeInGuard(energyClassification, chunk);
1766
+
1767
+ // Playback echo is neither user audio nor useful pre-roll. Dropping it
1768
+ // prevents the assistant's reply from reaching transcription as a ghost
1769
+ // follow-up turn.
1770
+ if (energyClassification === "echo") {
1771
+ return;
1772
+ }
1648
1773
 
1649
1774
  // Idle mic: hold silent chunks in the bounded pre-roll instead of
1650
1775
  // collecting or streaming them; flushed on speech onset so the
@@ -1722,6 +1847,193 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1722
1847
  await this.routeVadAudio(utterance, chunk);
1723
1848
  }
1724
1849
 
1850
+ /**
1851
+ * Classify microphone energy while keeping assistant playback echo out of
1852
+ * barge-in, turn detection, pre-roll, and transcription.
1853
+ *
1854
+ * A short onset probe must correlate with PCM sent to the speaker before its
1855
+ * microphone power can seed the adaptive threshold. Nonmatching probe audio
1856
+ * is replayed through VAD in original order, so a user who talks at playback
1857
+ * onset is neither learned as echo nor lost. Once seeded, the EMA follows
1858
+ * confirmed echo while speech above the learned margin remains frozen out.
1859
+ */
1860
+ private classifyVadEnergy(chunk: Buffer): VadClassifiedChunk[] {
1861
+ const baseThreshold =
1862
+ this.speechEnergyThreshold ?? DEFAULT_SPEECH_ENERGY_THRESHOLD;
1863
+ const meanAmplitude = pcm16MeanAmplitude(chunk);
1864
+ if (
1865
+ this.echoBargeInMargin <= 1 ||
1866
+ !this.isAssistantPlaybackEchoPossible()
1867
+ ) {
1868
+ this.resetEchoReference();
1869
+ return [
1870
+ this.classifyAtFixedThreshold(chunk, baseThreshold, meanAmplitude),
1871
+ ];
1872
+ }
1873
+
1874
+ if (this.echoWindowTotalAudioMs === 0) {
1875
+ this.echoWindowGuardCarryover =
1876
+ this.pendingBargeIn !== null && this.pendingBargeIn.speechMs > 0;
1877
+ } else if (this.pendingBargeIn === null) {
1878
+ this.echoWindowGuardCarryover = false;
1879
+ }
1880
+
1881
+ const chunkMs = pcm16DurationMs(
1882
+ chunk.byteLength,
1883
+ this.context.startFrame.audio.sampleRate,
1884
+ );
1885
+ const onsetWasEligible =
1886
+ !this.echoOnsetLapsed &&
1887
+ this.echoWindowTotalAudioMs < ECHO_ONSET_ELIGIBILITY_MS;
1888
+ this.echoWindowTotalAudioMs += chunkMs;
1889
+
1890
+ if (this.echoProbeChunks.length > 0) {
1891
+ this.echoProbeChunks.push(Buffer.from(chunk));
1892
+ return this.resolveEchoProbe(baseThreshold);
1893
+ }
1894
+
1895
+ if (meanAmplitude <= baseThreshold) {
1896
+ this.echoSubBaseRunMs += chunkMs;
1897
+ if (this.echoSubBaseRunMs >= ECHO_ONSET_ELIGIBILITY_MS) {
1898
+ this.echoEnergyEma = 0;
1899
+ this.echoOnsetLapsed = true;
1900
+ }
1901
+ return [{ chunk, classification: "silence" }];
1902
+ }
1903
+
1904
+ this.echoSubBaseRunMs = 0;
1905
+ if (
1906
+ this.echoEnergyEma === 0 &&
1907
+ onsetWasEligible &&
1908
+ !this.echoWindowGuardCarryover
1909
+ ) {
1910
+ this.echoProbeChunks.push(Buffer.from(chunk));
1911
+ return this.resolveEchoProbe(baseThreshold);
1912
+ }
1913
+
1914
+ if (this.echoEnergyEma === 0) {
1915
+ this.echoOnsetLapsed = true;
1916
+ return [{ chunk, classification: "speech" }];
1917
+ }
1918
+
1919
+ const speechThreshold = Math.max(
1920
+ baseThreshold,
1921
+ this.echoBargeInMargin * this.echoEnergyEma,
1922
+ );
1923
+ if (meanAmplitude > speechThreshold) {
1924
+ const guardHasSpeech =
1925
+ this.pendingBargeIn !== null && this.pendingBargeIn.speechMs > 0;
1926
+ if (!guardHasSpeech && this.echoMatchesAssistant(chunk)) {
1927
+ this.updateEchoEnergy(meanAmplitude, chunkMs);
1928
+ return [{ chunk, classification: "echo" }];
1929
+ }
1930
+ return [{ chunk, classification: "speech" }];
1931
+ }
1932
+
1933
+ this.updateEchoEnergy(meanAmplitude, chunkMs);
1934
+ return [{ chunk, classification: "echo" }];
1935
+ }
1936
+
1937
+ private resolveEchoProbe(baseThreshold: number): VadClassifiedChunk[] {
1938
+ const probe = Buffer.concat(this.echoProbeChunks);
1939
+ const probeAudioMs = pcm16DurationMs(
1940
+ probe.byteLength,
1941
+ this.context.startFrame.audio.sampleRate,
1942
+ );
1943
+ if (
1944
+ probeAudioMs >= ECHO_CORRELATION_MIN_MS &&
1945
+ this.echoMatchesAssistant(probe)
1946
+ ) {
1947
+ this.echoEnergyEma = Math.max(baseThreshold, pcm16MeanAmplitude(probe));
1948
+ const chunks = this.echoProbeChunks.splice(0);
1949
+ return chunks.map((chunk) => ({ chunk, classification: "echo" }));
1950
+ }
1951
+ if (probeAudioMs < ECHO_CORRELATION_PROBE_MS) {
1952
+ return [];
1953
+ }
1954
+
1955
+ this.echoOnsetLapsed = true;
1956
+ const chunks = this.echoProbeChunks.splice(0);
1957
+ return chunks.map((chunk) =>
1958
+ this.classifyAtFixedThreshold(chunk, baseThreshold),
1959
+ );
1960
+ }
1961
+
1962
+ private echoMatchesAssistant(chunk: Buffer): boolean {
1963
+ const sampleRate = this.context.startFrame.audio.sampleRate;
1964
+ const minimumBytes = Math.ceil(
1965
+ (sampleRate * ECHO_CORRELATION_MIN_MS * 2) / 1_000,
1966
+ );
1967
+ if (
1968
+ chunk.byteLength < minimumBytes ||
1969
+ this.echoReferenceAudio.byteLength < minimumBytes
1970
+ ) {
1971
+ return false;
1972
+ }
1973
+ const probeByteLength = Math.min(
1974
+ chunk.byteLength,
1975
+ Math.ceil((sampleRate * ECHO_CORRELATION_PROBE_MS * 2) / 1_000),
1976
+ );
1977
+ return (
1978
+ pcm16MaxNormalizedCorrelation(
1979
+ chunk.subarray(0, probeByteLength),
1980
+ this.echoReferenceAudio,
1981
+ ) >= ECHO_CORRELATION_THRESHOLD
1982
+ );
1983
+ }
1984
+
1985
+ private updateEchoEnergy(meanAmplitude: number, chunkMs: number): void {
1986
+ const alpha = 1 - 0.5 ** (chunkMs / this.echoEmaHalfLifeMs);
1987
+ this.echoEnergyEma =
1988
+ alpha * meanAmplitude + (1 - alpha) * this.echoEnergyEma;
1989
+ }
1990
+
1991
+ private classifyAtFixedThreshold(
1992
+ chunk: Buffer,
1993
+ baseThreshold: number,
1994
+ meanAmplitude = pcm16MeanAmplitude(chunk),
1995
+ ): VadClassifiedChunk {
1996
+ return {
1997
+ chunk,
1998
+ classification: meanAmplitude > baseThreshold ? "speech" : "silence",
1999
+ };
2000
+ }
2001
+
2002
+ private isAssistantPlaybackEchoPossible(): boolean {
2003
+ return (
2004
+ Date.now() < this.assistantPlaybackTailUntilMs + this.echoDrainSlackMs
2005
+ );
2006
+ }
2007
+
2008
+ private resetEchoReference(): void {
2009
+ this.echoEnergyEma = 0;
2010
+ this.echoProbeChunks = [];
2011
+ this.echoReferenceAudio = Buffer.alloc(0);
2012
+ this.echoWindowTotalAudioMs = 0;
2013
+ this.echoSubBaseRunMs = 0;
2014
+ this.echoOnsetLapsed = false;
2015
+ this.echoWindowGuardCarryover = false;
2016
+ }
2017
+
2018
+ private appendEchoReference(chunk: LiveVoiceTtsAudioChunk): void {
2019
+ if (
2020
+ chunk.contentType.split(";", 1)[0]?.trim().toLowerCase() !==
2021
+ "audio/pcm" ||
2022
+ chunk.sampleRate !== this.context.startFrame.audio.sampleRate
2023
+ ) {
2024
+ return;
2025
+ }
2026
+ const audio = Buffer.from(chunk.dataBase64, "base64");
2027
+ const maxBytes = Math.ceil(
2028
+ (chunk.sampleRate * ECHO_REFERENCE_MAX_MS * 2) / 1_000,
2029
+ );
2030
+ const combined = Buffer.concat([this.echoReferenceAudio, audio]);
2031
+ this.echoReferenceAudio =
2032
+ combined.byteLength > maxBytes
2033
+ ? combined.subarray(combined.byteLength - maxBytes)
2034
+ : combined;
2035
+ }
2036
+
1725
2037
  private async routeVadAudio(
1726
2038
  utterance: UtteranceCycle,
1727
2039
  chunk: Buffer,
@@ -1869,7 +2181,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1869
2181
  // The client can still be draining audible playback after tts_done
1870
2182
  // (the turn is already cleared server-side) — that tail deserves the
1871
2183
  // same guard, or a noise blip clips the reply's last words.
1872
- const drainingPlayback = Date.now() < this.assistantPlaybackTailUntilMs;
2184
+ const drainingPlayback = this.isAssistantPlaybackEchoPossible();
1873
2185
 
1874
2186
  if ((bargeableTurn || drainingPlayback) && this.bargeInMinSpeechMs > 0) {
1875
2187
  // Onset audio keeps flowing into the cycle/pre-roll while the guard
@@ -1891,13 +2203,14 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1891
2203
  }
1892
2204
  }
1893
2205
 
1894
- // Advances the sustained-speech barge-in guard by one server-VAD chunk:
1895
- // above-gate speech accumulates toward bargeInMinSpeechMs while brief
1896
- // sub-threshold gaps are tolerated (a single continuous silence longer than
1897
- // BARGE_IN_GAP_TOLERANCE_MS, or cumulative tolerated silence past the
1898
- // duty-cycle ceiling, zeroes the run), and once met the deferred
1899
- // speech_started + barge-in fire.
1900
- private trackBargeInGuard(hasSpeech: boolean, chunk: Buffer): void {
2206
+ // Advance the sustained-speech barge-in guard by one server-VAD chunk.
2207
+ // Speech accumulates toward bargeInMinSpeechMs, short true-silence gaps are
2208
+ // tolerated, and classified playback echo resets the run immediately.
2209
+ // Longer or mostly silent runs reset through the existing gap limits.
2210
+ private trackBargeInGuard(
2211
+ classification: VadEnergyClassification,
2212
+ chunk: Buffer,
2213
+ ): void {
1901
2214
  const guard = this.pendingBargeIn;
1902
2215
  if (!guard) {
1903
2216
  return;
@@ -1906,7 +2219,11 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1906
2219
  chunk.byteLength,
1907
2220
  this.context.startFrame.audio.sampleRate,
1908
2221
  );
1909
- if (!hasSpeech) {
2222
+ if (classification === "echo") {
2223
+ this.resetBargeInGuardRun();
2224
+ return;
2225
+ }
2226
+ if (classification === "silence") {
1910
2227
  guard.silenceMs += chunkMs;
1911
2228
  guard.toleratedSilenceMs += chunkMs;
1912
2229
  // Strictly greater on the per-gap check: a gap of exactly
@@ -1920,9 +2237,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1920
2237
  guard.toleratedSilenceMs >
1921
2238
  this.bargeInMinSpeechMs * BARGE_IN_MAX_TOLERATED_SILENCE_RATIO
1922
2239
  ) {
1923
- guard.speechMs = 0;
1924
- guard.silenceMs = 0;
1925
- guard.toleratedSilenceMs = 0;
2240
+ this.resetBargeInGuardRun();
1926
2241
  }
1927
2242
  return;
1928
2243
  }
@@ -1940,6 +2255,21 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1940
2255
  }
1941
2256
  }
1942
2257
 
2258
+ private resetBargeInGuardRun(): void {
2259
+ const guard = this.pendingBargeIn;
2260
+ if (!guard) {
2261
+ return;
2262
+ }
2263
+ guard.speechMs = 0;
2264
+ guard.silenceMs = 0;
2265
+ guard.toleratedSilenceMs = 0;
2266
+ if (this.echoWindowGuardCarryover) {
2267
+ this.echoWindowGuardCarryover = false;
2268
+ this.echoEnergyEma = 0;
2269
+ this.echoProbeChunks = [];
2270
+ }
2271
+ }
2272
+
1943
2273
  private bargeIn(turn: ActiveAssistantTurn): void {
1944
2274
  // Abort synchronously so no tts_audio frame can follow turn_cancelled,
1945
2275
  // and settle the cancelled turn's metrics so the next utterance's marks
@@ -2527,8 +2857,13 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
2527
2857
  // Spoken, because opening the room is only a cue for someone looking at
2528
2858
  // the screen, and the case this exists for is a phone the user has put
2529
2859
  // down. One line, not narration: the turn is not working, it is waiting,
2530
- // and it says which.
2531
- this.enqueueFillerPhrase(turn, APPROVAL_PENDING_PHRASE);
2860
+ // and it says which, in the turn's spoken language, like every other
2861
+ // filler phrase.
2862
+ this.enqueueFillerPhrase(
2863
+ turn,
2864
+ approvalPendingPhraseFor(turn.language),
2865
+ this.fixedPhraseLanguage(turn, APPROVAL_PENDING_PHRASE_BY_LANGUAGE),
2866
+ );
2532
2867
  }
2533
2868
 
2534
2869
  /** Clear the wait once a decision lands, so the turn narrates normally again. */
@@ -2851,6 +3186,13 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
2851
3186
  // are skipped; the thinking frame and timers still apply.
2852
3187
  const alreadyReleased = utterance.released;
2853
3188
  turn.speculativePending = false;
3189
+ // Finals can land between the speculative dispatch and this verdict.
3190
+ // Fill the language only when dispatch had none: the model request was
3191
+ // already issued with the dispatch language, so overwriting here would
3192
+ // hint TTS (and any voice override) in a different language than the
3193
+ // text it speaks. The tally still carries the corrected detection into
3194
+ // the next turn.
3195
+ turn.language ??= this.turnLanguageFor(utterance);
2854
3196
  if (turn.verdictDeadlineTimer !== null) {
2855
3197
  clearTimeout(turn.verdictDeadlineTimer);
2856
3198
  turn.verdictDeadlineTimer = null;
@@ -3189,11 +3531,16 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3189
3531
  switch (event.type) {
3190
3532
  case "partial":
3191
3533
  utterance.latestPartialText = event.text;
3534
+ this.capturePartialLanguages(utterance, event.languages);
3192
3535
  this.markFirstPartial(utterance);
3193
3536
  await this.sendFrame({ type: "stt_partial", text: event.text });
3194
3537
  return;
3195
3538
  case "final":
3196
- await this.recordFinalTranscript(utterance, event.text);
3539
+ await this.recordFinalTranscript(
3540
+ utterance,
3541
+ event.text,
3542
+ event.languages,
3543
+ );
3197
3544
  return;
3198
3545
  case "finalized":
3199
3546
  // Per-cycle transcribers are torn down with stop(); the finalize
@@ -3262,6 +3609,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3262
3609
  return;
3263
3610
  }
3264
3611
  target.latestPartialText = event.text;
3612
+ this.capturePartialLanguages(target, event.languages);
3265
3613
  this.markFirstPartial(target);
3266
3614
  await this.sendFrame({ type: "stt_partial", text: event.text });
3267
3615
  return;
@@ -3277,7 +3625,11 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3277
3625
  // newer cycle.
3278
3626
  const owner = this.finalizeQueue[0];
3279
3627
  if (owner && !owner.assistantTurnStarted && !owner.completed) {
3280
- await this.recordFinalTranscript(owner, event.text);
3628
+ await this.recordFinalTranscript(
3629
+ owner,
3630
+ event.text,
3631
+ event.languages,
3632
+ );
3281
3633
  } else {
3282
3634
  log.warn(
3283
3635
  "Dropping a late finalize flush segment: its assistant turn already dispatched",
@@ -3294,7 +3646,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3294
3646
  );
3295
3647
  return;
3296
3648
  }
3297
- await this.recordFinalTranscript(target, event.text);
3649
+ await this.recordFinalTranscript(target, event.text, event.languages);
3298
3650
  return;
3299
3651
  }
3300
3652
  case "finalized": {
@@ -3363,10 +3715,16 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3363
3715
  private async recordFinalTranscript(
3364
3716
  utterance: UtteranceCycle,
3365
3717
  text: string,
3718
+ languages?: readonly string[],
3366
3719
  ): Promise<void> {
3367
3720
  const transcript = text.trim();
3368
3721
  if (transcript.length > 0) {
3369
3722
  utterance.finalTranscriptSegments.push(transcript);
3723
+ // Tally only finals that committed transcript: empty silence frames
3724
+ // can still carry container-level language tags describing no emitted
3725
+ // words, and counting those would let silence outvote real speech
3726
+ // (same choice as the adapter's boundary-final aggregation).
3727
+ voteDominantLanguage(utterance.languageTally, languages);
3370
3728
  }
3371
3729
  // The final commits (and supersedes) whatever partial was trailing it.
3372
3730
  utterance.latestPartialText = null;
@@ -3405,6 +3763,59 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3405
3763
  await this.startAssistantTurnIfReady();
3406
3764
  }
3407
3765
 
3766
+ // Record a partial event's detected languages so speculative dispatch
3767
+ // has a detection before the first tagged final. The event contract
3768
+ // (stt/types.ts) guarantees the tags arrive as normalized base subtags
3769
+ // in dominance order, so they are stored as-is. Partials revise each
3770
+ // other, so this overwrites rather than tallies, and a tag-less partial
3771
+ // keeps the previous value.
3772
+ private capturePartialLanguages(
3773
+ utterance: UtteranceCycle,
3774
+ languages: readonly string[] | undefined,
3775
+ ): void {
3776
+ if (!languages || languages.length === 0) {
3777
+ return;
3778
+ }
3779
+ utterance.latestPartialLanguages = languages;
3780
+ }
3781
+
3782
+ /**
3783
+ * The caller's spoken language for a turn on this utterance, as a
3784
+ * lowercase base subtag: the dominant tallied STT-detected language
3785
+ * (most final-event counts, ties by first appearance), else the latest
3786
+ * tagged partial's dominant language (speculative turns dispatch from
3787
+ * partials), else a monolingual `services.stt.language` pin (a pinned
3788
+ * language IS the spoken language), else undefined ("multi" with no tags,
3789
+ * non-tagging providers, silence).
3790
+ */
3791
+ private turnLanguageFor(utterance: UtteranceCycle): string | undefined {
3792
+ const dominant = dominantLanguageTag(utterance.languageTally);
3793
+ if (dominant !== undefined) {
3794
+ return dominant;
3795
+ }
3796
+ // No tagged final yet (speculative turns dispatch from partials): the
3797
+ // latest tagged partial is the best detection available and outranks a
3798
+ // static pin for the same reason the tally does.
3799
+ const partialDominant = utterance.latestPartialLanguages?.[0];
3800
+ if (partialDominant !== undefined) {
3801
+ return partialDominant;
3802
+ }
3803
+ // A persisted pin only counts when the provider that actually
3804
+ // transcribed honors manual language selection (the shared
3805
+ // pinnedListeningLanguage gate). The DIALED transcriber's providerId
3806
+ // is authoritative, because the resolver silently falls back to
3807
+ // managed vellum (which honors the pin) when a BYOK provider has no
3808
+ // credential; the configured provider is only the last resort when no
3809
+ // transcriber reference survives.
3810
+ const { language: configured, provider: sttProvider } =
3811
+ getConfig().services.stt;
3812
+ const dialedProvider =
3813
+ utterance.dialedSttProvider ??
3814
+ this.sharedTranscriber?.providerId ??
3815
+ (sttProvider as SttProviderId);
3816
+ return pinnedListeningLanguage(dialedProvider, configured);
3817
+ }
3818
+
3408
3819
  // Providers emit `error` mid-stream and may keep streaming; `closed` /
3409
3820
  // `final` still drive turn lifecycle. Only transient categories are
3410
3821
  // recoverable — auth/rate-limit/invalid-audio will not self-heal, so
@@ -3660,6 +4071,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3660
4071
  token,
3661
4072
  turnId,
3662
4073
  utterance,
4074
+ language: this.turnLanguageFor(utterance),
3663
4075
  abortController,
3664
4076
  handle: null,
3665
4077
  launchedAtMs: Date.now(),
@@ -4312,7 +4724,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4312
4724
  // deleted row for a bridge the model never produced).
4313
4725
  const usesFallbackBridge = cappedBridge.length < MIN_SPOKEN_BRIDGE_CHARS;
4314
4726
  const spokenBridge = usesFallbackBridge
4315
- ? FALLBACK_ESCALATION_BRIDGE
4727
+ ? fallbackEscalationBridgeFor(activeTurn.language)
4316
4728
  : cappedBridge;
4317
4729
  if (!usesFallbackBridge) {
4318
4730
  this.markFirstAssistantDelta(activeTurn.utterance, activeTurn.turnId);
@@ -4321,12 +4733,25 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4321
4733
  { type: "assistant_text_delta", text: spokenBridge },
4322
4734
  () => !activeTurn.abortController.signal.aborted && !this.isClosed,
4323
4735
  );
4736
+ this.bufferAssistantTextForTts(activeTurn.token, `${spokenBridge} `);
4737
+ // Force-flush now: on the TTS path an unpunctuated bridge would
4738
+ // otherwise sit buffered until a sentence boundary and leave the
4739
+ // caller in silence during the escalated model's call.
4740
+ this.flushTtsBuffer(activeTurn.token, true);
4741
+ } else {
4742
+ // The canned bridge is a fixed localized-table phrase, enqueued
4743
+ // directly (it is already one complete sentence) so the segment can
4744
+ // carry the "en" override when the table lacks the turn's language.
4745
+ const speakable = sanitizeForTts(spokenBridge).trim();
4746
+ if (speakable.length > 0) {
4747
+ this.enqueueTtsSegment(activeTurn.token, speakable, {
4748
+ language: this.fixedPhraseLanguage(
4749
+ activeTurn,
4750
+ FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
4751
+ ),
4752
+ });
4753
+ }
4324
4754
  }
4325
- this.bufferAssistantTextForTts(activeTurn.token, `${spokenBridge} `);
4326
- // Force-flush now: on the TTS path an unpunctuated bridge would otherwise
4327
- // sit buffered until a sentence boundary and leave the caller in silence
4328
- // during the escalated model's call.
4329
- this.flushTtsBuffer(activeTurn.token, true);
4330
4755
 
4331
4756
  // No overrideProfile: the escalated leg runs on the call-site default —
4332
4757
  // the exact profile an un-routed voice turn would use (see
@@ -4690,6 +5115,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4690
5115
  : null,
4691
5116
  turnElapsedMs: now - turn.launchedAtMs,
4692
5117
  updateIndex: progress.updatesSpoken + 1,
5118
+ ...(turn.language !== undefined ? { languageHint: turn.language } : {}),
4693
5119
  };
4694
5120
  const generated = await frontDecider
4695
5121
  .generateProgressText(input, turn.abortController.signal)
@@ -4718,13 +5144,20 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4718
5144
  return;
4719
5145
  }
4720
5146
  let raw = generated;
5147
+ // Decider text is generated in the turn's language; only the static
5148
+ // fallback comes from a localized table and may need the "en" override.
5149
+ let fillerLanguage: string | undefined;
4721
5150
  if (raw === null) {
4722
5151
  if (trigger !== "idle") {
4723
5152
  return;
4724
5153
  }
4725
- raw = pickProgressPhrase(this.progressPhraseCounter++);
5154
+ raw = pickProgressPhrase(this.progressPhraseCounter++, turn.language);
5155
+ fillerLanguage = this.fixedPhraseLanguage(
5156
+ turn,
5157
+ PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE,
5158
+ );
4726
5159
  }
4727
- if (!this.enqueueFillerPhrase(turn, raw)) {
5160
+ if (!this.enqueueFillerPhrase(turn, raw, fillerLanguage)) {
4728
5161
  return;
4729
5162
  }
4730
5163
  progress.opsSinceNarration = 0;
@@ -4784,6 +5217,9 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4784
5217
  {
4785
5218
  transcriptSoFar: transcript,
4786
5219
  toolName,
5220
+ ...(activeTurn.language !== undefined
5221
+ ? { languageHint: activeTurn.language }
5222
+ : {}),
4787
5223
  },
4788
5224
  activeTurn.abortController.signal,
4789
5225
  )
@@ -4821,20 +5257,42 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4821
5257
  // Sanitize and enqueue one filler sentence (spoken ack or progress
4822
5258
  // narration) on the turn's ordered TTS queue — the shared tail of every
4823
5259
  // filler path. Returns whether a phrase actually enqueued; per-kind metric
4824
- // marks and bookkeeping are the caller's.
4825
- private enqueueFillerPhrase(turn: ActiveAssistantTurn, raw: string): boolean {
5260
+ // marks and bookkeeping are the caller's. `language` is a per-segment
5261
+ // hint override (see fixedPhraseLanguage); omit it for generated text,
5262
+ // which is already in the turn's language.
5263
+ private enqueueFillerPhrase(
5264
+ turn: ActiveAssistantTurn,
5265
+ raw: string,
5266
+ language?: string,
5267
+ ): boolean {
4826
5268
  const phrase = sanitizeForTts(raw).trim();
4827
5269
  if (phrase.length === 0) {
4828
5270
  return false;
4829
5271
  }
4830
5272
  this.enqueueTtsSegment(turn.token, phrase, {
4831
5273
  countsAsFirstSegment: false,
5274
+ ...(language !== undefined ? { language } : {}),
4832
5275
  });
4833
5276
  // A spoken filler holds the floor, so narration's minGapMs spaces from it.
4834
5277
  turn.progress.lastFloorHolderAtMs = Date.now();
4835
5278
  return true;
4836
5279
  }
4837
5280
 
5281
+ // The TTS hint override for a fixed phrase picked from a localized table:
5282
+ // "en" when the turn has a language the table does not cover (the picker
5283
+ // fell back to English text, which must not be synthesized under an
5284
+ // ar/ko/ta hint), undefined otherwise (the segment rides the turn's
5285
+ // language, or no hint at all when the language is unknown).
5286
+ private fixedPhraseLanguage(
5287
+ turn: ActiveAssistantTurn,
5288
+ table: Readonly<Record<string, unknown>>,
5289
+ ): string | undefined {
5290
+ return turn.language !== undefined &&
5291
+ !hasLocalizedEntry(table, turn.language)
5292
+ ? "en"
5293
+ : undefined;
5294
+ }
5295
+
4838
5296
  private bufferAssistantTextForTts(token: symbol, text: string): void {
4839
5297
  if (!this.streamTtsAudio || text.length === 0) {
4840
5298
  return;
@@ -4955,7 +5413,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4955
5413
  private enqueueTtsSegment(
4956
5414
  token: symbol,
4957
5415
  segment: string,
4958
- options: { countsAsFirstSegment?: boolean } = {},
5416
+ options: { countsAsFirstSegment?: boolean; language?: string } = {},
4959
5417
  ): void {
4960
5418
  const activeTurn = this.activeAssistantTurn;
4961
5419
  if (activeTurn?.token !== token || !this.streamTtsAudio) {
@@ -4969,6 +5427,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4969
5427
  }
4970
5428
  const job: TtsSegmentJob = {
4971
5429
  text: segment,
5430
+ language: options.language,
4972
5431
  started: false,
4973
5432
  settled: false,
4974
5433
  emitting: false,
@@ -5008,10 +5467,14 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
5008
5467
  return;
5009
5468
  }
5010
5469
  job.started = true;
5470
+ // The segment's own language override (fixed English fallback text)
5471
+ // wins over the turn's language.
5472
+ const language = job.language ?? activeTurn.language;
5011
5473
  let synthesis: Promise<void>;
5012
5474
  try {
5013
5475
  synthesis = streamTtsAudio({
5014
5476
  text: job.text,
5477
+ ...(language !== undefined ? { language } : {}),
5015
5478
  signal: activeTurn.abortController.signal,
5016
5479
  outputFormat: "pcm",
5017
5480
  sampleRate: this.context.startFrame.audio.sampleRate,
@@ -5151,6 +5614,10 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
5151
5614
  chunk.sampleRate,
5152
5615
  );
5153
5616
  const now = Date.now();
5617
+ if (!this.isAssistantPlaybackEchoPossible()) {
5618
+ this.resetEchoReference();
5619
+ }
5620
+ this.appendEchoReference(chunk);
5154
5621
  this.assistantPlaybackTailUntilMs =
5155
5622
  Math.max(now, this.assistantPlaybackTailUntilMs) + chunkMs;
5156
5623
  const turnAfterSend = this.activeAssistantTurn;
@@ -5655,6 +6122,11 @@ export function createLiveVoiceSession(
5655
6122
  options.speechEnergyThreshold ?? vadConfig?.speechEnergyThreshold,
5656
6123
  bargeInMinSpeechMs:
5657
6124
  options.bargeInMinSpeechMs ?? vadConfig?.bargeInMinSpeechMs,
6125
+ echoBargeInMargin:
6126
+ options.echoBargeInMargin ?? vadConfig?.echoBargeInMargin,
6127
+ echoEmaHalfLifeMs:
6128
+ options.echoEmaHalfLifeMs ?? vadConfig?.echoEmaHalfLifeMs,
6129
+ echoDrainSlackMs: options.echoDrainSlackMs ?? vadConfig?.echoDrainSlackMs,
5658
6130
  frontModelConfig,
5659
6131
  // Eager construction is safe even when the `liveVoice.frontModel` config
5660
6132
  // namespace is absent — schema defaults fill the tunables. An explicit