@vellumai/assistant 0.11.3-staging.2 → 0.11.3-staging.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/__tests__/call-controller.test.ts +120 -0
- package/src/__tests__/config-loader-backfill.test.ts +19 -0
- package/src/__tests__/config-schema.test.ts +142 -5
- package/src/__tests__/events-dev-bypass-actor.test.ts +112 -1
- package/src/__tests__/media-stream-output.test.ts +175 -0
- package/src/__tests__/media-stream-stt-session.test.ts +67 -0
- package/src/calls/__tests__/tts-text-sanitizer.test.ts +13 -0
- package/src/calls/__tests__/voice-session-bridge.test.ts +81 -12
- package/src/calls/__tests__/voice-triage-escalate.test.ts +106 -0
- package/src/calls/call-controller.ts +30 -2
- package/src/calls/call-speech-output.ts +12 -4
- package/src/calls/call-transport.ts +18 -1
- package/src/calls/media-stream-output.ts +53 -5
- package/src/calls/media-stream-server.ts +12 -0
- package/src/calls/media-stream-stt-session.ts +34 -0
- package/src/calls/telephony-synthesis-language.ts +84 -0
- package/src/calls/tts-text-sanitizer.ts +12 -5
- package/src/calls/voice-session-bridge.ts +31 -3
- package/src/calls/voice-triage-escalate.ts +52 -5
- package/src/config/bundled-skills/phone-calls/references/CONFIG.md +15 -15
- package/src/config/loader.ts +5 -0
- package/src/config/schemas/__tests__/live-voice.test.ts +35 -0
- package/src/config/schemas/calls.ts +0 -4
- package/src/config/schemas/live-voice.ts +25 -0
- package/src/config/schemas/tts.ts +63 -0
- package/src/live-voice/__tests__/front-decision.test.ts +120 -0
- package/src/live-voice/__tests__/live-voice-events.test.ts +1 -1
- package/src/live-voice/__tests__/live-voice-integration.test.ts +5 -0
- package/src/live-voice/__tests__/live-voice-progress.test.ts +178 -11
- package/src/live-voice/__tests__/live-voice-stt.test.ts +304 -1
- package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +102 -2
- package/src/live-voice/__tests__/live-voice-tts.test.ts +148 -2
- package/src/live-voice/__tests__/live-voice-vad.test.ts +278 -1
- package/src/live-voice/__tests__/progress-phrases.test.ts +167 -0
- package/src/live-voice/front-decision.ts +50 -3
- package/src/live-voice/live-voice-session.ts +517 -45
- package/src/live-voice/live-voice-tts.ts +18 -2
- package/src/live-voice/progress-phrases.ts +105 -2
- package/src/providers/speech-to-text/deepgram-realtime.test.ts +283 -1
- package/src/providers/speech-to-text/deepgram-realtime.ts +117 -3
- package/src/providers/speech-to-text/provider-catalog.ts +38 -0
- package/src/runtime/__tests__/local-actor-identity-force-refresh.test.ts +104 -0
- package/src/runtime/assistant-event-hub.ts +23 -0
- package/src/runtime/local-actor-identity.ts +18 -5
- package/src/runtime/routes/__tests__/sse-actor-principal-heal.test.ts +165 -0
- package/src/runtime/routes/__tests__/surface-action-routes.test.ts +2 -0
- package/src/runtime/routes/events-routes.ts +17 -16
- package/src/runtime/routes/sse-actor-principal-heal.ts +113 -0
- package/src/stt/__tests__/language-metadata.test.ts +85 -0
- package/src/stt/__tests__/speech-energy.test.ts +79 -0
- package/src/stt/language-metadata.ts +65 -0
- package/src/stt/speech-energy.ts +115 -12
- package/src/stt/types.ts +16 -0
- package/src/tts/__tests__/provider-adapters.test.ts +147 -0
- package/src/tts/__tests__/speakable-segments.test.ts +470 -0
- package/src/tts/language-voices.ts +23 -0
- package/src/tts/providers/deepgram-provider.ts +3 -1
- package/src/tts/providers/elevenlabs-provider.ts +73 -1
- package/src/tts/providers/xai-provider.ts +28 -2
- package/src/tts/speakable-segments.ts +293 -23
- package/src/tts/synthesis-stream.ts +7 -0
- package/src/tts/types.ts +7 -0
- package/src/util/__tests__/language-subtag.test.ts +54 -0
- package/src/util/language-subtag.ts +43 -0
- package/src/util/unicode.ts +1 -1
|
@@ -25,7 +25,8 @@ import {
|
|
|
25
25
|
classifyFrontDoorLeading,
|
|
26
26
|
ESCALATE_VERDICT_TOKEN,
|
|
27
27
|
ESCALATION_CONTINUATION_CONTENT,
|
|
28
|
-
|
|
28
|
+
FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
|
|
29
|
+
fallbackEscalationBridgeFor,
|
|
29
30
|
isEscalationBridgeComplete,
|
|
30
31
|
MIN_SPOKEN_BRIDGE_CHARS,
|
|
31
32
|
type VoiceRoutingLeg,
|
|
@@ -46,14 +47,24 @@ import { isInstalledStaticSkillLoad } from "../permissions/checker.js";
|
|
|
46
47
|
import { ensureConversationExists } from "../persistence/conversation-crud.js";
|
|
47
48
|
import {
|
|
48
49
|
listProviderIds,
|
|
50
|
+
pinnedListeningLanguage,
|
|
49
51
|
supportsBoundary,
|
|
50
52
|
} from "../providers/speech-to-text/provider-catalog.js";
|
|
51
53
|
import type { ResolveStreamingTranscriberOptions } from "../providers/speech-to-text/resolve.js";
|
|
52
54
|
import { broadcastMessage } from "../runtime/assistant-event-hub.js";
|
|
53
55
|
import { publishConversationListAndMetadataChanged } from "../runtime/sync/resource-sync-events.js";
|
|
54
|
-
import {
|
|
56
|
+
import {
|
|
57
|
+
dominantLanguageTag,
|
|
58
|
+
voteDominantLanguage,
|
|
59
|
+
} from "../stt/language-metadata.js";
|
|
60
|
+
import {
|
|
61
|
+
DEFAULT_SPEECH_ENERGY_THRESHOLD,
|
|
62
|
+
pcm16MaxNormalizedCorrelation,
|
|
63
|
+
pcm16MeanAmplitude,
|
|
64
|
+
} from "../stt/speech-energy.js";
|
|
55
65
|
import type {
|
|
56
66
|
StreamingTranscriber,
|
|
67
|
+
SttProviderId,
|
|
57
68
|
SttStreamServerErrorEvent,
|
|
58
69
|
SttStreamServerEvent,
|
|
59
70
|
} from "../stt/types.js";
|
|
@@ -62,6 +73,7 @@ import { liveVoiceEndScreen } from "../telemetry/live-voice-funnel.js";
|
|
|
62
73
|
import { getToolOwner } from "../tools/registry.js";
|
|
63
74
|
import { extractSpeakableSegments } from "../tts/speakable-segments.js";
|
|
64
75
|
import { createAbortReason } from "../util/abort-reasons.js";
|
|
76
|
+
import { hasLocalizedEntry } from "../util/language-subtag.js";
|
|
65
77
|
import { getLogger } from "../util/logger.js";
|
|
66
78
|
import {
|
|
67
79
|
activityLabelForTool,
|
|
@@ -101,7 +113,12 @@ import type {
|
|
|
101
113
|
LiveVoiceTtsOptions,
|
|
102
114
|
LiveVoiceTtsResult,
|
|
103
115
|
} from "./live-voice-tts.js";
|
|
104
|
-
import {
|
|
116
|
+
import {
|
|
117
|
+
APPROVAL_PENDING_PHRASE_BY_LANGUAGE,
|
|
118
|
+
approvalPendingPhraseFor,
|
|
119
|
+
pickProgressPhrase,
|
|
120
|
+
PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE,
|
|
121
|
+
} from "./progress-phrases.js";
|
|
105
122
|
import {
|
|
106
123
|
type LiveVoiceClientAttachImageFrame,
|
|
107
124
|
type LiveVoiceClientFrame,
|
|
@@ -119,6 +136,13 @@ type LiveVoiceSessionState =
|
|
|
119
136
|
| "failed"
|
|
120
137
|
| "closed";
|
|
121
138
|
|
|
139
|
+
type VadEnergyClassification = "speech" | "silence" | "echo";
|
|
140
|
+
|
|
141
|
+
interface VadClassifiedChunk {
|
|
142
|
+
readonly chunk: Buffer;
|
|
143
|
+
readonly classification: VadEnergyClassification;
|
|
144
|
+
}
|
|
145
|
+
|
|
122
146
|
// Cap on audio buffered while a server-VAD utterance waits for its
|
|
123
147
|
// transcriber (PCM16 mono seconds; oldest chunks are dropped past the cap).
|
|
124
148
|
const SERVER_VAD_PENDING_AUDIO_MAX_SECONDS = 10;
|
|
@@ -138,6 +162,25 @@ const FINALIZE_GRACE_MS = 1_000;
|
|
|
138
162
|
// liveVoice.vad.bargeInMinSpeechMs schema default; 0 disables the guard for
|
|
139
163
|
// instant barge-in.
|
|
140
164
|
const DEFAULT_BARGE_IN_MIN_SPEECH_MS = 250;
|
|
165
|
+
// The playback echo gate learns microphone energy while assistant audio is
|
|
166
|
+
// expected at the speaker. Input must rise above the learned level by this
|
|
167
|
+
// margin to count as user speech.
|
|
168
|
+
const DEFAULT_ECHO_BARGE_IN_MARGIN = 1.5;
|
|
169
|
+
const DEFAULT_ECHO_EMA_HALF_LIFE_MS = 400;
|
|
170
|
+
// Before learning a microphone power baseline, compare a short input window
|
|
171
|
+
// with the PCM sent to the speaker. This keeps a user's first interruption
|
|
172
|
+
// from becoming its own echo threshold.
|
|
173
|
+
const ECHO_CORRELATION_PROBE_MS = 100;
|
|
174
|
+
const ECHO_CORRELATION_MIN_MS = 50;
|
|
175
|
+
const ECHO_CORRELATION_THRESHOLD = 0.65;
|
|
176
|
+
const ECHO_REFERENCE_MAX_MS = 10_000;
|
|
177
|
+
// Echo should reach the microphone near playback onset. If no signal arrives
|
|
178
|
+
// within this much input audio, the gate returns to the fixed base threshold.
|
|
179
|
+
// The same interval expires a learned reference after a real silent gap.
|
|
180
|
+
const ECHO_ONSET_ELIGIBILITY_MS = 300;
|
|
181
|
+
// Client buffering makes audible playback trail the server's send-time
|
|
182
|
+
// estimate. Keep the echo window open briefly past that estimate.
|
|
183
|
+
const DEFAULT_ECHO_DRAIN_SLACK_MS = 300;
|
|
141
184
|
// Mirrors MediaTurnDetector's DEFAULT_SILENCE_THRESHOLD_MS: the session
|
|
142
185
|
// tracks the effective trailing-silence threshold (the detector keeps its own
|
|
143
186
|
// copy private) so the endpoint decider can report the pause length.
|
|
@@ -268,6 +311,17 @@ export interface LiveVoiceSessionOptions {
|
|
|
268
311
|
* defaults to `DEFAULT_BARGE_IN_MIN_SPEECH_MS`.
|
|
269
312
|
*/
|
|
270
313
|
bargeInMinSpeechMs?: number;
|
|
314
|
+
/**
|
|
315
|
+
* Multiplier over the learned playback echo level that input must exceed
|
|
316
|
+
* to count as speech while assistant audio is playing. Values at or below
|
|
317
|
+
* 1 disable adaptation for internal fixed-gate callers; workspace config
|
|
318
|
+
* requires a value greater than 1.
|
|
319
|
+
*/
|
|
320
|
+
echoBargeInMargin?: number;
|
|
321
|
+
/** Half-life in milliseconds for the learned playback echo level. */
|
|
322
|
+
echoEmaHalfLifeMs?: number;
|
|
323
|
+
/** Extra time after the playback estimate during which echo is expected. */
|
|
324
|
+
echoDrainSlackMs?: number;
|
|
271
325
|
/**
|
|
272
326
|
* Overrides the bounded wait for the shared transcriber's finalize
|
|
273
327
|
* flush in persistent mode (test hook). Defaults to `FINALIZE_GRACE_MS`.
|
|
@@ -369,6 +423,23 @@ interface UtteranceCycle {
|
|
|
369
423
|
// text replays the boundary immediately — the hold was judged on stale
|
|
370
424
|
// text, so waiting out the extension only adds silence.
|
|
371
425
|
heldSpeculativeContent: string | null;
|
|
426
|
+
// Count per detected-language base subtag (see voteDominantLanguage)
|
|
427
|
+
// across this cycle's final transcript events. Resolves the turn's spoken
|
|
428
|
+
// language (see turnLanguageFor); empty when the provider tags nothing.
|
|
429
|
+
languageTally: Map<string, number>;
|
|
430
|
+
// Detected languages of the most recent partial that carried any, already
|
|
431
|
+
// normalized, dominance order. Speculative turns dispatch from partials
|
|
432
|
+
// before the first tagged final lands, so turnLanguageFor falls back to
|
|
433
|
+
// this when the final tally is still empty. Never cleared: the tally
|
|
434
|
+
// outranks it once finals arrive, and a revising partial without tags
|
|
435
|
+
// must not wipe an earlier partial's detection.
|
|
436
|
+
latestPartialLanguages: readonly string[] | null;
|
|
437
|
+
// The provider that actually transcribed this cycle, recorded when its
|
|
438
|
+
// transcriber is assigned and kept after teardown nulls `transcriber`.
|
|
439
|
+
// The resolver can silently dial managed vellum when a BYOK provider has
|
|
440
|
+
// no credential, so the language-pin gate in turnLanguageFor must follow
|
|
441
|
+
// this, not the configured provider.
|
|
442
|
+
dialedSttProvider: SttProviderId | null;
|
|
372
443
|
turnId: string | null;
|
|
373
444
|
userMessageId: string | null;
|
|
374
445
|
userAudioChunks: Buffer[];
|
|
@@ -399,6 +470,11 @@ type UtteranceStartResult =
|
|
|
399
470
|
// client in job-list order.
|
|
400
471
|
interface TtsSegmentJob {
|
|
401
472
|
readonly text: string;
|
|
473
|
+
// Per-segment language-hint override, preferred over the turn's language.
|
|
474
|
+
// Set on fixed phrases whose localized table lacks the turn's language:
|
|
475
|
+
// the English fallback text carries "en" so an enforcing provider never
|
|
476
|
+
// renders English words as ar/ko/ta. Undefined means the turn language.
|
|
477
|
+
readonly language: string | undefined;
|
|
402
478
|
// The provider stream was started (the job holds an open-job slot).
|
|
403
479
|
started: boolean;
|
|
404
480
|
// Emission finished; the slot is free for the next queued segment.
|
|
@@ -483,6 +559,13 @@ interface ActiveAssistantTurn {
|
|
|
483
559
|
token: symbol;
|
|
484
560
|
turnId: string;
|
|
485
561
|
utterance: UtteranceCycle;
|
|
562
|
+
// The caller's spoken language for this turn as a lowercase base subtag
|
|
563
|
+
// (see turnLanguageFor): the dominant STT-detected language, else a
|
|
564
|
+
// monolingual services.stt.language pin. Undefined when unknown, which
|
|
565
|
+
// disables every language-aware path (prompt note, TTS hint, localized
|
|
566
|
+
// fallbacks). Re-resolved when a speculative turn commits, since finals
|
|
567
|
+
// can land between dispatch and verdict.
|
|
568
|
+
language: string | undefined;
|
|
486
569
|
abortController: AbortController;
|
|
487
570
|
handle: VoiceTurnHandle | null;
|
|
488
571
|
// When the turn launched, for narration's turnElapsedMs.
|
|
@@ -681,14 +764,8 @@ function createControlMarkerHoldback(
|
|
|
681
764
|
// barge-in, the interruption merge note is appended to it (see
|
|
682
765
|
// buildInterruptionMergeNote) so the model reconciles the interrupted request
|
|
683
766
|
// with the new utterance.
|
|
684
|
-
// Spoken once when a turn starts waiting on the user's decision. Fixed rather
|
|
685
|
-
// than generated: this is a statement about the system's state, not about the
|
|
686
|
-
// work, and it has to be true every time. Kept in the shape of the progress
|
|
687
|
-
// phrases it displaces (short, neutral, no claim about tools).
|
|
688
|
-
const APPROVAL_PENDING_PHRASE = "I need your okay for that one. Take a look.";
|
|
689
|
-
|
|
690
767
|
const LIVE_VOICE_CONTROL_PROMPT_BASE =
|
|
691
|
-
"You are speaking in a local live voice session. Keep replies brief and conversational. Speech is the main channel: say the answer, and do not narrate a surface instead of answering. You can also put something on screen when it genuinely helps (a form, a list to pick from, a progress card for long work); the call overlay minimizes by itself once you finish speaking, so the user sees it without doing anything. Never tell the user you cannot show them something. ";
|
|
768
|
+
"You are speaking in a local live voice session. Keep replies brief and conversational. Speech is the main channel: say the answer, and do not narrate a surface instead of answering. You can also put something on screen when it genuinely helps (a form, a list to pick from, a progress card for long work); the call overlay minimizes by itself once you finish speaking, so the user sees it without doing anything. Never tell the user you cannot show them something. Reply in the language the caller is speaking; if they switch languages, switch with them. ";
|
|
692
769
|
|
|
693
770
|
// Appended for the legs that can actually put something on screen: the main
|
|
694
771
|
// leg and the escalated leg. The front-door (fast) leg never receives it, for
|
|
@@ -777,6 +854,9 @@ function buildVoiceControlPrompt(
|
|
|
777
854
|
LIVE_VOICE_CONTROL_PROMPT_BASE +
|
|
778
855
|
(leg.frontDoor === true ? "" : LIVE_VOICE_SCREEN_REVEAL_TEACHING) +
|
|
779
856
|
VOICE_NO_SETUP_FLOWS_RULE;
|
|
857
|
+
if (turn.language !== undefined) {
|
|
858
|
+
prompt = `${prompt}\n\nThe caller has been speaking the language with code "${turn.language}" this turn. Reply in that language unless they clearly switch to another.`;
|
|
859
|
+
}
|
|
780
860
|
if (turn.interruptedRequest) {
|
|
781
861
|
prompt = `${prompt}\n\n${buildInterruptionMergeNote(turn.interruptedRequest)}`;
|
|
782
862
|
}
|
|
@@ -825,6 +905,9 @@ function createUtteranceCycle(): UtteranceCycle {
|
|
|
825
905
|
latestPartialText: null,
|
|
826
906
|
endpointExtensionCount: 0,
|
|
827
907
|
heldSpeculativeContent: null,
|
|
908
|
+
languageTally: new Map(),
|
|
909
|
+
latestPartialLanguages: null,
|
|
910
|
+
dialedSttProvider: null,
|
|
828
911
|
turnId: null,
|
|
829
912
|
userMessageId: null,
|
|
830
913
|
userAudioChunks: [],
|
|
@@ -991,9 +1074,29 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
991
1074
|
private failureCode: LiveVoiceProtocolErrorCode | null = null;
|
|
992
1075
|
// Non-null iff the start frame requested turnDetection "server_vad".
|
|
993
1076
|
private readonly turnDetector: MediaTurnDetector | null;
|
|
994
|
-
//
|
|
995
|
-
//
|
|
1077
|
+
// Base energy gate for server-VAD speech classification. During estimated
|
|
1078
|
+
// playback, classifyVadEnergy raises this above the learned echo level.
|
|
996
1079
|
private readonly speechEnergyThreshold: number | undefined;
|
|
1080
|
+
private readonly echoBargeInMargin: number;
|
|
1081
|
+
private readonly echoEmaHalfLifeMs: number;
|
|
1082
|
+
private readonly echoDrainSlackMs: number;
|
|
1083
|
+
// Learned microphone energy attributable to assistant playback.
|
|
1084
|
+
private echoEnergyEma = 0;
|
|
1085
|
+
// Signal-bearing microphone audio held until it can be compared with the
|
|
1086
|
+
// assistant PCM. A nonmatch is replayed through VAD in original order.
|
|
1087
|
+
private echoProbeChunks: Buffer[] = [];
|
|
1088
|
+
// Recent raw assistant PCM from the current playback burst.
|
|
1089
|
+
private echoReferenceAudio = Buffer.alloc(0);
|
|
1090
|
+
private echoWindowTotalAudioMs = 0;
|
|
1091
|
+
// Consecutive sub-base input expires a reference that can no longer
|
|
1092
|
+
// describe audible playback.
|
|
1093
|
+
private echoSubBaseRunMs = 0;
|
|
1094
|
+
// Once onset eligibility lapses, later user speech cannot seed a new echo
|
|
1095
|
+
// reference in the same playback window.
|
|
1096
|
+
private echoOnsetLapsed = false;
|
|
1097
|
+
// A live speech run that predates playback belongs to the user and bypasses
|
|
1098
|
+
// echo warm-up until that run genuinely resets.
|
|
1099
|
+
private echoWindowGuardCarryover = false;
|
|
997
1100
|
// Mutable so a mid-session `update_config` frame can retune "interrupt
|
|
998
1101
|
// sensitivity" live (see applyConfigUpdate).
|
|
999
1102
|
private bargeInMinSpeechMs: number;
|
|
@@ -1166,6 +1269,12 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
1166
1269
|
context.startFrame.bargeInMinSpeechMs ??
|
|
1167
1270
|
options.bargeInMinSpeechMs ??
|
|
1168
1271
|
DEFAULT_BARGE_IN_MIN_SPEECH_MS;
|
|
1272
|
+
this.echoBargeInMargin =
|
|
1273
|
+
options.echoBargeInMargin ?? DEFAULT_ECHO_BARGE_IN_MARGIN;
|
|
1274
|
+
this.echoEmaHalfLifeMs =
|
|
1275
|
+
options.echoEmaHalfLifeMs ?? DEFAULT_ECHO_EMA_HALF_LIFE_MS;
|
|
1276
|
+
this.echoDrainSlackMs =
|
|
1277
|
+
options.echoDrainSlackMs ?? DEFAULT_ECHO_DRAIN_SLACK_MS;
|
|
1169
1278
|
this.finalizeGraceMs = options.finalizeGraceMs ?? FINALIZE_GRACE_MS;
|
|
1170
1279
|
this.frontDecider = options.frontDecider ?? null;
|
|
1171
1280
|
this.frontModelConfig = LiveVoiceFrontModelConfigSchema.parse(
|
|
@@ -1387,6 +1496,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
1387
1496
|
// Persistent re-arm: the shared stream is already open, so the cycle
|
|
1388
1497
|
// goes straight to streaming with no resolve/start round-trip.
|
|
1389
1498
|
utterance.transcriber = shared;
|
|
1499
|
+
utterance.dialedSttProvider = shared.providerId;
|
|
1390
1500
|
return await this.activateUtterance(utterance, replayTurnEnd);
|
|
1391
1501
|
}
|
|
1392
1502
|
// The shared stream is pinned to the old language, so retire it and
|
|
@@ -1419,6 +1529,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
1419
1529
|
}
|
|
1420
1530
|
|
|
1421
1531
|
utterance.transcriber = transcriber;
|
|
1532
|
+
utterance.dialedSttProvider = transcriber.providerId;
|
|
1422
1533
|
if (
|
|
1423
1534
|
this.turnDetector &&
|
|
1424
1535
|
typeof transcriber.finalizeUtterance === "function"
|
|
@@ -1639,12 +1750,26 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
1639
1750
|
return;
|
|
1640
1751
|
}
|
|
1641
1752
|
|
|
1642
|
-
const
|
|
1643
|
-
|
|
1644
|
-
|
|
1645
|
-
|
|
1753
|
+
for (const classified of this.classifyVadEnergy(chunk)) {
|
|
1754
|
+
await this.handleClassifiedVadAudio(detector, classified);
|
|
1755
|
+
}
|
|
1756
|
+
}
|
|
1757
|
+
|
|
1758
|
+
private async handleClassifiedVadAudio(
|
|
1759
|
+
detector: MediaTurnDetector,
|
|
1760
|
+
classified: VadClassifiedChunk,
|
|
1761
|
+
): Promise<void> {
|
|
1762
|
+
const { chunk, classification: energyClassification } = classified;
|
|
1763
|
+
const hasSpeech = energyClassification === "speech";
|
|
1646
1764
|
detector.onMediaChunk(hasSpeech);
|
|
1647
|
-
this.trackBargeInGuard(
|
|
1765
|
+
this.trackBargeInGuard(energyClassification, chunk);
|
|
1766
|
+
|
|
1767
|
+
// Playback echo is neither user audio nor useful pre-roll. Dropping it
|
|
1768
|
+
// prevents the assistant's reply from reaching transcription as a ghost
|
|
1769
|
+
// follow-up turn.
|
|
1770
|
+
if (energyClassification === "echo") {
|
|
1771
|
+
return;
|
|
1772
|
+
}
|
|
1648
1773
|
|
|
1649
1774
|
// Idle mic: hold silent chunks in the bounded pre-roll instead of
|
|
1650
1775
|
// collecting or streaming them; flushed on speech onset so the
|
|
@@ -1722,6 +1847,193 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
1722
1847
|
await this.routeVadAudio(utterance, chunk);
|
|
1723
1848
|
}
|
|
1724
1849
|
|
|
1850
|
+
/**
|
|
1851
|
+
* Classify microphone energy while keeping assistant playback echo out of
|
|
1852
|
+
* barge-in, turn detection, pre-roll, and transcription.
|
|
1853
|
+
*
|
|
1854
|
+
* A short onset probe must correlate with PCM sent to the speaker before its
|
|
1855
|
+
* microphone power can seed the adaptive threshold. Nonmatching probe audio
|
|
1856
|
+
* is replayed through VAD in original order, so a user who talks at playback
|
|
1857
|
+
* onset is neither learned as echo nor lost. Once seeded, the EMA follows
|
|
1858
|
+
* confirmed echo while speech above the learned margin remains frozen out.
|
|
1859
|
+
*/
|
|
1860
|
+
private classifyVadEnergy(chunk: Buffer): VadClassifiedChunk[] {
|
|
1861
|
+
const baseThreshold =
|
|
1862
|
+
this.speechEnergyThreshold ?? DEFAULT_SPEECH_ENERGY_THRESHOLD;
|
|
1863
|
+
const meanAmplitude = pcm16MeanAmplitude(chunk);
|
|
1864
|
+
if (
|
|
1865
|
+
this.echoBargeInMargin <= 1 ||
|
|
1866
|
+
!this.isAssistantPlaybackEchoPossible()
|
|
1867
|
+
) {
|
|
1868
|
+
this.resetEchoReference();
|
|
1869
|
+
return [
|
|
1870
|
+
this.classifyAtFixedThreshold(chunk, baseThreshold, meanAmplitude),
|
|
1871
|
+
];
|
|
1872
|
+
}
|
|
1873
|
+
|
|
1874
|
+
if (this.echoWindowTotalAudioMs === 0) {
|
|
1875
|
+
this.echoWindowGuardCarryover =
|
|
1876
|
+
this.pendingBargeIn !== null && this.pendingBargeIn.speechMs > 0;
|
|
1877
|
+
} else if (this.pendingBargeIn === null) {
|
|
1878
|
+
this.echoWindowGuardCarryover = false;
|
|
1879
|
+
}
|
|
1880
|
+
|
|
1881
|
+
const chunkMs = pcm16DurationMs(
|
|
1882
|
+
chunk.byteLength,
|
|
1883
|
+
this.context.startFrame.audio.sampleRate,
|
|
1884
|
+
);
|
|
1885
|
+
const onsetWasEligible =
|
|
1886
|
+
!this.echoOnsetLapsed &&
|
|
1887
|
+
this.echoWindowTotalAudioMs < ECHO_ONSET_ELIGIBILITY_MS;
|
|
1888
|
+
this.echoWindowTotalAudioMs += chunkMs;
|
|
1889
|
+
|
|
1890
|
+
if (this.echoProbeChunks.length > 0) {
|
|
1891
|
+
this.echoProbeChunks.push(Buffer.from(chunk));
|
|
1892
|
+
return this.resolveEchoProbe(baseThreshold);
|
|
1893
|
+
}
|
|
1894
|
+
|
|
1895
|
+
if (meanAmplitude <= baseThreshold) {
|
|
1896
|
+
this.echoSubBaseRunMs += chunkMs;
|
|
1897
|
+
if (this.echoSubBaseRunMs >= ECHO_ONSET_ELIGIBILITY_MS) {
|
|
1898
|
+
this.echoEnergyEma = 0;
|
|
1899
|
+
this.echoOnsetLapsed = true;
|
|
1900
|
+
}
|
|
1901
|
+
return [{ chunk, classification: "silence" }];
|
|
1902
|
+
}
|
|
1903
|
+
|
|
1904
|
+
this.echoSubBaseRunMs = 0;
|
|
1905
|
+
if (
|
|
1906
|
+
this.echoEnergyEma === 0 &&
|
|
1907
|
+
onsetWasEligible &&
|
|
1908
|
+
!this.echoWindowGuardCarryover
|
|
1909
|
+
) {
|
|
1910
|
+
this.echoProbeChunks.push(Buffer.from(chunk));
|
|
1911
|
+
return this.resolveEchoProbe(baseThreshold);
|
|
1912
|
+
}
|
|
1913
|
+
|
|
1914
|
+
if (this.echoEnergyEma === 0) {
|
|
1915
|
+
this.echoOnsetLapsed = true;
|
|
1916
|
+
return [{ chunk, classification: "speech" }];
|
|
1917
|
+
}
|
|
1918
|
+
|
|
1919
|
+
const speechThreshold = Math.max(
|
|
1920
|
+
baseThreshold,
|
|
1921
|
+
this.echoBargeInMargin * this.echoEnergyEma,
|
|
1922
|
+
);
|
|
1923
|
+
if (meanAmplitude > speechThreshold) {
|
|
1924
|
+
const guardHasSpeech =
|
|
1925
|
+
this.pendingBargeIn !== null && this.pendingBargeIn.speechMs > 0;
|
|
1926
|
+
if (!guardHasSpeech && this.echoMatchesAssistant(chunk)) {
|
|
1927
|
+
this.updateEchoEnergy(meanAmplitude, chunkMs);
|
|
1928
|
+
return [{ chunk, classification: "echo" }];
|
|
1929
|
+
}
|
|
1930
|
+
return [{ chunk, classification: "speech" }];
|
|
1931
|
+
}
|
|
1932
|
+
|
|
1933
|
+
this.updateEchoEnergy(meanAmplitude, chunkMs);
|
|
1934
|
+
return [{ chunk, classification: "echo" }];
|
|
1935
|
+
}
|
|
1936
|
+
|
|
1937
|
+
private resolveEchoProbe(baseThreshold: number): VadClassifiedChunk[] {
|
|
1938
|
+
const probe = Buffer.concat(this.echoProbeChunks);
|
|
1939
|
+
const probeAudioMs = pcm16DurationMs(
|
|
1940
|
+
probe.byteLength,
|
|
1941
|
+
this.context.startFrame.audio.sampleRate,
|
|
1942
|
+
);
|
|
1943
|
+
if (
|
|
1944
|
+
probeAudioMs >= ECHO_CORRELATION_MIN_MS &&
|
|
1945
|
+
this.echoMatchesAssistant(probe)
|
|
1946
|
+
) {
|
|
1947
|
+
this.echoEnergyEma = Math.max(baseThreshold, pcm16MeanAmplitude(probe));
|
|
1948
|
+
const chunks = this.echoProbeChunks.splice(0);
|
|
1949
|
+
return chunks.map((chunk) => ({ chunk, classification: "echo" }));
|
|
1950
|
+
}
|
|
1951
|
+
if (probeAudioMs < ECHO_CORRELATION_PROBE_MS) {
|
|
1952
|
+
return [];
|
|
1953
|
+
}
|
|
1954
|
+
|
|
1955
|
+
this.echoOnsetLapsed = true;
|
|
1956
|
+
const chunks = this.echoProbeChunks.splice(0);
|
|
1957
|
+
return chunks.map((chunk) =>
|
|
1958
|
+
this.classifyAtFixedThreshold(chunk, baseThreshold),
|
|
1959
|
+
);
|
|
1960
|
+
}
|
|
1961
|
+
|
|
1962
|
+
private echoMatchesAssistant(chunk: Buffer): boolean {
|
|
1963
|
+
const sampleRate = this.context.startFrame.audio.sampleRate;
|
|
1964
|
+
const minimumBytes = Math.ceil(
|
|
1965
|
+
(sampleRate * ECHO_CORRELATION_MIN_MS * 2) / 1_000,
|
|
1966
|
+
);
|
|
1967
|
+
if (
|
|
1968
|
+
chunk.byteLength < minimumBytes ||
|
|
1969
|
+
this.echoReferenceAudio.byteLength < minimumBytes
|
|
1970
|
+
) {
|
|
1971
|
+
return false;
|
|
1972
|
+
}
|
|
1973
|
+
const probeByteLength = Math.min(
|
|
1974
|
+
chunk.byteLength,
|
|
1975
|
+
Math.ceil((sampleRate * ECHO_CORRELATION_PROBE_MS * 2) / 1_000),
|
|
1976
|
+
);
|
|
1977
|
+
return (
|
|
1978
|
+
pcm16MaxNormalizedCorrelation(
|
|
1979
|
+
chunk.subarray(0, probeByteLength),
|
|
1980
|
+
this.echoReferenceAudio,
|
|
1981
|
+
) >= ECHO_CORRELATION_THRESHOLD
|
|
1982
|
+
);
|
|
1983
|
+
}
|
|
1984
|
+
|
|
1985
|
+
private updateEchoEnergy(meanAmplitude: number, chunkMs: number): void {
|
|
1986
|
+
const alpha = 1 - 0.5 ** (chunkMs / this.echoEmaHalfLifeMs);
|
|
1987
|
+
this.echoEnergyEma =
|
|
1988
|
+
alpha * meanAmplitude + (1 - alpha) * this.echoEnergyEma;
|
|
1989
|
+
}
|
|
1990
|
+
|
|
1991
|
+
private classifyAtFixedThreshold(
|
|
1992
|
+
chunk: Buffer,
|
|
1993
|
+
baseThreshold: number,
|
|
1994
|
+
meanAmplitude = pcm16MeanAmplitude(chunk),
|
|
1995
|
+
): VadClassifiedChunk {
|
|
1996
|
+
return {
|
|
1997
|
+
chunk,
|
|
1998
|
+
classification: meanAmplitude > baseThreshold ? "speech" : "silence",
|
|
1999
|
+
};
|
|
2000
|
+
}
|
|
2001
|
+
|
|
2002
|
+
private isAssistantPlaybackEchoPossible(): boolean {
|
|
2003
|
+
return (
|
|
2004
|
+
Date.now() < this.assistantPlaybackTailUntilMs + this.echoDrainSlackMs
|
|
2005
|
+
);
|
|
2006
|
+
}
|
|
2007
|
+
|
|
2008
|
+
private resetEchoReference(): void {
|
|
2009
|
+
this.echoEnergyEma = 0;
|
|
2010
|
+
this.echoProbeChunks = [];
|
|
2011
|
+
this.echoReferenceAudio = Buffer.alloc(0);
|
|
2012
|
+
this.echoWindowTotalAudioMs = 0;
|
|
2013
|
+
this.echoSubBaseRunMs = 0;
|
|
2014
|
+
this.echoOnsetLapsed = false;
|
|
2015
|
+
this.echoWindowGuardCarryover = false;
|
|
2016
|
+
}
|
|
2017
|
+
|
|
2018
|
+
private appendEchoReference(chunk: LiveVoiceTtsAudioChunk): void {
|
|
2019
|
+
if (
|
|
2020
|
+
chunk.contentType.split(";", 1)[0]?.trim().toLowerCase() !==
|
|
2021
|
+
"audio/pcm" ||
|
|
2022
|
+
chunk.sampleRate !== this.context.startFrame.audio.sampleRate
|
|
2023
|
+
) {
|
|
2024
|
+
return;
|
|
2025
|
+
}
|
|
2026
|
+
const audio = Buffer.from(chunk.dataBase64, "base64");
|
|
2027
|
+
const maxBytes = Math.ceil(
|
|
2028
|
+
(chunk.sampleRate * ECHO_REFERENCE_MAX_MS * 2) / 1_000,
|
|
2029
|
+
);
|
|
2030
|
+
const combined = Buffer.concat([this.echoReferenceAudio, audio]);
|
|
2031
|
+
this.echoReferenceAudio =
|
|
2032
|
+
combined.byteLength > maxBytes
|
|
2033
|
+
? combined.subarray(combined.byteLength - maxBytes)
|
|
2034
|
+
: combined;
|
|
2035
|
+
}
|
|
2036
|
+
|
|
1725
2037
|
private async routeVadAudio(
|
|
1726
2038
|
utterance: UtteranceCycle,
|
|
1727
2039
|
chunk: Buffer,
|
|
@@ -1869,7 +2181,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
1869
2181
|
// The client can still be draining audible playback after tts_done
|
|
1870
2182
|
// (the turn is already cleared server-side) — that tail deserves the
|
|
1871
2183
|
// same guard, or a noise blip clips the reply's last words.
|
|
1872
|
-
const drainingPlayback =
|
|
2184
|
+
const drainingPlayback = this.isAssistantPlaybackEchoPossible();
|
|
1873
2185
|
|
|
1874
2186
|
if ((bargeableTurn || drainingPlayback) && this.bargeInMinSpeechMs > 0) {
|
|
1875
2187
|
// Onset audio keeps flowing into the cycle/pre-roll while the guard
|
|
@@ -1891,13 +2203,14 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
1891
2203
|
}
|
|
1892
2204
|
}
|
|
1893
2205
|
|
|
1894
|
-
//
|
|
1895
|
-
//
|
|
1896
|
-
//
|
|
1897
|
-
//
|
|
1898
|
-
|
|
1899
|
-
|
|
1900
|
-
|
|
2206
|
+
// Advance the sustained-speech barge-in guard by one server-VAD chunk.
|
|
2207
|
+
// Speech accumulates toward bargeInMinSpeechMs, short true-silence gaps are
|
|
2208
|
+
// tolerated, and classified playback echo resets the run immediately.
|
|
2209
|
+
// Longer or mostly silent runs reset through the existing gap limits.
|
|
2210
|
+
private trackBargeInGuard(
|
|
2211
|
+
classification: VadEnergyClassification,
|
|
2212
|
+
chunk: Buffer,
|
|
2213
|
+
): void {
|
|
1901
2214
|
const guard = this.pendingBargeIn;
|
|
1902
2215
|
if (!guard) {
|
|
1903
2216
|
return;
|
|
@@ -1906,7 +2219,11 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
1906
2219
|
chunk.byteLength,
|
|
1907
2220
|
this.context.startFrame.audio.sampleRate,
|
|
1908
2221
|
);
|
|
1909
|
-
if (
|
|
2222
|
+
if (classification === "echo") {
|
|
2223
|
+
this.resetBargeInGuardRun();
|
|
2224
|
+
return;
|
|
2225
|
+
}
|
|
2226
|
+
if (classification === "silence") {
|
|
1910
2227
|
guard.silenceMs += chunkMs;
|
|
1911
2228
|
guard.toleratedSilenceMs += chunkMs;
|
|
1912
2229
|
// Strictly greater on the per-gap check: a gap of exactly
|
|
@@ -1920,9 +2237,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
1920
2237
|
guard.toleratedSilenceMs >
|
|
1921
2238
|
this.bargeInMinSpeechMs * BARGE_IN_MAX_TOLERATED_SILENCE_RATIO
|
|
1922
2239
|
) {
|
|
1923
|
-
|
|
1924
|
-
guard.silenceMs = 0;
|
|
1925
|
-
guard.toleratedSilenceMs = 0;
|
|
2240
|
+
this.resetBargeInGuardRun();
|
|
1926
2241
|
}
|
|
1927
2242
|
return;
|
|
1928
2243
|
}
|
|
@@ -1940,6 +2255,21 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
1940
2255
|
}
|
|
1941
2256
|
}
|
|
1942
2257
|
|
|
2258
|
+
private resetBargeInGuardRun(): void {
|
|
2259
|
+
const guard = this.pendingBargeIn;
|
|
2260
|
+
if (!guard) {
|
|
2261
|
+
return;
|
|
2262
|
+
}
|
|
2263
|
+
guard.speechMs = 0;
|
|
2264
|
+
guard.silenceMs = 0;
|
|
2265
|
+
guard.toleratedSilenceMs = 0;
|
|
2266
|
+
if (this.echoWindowGuardCarryover) {
|
|
2267
|
+
this.echoWindowGuardCarryover = false;
|
|
2268
|
+
this.echoEnergyEma = 0;
|
|
2269
|
+
this.echoProbeChunks = [];
|
|
2270
|
+
}
|
|
2271
|
+
}
|
|
2272
|
+
|
|
1943
2273
|
private bargeIn(turn: ActiveAssistantTurn): void {
|
|
1944
2274
|
// Abort synchronously so no tts_audio frame can follow turn_cancelled,
|
|
1945
2275
|
// and settle the cancelled turn's metrics so the next utterance's marks
|
|
@@ -2527,8 +2857,13 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
2527
2857
|
// Spoken, because opening the room is only a cue for someone looking at
|
|
2528
2858
|
// the screen, and the case this exists for is a phone the user has put
|
|
2529
2859
|
// down. One line, not narration: the turn is not working, it is waiting,
|
|
2530
|
-
// and it says which
|
|
2531
|
-
|
|
2860
|
+
// and it says which, in the turn's spoken language, like every other
|
|
2861
|
+
// filler phrase.
|
|
2862
|
+
this.enqueueFillerPhrase(
|
|
2863
|
+
turn,
|
|
2864
|
+
approvalPendingPhraseFor(turn.language),
|
|
2865
|
+
this.fixedPhraseLanguage(turn, APPROVAL_PENDING_PHRASE_BY_LANGUAGE),
|
|
2866
|
+
);
|
|
2532
2867
|
}
|
|
2533
2868
|
|
|
2534
2869
|
/** Clear the wait once a decision lands, so the turn narrates normally again. */
|
|
@@ -2851,6 +3186,13 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
2851
3186
|
// are skipped; the thinking frame and timers still apply.
|
|
2852
3187
|
const alreadyReleased = utterance.released;
|
|
2853
3188
|
turn.speculativePending = false;
|
|
3189
|
+
// Finals can land between the speculative dispatch and this verdict.
|
|
3190
|
+
// Fill the language only when dispatch had none: the model request was
|
|
3191
|
+
// already issued with the dispatch language, so overwriting here would
|
|
3192
|
+
// hint TTS (and any voice override) in a different language than the
|
|
3193
|
+
// text it speaks. The tally still carries the corrected detection into
|
|
3194
|
+
// the next turn.
|
|
3195
|
+
turn.language ??= this.turnLanguageFor(utterance);
|
|
2854
3196
|
if (turn.verdictDeadlineTimer !== null) {
|
|
2855
3197
|
clearTimeout(turn.verdictDeadlineTimer);
|
|
2856
3198
|
turn.verdictDeadlineTimer = null;
|
|
@@ -3189,11 +3531,16 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3189
3531
|
switch (event.type) {
|
|
3190
3532
|
case "partial":
|
|
3191
3533
|
utterance.latestPartialText = event.text;
|
|
3534
|
+
this.capturePartialLanguages(utterance, event.languages);
|
|
3192
3535
|
this.markFirstPartial(utterance);
|
|
3193
3536
|
await this.sendFrame({ type: "stt_partial", text: event.text });
|
|
3194
3537
|
return;
|
|
3195
3538
|
case "final":
|
|
3196
|
-
await this.recordFinalTranscript(
|
|
3539
|
+
await this.recordFinalTranscript(
|
|
3540
|
+
utterance,
|
|
3541
|
+
event.text,
|
|
3542
|
+
event.languages,
|
|
3543
|
+
);
|
|
3197
3544
|
return;
|
|
3198
3545
|
case "finalized":
|
|
3199
3546
|
// Per-cycle transcribers are torn down with stop(); the finalize
|
|
@@ -3262,6 +3609,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3262
3609
|
return;
|
|
3263
3610
|
}
|
|
3264
3611
|
target.latestPartialText = event.text;
|
|
3612
|
+
this.capturePartialLanguages(target, event.languages);
|
|
3265
3613
|
this.markFirstPartial(target);
|
|
3266
3614
|
await this.sendFrame({ type: "stt_partial", text: event.text });
|
|
3267
3615
|
return;
|
|
@@ -3277,7 +3625,11 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3277
3625
|
// newer cycle.
|
|
3278
3626
|
const owner = this.finalizeQueue[0];
|
|
3279
3627
|
if (owner && !owner.assistantTurnStarted && !owner.completed) {
|
|
3280
|
-
await this.recordFinalTranscript(
|
|
3628
|
+
await this.recordFinalTranscript(
|
|
3629
|
+
owner,
|
|
3630
|
+
event.text,
|
|
3631
|
+
event.languages,
|
|
3632
|
+
);
|
|
3281
3633
|
} else {
|
|
3282
3634
|
log.warn(
|
|
3283
3635
|
"Dropping a late finalize flush segment: its assistant turn already dispatched",
|
|
@@ -3294,7 +3646,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3294
3646
|
);
|
|
3295
3647
|
return;
|
|
3296
3648
|
}
|
|
3297
|
-
await this.recordFinalTranscript(target, event.text);
|
|
3649
|
+
await this.recordFinalTranscript(target, event.text, event.languages);
|
|
3298
3650
|
return;
|
|
3299
3651
|
}
|
|
3300
3652
|
case "finalized": {
|
|
@@ -3363,10 +3715,16 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3363
3715
|
private async recordFinalTranscript(
|
|
3364
3716
|
utterance: UtteranceCycle,
|
|
3365
3717
|
text: string,
|
|
3718
|
+
languages?: readonly string[],
|
|
3366
3719
|
): Promise<void> {
|
|
3367
3720
|
const transcript = text.trim();
|
|
3368
3721
|
if (transcript.length > 0) {
|
|
3369
3722
|
utterance.finalTranscriptSegments.push(transcript);
|
|
3723
|
+
// Tally only finals that committed transcript: empty silence frames
|
|
3724
|
+
// can still carry container-level language tags describing no emitted
|
|
3725
|
+
// words, and counting those would let silence outvote real speech
|
|
3726
|
+
// (same choice as the adapter's boundary-final aggregation).
|
|
3727
|
+
voteDominantLanguage(utterance.languageTally, languages);
|
|
3370
3728
|
}
|
|
3371
3729
|
// The final commits (and supersedes) whatever partial was trailing it.
|
|
3372
3730
|
utterance.latestPartialText = null;
|
|
@@ -3405,6 +3763,59 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3405
3763
|
await this.startAssistantTurnIfReady();
|
|
3406
3764
|
}
|
|
3407
3765
|
|
|
3766
|
+
// Record a partial event's detected languages so speculative dispatch
|
|
3767
|
+
// has a detection before the first tagged final. The event contract
|
|
3768
|
+
// (stt/types.ts) guarantees the tags arrive as normalized base subtags
|
|
3769
|
+
// in dominance order, so they are stored as-is. Partials revise each
|
|
3770
|
+
// other, so this overwrites rather than tallies, and a tag-less partial
|
|
3771
|
+
// keeps the previous value.
|
|
3772
|
+
private capturePartialLanguages(
|
|
3773
|
+
utterance: UtteranceCycle,
|
|
3774
|
+
languages: readonly string[] | undefined,
|
|
3775
|
+
): void {
|
|
3776
|
+
if (!languages || languages.length === 0) {
|
|
3777
|
+
return;
|
|
3778
|
+
}
|
|
3779
|
+
utterance.latestPartialLanguages = languages;
|
|
3780
|
+
}
|
|
3781
|
+
|
|
3782
|
+
/**
|
|
3783
|
+
* The caller's spoken language for a turn on this utterance, as a
|
|
3784
|
+
* lowercase base subtag: the dominant tallied STT-detected language
|
|
3785
|
+
* (most final-event counts, ties by first appearance), else the latest
|
|
3786
|
+
* tagged partial's dominant language (speculative turns dispatch from
|
|
3787
|
+
* partials), else a monolingual `services.stt.language` pin (a pinned
|
|
3788
|
+
* language IS the spoken language), else undefined ("multi" with no tags,
|
|
3789
|
+
* non-tagging providers, silence).
|
|
3790
|
+
*/
|
|
3791
|
+
private turnLanguageFor(utterance: UtteranceCycle): string | undefined {
|
|
3792
|
+
const dominant = dominantLanguageTag(utterance.languageTally);
|
|
3793
|
+
if (dominant !== undefined) {
|
|
3794
|
+
return dominant;
|
|
3795
|
+
}
|
|
3796
|
+
// No tagged final yet (speculative turns dispatch from partials): the
|
|
3797
|
+
// latest tagged partial is the best detection available and outranks a
|
|
3798
|
+
// static pin for the same reason the tally does.
|
|
3799
|
+
const partialDominant = utterance.latestPartialLanguages?.[0];
|
|
3800
|
+
if (partialDominant !== undefined) {
|
|
3801
|
+
return partialDominant;
|
|
3802
|
+
}
|
|
3803
|
+
// A persisted pin only counts when the provider that actually
|
|
3804
|
+
// transcribed honors manual language selection (the shared
|
|
3805
|
+
// pinnedListeningLanguage gate). The DIALED transcriber's providerId
|
|
3806
|
+
// is authoritative, because the resolver silently falls back to
|
|
3807
|
+
// managed vellum (which honors the pin) when a BYOK provider has no
|
|
3808
|
+
// credential; the configured provider is only the last resort when no
|
|
3809
|
+
// transcriber reference survives.
|
|
3810
|
+
const { language: configured, provider: sttProvider } =
|
|
3811
|
+
getConfig().services.stt;
|
|
3812
|
+
const dialedProvider =
|
|
3813
|
+
utterance.dialedSttProvider ??
|
|
3814
|
+
this.sharedTranscriber?.providerId ??
|
|
3815
|
+
(sttProvider as SttProviderId);
|
|
3816
|
+
return pinnedListeningLanguage(dialedProvider, configured);
|
|
3817
|
+
}
|
|
3818
|
+
|
|
3408
3819
|
// Providers emit `error` mid-stream and may keep streaming; `closed` /
|
|
3409
3820
|
// `final` still drive turn lifecycle. Only transient categories are
|
|
3410
3821
|
// recoverable — auth/rate-limit/invalid-audio will not self-heal, so
|
|
@@ -3660,6 +4071,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3660
4071
|
token,
|
|
3661
4072
|
turnId,
|
|
3662
4073
|
utterance,
|
|
4074
|
+
language: this.turnLanguageFor(utterance),
|
|
3663
4075
|
abortController,
|
|
3664
4076
|
handle: null,
|
|
3665
4077
|
launchedAtMs: Date.now(),
|
|
@@ -4312,7 +4724,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4312
4724
|
// deleted row for a bridge the model never produced).
|
|
4313
4725
|
const usesFallbackBridge = cappedBridge.length < MIN_SPOKEN_BRIDGE_CHARS;
|
|
4314
4726
|
const spokenBridge = usesFallbackBridge
|
|
4315
|
-
?
|
|
4727
|
+
? fallbackEscalationBridgeFor(activeTurn.language)
|
|
4316
4728
|
: cappedBridge;
|
|
4317
4729
|
if (!usesFallbackBridge) {
|
|
4318
4730
|
this.markFirstAssistantDelta(activeTurn.utterance, activeTurn.turnId);
|
|
@@ -4321,12 +4733,25 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4321
4733
|
{ type: "assistant_text_delta", text: spokenBridge },
|
|
4322
4734
|
() => !activeTurn.abortController.signal.aborted && !this.isClosed,
|
|
4323
4735
|
);
|
|
4736
|
+
this.bufferAssistantTextForTts(activeTurn.token, `${spokenBridge} `);
|
|
4737
|
+
// Force-flush now: on the TTS path an unpunctuated bridge would
|
|
4738
|
+
// otherwise sit buffered until a sentence boundary and leave the
|
|
4739
|
+
// caller in silence during the escalated model's call.
|
|
4740
|
+
this.flushTtsBuffer(activeTurn.token, true);
|
|
4741
|
+
} else {
|
|
4742
|
+
// The canned bridge is a fixed localized-table phrase, enqueued
|
|
4743
|
+
// directly (it is already one complete sentence) so the segment can
|
|
4744
|
+
// carry the "en" override when the table lacks the turn's language.
|
|
4745
|
+
const speakable = sanitizeForTts(spokenBridge).trim();
|
|
4746
|
+
if (speakable.length > 0) {
|
|
4747
|
+
this.enqueueTtsSegment(activeTurn.token, speakable, {
|
|
4748
|
+
language: this.fixedPhraseLanguage(
|
|
4749
|
+
activeTurn,
|
|
4750
|
+
FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
|
|
4751
|
+
),
|
|
4752
|
+
});
|
|
4753
|
+
}
|
|
4324
4754
|
}
|
|
4325
|
-
this.bufferAssistantTextForTts(activeTurn.token, `${spokenBridge} `);
|
|
4326
|
-
// Force-flush now: on the TTS path an unpunctuated bridge would otherwise
|
|
4327
|
-
// sit buffered until a sentence boundary and leave the caller in silence
|
|
4328
|
-
// during the escalated model's call.
|
|
4329
|
-
this.flushTtsBuffer(activeTurn.token, true);
|
|
4330
4755
|
|
|
4331
4756
|
// No overrideProfile: the escalated leg runs on the call-site default —
|
|
4332
4757
|
// the exact profile an un-routed voice turn would use (see
|
|
@@ -4690,6 +5115,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4690
5115
|
: null,
|
|
4691
5116
|
turnElapsedMs: now - turn.launchedAtMs,
|
|
4692
5117
|
updateIndex: progress.updatesSpoken + 1,
|
|
5118
|
+
...(turn.language !== undefined ? { languageHint: turn.language } : {}),
|
|
4693
5119
|
};
|
|
4694
5120
|
const generated = await frontDecider
|
|
4695
5121
|
.generateProgressText(input, turn.abortController.signal)
|
|
@@ -4718,13 +5144,20 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4718
5144
|
return;
|
|
4719
5145
|
}
|
|
4720
5146
|
let raw = generated;
|
|
5147
|
+
// Decider text is generated in the turn's language; only the static
|
|
5148
|
+
// fallback comes from a localized table and may need the "en" override.
|
|
5149
|
+
let fillerLanguage: string | undefined;
|
|
4721
5150
|
if (raw === null) {
|
|
4722
5151
|
if (trigger !== "idle") {
|
|
4723
5152
|
return;
|
|
4724
5153
|
}
|
|
4725
|
-
raw = pickProgressPhrase(this.progressPhraseCounter
|
|
5154
|
+
raw = pickProgressPhrase(this.progressPhraseCounter++, turn.language);
|
|
5155
|
+
fillerLanguage = this.fixedPhraseLanguage(
|
|
5156
|
+
turn,
|
|
5157
|
+
PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE,
|
|
5158
|
+
);
|
|
4726
5159
|
}
|
|
4727
|
-
if (!this.enqueueFillerPhrase(turn, raw)) {
|
|
5160
|
+
if (!this.enqueueFillerPhrase(turn, raw, fillerLanguage)) {
|
|
4728
5161
|
return;
|
|
4729
5162
|
}
|
|
4730
5163
|
progress.opsSinceNarration = 0;
|
|
@@ -4784,6 +5217,9 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4784
5217
|
{
|
|
4785
5218
|
transcriptSoFar: transcript,
|
|
4786
5219
|
toolName,
|
|
5220
|
+
...(activeTurn.language !== undefined
|
|
5221
|
+
? { languageHint: activeTurn.language }
|
|
5222
|
+
: {}),
|
|
4787
5223
|
},
|
|
4788
5224
|
activeTurn.abortController.signal,
|
|
4789
5225
|
)
|
|
@@ -4821,20 +5257,42 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4821
5257
|
// Sanitize and enqueue one filler sentence (spoken ack or progress
|
|
4822
5258
|
// narration) on the turn's ordered TTS queue — the shared tail of every
|
|
4823
5259
|
// filler path. Returns whether a phrase actually enqueued; per-kind metric
|
|
4824
|
-
// marks and bookkeeping are the caller's.
|
|
4825
|
-
|
|
5260
|
+
// marks and bookkeeping are the caller's. `language` is a per-segment
|
|
5261
|
+
// hint override (see fixedPhraseLanguage); omit it for generated text,
|
|
5262
|
+
// which is already in the turn's language.
|
|
5263
|
+
private enqueueFillerPhrase(
|
|
5264
|
+
turn: ActiveAssistantTurn,
|
|
5265
|
+
raw: string,
|
|
5266
|
+
language?: string,
|
|
5267
|
+
): boolean {
|
|
4826
5268
|
const phrase = sanitizeForTts(raw).trim();
|
|
4827
5269
|
if (phrase.length === 0) {
|
|
4828
5270
|
return false;
|
|
4829
5271
|
}
|
|
4830
5272
|
this.enqueueTtsSegment(turn.token, phrase, {
|
|
4831
5273
|
countsAsFirstSegment: false,
|
|
5274
|
+
...(language !== undefined ? { language } : {}),
|
|
4832
5275
|
});
|
|
4833
5276
|
// A spoken filler holds the floor, so narration's minGapMs spaces from it.
|
|
4834
5277
|
turn.progress.lastFloorHolderAtMs = Date.now();
|
|
4835
5278
|
return true;
|
|
4836
5279
|
}
|
|
4837
5280
|
|
|
5281
|
+
// The TTS hint override for a fixed phrase picked from a localized table:
|
|
5282
|
+
// "en" when the turn has a language the table does not cover (the picker
|
|
5283
|
+
// fell back to English text, which must not be synthesized under an
|
|
5284
|
+
// ar/ko/ta hint), undefined otherwise (the segment rides the turn's
|
|
5285
|
+
// language, or no hint at all when the language is unknown).
|
|
5286
|
+
private fixedPhraseLanguage(
|
|
5287
|
+
turn: ActiveAssistantTurn,
|
|
5288
|
+
table: Readonly<Record<string, unknown>>,
|
|
5289
|
+
): string | undefined {
|
|
5290
|
+
return turn.language !== undefined &&
|
|
5291
|
+
!hasLocalizedEntry(table, turn.language)
|
|
5292
|
+
? "en"
|
|
5293
|
+
: undefined;
|
|
5294
|
+
}
|
|
5295
|
+
|
|
4838
5296
|
private bufferAssistantTextForTts(token: symbol, text: string): void {
|
|
4839
5297
|
if (!this.streamTtsAudio || text.length === 0) {
|
|
4840
5298
|
return;
|
|
@@ -4955,7 +5413,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4955
5413
|
private enqueueTtsSegment(
|
|
4956
5414
|
token: symbol,
|
|
4957
5415
|
segment: string,
|
|
4958
|
-
options: { countsAsFirstSegment?: boolean } = {},
|
|
5416
|
+
options: { countsAsFirstSegment?: boolean; language?: string } = {},
|
|
4959
5417
|
): void {
|
|
4960
5418
|
const activeTurn = this.activeAssistantTurn;
|
|
4961
5419
|
if (activeTurn?.token !== token || !this.streamTtsAudio) {
|
|
@@ -4969,6 +5427,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4969
5427
|
}
|
|
4970
5428
|
const job: TtsSegmentJob = {
|
|
4971
5429
|
text: segment,
|
|
5430
|
+
language: options.language,
|
|
4972
5431
|
started: false,
|
|
4973
5432
|
settled: false,
|
|
4974
5433
|
emitting: false,
|
|
@@ -5008,10 +5467,14 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
5008
5467
|
return;
|
|
5009
5468
|
}
|
|
5010
5469
|
job.started = true;
|
|
5470
|
+
// The segment's own language override (fixed English fallback text)
|
|
5471
|
+
// wins over the turn's language.
|
|
5472
|
+
const language = job.language ?? activeTurn.language;
|
|
5011
5473
|
let synthesis: Promise<void>;
|
|
5012
5474
|
try {
|
|
5013
5475
|
synthesis = streamTtsAudio({
|
|
5014
5476
|
text: job.text,
|
|
5477
|
+
...(language !== undefined ? { language } : {}),
|
|
5015
5478
|
signal: activeTurn.abortController.signal,
|
|
5016
5479
|
outputFormat: "pcm",
|
|
5017
5480
|
sampleRate: this.context.startFrame.audio.sampleRate,
|
|
@@ -5151,6 +5614,10 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
5151
5614
|
chunk.sampleRate,
|
|
5152
5615
|
);
|
|
5153
5616
|
const now = Date.now();
|
|
5617
|
+
if (!this.isAssistantPlaybackEchoPossible()) {
|
|
5618
|
+
this.resetEchoReference();
|
|
5619
|
+
}
|
|
5620
|
+
this.appendEchoReference(chunk);
|
|
5154
5621
|
this.assistantPlaybackTailUntilMs =
|
|
5155
5622
|
Math.max(now, this.assistantPlaybackTailUntilMs) + chunkMs;
|
|
5156
5623
|
const turnAfterSend = this.activeAssistantTurn;
|
|
@@ -5655,6 +6122,11 @@ export function createLiveVoiceSession(
|
|
|
5655
6122
|
options.speechEnergyThreshold ?? vadConfig?.speechEnergyThreshold,
|
|
5656
6123
|
bargeInMinSpeechMs:
|
|
5657
6124
|
options.bargeInMinSpeechMs ?? vadConfig?.bargeInMinSpeechMs,
|
|
6125
|
+
echoBargeInMargin:
|
|
6126
|
+
options.echoBargeInMargin ?? vadConfig?.echoBargeInMargin,
|
|
6127
|
+
echoEmaHalfLifeMs:
|
|
6128
|
+
options.echoEmaHalfLifeMs ?? vadConfig?.echoEmaHalfLifeMs,
|
|
6129
|
+
echoDrainSlackMs: options.echoDrainSlackMs ?? vadConfig?.echoDrainSlackMs,
|
|
5658
6130
|
frontModelConfig,
|
|
5659
6131
|
// Eager construction is safe even when the `liveVoice.frontModel` config
|
|
5660
6132
|
// namespace is absent — schema defaults fill the tunables. An explicit
|