@vellumai/assistant 0.11.3-staging.2 → 0.11.3-staging.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/__tests__/call-controller.test.ts +120 -0
- package/src/__tests__/config-loader-backfill.test.ts +19 -0
- package/src/__tests__/config-schema.test.ts +139 -5
- package/src/__tests__/events-dev-bypass-actor.test.ts +112 -1
- package/src/__tests__/media-stream-output.test.ts +175 -0
- package/src/__tests__/media-stream-stt-session.test.ts +67 -0
- package/src/calls/__tests__/tts-text-sanitizer.test.ts +13 -0
- package/src/calls/__tests__/voice-session-bridge.test.ts +81 -12
- package/src/calls/__tests__/voice-triage-escalate.test.ts +106 -0
- package/src/calls/call-controller.ts +30 -2
- package/src/calls/call-speech-output.ts +12 -4
- package/src/calls/call-transport.ts +18 -1
- package/src/calls/media-stream-output.ts +53 -5
- package/src/calls/media-stream-server.ts +12 -0
- package/src/calls/media-stream-stt-session.ts +34 -0
- package/src/calls/telephony-synthesis-language.ts +84 -0
- package/src/calls/tts-text-sanitizer.ts +12 -5
- package/src/calls/voice-session-bridge.ts +31 -3
- package/src/calls/voice-triage-escalate.ts +52 -5
- package/src/config/bundled-skills/phone-calls/references/CONFIG.md +15 -15
- package/src/config/loader.ts +5 -0
- package/src/config/schemas/calls.ts +0 -4
- package/src/config/schemas/tts.ts +63 -0
- package/src/live-voice/__tests__/front-decision.test.ts +120 -0
- package/src/live-voice/__tests__/live-voice-events.test.ts +1 -1
- package/src/live-voice/__tests__/live-voice-progress.test.ts +178 -11
- package/src/live-voice/__tests__/live-voice-stt.test.ts +304 -1
- package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +102 -2
- package/src/live-voice/__tests__/live-voice-tts.test.ts +148 -2
- package/src/live-voice/__tests__/progress-phrases.test.ts +167 -0
- package/src/live-voice/front-decision.ts +50 -3
- package/src/live-voice/live-voice-session.ts +202 -25
- package/src/live-voice/live-voice-tts.ts +18 -2
- package/src/live-voice/progress-phrases.ts +105 -2
- package/src/providers/speech-to-text/deepgram-realtime.test.ts +283 -1
- package/src/providers/speech-to-text/deepgram-realtime.ts +117 -3
- package/src/providers/speech-to-text/provider-catalog.ts +38 -0
- package/src/runtime/__tests__/local-actor-identity-force-refresh.test.ts +104 -0
- package/src/runtime/assistant-event-hub.ts +23 -0
- package/src/runtime/local-actor-identity.ts +18 -5
- package/src/runtime/routes/__tests__/sse-actor-principal-heal.test.ts +165 -0
- package/src/runtime/routes/__tests__/surface-action-routes.test.ts +2 -0
- package/src/runtime/routes/events-routes.ts +17 -16
- package/src/runtime/routes/sse-actor-principal-heal.ts +113 -0
- package/src/stt/__tests__/language-metadata.test.ts +85 -0
- package/src/stt/language-metadata.ts +65 -0
- package/src/stt/types.ts +16 -0
- package/src/tts/__tests__/provider-adapters.test.ts +147 -0
- package/src/tts/__tests__/speakable-segments.test.ts +470 -0
- package/src/tts/language-voices.ts +23 -0
- package/src/tts/providers/deepgram-provider.ts +3 -1
- package/src/tts/providers/elevenlabs-provider.ts +73 -1
- package/src/tts/providers/xai-provider.ts +28 -2
- package/src/tts/speakable-segments.ts +293 -23
- package/src/tts/synthesis-stream.ts +7 -0
- package/src/tts/types.ts +7 -0
- package/src/util/__tests__/language-subtag.test.ts +54 -0
- package/src/util/language-subtag.ts +43 -0
- package/src/util/unicode.ts +1 -1
|
@@ -25,7 +25,8 @@ import {
|
|
|
25
25
|
classifyFrontDoorLeading,
|
|
26
26
|
ESCALATE_VERDICT_TOKEN,
|
|
27
27
|
ESCALATION_CONTINUATION_CONTENT,
|
|
28
|
-
|
|
28
|
+
FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
|
|
29
|
+
fallbackEscalationBridgeFor,
|
|
29
30
|
isEscalationBridgeComplete,
|
|
30
31
|
MIN_SPOKEN_BRIDGE_CHARS,
|
|
31
32
|
type VoiceRoutingLeg,
|
|
@@ -46,14 +47,20 @@ import { isInstalledStaticSkillLoad } from "../permissions/checker.js";
|
|
|
46
47
|
import { ensureConversationExists } from "../persistence/conversation-crud.js";
|
|
47
48
|
import {
|
|
48
49
|
listProviderIds,
|
|
50
|
+
pinnedListeningLanguage,
|
|
49
51
|
supportsBoundary,
|
|
50
52
|
} from "../providers/speech-to-text/provider-catalog.js";
|
|
51
53
|
import type { ResolveStreamingTranscriberOptions } from "../providers/speech-to-text/resolve.js";
|
|
52
54
|
import { broadcastMessage } from "../runtime/assistant-event-hub.js";
|
|
53
55
|
import { publishConversationListAndMetadataChanged } from "../runtime/sync/resource-sync-events.js";
|
|
56
|
+
import {
|
|
57
|
+
dominantLanguageTag,
|
|
58
|
+
voteDominantLanguage,
|
|
59
|
+
} from "../stt/language-metadata.js";
|
|
54
60
|
import { detectPcm16SpeechActivity } from "../stt/speech-energy.js";
|
|
55
61
|
import type {
|
|
56
62
|
StreamingTranscriber,
|
|
63
|
+
SttProviderId,
|
|
57
64
|
SttStreamServerErrorEvent,
|
|
58
65
|
SttStreamServerEvent,
|
|
59
66
|
} from "../stt/types.js";
|
|
@@ -62,6 +69,7 @@ import { liveVoiceEndScreen } from "../telemetry/live-voice-funnel.js";
|
|
|
62
69
|
import { getToolOwner } from "../tools/registry.js";
|
|
63
70
|
import { extractSpeakableSegments } from "../tts/speakable-segments.js";
|
|
64
71
|
import { createAbortReason } from "../util/abort-reasons.js";
|
|
72
|
+
import { hasLocalizedEntry } from "../util/language-subtag.js";
|
|
65
73
|
import { getLogger } from "../util/logger.js";
|
|
66
74
|
import {
|
|
67
75
|
activityLabelForTool,
|
|
@@ -101,7 +109,12 @@ import type {
|
|
|
101
109
|
LiveVoiceTtsOptions,
|
|
102
110
|
LiveVoiceTtsResult,
|
|
103
111
|
} from "./live-voice-tts.js";
|
|
104
|
-
import {
|
|
112
|
+
import {
|
|
113
|
+
APPROVAL_PENDING_PHRASE_BY_LANGUAGE,
|
|
114
|
+
approvalPendingPhraseFor,
|
|
115
|
+
pickProgressPhrase,
|
|
116
|
+
PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE,
|
|
117
|
+
} from "./progress-phrases.js";
|
|
105
118
|
import {
|
|
106
119
|
type LiveVoiceClientAttachImageFrame,
|
|
107
120
|
type LiveVoiceClientFrame,
|
|
@@ -369,6 +382,23 @@ interface UtteranceCycle {
|
|
|
369
382
|
// text replays the boundary immediately — the hold was judged on stale
|
|
370
383
|
// text, so waiting out the extension only adds silence.
|
|
371
384
|
heldSpeculativeContent: string | null;
|
|
385
|
+
// Count per detected-language base subtag (see voteDominantLanguage)
|
|
386
|
+
// across this cycle's final transcript events. Resolves the turn's spoken
|
|
387
|
+
// language (see turnLanguageFor); empty when the provider tags nothing.
|
|
388
|
+
languageTally: Map<string, number>;
|
|
389
|
+
// Detected languages of the most recent partial that carried any, already
|
|
390
|
+
// normalized, dominance order. Speculative turns dispatch from partials
|
|
391
|
+
// before the first tagged final lands, so turnLanguageFor falls back to
|
|
392
|
+
// this when the final tally is still empty. Never cleared: the tally
|
|
393
|
+
// outranks it once finals arrive, and a revising partial without tags
|
|
394
|
+
// must not wipe an earlier partial's detection.
|
|
395
|
+
latestPartialLanguages: readonly string[] | null;
|
|
396
|
+
// The provider that actually transcribed this cycle, recorded when its
|
|
397
|
+
// transcriber is assigned and kept after teardown nulls `transcriber`.
|
|
398
|
+
// The resolver can silently dial managed vellum when a BYOK provider has
|
|
399
|
+
// no credential, so the language-pin gate in turnLanguageFor must follow
|
|
400
|
+
// this, not the configured provider.
|
|
401
|
+
dialedSttProvider: SttProviderId | null;
|
|
372
402
|
turnId: string | null;
|
|
373
403
|
userMessageId: string | null;
|
|
374
404
|
userAudioChunks: Buffer[];
|
|
@@ -399,6 +429,11 @@ type UtteranceStartResult =
|
|
|
399
429
|
// client in job-list order.
|
|
400
430
|
interface TtsSegmentJob {
|
|
401
431
|
readonly text: string;
|
|
432
|
+
// Per-segment language-hint override, preferred over the turn's language.
|
|
433
|
+
// Set on fixed phrases whose localized table lacks the turn's language:
|
|
434
|
+
// the English fallback text carries "en" so an enforcing provider never
|
|
435
|
+
// renders English words as ar/ko/ta. Undefined means the turn language.
|
|
436
|
+
readonly language: string | undefined;
|
|
402
437
|
// The provider stream was started (the job holds an open-job slot).
|
|
403
438
|
started: boolean;
|
|
404
439
|
// Emission finished; the slot is free for the next queued segment.
|
|
@@ -483,6 +518,13 @@ interface ActiveAssistantTurn {
|
|
|
483
518
|
token: symbol;
|
|
484
519
|
turnId: string;
|
|
485
520
|
utterance: UtteranceCycle;
|
|
521
|
+
// The caller's spoken language for this turn as a lowercase base subtag
|
|
522
|
+
// (see turnLanguageFor): the dominant STT-detected language, else a
|
|
523
|
+
// monolingual services.stt.language pin. Undefined when unknown, which
|
|
524
|
+
// disables every language-aware path (prompt note, TTS hint, localized
|
|
525
|
+
// fallbacks). Re-resolved when a speculative turn commits, since finals
|
|
526
|
+
// can land between dispatch and verdict.
|
|
527
|
+
language: string | undefined;
|
|
486
528
|
abortController: AbortController;
|
|
487
529
|
handle: VoiceTurnHandle | null;
|
|
488
530
|
// When the turn launched, for narration's turnElapsedMs.
|
|
@@ -681,14 +723,8 @@ function createControlMarkerHoldback(
|
|
|
681
723
|
// barge-in, the interruption merge note is appended to it (see
|
|
682
724
|
// buildInterruptionMergeNote) so the model reconciles the interrupted request
|
|
683
725
|
// with the new utterance.
|
|
684
|
-
// Spoken once when a turn starts waiting on the user's decision. Fixed rather
|
|
685
|
-
// than generated: this is a statement about the system's state, not about the
|
|
686
|
-
// work, and it has to be true every time. Kept in the shape of the progress
|
|
687
|
-
// phrases it displaces (short, neutral, no claim about tools).
|
|
688
|
-
const APPROVAL_PENDING_PHRASE = "I need your okay for that one. Take a look.";
|
|
689
|
-
|
|
690
726
|
const LIVE_VOICE_CONTROL_PROMPT_BASE =
|
|
691
|
-
"You are speaking in a local live voice session. Keep replies brief and conversational. Speech is the main channel: say the answer, and do not narrate a surface instead of answering. You can also put something on screen when it genuinely helps (a form, a list to pick from, a progress card for long work); the call overlay minimizes by itself once you finish speaking, so the user sees it without doing anything. Never tell the user you cannot show them something. ";
|
|
727
|
+
"You are speaking in a local live voice session. Keep replies brief and conversational. Speech is the main channel: say the answer, and do not narrate a surface instead of answering. You can also put something on screen when it genuinely helps (a form, a list to pick from, a progress card for long work); the call overlay minimizes by itself once you finish speaking, so the user sees it without doing anything. Never tell the user you cannot show them something. Reply in the language the caller is speaking; if they switch languages, switch with them. ";
|
|
692
728
|
|
|
693
729
|
// Appended for the legs that can actually put something on screen: the main
|
|
694
730
|
// leg and the escalated leg. The front-door (fast) leg never receives it, for
|
|
@@ -777,6 +813,9 @@ function buildVoiceControlPrompt(
|
|
|
777
813
|
LIVE_VOICE_CONTROL_PROMPT_BASE +
|
|
778
814
|
(leg.frontDoor === true ? "" : LIVE_VOICE_SCREEN_REVEAL_TEACHING) +
|
|
779
815
|
VOICE_NO_SETUP_FLOWS_RULE;
|
|
816
|
+
if (turn.language !== undefined) {
|
|
817
|
+
prompt = `${prompt}\n\nThe caller has been speaking the language with code "${turn.language}" this turn. Reply in that language unless they clearly switch to another.`;
|
|
818
|
+
}
|
|
780
819
|
if (turn.interruptedRequest) {
|
|
781
820
|
prompt = `${prompt}\n\n${buildInterruptionMergeNote(turn.interruptedRequest)}`;
|
|
782
821
|
}
|
|
@@ -825,6 +864,9 @@ function createUtteranceCycle(): UtteranceCycle {
|
|
|
825
864
|
latestPartialText: null,
|
|
826
865
|
endpointExtensionCount: 0,
|
|
827
866
|
heldSpeculativeContent: null,
|
|
867
|
+
languageTally: new Map(),
|
|
868
|
+
latestPartialLanguages: null,
|
|
869
|
+
dialedSttProvider: null,
|
|
828
870
|
turnId: null,
|
|
829
871
|
userMessageId: null,
|
|
830
872
|
userAudioChunks: [],
|
|
@@ -1387,6 +1429,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
1387
1429
|
// Persistent re-arm: the shared stream is already open, so the cycle
|
|
1388
1430
|
// goes straight to streaming with no resolve/start round-trip.
|
|
1389
1431
|
utterance.transcriber = shared;
|
|
1432
|
+
utterance.dialedSttProvider = shared.providerId;
|
|
1390
1433
|
return await this.activateUtterance(utterance, replayTurnEnd);
|
|
1391
1434
|
}
|
|
1392
1435
|
// The shared stream is pinned to the old language, so retire it and
|
|
@@ -1419,6 +1462,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
1419
1462
|
}
|
|
1420
1463
|
|
|
1421
1464
|
utterance.transcriber = transcriber;
|
|
1465
|
+
utterance.dialedSttProvider = transcriber.providerId;
|
|
1422
1466
|
if (
|
|
1423
1467
|
this.turnDetector &&
|
|
1424
1468
|
typeof transcriber.finalizeUtterance === "function"
|
|
@@ -2527,8 +2571,13 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
2527
2571
|
// Spoken, because opening the room is only a cue for someone looking at
|
|
2528
2572
|
// the screen, and the case this exists for is a phone the user has put
|
|
2529
2573
|
// down. One line, not narration: the turn is not working, it is waiting,
|
|
2530
|
-
// and it says which
|
|
2531
|
-
|
|
2574
|
+
// and it says which, in the turn's spoken language, like every other
|
|
2575
|
+
// filler phrase.
|
|
2576
|
+
this.enqueueFillerPhrase(
|
|
2577
|
+
turn,
|
|
2578
|
+
approvalPendingPhraseFor(turn.language),
|
|
2579
|
+
this.fixedPhraseLanguage(turn, APPROVAL_PENDING_PHRASE_BY_LANGUAGE),
|
|
2580
|
+
);
|
|
2532
2581
|
}
|
|
2533
2582
|
|
|
2534
2583
|
/** Clear the wait once a decision lands, so the turn narrates normally again. */
|
|
@@ -2851,6 +2900,13 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
2851
2900
|
// are skipped; the thinking frame and timers still apply.
|
|
2852
2901
|
const alreadyReleased = utterance.released;
|
|
2853
2902
|
turn.speculativePending = false;
|
|
2903
|
+
// Finals can land between the speculative dispatch and this verdict.
|
|
2904
|
+
// Fill the language only when dispatch had none: the model request was
|
|
2905
|
+
// already issued with the dispatch language, so overwriting here would
|
|
2906
|
+
// hint TTS (and any voice override) in a different language than the
|
|
2907
|
+
// text it speaks. The tally still carries the corrected detection into
|
|
2908
|
+
// the next turn.
|
|
2909
|
+
turn.language ??= this.turnLanguageFor(utterance);
|
|
2854
2910
|
if (turn.verdictDeadlineTimer !== null) {
|
|
2855
2911
|
clearTimeout(turn.verdictDeadlineTimer);
|
|
2856
2912
|
turn.verdictDeadlineTimer = null;
|
|
@@ -3189,11 +3245,16 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3189
3245
|
switch (event.type) {
|
|
3190
3246
|
case "partial":
|
|
3191
3247
|
utterance.latestPartialText = event.text;
|
|
3248
|
+
this.capturePartialLanguages(utterance, event.languages);
|
|
3192
3249
|
this.markFirstPartial(utterance);
|
|
3193
3250
|
await this.sendFrame({ type: "stt_partial", text: event.text });
|
|
3194
3251
|
return;
|
|
3195
3252
|
case "final":
|
|
3196
|
-
await this.recordFinalTranscript(
|
|
3253
|
+
await this.recordFinalTranscript(
|
|
3254
|
+
utterance,
|
|
3255
|
+
event.text,
|
|
3256
|
+
event.languages,
|
|
3257
|
+
);
|
|
3197
3258
|
return;
|
|
3198
3259
|
case "finalized":
|
|
3199
3260
|
// Per-cycle transcribers are torn down with stop(); the finalize
|
|
@@ -3262,6 +3323,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3262
3323
|
return;
|
|
3263
3324
|
}
|
|
3264
3325
|
target.latestPartialText = event.text;
|
|
3326
|
+
this.capturePartialLanguages(target, event.languages);
|
|
3265
3327
|
this.markFirstPartial(target);
|
|
3266
3328
|
await this.sendFrame({ type: "stt_partial", text: event.text });
|
|
3267
3329
|
return;
|
|
@@ -3277,7 +3339,11 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3277
3339
|
// newer cycle.
|
|
3278
3340
|
const owner = this.finalizeQueue[0];
|
|
3279
3341
|
if (owner && !owner.assistantTurnStarted && !owner.completed) {
|
|
3280
|
-
await this.recordFinalTranscript(
|
|
3342
|
+
await this.recordFinalTranscript(
|
|
3343
|
+
owner,
|
|
3344
|
+
event.text,
|
|
3345
|
+
event.languages,
|
|
3346
|
+
);
|
|
3281
3347
|
} else {
|
|
3282
3348
|
log.warn(
|
|
3283
3349
|
"Dropping a late finalize flush segment: its assistant turn already dispatched",
|
|
@@ -3294,7 +3360,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3294
3360
|
);
|
|
3295
3361
|
return;
|
|
3296
3362
|
}
|
|
3297
|
-
await this.recordFinalTranscript(target, event.text);
|
|
3363
|
+
await this.recordFinalTranscript(target, event.text, event.languages);
|
|
3298
3364
|
return;
|
|
3299
3365
|
}
|
|
3300
3366
|
case "finalized": {
|
|
@@ -3363,10 +3429,16 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3363
3429
|
private async recordFinalTranscript(
|
|
3364
3430
|
utterance: UtteranceCycle,
|
|
3365
3431
|
text: string,
|
|
3432
|
+
languages?: readonly string[],
|
|
3366
3433
|
): Promise<void> {
|
|
3367
3434
|
const transcript = text.trim();
|
|
3368
3435
|
if (transcript.length > 0) {
|
|
3369
3436
|
utterance.finalTranscriptSegments.push(transcript);
|
|
3437
|
+
// Tally only finals that committed transcript: empty silence frames
|
|
3438
|
+
// can still carry container-level language tags describing no emitted
|
|
3439
|
+
// words, and counting those would let silence outvote real speech
|
|
3440
|
+
// (same choice as the adapter's boundary-final aggregation).
|
|
3441
|
+
voteDominantLanguage(utterance.languageTally, languages);
|
|
3370
3442
|
}
|
|
3371
3443
|
// The final commits (and supersedes) whatever partial was trailing it.
|
|
3372
3444
|
utterance.latestPartialText = null;
|
|
@@ -3405,6 +3477,59 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3405
3477
|
await this.startAssistantTurnIfReady();
|
|
3406
3478
|
}
|
|
3407
3479
|
|
|
3480
|
+
// Record a partial event's detected languages so speculative dispatch
|
|
3481
|
+
// has a detection before the first tagged final. The event contract
|
|
3482
|
+
// (stt/types.ts) guarantees the tags arrive as normalized base subtags
|
|
3483
|
+
// in dominance order, so they are stored as-is. Partials revise each
|
|
3484
|
+
// other, so this overwrites rather than tallies, and a tag-less partial
|
|
3485
|
+
// keeps the previous value.
|
|
3486
|
+
private capturePartialLanguages(
|
|
3487
|
+
utterance: UtteranceCycle,
|
|
3488
|
+
languages: readonly string[] | undefined,
|
|
3489
|
+
): void {
|
|
3490
|
+
if (!languages || languages.length === 0) {
|
|
3491
|
+
return;
|
|
3492
|
+
}
|
|
3493
|
+
utterance.latestPartialLanguages = languages;
|
|
3494
|
+
}
|
|
3495
|
+
|
|
3496
|
+
/**
|
|
3497
|
+
* The caller's spoken language for a turn on this utterance, as a
|
|
3498
|
+
* lowercase base subtag: the dominant tallied STT-detected language
|
|
3499
|
+
* (most final-event counts, ties by first appearance), else the latest
|
|
3500
|
+
* tagged partial's dominant language (speculative turns dispatch from
|
|
3501
|
+
* partials), else a monolingual `services.stt.language` pin (a pinned
|
|
3502
|
+
* language IS the spoken language), else undefined ("multi" with no tags,
|
|
3503
|
+
* non-tagging providers, silence).
|
|
3504
|
+
*/
|
|
3505
|
+
private turnLanguageFor(utterance: UtteranceCycle): string | undefined {
|
|
3506
|
+
const dominant = dominantLanguageTag(utterance.languageTally);
|
|
3507
|
+
if (dominant !== undefined) {
|
|
3508
|
+
return dominant;
|
|
3509
|
+
}
|
|
3510
|
+
// No tagged final yet (speculative turns dispatch from partials): the
|
|
3511
|
+
// latest tagged partial is the best detection available and outranks a
|
|
3512
|
+
// static pin for the same reason the tally does.
|
|
3513
|
+
const partialDominant = utterance.latestPartialLanguages?.[0];
|
|
3514
|
+
if (partialDominant !== undefined) {
|
|
3515
|
+
return partialDominant;
|
|
3516
|
+
}
|
|
3517
|
+
// A persisted pin only counts when the provider that actually
|
|
3518
|
+
// transcribed honors manual language selection (the shared
|
|
3519
|
+
// pinnedListeningLanguage gate). The DIALED transcriber's providerId
|
|
3520
|
+
// is authoritative, because the resolver silently falls back to
|
|
3521
|
+
// managed vellum (which honors the pin) when a BYOK provider has no
|
|
3522
|
+
// credential; the configured provider is only the last resort when no
|
|
3523
|
+
// transcriber reference survives.
|
|
3524
|
+
const { language: configured, provider: sttProvider } =
|
|
3525
|
+
getConfig().services.stt;
|
|
3526
|
+
const dialedProvider =
|
|
3527
|
+
utterance.dialedSttProvider ??
|
|
3528
|
+
this.sharedTranscriber?.providerId ??
|
|
3529
|
+
(sttProvider as SttProviderId);
|
|
3530
|
+
return pinnedListeningLanguage(dialedProvider, configured);
|
|
3531
|
+
}
|
|
3532
|
+
|
|
3408
3533
|
// Providers emit `error` mid-stream and may keep streaming; `closed` /
|
|
3409
3534
|
// `final` still drive turn lifecycle. Only transient categories are
|
|
3410
3535
|
// recoverable — auth/rate-limit/invalid-audio will not self-heal, so
|
|
@@ -3660,6 +3785,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
3660
3785
|
token,
|
|
3661
3786
|
turnId,
|
|
3662
3787
|
utterance,
|
|
3788
|
+
language: this.turnLanguageFor(utterance),
|
|
3663
3789
|
abortController,
|
|
3664
3790
|
handle: null,
|
|
3665
3791
|
launchedAtMs: Date.now(),
|
|
@@ -4312,7 +4438,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4312
4438
|
// deleted row for a bridge the model never produced).
|
|
4313
4439
|
const usesFallbackBridge = cappedBridge.length < MIN_SPOKEN_BRIDGE_CHARS;
|
|
4314
4440
|
const spokenBridge = usesFallbackBridge
|
|
4315
|
-
?
|
|
4441
|
+
? fallbackEscalationBridgeFor(activeTurn.language)
|
|
4316
4442
|
: cappedBridge;
|
|
4317
4443
|
if (!usesFallbackBridge) {
|
|
4318
4444
|
this.markFirstAssistantDelta(activeTurn.utterance, activeTurn.turnId);
|
|
@@ -4321,12 +4447,25 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4321
4447
|
{ type: "assistant_text_delta", text: spokenBridge },
|
|
4322
4448
|
() => !activeTurn.abortController.signal.aborted && !this.isClosed,
|
|
4323
4449
|
);
|
|
4450
|
+
this.bufferAssistantTextForTts(activeTurn.token, `${spokenBridge} `);
|
|
4451
|
+
// Force-flush now: on the TTS path an unpunctuated bridge would
|
|
4452
|
+
// otherwise sit buffered until a sentence boundary and leave the
|
|
4453
|
+
// caller in silence during the escalated model's call.
|
|
4454
|
+
this.flushTtsBuffer(activeTurn.token, true);
|
|
4455
|
+
} else {
|
|
4456
|
+
// The canned bridge is a fixed localized-table phrase, enqueued
|
|
4457
|
+
// directly (it is already one complete sentence) so the segment can
|
|
4458
|
+
// carry the "en" override when the table lacks the turn's language.
|
|
4459
|
+
const speakable = sanitizeForTts(spokenBridge).trim();
|
|
4460
|
+
if (speakable.length > 0) {
|
|
4461
|
+
this.enqueueTtsSegment(activeTurn.token, speakable, {
|
|
4462
|
+
language: this.fixedPhraseLanguage(
|
|
4463
|
+
activeTurn,
|
|
4464
|
+
FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
|
|
4465
|
+
),
|
|
4466
|
+
});
|
|
4467
|
+
}
|
|
4324
4468
|
}
|
|
4325
|
-
this.bufferAssistantTextForTts(activeTurn.token, `${spokenBridge} `);
|
|
4326
|
-
// Force-flush now: on the TTS path an unpunctuated bridge would otherwise
|
|
4327
|
-
// sit buffered until a sentence boundary and leave the caller in silence
|
|
4328
|
-
// during the escalated model's call.
|
|
4329
|
-
this.flushTtsBuffer(activeTurn.token, true);
|
|
4330
4469
|
|
|
4331
4470
|
// No overrideProfile: the escalated leg runs on the call-site default —
|
|
4332
4471
|
// the exact profile an un-routed voice turn would use (see
|
|
@@ -4690,6 +4829,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4690
4829
|
: null,
|
|
4691
4830
|
turnElapsedMs: now - turn.launchedAtMs,
|
|
4692
4831
|
updateIndex: progress.updatesSpoken + 1,
|
|
4832
|
+
...(turn.language !== undefined ? { languageHint: turn.language } : {}),
|
|
4693
4833
|
};
|
|
4694
4834
|
const generated = await frontDecider
|
|
4695
4835
|
.generateProgressText(input, turn.abortController.signal)
|
|
@@ -4718,13 +4858,20 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4718
4858
|
return;
|
|
4719
4859
|
}
|
|
4720
4860
|
let raw = generated;
|
|
4861
|
+
// Decider text is generated in the turn's language; only the static
|
|
4862
|
+
// fallback comes from a localized table and may need the "en" override.
|
|
4863
|
+
let fillerLanguage: string | undefined;
|
|
4721
4864
|
if (raw === null) {
|
|
4722
4865
|
if (trigger !== "idle") {
|
|
4723
4866
|
return;
|
|
4724
4867
|
}
|
|
4725
|
-
raw = pickProgressPhrase(this.progressPhraseCounter
|
|
4868
|
+
raw = pickProgressPhrase(this.progressPhraseCounter++, turn.language);
|
|
4869
|
+
fillerLanguage = this.fixedPhraseLanguage(
|
|
4870
|
+
turn,
|
|
4871
|
+
PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE,
|
|
4872
|
+
);
|
|
4726
4873
|
}
|
|
4727
|
-
if (!this.enqueueFillerPhrase(turn, raw)) {
|
|
4874
|
+
if (!this.enqueueFillerPhrase(turn, raw, fillerLanguage)) {
|
|
4728
4875
|
return;
|
|
4729
4876
|
}
|
|
4730
4877
|
progress.opsSinceNarration = 0;
|
|
@@ -4784,6 +4931,9 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4784
4931
|
{
|
|
4785
4932
|
transcriptSoFar: transcript,
|
|
4786
4933
|
toolName,
|
|
4934
|
+
...(activeTurn.language !== undefined
|
|
4935
|
+
? { languageHint: activeTurn.language }
|
|
4936
|
+
: {}),
|
|
4787
4937
|
},
|
|
4788
4938
|
activeTurn.abortController.signal,
|
|
4789
4939
|
)
|
|
@@ -4821,20 +4971,42 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4821
4971
|
// Sanitize and enqueue one filler sentence (spoken ack or progress
|
|
4822
4972
|
// narration) on the turn's ordered TTS queue — the shared tail of every
|
|
4823
4973
|
// filler path. Returns whether a phrase actually enqueued; per-kind metric
|
|
4824
|
-
// marks and bookkeeping are the caller's.
|
|
4825
|
-
|
|
4974
|
+
// marks and bookkeeping are the caller's. `language` is a per-segment
|
|
4975
|
+
// hint override (see fixedPhraseLanguage); omit it for generated text,
|
|
4976
|
+
// which is already in the turn's language.
|
|
4977
|
+
private enqueueFillerPhrase(
|
|
4978
|
+
turn: ActiveAssistantTurn,
|
|
4979
|
+
raw: string,
|
|
4980
|
+
language?: string,
|
|
4981
|
+
): boolean {
|
|
4826
4982
|
const phrase = sanitizeForTts(raw).trim();
|
|
4827
4983
|
if (phrase.length === 0) {
|
|
4828
4984
|
return false;
|
|
4829
4985
|
}
|
|
4830
4986
|
this.enqueueTtsSegment(turn.token, phrase, {
|
|
4831
4987
|
countsAsFirstSegment: false,
|
|
4988
|
+
...(language !== undefined ? { language } : {}),
|
|
4832
4989
|
});
|
|
4833
4990
|
// A spoken filler holds the floor, so narration's minGapMs spaces from it.
|
|
4834
4991
|
turn.progress.lastFloorHolderAtMs = Date.now();
|
|
4835
4992
|
return true;
|
|
4836
4993
|
}
|
|
4837
4994
|
|
|
4995
|
+
// The TTS hint override for a fixed phrase picked from a localized table:
|
|
4996
|
+
// "en" when the turn has a language the table does not cover (the picker
|
|
4997
|
+
// fell back to English text, which must not be synthesized under an
|
|
4998
|
+
// ar/ko/ta hint), undefined otherwise (the segment rides the turn's
|
|
4999
|
+
// language, or no hint at all when the language is unknown).
|
|
5000
|
+
private fixedPhraseLanguage(
|
|
5001
|
+
turn: ActiveAssistantTurn,
|
|
5002
|
+
table: Readonly<Record<string, unknown>>,
|
|
5003
|
+
): string | undefined {
|
|
5004
|
+
return turn.language !== undefined &&
|
|
5005
|
+
!hasLocalizedEntry(table, turn.language)
|
|
5006
|
+
? "en"
|
|
5007
|
+
: undefined;
|
|
5008
|
+
}
|
|
5009
|
+
|
|
4838
5010
|
private bufferAssistantTextForTts(token: symbol, text: string): void {
|
|
4839
5011
|
if (!this.streamTtsAudio || text.length === 0) {
|
|
4840
5012
|
return;
|
|
@@ -4955,7 +5127,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4955
5127
|
private enqueueTtsSegment(
|
|
4956
5128
|
token: symbol,
|
|
4957
5129
|
segment: string,
|
|
4958
|
-
options: { countsAsFirstSegment?: boolean } = {},
|
|
5130
|
+
options: { countsAsFirstSegment?: boolean; language?: string } = {},
|
|
4959
5131
|
): void {
|
|
4960
5132
|
const activeTurn = this.activeAssistantTurn;
|
|
4961
5133
|
if (activeTurn?.token !== token || !this.streamTtsAudio) {
|
|
@@ -4969,6 +5141,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
4969
5141
|
}
|
|
4970
5142
|
const job: TtsSegmentJob = {
|
|
4971
5143
|
text: segment,
|
|
5144
|
+
language: options.language,
|
|
4972
5145
|
started: false,
|
|
4973
5146
|
settled: false,
|
|
4974
5147
|
emitting: false,
|
|
@@ -5008,10 +5181,14 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
|
|
|
5008
5181
|
return;
|
|
5009
5182
|
}
|
|
5010
5183
|
job.started = true;
|
|
5184
|
+
// The segment's own language override (fixed English fallback text)
|
|
5185
|
+
// wins over the turn's language.
|
|
5186
|
+
const language = job.language ?? activeTurn.language;
|
|
5011
5187
|
let synthesis: Promise<void>;
|
|
5012
5188
|
try {
|
|
5013
5189
|
synthesis = streamTtsAudio({
|
|
5014
5190
|
text: job.text,
|
|
5191
|
+
...(language !== undefined ? { language } : {}),
|
|
5015
5192
|
signal: activeTurn.abortController.signal,
|
|
5016
5193
|
outputFormat: "pcm",
|
|
5017
5194
|
sampleRate: this.context.startFrame.audio.sampleRate,
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { resolveLanguageVoiceOverride } from "../tts/language-voices.js";
|
|
1
2
|
import { createPcmChunkAligner } from "../tts/pcm-chunk-aligner.js";
|
|
2
3
|
import { getTtsProvider } from "../tts/provider-catalog.js";
|
|
3
4
|
import { synthesizeAndEmit } from "../tts/synthesis-stream.js";
|
|
@@ -30,6 +31,7 @@ export interface LiveVoiceTtsOptions {
|
|
|
30
31
|
useCase?: TtsUseCase;
|
|
31
32
|
outputFormat?: TtsSynthesisRequest["outputFormat"];
|
|
32
33
|
sampleRate?: number;
|
|
34
|
+
language?: string;
|
|
33
35
|
config?: LiveVoiceTtsConfig;
|
|
34
36
|
onAudioChunk: (chunk: LiveVoiceTtsAudioChunk) => void;
|
|
35
37
|
}
|
|
@@ -72,11 +74,23 @@ interface ResolvedStreamingTtsProvider {
|
|
|
72
74
|
providerConfig: Record<string, unknown>;
|
|
73
75
|
}
|
|
74
76
|
|
|
77
|
+
export { resolveLanguageVoiceOverride };
|
|
78
|
+
|
|
75
79
|
export async function streamLiveVoiceTtsAudio(
|
|
76
80
|
options: LiveVoiceTtsOptions,
|
|
77
81
|
): Promise<LiveVoiceTtsResult> {
|
|
78
82
|
const { provider, providerId, providerConfig } =
|
|
79
83
|
await resolveLiveVoiceStreamingTtsProvider(options.config);
|
|
84
|
+
// An explicit request voice wins outright; otherwise a language-known
|
|
85
|
+
// turn may select the provider's configured per-language voice. The cast
|
|
86
|
+
// recovers the schema-typed map that resolveTtsConfig's generic
|
|
87
|
+
// provider-block lookup erases.
|
|
88
|
+
const voiceId =
|
|
89
|
+
options.voiceId ??
|
|
90
|
+
resolveLanguageVoiceOverride(
|
|
91
|
+
providerConfig.languageVoices as Record<string, string> | undefined,
|
|
92
|
+
options.language,
|
|
93
|
+
);
|
|
80
94
|
const useCase = options.useCase ?? "phone-call";
|
|
81
95
|
const requestedSampleRate = resolveSampleRate(
|
|
82
96
|
options.sampleRate,
|
|
@@ -89,9 +103,10 @@ export async function streamLiveVoiceTtsAudio(
|
|
|
89
103
|
const providerSampleRate = provider.resolveOutputSampleRateHz?.({
|
|
90
104
|
text: options.text,
|
|
91
105
|
useCase,
|
|
92
|
-
voiceId
|
|
106
|
+
voiceId,
|
|
93
107
|
outputFormat: options.outputFormat,
|
|
94
108
|
sampleRateHz: requestedSampleRate,
|
|
109
|
+
language: options.language,
|
|
95
110
|
signal: options.signal,
|
|
96
111
|
});
|
|
97
112
|
const sampleRate = providerSampleRate ?? requestedSampleRate;
|
|
@@ -138,9 +153,10 @@ export async function streamLiveVoiceTtsAudio(
|
|
|
138
153
|
provider,
|
|
139
154
|
text: options.text,
|
|
140
155
|
useCase,
|
|
141
|
-
voiceId
|
|
156
|
+
voiceId,
|
|
142
157
|
outputFormat: options.outputFormat,
|
|
143
158
|
sampleRateHz: requestedSampleRate,
|
|
159
|
+
language: options.language,
|
|
144
160
|
signal: options.signal,
|
|
145
161
|
onChunk: (chunk) => {
|
|
146
162
|
if (canStreamChunks) {
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { localizedOrDefault } from "../util/language-subtag.js";
|
|
2
|
+
|
|
1
3
|
// Static fallbacks for an idle-triggered progress narration whose LLM
|
|
2
4
|
// phrasing failed — the one case where prolonged silence is actively harmful.
|
|
3
5
|
// The idle trigger can fire on a slow turn with zero tool activity, so every
|
|
@@ -10,8 +12,109 @@ export const PROGRESS_FALLBACK_PHRASES: readonly string[] = [
|
|
|
10
12
|
"Almost there — thanks for waiting.",
|
|
11
13
|
];
|
|
12
14
|
|
|
15
|
+
// Per-language fallback phrases, keyed by lowercased BCP 47 base subtag,
|
|
16
|
+
// covering the Deepgram code-switching roster (DEEPGRAM_MULTI_LANGUAGE_CODES
|
|
17
|
+
// in providers/speech-to-text/deepgram.ts). Every list carries the same
|
|
18
|
+
// invariants as the English one above: persona-neutral, no claims about
|
|
19
|
+
// running tools or tasks, at most 8 words per phrase.
|
|
20
|
+
export const PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE: Readonly<
|
|
21
|
+
Record<string, readonly string[]>
|
|
22
|
+
> = {
|
|
23
|
+
en: PROGRESS_FALLBACK_PHRASES,
|
|
24
|
+
es: [
|
|
25
|
+
"Sigo en ello, un momento.",
|
|
26
|
+
"Todavía lo estoy pensando.",
|
|
27
|
+
"Casi listo, gracias por esperar.",
|
|
28
|
+
],
|
|
29
|
+
fr: [
|
|
30
|
+
"J'y suis encore, un instant.",
|
|
31
|
+
"J'y réfléchis encore.",
|
|
32
|
+
"Presque fini, merci de patienter.",
|
|
33
|
+
],
|
|
34
|
+
de: [
|
|
35
|
+
"Bin noch dabei, einen Moment.",
|
|
36
|
+
"Ich denke noch darüber nach.",
|
|
37
|
+
"Fast fertig, danke fürs Warten.",
|
|
38
|
+
],
|
|
39
|
+
hi: [
|
|
40
|
+
"बस एक पल रुकिए।",
|
|
41
|
+
"अभी इस पर विचार चल रहा है।",
|
|
42
|
+
"बस थोड़ा और इंतज़ार कीजिए, धन्यवाद।",
|
|
43
|
+
],
|
|
44
|
+
ru: [
|
|
45
|
+
"Секундочку, я ещё здесь.",
|
|
46
|
+
"Я всё ещё думаю над этим.",
|
|
47
|
+
"Почти готово, спасибо за ожидание.",
|
|
48
|
+
],
|
|
49
|
+
pt: [
|
|
50
|
+
"Ainda estou nisso, um momento.",
|
|
51
|
+
"Ainda estou pensando nisso.",
|
|
52
|
+
"Quase lá, agradeço a espera.",
|
|
53
|
+
],
|
|
54
|
+
ja: [
|
|
55
|
+
"まだ対応中です。少々お待ちください。",
|
|
56
|
+
"まだ考えているところです。",
|
|
57
|
+
"もうすぐです。お待ちいただきありがとうございます。",
|
|
58
|
+
],
|
|
59
|
+
it: [
|
|
60
|
+
"Ancora un attimo, per favore.",
|
|
61
|
+
"Ci sto ancora pensando.",
|
|
62
|
+
"Quasi fatto, grazie per l'attesa.",
|
|
63
|
+
],
|
|
64
|
+
nl: [
|
|
65
|
+
"Ik ben er nog mee bezig.",
|
|
66
|
+
"Ik denk er nog over na.",
|
|
67
|
+
"Bijna klaar, bedankt voor het wachten.",
|
|
68
|
+
],
|
|
69
|
+
};
|
|
70
|
+
|
|
13
71
|
// Deterministic rotation through the phrase list: callers hold a nonnegative
|
|
14
72
|
// monotonic counter, so consecutive picks vary while tests stay reproducible.
|
|
15
|
-
|
|
16
|
-
|
|
73
|
+
// `language` selects the per-language list by its lowercased base subtag
|
|
74
|
+
// (e.g. "pt-BR" -> "pt"); unknown or absent languages fall back to English.
|
|
75
|
+
export function pickProgressPhrase(counter: number, language?: string): string {
|
|
76
|
+
const phrases = localizedOrDefault(
|
|
77
|
+
PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE,
|
|
78
|
+
language,
|
|
79
|
+
PROGRESS_FALLBACK_PHRASES,
|
|
80
|
+
);
|
|
81
|
+
return phrases[counter % phrases.length];
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// Spoken once when a turn starts waiting on the user's approval decision.
|
|
85
|
+
// Fixed rather than generated: this is a statement about the system's
|
|
86
|
+
// state, not about the work, and it has to be true every time. Kept in the
|
|
87
|
+
// shape of the progress phrases it displaces (short, neutral, no claim
|
|
88
|
+
// about tools).
|
|
89
|
+
export const APPROVAL_PENDING_PHRASE =
|
|
90
|
+
"I need your okay for that one. Take a look.";
|
|
91
|
+
|
|
92
|
+
// Per-language spellings of the approval-pending phrase, keyed by lowercased
|
|
93
|
+
// BCP 47 base subtag, covering the Deepgram code-switching roster
|
|
94
|
+
// (DEEPGRAM_MULTI_LANGUAGE_CODES in providers/speech-to-text/deepgram.ts).
|
|
95
|
+
// Same invariants as the progress phrases: persona-neutral, no claims about
|
|
96
|
+
// running tools or tasks.
|
|
97
|
+
export const APPROVAL_PENDING_PHRASE_BY_LANGUAGE: Readonly<
|
|
98
|
+
Record<string, string>
|
|
99
|
+
> = {
|
|
100
|
+
en: APPROVAL_PENDING_PHRASE,
|
|
101
|
+
es: "Necesito tu visto bueno para eso. Échale un vistazo.",
|
|
102
|
+
fr: "J'ai besoin de ton accord pour ça. Jette un œil.",
|
|
103
|
+
de: "Dafür brauche ich dein Okay. Schau mal drauf.",
|
|
104
|
+
hi: "इसके लिए मुझे आपकी मंज़ूरी चाहिए। एक नज़र डाल लीजिए।",
|
|
105
|
+
ru: "Для этого мне нужно твоё согласие. Взгляни, пожалуйста.",
|
|
106
|
+
pt: "Preciso do seu ok para isso. Dê uma olhada.",
|
|
107
|
+
ja: "これには許可が必要です。ご確認ください。",
|
|
108
|
+
it: "Mi serve il tuo via libera per questo. Dai un'occhiata.",
|
|
109
|
+
nl: "Hiervoor heb ik je akkoord nodig. Kijk even mee.",
|
|
110
|
+
};
|
|
111
|
+
|
|
112
|
+
// The approval-pending phrase in the turn's spoken language, defaulting to
|
|
113
|
+
// English for unknown or absent languages.
|
|
114
|
+
export function approvalPendingPhraseFor(language?: string): string {
|
|
115
|
+
return localizedOrDefault(
|
|
116
|
+
APPROVAL_PENDING_PHRASE_BY_LANGUAGE,
|
|
117
|
+
language,
|
|
118
|
+
APPROVAL_PENDING_PHRASE,
|
|
119
|
+
);
|
|
17
120
|
}
|