@vellumai/assistant 0.11.3-staging.2 → 0.11.3-staging.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/package.json +1 -1
  2. package/src/__tests__/call-controller.test.ts +120 -0
  3. package/src/__tests__/config-loader-backfill.test.ts +19 -0
  4. package/src/__tests__/config-schema.test.ts +139 -5
  5. package/src/__tests__/events-dev-bypass-actor.test.ts +112 -1
  6. package/src/__tests__/media-stream-output.test.ts +175 -0
  7. package/src/__tests__/media-stream-stt-session.test.ts +67 -0
  8. package/src/calls/__tests__/tts-text-sanitizer.test.ts +13 -0
  9. package/src/calls/__tests__/voice-session-bridge.test.ts +81 -12
  10. package/src/calls/__tests__/voice-triage-escalate.test.ts +106 -0
  11. package/src/calls/call-controller.ts +30 -2
  12. package/src/calls/call-speech-output.ts +12 -4
  13. package/src/calls/call-transport.ts +18 -1
  14. package/src/calls/media-stream-output.ts +53 -5
  15. package/src/calls/media-stream-server.ts +12 -0
  16. package/src/calls/media-stream-stt-session.ts +34 -0
  17. package/src/calls/telephony-synthesis-language.ts +84 -0
  18. package/src/calls/tts-text-sanitizer.ts +12 -5
  19. package/src/calls/voice-session-bridge.ts +31 -3
  20. package/src/calls/voice-triage-escalate.ts +52 -5
  21. package/src/config/bundled-skills/phone-calls/references/CONFIG.md +15 -15
  22. package/src/config/loader.ts +5 -0
  23. package/src/config/schemas/calls.ts +0 -4
  24. package/src/config/schemas/tts.ts +63 -0
  25. package/src/live-voice/__tests__/front-decision.test.ts +120 -0
  26. package/src/live-voice/__tests__/live-voice-events.test.ts +1 -1
  27. package/src/live-voice/__tests__/live-voice-progress.test.ts +178 -11
  28. package/src/live-voice/__tests__/live-voice-stt.test.ts +304 -1
  29. package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +102 -2
  30. package/src/live-voice/__tests__/live-voice-tts.test.ts +148 -2
  31. package/src/live-voice/__tests__/progress-phrases.test.ts +167 -0
  32. package/src/live-voice/front-decision.ts +50 -3
  33. package/src/live-voice/live-voice-session.ts +202 -25
  34. package/src/live-voice/live-voice-tts.ts +18 -2
  35. package/src/live-voice/progress-phrases.ts +105 -2
  36. package/src/providers/speech-to-text/deepgram-realtime.test.ts +283 -1
  37. package/src/providers/speech-to-text/deepgram-realtime.ts +117 -3
  38. package/src/providers/speech-to-text/provider-catalog.ts +38 -0
  39. package/src/runtime/__tests__/local-actor-identity-force-refresh.test.ts +104 -0
  40. package/src/runtime/assistant-event-hub.ts +23 -0
  41. package/src/runtime/local-actor-identity.ts +18 -5
  42. package/src/runtime/routes/__tests__/sse-actor-principal-heal.test.ts +165 -0
  43. package/src/runtime/routes/__tests__/surface-action-routes.test.ts +2 -0
  44. package/src/runtime/routes/events-routes.ts +17 -16
  45. package/src/runtime/routes/sse-actor-principal-heal.ts +113 -0
  46. package/src/stt/__tests__/language-metadata.test.ts +85 -0
  47. package/src/stt/language-metadata.ts +65 -0
  48. package/src/stt/types.ts +16 -0
  49. package/src/tts/__tests__/provider-adapters.test.ts +147 -0
  50. package/src/tts/__tests__/speakable-segments.test.ts +470 -0
  51. package/src/tts/language-voices.ts +23 -0
  52. package/src/tts/providers/deepgram-provider.ts +3 -1
  53. package/src/tts/providers/elevenlabs-provider.ts +73 -1
  54. package/src/tts/providers/xai-provider.ts +28 -2
  55. package/src/tts/speakable-segments.ts +293 -23
  56. package/src/tts/synthesis-stream.ts +7 -0
  57. package/src/tts/types.ts +7 -0
  58. package/src/util/__tests__/language-subtag.test.ts +54 -0
  59. package/src/util/language-subtag.ts +43 -0
  60. package/src/util/unicode.ts +1 -1
@@ -25,7 +25,8 @@ import {
25
25
  classifyFrontDoorLeading,
26
26
  ESCALATE_VERDICT_TOKEN,
27
27
  ESCALATION_CONTINUATION_CONTENT,
28
- FALLBACK_ESCALATION_BRIDGE,
28
+ FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
29
+ fallbackEscalationBridgeFor,
29
30
  isEscalationBridgeComplete,
30
31
  MIN_SPOKEN_BRIDGE_CHARS,
31
32
  type VoiceRoutingLeg,
@@ -46,14 +47,20 @@ import { isInstalledStaticSkillLoad } from "../permissions/checker.js";
46
47
  import { ensureConversationExists } from "../persistence/conversation-crud.js";
47
48
  import {
48
49
  listProviderIds,
50
+ pinnedListeningLanguage,
49
51
  supportsBoundary,
50
52
  } from "../providers/speech-to-text/provider-catalog.js";
51
53
  import type { ResolveStreamingTranscriberOptions } from "../providers/speech-to-text/resolve.js";
52
54
  import { broadcastMessage } from "../runtime/assistant-event-hub.js";
53
55
  import { publishConversationListAndMetadataChanged } from "../runtime/sync/resource-sync-events.js";
56
+ import {
57
+ dominantLanguageTag,
58
+ voteDominantLanguage,
59
+ } from "../stt/language-metadata.js";
54
60
  import { detectPcm16SpeechActivity } from "../stt/speech-energy.js";
55
61
  import type {
56
62
  StreamingTranscriber,
63
+ SttProviderId,
57
64
  SttStreamServerErrorEvent,
58
65
  SttStreamServerEvent,
59
66
  } from "../stt/types.js";
@@ -62,6 +69,7 @@ import { liveVoiceEndScreen } from "../telemetry/live-voice-funnel.js";
62
69
  import { getToolOwner } from "../tools/registry.js";
63
70
  import { extractSpeakableSegments } from "../tts/speakable-segments.js";
64
71
  import { createAbortReason } from "../util/abort-reasons.js";
72
+ import { hasLocalizedEntry } from "../util/language-subtag.js";
65
73
  import { getLogger } from "../util/logger.js";
66
74
  import {
67
75
  activityLabelForTool,
@@ -101,7 +109,12 @@ import type {
101
109
  LiveVoiceTtsOptions,
102
110
  LiveVoiceTtsResult,
103
111
  } from "./live-voice-tts.js";
104
- import { pickProgressPhrase } from "./progress-phrases.js";
112
+ import {
113
+ APPROVAL_PENDING_PHRASE_BY_LANGUAGE,
114
+ approvalPendingPhraseFor,
115
+ pickProgressPhrase,
116
+ PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE,
117
+ } from "./progress-phrases.js";
105
118
  import {
106
119
  type LiveVoiceClientAttachImageFrame,
107
120
  type LiveVoiceClientFrame,
@@ -369,6 +382,23 @@ interface UtteranceCycle {
369
382
  // text replays the boundary immediately — the hold was judged on stale
370
383
  // text, so waiting out the extension only adds silence.
371
384
  heldSpeculativeContent: string | null;
385
+ // Count per detected-language base subtag (see voteDominantLanguage)
386
+ // across this cycle's final transcript events. Resolves the turn's spoken
387
+ // language (see turnLanguageFor); empty when the provider tags nothing.
388
+ languageTally: Map<string, number>;
389
+ // Detected languages of the most recent partial that carried any, already
390
+ // normalized, dominance order. Speculative turns dispatch from partials
391
+ // before the first tagged final lands, so turnLanguageFor falls back to
392
+ // this when the final tally is still empty. Never cleared: the tally
393
+ // outranks it once finals arrive, and a revising partial without tags
394
+ // must not wipe an earlier partial's detection.
395
+ latestPartialLanguages: readonly string[] | null;
396
+ // The provider that actually transcribed this cycle, recorded when its
397
+ // transcriber is assigned and kept after teardown nulls `transcriber`.
398
+ // The resolver can silently dial managed vellum when a BYOK provider has
399
+ // no credential, so the language-pin gate in turnLanguageFor must follow
400
+ // this, not the configured provider.
401
+ dialedSttProvider: SttProviderId | null;
372
402
  turnId: string | null;
373
403
  userMessageId: string | null;
374
404
  userAudioChunks: Buffer[];
@@ -399,6 +429,11 @@ type UtteranceStartResult =
399
429
  // client in job-list order.
400
430
  interface TtsSegmentJob {
401
431
  readonly text: string;
432
+ // Per-segment language-hint override, preferred over the turn's language.
433
+ // Set on fixed phrases whose localized table lacks the turn's language:
434
+ // the English fallback text carries "en" so an enforcing provider never
435
+ // renders English words as ar/ko/ta. Undefined means the turn language.
436
+ readonly language: string | undefined;
402
437
  // The provider stream was started (the job holds an open-job slot).
403
438
  started: boolean;
404
439
  // Emission finished; the slot is free for the next queued segment.
@@ -483,6 +518,13 @@ interface ActiveAssistantTurn {
483
518
  token: symbol;
484
519
  turnId: string;
485
520
  utterance: UtteranceCycle;
521
+ // The caller's spoken language for this turn as a lowercase base subtag
522
+ // (see turnLanguageFor): the dominant STT-detected language, else a
523
+ // monolingual services.stt.language pin. Undefined when unknown, which
524
+ // disables every language-aware path (prompt note, TTS hint, localized
525
+ // fallbacks). Re-resolved when a speculative turn commits, since finals
526
+ // can land between dispatch and verdict.
527
+ language: string | undefined;
486
528
  abortController: AbortController;
487
529
  handle: VoiceTurnHandle | null;
488
530
  // When the turn launched, for narration's turnElapsedMs.
@@ -681,14 +723,8 @@ function createControlMarkerHoldback(
681
723
  // barge-in, the interruption merge note is appended to it (see
682
724
  // buildInterruptionMergeNote) so the model reconciles the interrupted request
683
725
  // with the new utterance.
684
- // Spoken once when a turn starts waiting on the user's decision. Fixed rather
685
- // than generated: this is a statement about the system's state, not about the
686
- // work, and it has to be true every time. Kept in the shape of the progress
687
- // phrases it displaces (short, neutral, no claim about tools).
688
- const APPROVAL_PENDING_PHRASE = "I need your okay for that one. Take a look.";
689
-
690
726
  const LIVE_VOICE_CONTROL_PROMPT_BASE =
691
- "You are speaking in a local live voice session. Keep replies brief and conversational. Speech is the main channel: say the answer, and do not narrate a surface instead of answering. You can also put something on screen when it genuinely helps (a form, a list to pick from, a progress card for long work); the call overlay minimizes by itself once you finish speaking, so the user sees it without doing anything. Never tell the user you cannot show them something. ";
727
+ "You are speaking in a local live voice session. Keep replies brief and conversational. Speech is the main channel: say the answer, and do not narrate a surface instead of answering. You can also put something on screen when it genuinely helps (a form, a list to pick from, a progress card for long work); the call overlay minimizes by itself once you finish speaking, so the user sees it without doing anything. Never tell the user you cannot show them something. Reply in the language the caller is speaking; if they switch languages, switch with them. ";
692
728
 
693
729
  // Appended for the legs that can actually put something on screen: the main
694
730
  // leg and the escalated leg. The front-door (fast) leg never receives it, for
@@ -777,6 +813,9 @@ function buildVoiceControlPrompt(
777
813
  LIVE_VOICE_CONTROL_PROMPT_BASE +
778
814
  (leg.frontDoor === true ? "" : LIVE_VOICE_SCREEN_REVEAL_TEACHING) +
779
815
  VOICE_NO_SETUP_FLOWS_RULE;
816
+ if (turn.language !== undefined) {
817
+ prompt = `${prompt}\n\nThe caller has been speaking the language with code "${turn.language}" this turn. Reply in that language unless they clearly switch to another.`;
818
+ }
780
819
  if (turn.interruptedRequest) {
781
820
  prompt = `${prompt}\n\n${buildInterruptionMergeNote(turn.interruptedRequest)}`;
782
821
  }
@@ -825,6 +864,9 @@ function createUtteranceCycle(): UtteranceCycle {
825
864
  latestPartialText: null,
826
865
  endpointExtensionCount: 0,
827
866
  heldSpeculativeContent: null,
867
+ languageTally: new Map(),
868
+ latestPartialLanguages: null,
869
+ dialedSttProvider: null,
828
870
  turnId: null,
829
871
  userMessageId: null,
830
872
  userAudioChunks: [],
@@ -1387,6 +1429,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1387
1429
  // Persistent re-arm: the shared stream is already open, so the cycle
1388
1430
  // goes straight to streaming with no resolve/start round-trip.
1389
1431
  utterance.transcriber = shared;
1432
+ utterance.dialedSttProvider = shared.providerId;
1390
1433
  return await this.activateUtterance(utterance, replayTurnEnd);
1391
1434
  }
1392
1435
  // The shared stream is pinned to the old language, so retire it and
@@ -1419,6 +1462,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
1419
1462
  }
1420
1463
 
1421
1464
  utterance.transcriber = transcriber;
1465
+ utterance.dialedSttProvider = transcriber.providerId;
1422
1466
  if (
1423
1467
  this.turnDetector &&
1424
1468
  typeof transcriber.finalizeUtterance === "function"
@@ -2527,8 +2571,13 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
2527
2571
  // Spoken, because opening the room is only a cue for someone looking at
2528
2572
  // the screen, and the case this exists for is a phone the user has put
2529
2573
  // down. One line, not narration: the turn is not working, it is waiting,
2530
- // and it says which.
2531
- this.enqueueFillerPhrase(turn, APPROVAL_PENDING_PHRASE);
2574
+ // and it says which, in the turn's spoken language, like every other
2575
+ // filler phrase.
2576
+ this.enqueueFillerPhrase(
2577
+ turn,
2578
+ approvalPendingPhraseFor(turn.language),
2579
+ this.fixedPhraseLanguage(turn, APPROVAL_PENDING_PHRASE_BY_LANGUAGE),
2580
+ );
2532
2581
  }
2533
2582
 
2534
2583
  /** Clear the wait once a decision lands, so the turn narrates normally again. */
@@ -2851,6 +2900,13 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
2851
2900
  // are skipped; the thinking frame and timers still apply.
2852
2901
  const alreadyReleased = utterance.released;
2853
2902
  turn.speculativePending = false;
2903
+ // Finals can land between the speculative dispatch and this verdict.
2904
+ // Fill the language only when dispatch had none: the model request was
2905
+ // already issued with the dispatch language, so overwriting here would
2906
+ // hint TTS (and any voice override) in a different language than the
2907
+ // text it speaks. The tally still carries the corrected detection into
2908
+ // the next turn.
2909
+ turn.language ??= this.turnLanguageFor(utterance);
2854
2910
  if (turn.verdictDeadlineTimer !== null) {
2855
2911
  clearTimeout(turn.verdictDeadlineTimer);
2856
2912
  turn.verdictDeadlineTimer = null;
@@ -3189,11 +3245,16 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3189
3245
  switch (event.type) {
3190
3246
  case "partial":
3191
3247
  utterance.latestPartialText = event.text;
3248
+ this.capturePartialLanguages(utterance, event.languages);
3192
3249
  this.markFirstPartial(utterance);
3193
3250
  await this.sendFrame({ type: "stt_partial", text: event.text });
3194
3251
  return;
3195
3252
  case "final":
3196
- await this.recordFinalTranscript(utterance, event.text);
3253
+ await this.recordFinalTranscript(
3254
+ utterance,
3255
+ event.text,
3256
+ event.languages,
3257
+ );
3197
3258
  return;
3198
3259
  case "finalized":
3199
3260
  // Per-cycle transcribers are torn down with stop(); the finalize
@@ -3262,6 +3323,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3262
3323
  return;
3263
3324
  }
3264
3325
  target.latestPartialText = event.text;
3326
+ this.capturePartialLanguages(target, event.languages);
3265
3327
  this.markFirstPartial(target);
3266
3328
  await this.sendFrame({ type: "stt_partial", text: event.text });
3267
3329
  return;
@@ -3277,7 +3339,11 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3277
3339
  // newer cycle.
3278
3340
  const owner = this.finalizeQueue[0];
3279
3341
  if (owner && !owner.assistantTurnStarted && !owner.completed) {
3280
- await this.recordFinalTranscript(owner, event.text);
3342
+ await this.recordFinalTranscript(
3343
+ owner,
3344
+ event.text,
3345
+ event.languages,
3346
+ );
3281
3347
  } else {
3282
3348
  log.warn(
3283
3349
  "Dropping a late finalize flush segment: its assistant turn already dispatched",
@@ -3294,7 +3360,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3294
3360
  );
3295
3361
  return;
3296
3362
  }
3297
- await this.recordFinalTranscript(target, event.text);
3363
+ await this.recordFinalTranscript(target, event.text, event.languages);
3298
3364
  return;
3299
3365
  }
3300
3366
  case "finalized": {
@@ -3363,10 +3429,16 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3363
3429
  private async recordFinalTranscript(
3364
3430
  utterance: UtteranceCycle,
3365
3431
  text: string,
3432
+ languages?: readonly string[],
3366
3433
  ): Promise<void> {
3367
3434
  const transcript = text.trim();
3368
3435
  if (transcript.length > 0) {
3369
3436
  utterance.finalTranscriptSegments.push(transcript);
3437
+ // Tally only finals that committed transcript: empty silence frames
3438
+ // can still carry container-level language tags describing no emitted
3439
+ // words, and counting those would let silence outvote real speech
3440
+ // (same choice as the adapter's boundary-final aggregation).
3441
+ voteDominantLanguage(utterance.languageTally, languages);
3370
3442
  }
3371
3443
  // The final commits (and supersedes) whatever partial was trailing it.
3372
3444
  utterance.latestPartialText = null;
@@ -3405,6 +3477,59 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3405
3477
  await this.startAssistantTurnIfReady();
3406
3478
  }
3407
3479
 
3480
+ // Record a partial event's detected languages so speculative dispatch
3481
+ // has a detection before the first tagged final. The event contract
3482
+ // (stt/types.ts) guarantees the tags arrive as normalized base subtags
3483
+ // in dominance order, so they are stored as-is. Partials revise each
3484
+ // other, so this overwrites rather than tallies, and a tag-less partial
3485
+ // keeps the previous value.
3486
+ private capturePartialLanguages(
3487
+ utterance: UtteranceCycle,
3488
+ languages: readonly string[] | undefined,
3489
+ ): void {
3490
+ if (!languages || languages.length === 0) {
3491
+ return;
3492
+ }
3493
+ utterance.latestPartialLanguages = languages;
3494
+ }
3495
+
3496
+ /**
3497
+ * The caller's spoken language for a turn on this utterance, as a
3498
+ * lowercase base subtag: the dominant tallied STT-detected language
3499
+ * (most final-event counts, ties by first appearance), else the latest
3500
+ * tagged partial's dominant language (speculative turns dispatch from
3501
+ * partials), else a monolingual `services.stt.language` pin (a pinned
3502
+ * language IS the spoken language), else undefined ("multi" with no tags,
3503
+ * non-tagging providers, silence).
3504
+ */
3505
+ private turnLanguageFor(utterance: UtteranceCycle): string | undefined {
3506
+ const dominant = dominantLanguageTag(utterance.languageTally);
3507
+ if (dominant !== undefined) {
3508
+ return dominant;
3509
+ }
3510
+ // No tagged final yet (speculative turns dispatch from partials): the
3511
+ // latest tagged partial is the best detection available and outranks a
3512
+ // static pin for the same reason the tally does.
3513
+ const partialDominant = utterance.latestPartialLanguages?.[0];
3514
+ if (partialDominant !== undefined) {
3515
+ return partialDominant;
3516
+ }
3517
+ // A persisted pin only counts when the provider that actually
3518
+ // transcribed honors manual language selection (the shared
3519
+ // pinnedListeningLanguage gate). The DIALED transcriber's providerId
3520
+ // is authoritative, because the resolver silently falls back to
3521
+ // managed vellum (which honors the pin) when a BYOK provider has no
3522
+ // credential; the configured provider is only the last resort when no
3523
+ // transcriber reference survives.
3524
+ const { language: configured, provider: sttProvider } =
3525
+ getConfig().services.stt;
3526
+ const dialedProvider =
3527
+ utterance.dialedSttProvider ??
3528
+ this.sharedTranscriber?.providerId ??
3529
+ (sttProvider as SttProviderId);
3530
+ return pinnedListeningLanguage(dialedProvider, configured);
3531
+ }
3532
+
3408
3533
  // Providers emit `error` mid-stream and may keep streaming; `closed` /
3409
3534
  // `final` still drive turn lifecycle. Only transient categories are
3410
3535
  // recoverable — auth/rate-limit/invalid-audio will not self-heal, so
@@ -3660,6 +3785,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
3660
3785
  token,
3661
3786
  turnId,
3662
3787
  utterance,
3788
+ language: this.turnLanguageFor(utterance),
3663
3789
  abortController,
3664
3790
  handle: null,
3665
3791
  launchedAtMs: Date.now(),
@@ -4312,7 +4438,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4312
4438
  // deleted row for a bridge the model never produced).
4313
4439
  const usesFallbackBridge = cappedBridge.length < MIN_SPOKEN_BRIDGE_CHARS;
4314
4440
  const spokenBridge = usesFallbackBridge
4315
- ? FALLBACK_ESCALATION_BRIDGE
4441
+ ? fallbackEscalationBridgeFor(activeTurn.language)
4316
4442
  : cappedBridge;
4317
4443
  if (!usesFallbackBridge) {
4318
4444
  this.markFirstAssistantDelta(activeTurn.utterance, activeTurn.turnId);
@@ -4321,12 +4447,25 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4321
4447
  { type: "assistant_text_delta", text: spokenBridge },
4322
4448
  () => !activeTurn.abortController.signal.aborted && !this.isClosed,
4323
4449
  );
4450
+ this.bufferAssistantTextForTts(activeTurn.token, `${spokenBridge} `);
4451
+ // Force-flush now: on the TTS path an unpunctuated bridge would
4452
+ // otherwise sit buffered until a sentence boundary and leave the
4453
+ // caller in silence during the escalated model's call.
4454
+ this.flushTtsBuffer(activeTurn.token, true);
4455
+ } else {
4456
+ // The canned bridge is a fixed localized-table phrase, enqueued
4457
+ // directly (it is already one complete sentence) so the segment can
4458
+ // carry the "en" override when the table lacks the turn's language.
4459
+ const speakable = sanitizeForTts(spokenBridge).trim();
4460
+ if (speakable.length > 0) {
4461
+ this.enqueueTtsSegment(activeTurn.token, speakable, {
4462
+ language: this.fixedPhraseLanguage(
4463
+ activeTurn,
4464
+ FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
4465
+ ),
4466
+ });
4467
+ }
4324
4468
  }
4325
- this.bufferAssistantTextForTts(activeTurn.token, `${spokenBridge} `);
4326
- // Force-flush now: on the TTS path an unpunctuated bridge would otherwise
4327
- // sit buffered until a sentence boundary and leave the caller in silence
4328
- // during the escalated model's call.
4329
- this.flushTtsBuffer(activeTurn.token, true);
4330
4469
 
4331
4470
  // No overrideProfile: the escalated leg runs on the call-site default —
4332
4471
  // the exact profile an un-routed voice turn would use (see
@@ -4690,6 +4829,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4690
4829
  : null,
4691
4830
  turnElapsedMs: now - turn.launchedAtMs,
4692
4831
  updateIndex: progress.updatesSpoken + 1,
4832
+ ...(turn.language !== undefined ? { languageHint: turn.language } : {}),
4693
4833
  };
4694
4834
  const generated = await frontDecider
4695
4835
  .generateProgressText(input, turn.abortController.signal)
@@ -4718,13 +4858,20 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4718
4858
  return;
4719
4859
  }
4720
4860
  let raw = generated;
4861
+ // Decider text is generated in the turn's language; only the static
4862
+ // fallback comes from a localized table and may need the "en" override.
4863
+ let fillerLanguage: string | undefined;
4721
4864
  if (raw === null) {
4722
4865
  if (trigger !== "idle") {
4723
4866
  return;
4724
4867
  }
4725
- raw = pickProgressPhrase(this.progressPhraseCounter++);
4868
+ raw = pickProgressPhrase(this.progressPhraseCounter++, turn.language);
4869
+ fillerLanguage = this.fixedPhraseLanguage(
4870
+ turn,
4871
+ PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE,
4872
+ );
4726
4873
  }
4727
- if (!this.enqueueFillerPhrase(turn, raw)) {
4874
+ if (!this.enqueueFillerPhrase(turn, raw, fillerLanguage)) {
4728
4875
  return;
4729
4876
  }
4730
4877
  progress.opsSinceNarration = 0;
@@ -4784,6 +4931,9 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4784
4931
  {
4785
4932
  transcriptSoFar: transcript,
4786
4933
  toolName,
4934
+ ...(activeTurn.language !== undefined
4935
+ ? { languageHint: activeTurn.language }
4936
+ : {}),
4787
4937
  },
4788
4938
  activeTurn.abortController.signal,
4789
4939
  )
@@ -4821,20 +4971,42 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4821
4971
  // Sanitize and enqueue one filler sentence (spoken ack or progress
4822
4972
  // narration) on the turn's ordered TTS queue — the shared tail of every
4823
4973
  // filler path. Returns whether a phrase actually enqueued; per-kind metric
4824
- // marks and bookkeeping are the caller's.
4825
- private enqueueFillerPhrase(turn: ActiveAssistantTurn, raw: string): boolean {
4974
+ // marks and bookkeeping are the caller's. `language` is a per-segment
4975
+ // hint override (see fixedPhraseLanguage); omit it for generated text,
4976
+ // which is already in the turn's language.
4977
+ private enqueueFillerPhrase(
4978
+ turn: ActiveAssistantTurn,
4979
+ raw: string,
4980
+ language?: string,
4981
+ ): boolean {
4826
4982
  const phrase = sanitizeForTts(raw).trim();
4827
4983
  if (phrase.length === 0) {
4828
4984
  return false;
4829
4985
  }
4830
4986
  this.enqueueTtsSegment(turn.token, phrase, {
4831
4987
  countsAsFirstSegment: false,
4988
+ ...(language !== undefined ? { language } : {}),
4832
4989
  });
4833
4990
  // A spoken filler holds the floor, so narration's minGapMs spaces from it.
4834
4991
  turn.progress.lastFloorHolderAtMs = Date.now();
4835
4992
  return true;
4836
4993
  }
4837
4994
 
4995
+ // The TTS hint override for a fixed phrase picked from a localized table:
4996
+ // "en" when the turn has a language the table does not cover (the picker
4997
+ // fell back to English text, which must not be synthesized under an
4998
+ // ar/ko/ta hint), undefined otherwise (the segment rides the turn's
4999
+ // language, or no hint at all when the language is unknown).
5000
+ private fixedPhraseLanguage(
5001
+ turn: ActiveAssistantTurn,
5002
+ table: Readonly<Record<string, unknown>>,
5003
+ ): string | undefined {
5004
+ return turn.language !== undefined &&
5005
+ !hasLocalizedEntry(table, turn.language)
5006
+ ? "en"
5007
+ : undefined;
5008
+ }
5009
+
4838
5010
  private bufferAssistantTextForTts(token: symbol, text: string): void {
4839
5011
  if (!this.streamTtsAudio || text.length === 0) {
4840
5012
  return;
@@ -4955,7 +5127,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4955
5127
  private enqueueTtsSegment(
4956
5128
  token: symbol,
4957
5129
  segment: string,
4958
- options: { countsAsFirstSegment?: boolean } = {},
5130
+ options: { countsAsFirstSegment?: boolean; language?: string } = {},
4959
5131
  ): void {
4960
5132
  const activeTurn = this.activeAssistantTurn;
4961
5133
  if (activeTurn?.token !== token || !this.streamTtsAudio) {
@@ -4969,6 +5141,7 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
4969
5141
  }
4970
5142
  const job: TtsSegmentJob = {
4971
5143
  text: segment,
5144
+ language: options.language,
4972
5145
  started: false,
4973
5146
  settled: false,
4974
5147
  emitting: false,
@@ -5008,10 +5181,14 @@ export class LiveVoiceSession implements LiveVoiceSessionContract {
5008
5181
  return;
5009
5182
  }
5010
5183
  job.started = true;
5184
+ // The segment's own language override (fixed English fallback text)
5185
+ // wins over the turn's language.
5186
+ const language = job.language ?? activeTurn.language;
5011
5187
  let synthesis: Promise<void>;
5012
5188
  try {
5013
5189
  synthesis = streamTtsAudio({
5014
5190
  text: job.text,
5191
+ ...(language !== undefined ? { language } : {}),
5015
5192
  signal: activeTurn.abortController.signal,
5016
5193
  outputFormat: "pcm",
5017
5194
  sampleRate: this.context.startFrame.audio.sampleRate,
@@ -1,3 +1,4 @@
1
+ import { resolveLanguageVoiceOverride } from "../tts/language-voices.js";
1
2
  import { createPcmChunkAligner } from "../tts/pcm-chunk-aligner.js";
2
3
  import { getTtsProvider } from "../tts/provider-catalog.js";
3
4
  import { synthesizeAndEmit } from "../tts/synthesis-stream.js";
@@ -30,6 +31,7 @@ export interface LiveVoiceTtsOptions {
30
31
  useCase?: TtsUseCase;
31
32
  outputFormat?: TtsSynthesisRequest["outputFormat"];
32
33
  sampleRate?: number;
34
+ language?: string;
33
35
  config?: LiveVoiceTtsConfig;
34
36
  onAudioChunk: (chunk: LiveVoiceTtsAudioChunk) => void;
35
37
  }
@@ -72,11 +74,23 @@ interface ResolvedStreamingTtsProvider {
72
74
  providerConfig: Record<string, unknown>;
73
75
  }
74
76
 
77
+ export { resolveLanguageVoiceOverride };
78
+
75
79
  export async function streamLiveVoiceTtsAudio(
76
80
  options: LiveVoiceTtsOptions,
77
81
  ): Promise<LiveVoiceTtsResult> {
78
82
  const { provider, providerId, providerConfig } =
79
83
  await resolveLiveVoiceStreamingTtsProvider(options.config);
84
+ // An explicit request voice wins outright; otherwise a language-known
85
+ // turn may select the provider's configured per-language voice. The cast
86
+ // recovers the schema-typed map that resolveTtsConfig's generic
87
+ // provider-block lookup erases.
88
+ const voiceId =
89
+ options.voiceId ??
90
+ resolveLanguageVoiceOverride(
91
+ providerConfig.languageVoices as Record<string, string> | undefined,
92
+ options.language,
93
+ );
80
94
  const useCase = options.useCase ?? "phone-call";
81
95
  const requestedSampleRate = resolveSampleRate(
82
96
  options.sampleRate,
@@ -89,9 +103,10 @@ export async function streamLiveVoiceTtsAudio(
89
103
  const providerSampleRate = provider.resolveOutputSampleRateHz?.({
90
104
  text: options.text,
91
105
  useCase,
92
- voiceId: options.voiceId,
106
+ voiceId,
93
107
  outputFormat: options.outputFormat,
94
108
  sampleRateHz: requestedSampleRate,
109
+ language: options.language,
95
110
  signal: options.signal,
96
111
  });
97
112
  const sampleRate = providerSampleRate ?? requestedSampleRate;
@@ -138,9 +153,10 @@ export async function streamLiveVoiceTtsAudio(
138
153
  provider,
139
154
  text: options.text,
140
155
  useCase,
141
- voiceId: options.voiceId,
156
+ voiceId,
142
157
  outputFormat: options.outputFormat,
143
158
  sampleRateHz: requestedSampleRate,
159
+ language: options.language,
144
160
  signal: options.signal,
145
161
  onChunk: (chunk) => {
146
162
  if (canStreamChunks) {
@@ -1,3 +1,5 @@
1
+ import { localizedOrDefault } from "../util/language-subtag.js";
2
+
1
3
  // Static fallbacks for an idle-triggered progress narration whose LLM
2
4
  // phrasing failed — the one case where prolonged silence is actively harmful.
3
5
  // The idle trigger can fire on a slow turn with zero tool activity, so every
@@ -10,8 +12,109 @@ export const PROGRESS_FALLBACK_PHRASES: readonly string[] = [
10
12
  "Almost there — thanks for waiting.",
11
13
  ];
12
14
 
15
+ // Per-language fallback phrases, keyed by lowercased BCP 47 base subtag,
16
+ // covering the Deepgram code-switching roster (DEEPGRAM_MULTI_LANGUAGE_CODES
17
+ // in providers/speech-to-text/deepgram.ts). Every list carries the same
18
+ // invariants as the English one above: persona-neutral, no claims about
19
+ // running tools or tasks, at most 8 words per phrase.
20
+ export const PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE: Readonly<
21
+ Record<string, readonly string[]>
22
+ > = {
23
+ en: PROGRESS_FALLBACK_PHRASES,
24
+ es: [
25
+ "Sigo en ello, un momento.",
26
+ "Todavía lo estoy pensando.",
27
+ "Casi listo, gracias por esperar.",
28
+ ],
29
+ fr: [
30
+ "J'y suis encore, un instant.",
31
+ "J'y réfléchis encore.",
32
+ "Presque fini, merci de patienter.",
33
+ ],
34
+ de: [
35
+ "Bin noch dabei, einen Moment.",
36
+ "Ich denke noch darüber nach.",
37
+ "Fast fertig, danke fürs Warten.",
38
+ ],
39
+ hi: [
40
+ "बस एक पल रुकिए।",
41
+ "अभी इस पर विचार चल रहा है।",
42
+ "बस थोड़ा और इंतज़ार कीजिए, धन्यवाद।",
43
+ ],
44
+ ru: [
45
+ "Секундочку, я ещё здесь.",
46
+ "Я всё ещё думаю над этим.",
47
+ "Почти готово, спасибо за ожидание.",
48
+ ],
49
+ pt: [
50
+ "Ainda estou nisso, um momento.",
51
+ "Ainda estou pensando nisso.",
52
+ "Quase lá, agradeço a espera.",
53
+ ],
54
+ ja: [
55
+ "まだ対応中です。少々お待ちください。",
56
+ "まだ考えているところです。",
57
+ "もうすぐです。お待ちいただきありがとうございます。",
58
+ ],
59
+ it: [
60
+ "Ancora un attimo, per favore.",
61
+ "Ci sto ancora pensando.",
62
+ "Quasi fatto, grazie per l'attesa.",
63
+ ],
64
+ nl: [
65
+ "Ik ben er nog mee bezig.",
66
+ "Ik denk er nog over na.",
67
+ "Bijna klaar, bedankt voor het wachten.",
68
+ ],
69
+ };
70
+
13
71
  // Deterministic rotation through the phrase list: callers hold a nonnegative
14
72
  // monotonic counter, so consecutive picks vary while tests stay reproducible.
15
- export function pickProgressPhrase(counter: number): string {
16
- return PROGRESS_FALLBACK_PHRASES[counter % PROGRESS_FALLBACK_PHRASES.length];
73
+ // `language` selects the per-language list by its lowercased base subtag
74
+ // (e.g. "pt-BR" -> "pt"); unknown or absent languages fall back to English.
75
+ export function pickProgressPhrase(counter: number, language?: string): string {
76
+ const phrases = localizedOrDefault(
77
+ PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE,
78
+ language,
79
+ PROGRESS_FALLBACK_PHRASES,
80
+ );
81
+ return phrases[counter % phrases.length];
82
+ }
83
+
84
+ // Spoken once when a turn starts waiting on the user's approval decision.
85
+ // Fixed rather than generated: this is a statement about the system's
86
+ // state, not about the work, and it has to be true every time. Kept in the
87
+ // shape of the progress phrases it displaces (short, neutral, no claim
88
+ // about tools).
89
+ export const APPROVAL_PENDING_PHRASE =
90
+ "I need your okay for that one. Take a look.";
91
+
92
+ // Per-language spellings of the approval-pending phrase, keyed by lowercased
93
+ // BCP 47 base subtag, covering the Deepgram code-switching roster
94
+ // (DEEPGRAM_MULTI_LANGUAGE_CODES in providers/speech-to-text/deepgram.ts).
95
+ // Same invariants as the progress phrases: persona-neutral, no claims about
96
+ // running tools or tasks.
97
+ export const APPROVAL_PENDING_PHRASE_BY_LANGUAGE: Readonly<
98
+ Record<string, string>
99
+ > = {
100
+ en: APPROVAL_PENDING_PHRASE,
101
+ es: "Necesito tu visto bueno para eso. Échale un vistazo.",
102
+ fr: "J'ai besoin de ton accord pour ça. Jette un œil.",
103
+ de: "Dafür brauche ich dein Okay. Schau mal drauf.",
104
+ hi: "इसके लिए मुझे आपकी मंज़ूरी चाहिए। एक नज़र डाल लीजिए।",
105
+ ru: "Для этого мне нужно твоё согласие. Взгляни, пожалуйста.",
106
+ pt: "Preciso do seu ok para isso. Dê uma olhada.",
107
+ ja: "これには許可が必要です。ご確認ください。",
108
+ it: "Mi serve il tuo via libera per questo. Dai un'occhiata.",
109
+ nl: "Hiervoor heb ik je akkoord nodig. Kijk even mee.",
110
+ };
111
+
112
+ // The approval-pending phrase in the turn's spoken language, defaulting to
113
+ // English for unknown or absent languages.
114
+ export function approvalPendingPhraseFor(language?: string): string {
115
+ return localizedOrDefault(
116
+ APPROVAL_PENDING_PHRASE_BY_LANGUAGE,
117
+ language,
118
+ APPROVAL_PENDING_PHRASE,
119
+ );
17
120
  }