@vellumai/assistant 0.11.3-staging.2 → 0.11.3-staging.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/package.json +1 -1
  2. package/src/__tests__/call-controller.test.ts +120 -0
  3. package/src/__tests__/config-loader-backfill.test.ts +19 -0
  4. package/src/__tests__/config-schema.test.ts +139 -5
  5. package/src/__tests__/events-dev-bypass-actor.test.ts +112 -1
  6. package/src/__tests__/media-stream-output.test.ts +175 -0
  7. package/src/__tests__/media-stream-stt-session.test.ts +67 -0
  8. package/src/calls/__tests__/tts-text-sanitizer.test.ts +13 -0
  9. package/src/calls/__tests__/voice-session-bridge.test.ts +81 -12
  10. package/src/calls/__tests__/voice-triage-escalate.test.ts +106 -0
  11. package/src/calls/call-controller.ts +30 -2
  12. package/src/calls/call-speech-output.ts +12 -4
  13. package/src/calls/call-transport.ts +18 -1
  14. package/src/calls/media-stream-output.ts +53 -5
  15. package/src/calls/media-stream-server.ts +12 -0
  16. package/src/calls/media-stream-stt-session.ts +34 -0
  17. package/src/calls/telephony-synthesis-language.ts +84 -0
  18. package/src/calls/tts-text-sanitizer.ts +12 -5
  19. package/src/calls/voice-session-bridge.ts +31 -3
  20. package/src/calls/voice-triage-escalate.ts +52 -5
  21. package/src/config/bundled-skills/phone-calls/references/CONFIG.md +15 -15
  22. package/src/config/loader.ts +5 -0
  23. package/src/config/schemas/calls.ts +0 -4
  24. package/src/config/schemas/tts.ts +63 -0
  25. package/src/live-voice/__tests__/front-decision.test.ts +120 -0
  26. package/src/live-voice/__tests__/live-voice-events.test.ts +1 -1
  27. package/src/live-voice/__tests__/live-voice-progress.test.ts +178 -11
  28. package/src/live-voice/__tests__/live-voice-stt.test.ts +304 -1
  29. package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +102 -2
  30. package/src/live-voice/__tests__/live-voice-tts.test.ts +148 -2
  31. package/src/live-voice/__tests__/progress-phrases.test.ts +167 -0
  32. package/src/live-voice/front-decision.ts +50 -3
  33. package/src/live-voice/live-voice-session.ts +202 -25
  34. package/src/live-voice/live-voice-tts.ts +18 -2
  35. package/src/live-voice/progress-phrases.ts +105 -2
  36. package/src/providers/speech-to-text/deepgram-realtime.test.ts +283 -1
  37. package/src/providers/speech-to-text/deepgram-realtime.ts +117 -3
  38. package/src/providers/speech-to-text/provider-catalog.ts +38 -0
  39. package/src/runtime/__tests__/local-actor-identity-force-refresh.test.ts +104 -0
  40. package/src/runtime/assistant-event-hub.ts +23 -0
  41. package/src/runtime/local-actor-identity.ts +18 -5
  42. package/src/runtime/routes/__tests__/sse-actor-principal-heal.test.ts +165 -0
  43. package/src/runtime/routes/__tests__/surface-action-routes.test.ts +2 -0
  44. package/src/runtime/routes/events-routes.ts +17 -16
  45. package/src/runtime/routes/sse-actor-principal-heal.ts +113 -0
  46. package/src/stt/__tests__/language-metadata.test.ts +85 -0
  47. package/src/stt/language-metadata.ts +65 -0
  48. package/src/stt/types.ts +16 -0
  49. package/src/tts/__tests__/provider-adapters.test.ts +147 -0
  50. package/src/tts/__tests__/speakable-segments.test.ts +470 -0
  51. package/src/tts/language-voices.ts +23 -0
  52. package/src/tts/providers/deepgram-provider.ts +3 -1
  53. package/src/tts/providers/elevenlabs-provider.ts +73 -1
  54. package/src/tts/providers/xai-provider.ts +28 -2
  55. package/src/tts/speakable-segments.ts +293 -23
  56. package/src/tts/synthesis-stream.ts +7 -0
  57. package/src/tts/types.ts +7 -0
  58. package/src/util/__tests__/language-subtag.test.ts +54 -0
  59. package/src/util/language-subtag.ts +43 -0
  60. package/src/util/unicode.ts +1 -1
@@ -26,6 +26,7 @@ import {
26
26
  } from "../calls/media-stream-audio-transcode.js";
27
27
  import { MediaStreamOutput } from "../calls/media-stream-output.js";
28
28
  import { resolveCallTtsProvider } from "../calls/resolve-call-tts-provider.js";
29
+ import { setConfig } from "./helpers/set-config.js";
29
30
 
30
31
  const mockResolveCallTtsProvider = resolveCallTtsProvider as ReturnType<
31
32
  typeof jest.fn
@@ -1213,6 +1214,180 @@ describe("MediaStreamOutput", () => {
1213
1214
  });
1214
1215
  });
1215
1216
 
1217
+ // ---------------------------------------------------------------------------
1218
+ // Synthesis language hint
1219
+ // ---------------------------------------------------------------------------
1220
+
1221
+ describe("synthesis language hint", () => {
1222
+ /** Synthesize one turn and return the provider request it produced. */
1223
+ async function synthesizeOneTurn(
1224
+ output: MediaStreamOutput,
1225
+ ): Promise<{ language?: string }> {
1226
+ mockSynthesize.mockResolvedValue({
1227
+ audio: makeWavBuffer([1000, 2000, 3000, 4000]),
1228
+ contentType: "audio/wav",
1229
+ });
1230
+ output.sendTextToken("Hello caller.", true);
1231
+ await drain(() => mockSynthesize.mock.calls.length > 0);
1232
+ return mockSynthesize.mock.calls[0][0] as { language?: string };
1233
+ }
1234
+
1235
+ test("the resolver's language rides the provider request", async () => {
1236
+ const { ws } = createMockWs();
1237
+ const output = makeOutput(ws, "stream-lang");
1238
+ output.setSynthesisLanguageResolver(() => "es");
1239
+
1240
+ const request = await synthesizeOneTurn(output);
1241
+
1242
+ expect(request.language).toBe("es");
1243
+ });
1244
+
1245
+ test("fixed system copy synthesizes without the caller-language hint while model text keeps it", async () => {
1246
+ const { ws } = createMockWs();
1247
+ const output = makeOutput(ws, "stream-lang");
1248
+ output.setSynthesisLanguageResolver(() => "ja");
1249
+ mockSynthesize.mockResolvedValue({
1250
+ audio: makeWavBuffer([1000, 2000, 3000, 4000]),
1251
+ contentType: "audio/wav",
1252
+ });
1253
+
1254
+ output.sendTextToken("Model reply.", true);
1255
+ output.sendTextToken("Are you still there?", true, { systemCopy: true });
1256
+ await drain(() => mockSynthesize.mock.calls.length >= 2);
1257
+
1258
+ const requests = mockSynthesize.mock.calls.map(
1259
+ (call) => call[0] as { text: string; language?: string },
1260
+ );
1261
+ expect(requests[0]?.text).toBe("Model reply.");
1262
+ expect(requests[0]?.language).toBe("ja");
1263
+ expect(requests[1]?.text).toBe("Are you still there?");
1264
+ expect(requests[1]?.language).toBeUndefined();
1265
+ });
1266
+
1267
+ test("no language when nothing resolves (default multilingual pin)", async () => {
1268
+ const { ws } = createMockWs();
1269
+ const output = makeOutput(ws, "stream-lang");
1270
+
1271
+ const request = await synthesizeOneTurn(output);
1272
+
1273
+ expect(request.language).toBeUndefined();
1274
+ });
1275
+
1276
+ test("a monolingual pin on a manual-selection provider resolves as the hint", async () => {
1277
+ setConfig("services", {
1278
+ stt: { provider: "deepgram", language: "es-ES" },
1279
+ });
1280
+ try {
1281
+ const { ws } = createMockWs();
1282
+ const output = makeOutput(ws, "stream-lang");
1283
+
1284
+ const request = await synthesizeOneTurn(output);
1285
+
1286
+ expect(request.language).toBe("es");
1287
+ } finally {
1288
+ setConfig("services", {});
1289
+ }
1290
+ });
1291
+
1292
+ test("the pin is ignored for auto-detecting providers", async () => {
1293
+ setConfig("services", {
1294
+ stt: { provider: "openai-whisper", language: "es" },
1295
+ });
1296
+ try {
1297
+ const { ws } = createMockWs();
1298
+ const output = makeOutput(ws, "stream-lang");
1299
+
1300
+ const request = await synthesizeOneTurn(output);
1301
+
1302
+ expect(request.language).toBeUndefined();
1303
+ } finally {
1304
+ setConfig("services", {});
1305
+ }
1306
+ });
1307
+ });
1308
+
1309
+ // ---------------------------------------------------------------------------
1310
+ // Per-language voice override (languageVoices)
1311
+ // ---------------------------------------------------------------------------
1312
+
1313
+ describe("per-language voice override", () => {
1314
+ afterEach(() => {
1315
+ setConfig("services", {});
1316
+ });
1317
+
1318
+ /** Seed a deepgram languageVoices entry and resolve deepgram as the provider. */
1319
+ function useDeepgramWithHindiVoice(): void {
1320
+ setConfig("services", {
1321
+ tts: {
1322
+ provider: "deepgram",
1323
+ providers: {
1324
+ deepgram: { languageVoices: { hi: "aura-2-hindi-voice" } },
1325
+ },
1326
+ },
1327
+ });
1328
+ useProvider({
1329
+ id: "deepgram",
1330
+ capabilities: { supportsStreaming: false, supportedFormats: ["wav"] },
1331
+ synthesize: mockSynthesize,
1332
+ });
1333
+ mockSynthesize.mockResolvedValue({
1334
+ audio: makeWavBuffer([1000, 2000, 3000, 4000]),
1335
+ contentType: "audio/wav",
1336
+ });
1337
+ }
1338
+
1339
+ test("a configured languageVoices entry for the resolved language rides the request as voiceId", async () => {
1340
+ useDeepgramWithHindiVoice();
1341
+ const { ws } = createMockWs();
1342
+ const output = makeOutput(ws, "stream-voice");
1343
+ output.setSynthesisLanguageResolver(() => "hi");
1344
+
1345
+ output.sendTextToken("Namaste.", true);
1346
+ await drain(() => mockSynthesize.mock.calls.length > 0);
1347
+
1348
+ const request = mockSynthesize.mock.calls[0][0] as {
1349
+ voiceId?: string;
1350
+ language?: string;
1351
+ };
1352
+ expect(request.language).toBe("hi");
1353
+ expect(request.voiceId).toBe("aura-2-hindi-voice");
1354
+ });
1355
+
1356
+ test("no voiceId when the map has no entry for the resolved language", async () => {
1357
+ useDeepgramWithHindiVoice();
1358
+ const { ws } = createMockWs();
1359
+ const output = makeOutput(ws, "stream-voice");
1360
+ output.setSynthesisLanguageResolver(() => "es");
1361
+
1362
+ output.sendTextToken("Hola.", true);
1363
+ await drain(() => mockSynthesize.mock.calls.length > 0);
1364
+
1365
+ const request = mockSynthesize.mock.calls[0][0] as {
1366
+ voiceId?: string;
1367
+ language?: string;
1368
+ };
1369
+ expect(request.language).toBe("es");
1370
+ expect(request.voiceId).toBeUndefined();
1371
+ });
1372
+
1373
+ test("system copy gets neither the language hint nor the language voice", async () => {
1374
+ useDeepgramWithHindiVoice();
1375
+ const { ws } = createMockWs();
1376
+ const output = makeOutput(ws, "stream-voice");
1377
+ output.setSynthesisLanguageResolver(() => "hi");
1378
+
1379
+ output.sendTextToken("Are you still there?", true, { systemCopy: true });
1380
+ await drain(() => mockSynthesize.mock.calls.length > 0);
1381
+
1382
+ const request = mockSynthesize.mock.calls[0][0] as {
1383
+ voiceId?: string;
1384
+ language?: string;
1385
+ };
1386
+ expect(request.language).toBeUndefined();
1387
+ expect(request.voiceId).toBeUndefined();
1388
+ });
1389
+ });
1390
+
1216
1391
  // ---------------------------------------------------------------------------
1217
1392
  // Streaming PCM synthesis (incremental transcode)
1218
1393
  // ---------------------------------------------------------------------------
@@ -897,6 +897,73 @@ describe("MediaStreamSttSession", () => {
897
897
  session.dispose();
898
898
  });
899
899
 
900
+ // ── Detected language (latest tagged utterance) ─────────────────
901
+
902
+ test("a language switch retargets on the first final in the new language", async () => {
903
+ const { session, fake } = await startStreamingSession();
904
+
905
+ expect(session.currentLanguage()).toBeUndefined();
906
+
907
+ fake.emit({ type: "final", text: "hola", languages: ["es", "en"] });
908
+ fake.emit({ type: "final", text: "buenos dias", languages: ["es"] });
909
+ expect(session.currentLanguage()).toBe("es");
910
+
911
+ // The first ja-tagged utterance wins immediately: the caller does
912
+ // not have to outvote the session's es history. The secondary "es"
913
+ // tag does not dilute the utterance's dominant tag.
914
+ fake.emit({ type: "final", text: "konnichiwa", languages: ["ja", "es"] });
915
+ expect(session.currentLanguage()).toBe("ja");
916
+
917
+ session.dispose();
918
+ });
919
+
920
+ test("regional variants normalize to their base subtag", async () => {
921
+ const { session, fake } = await startStreamingSession();
922
+
923
+ fake.emit({ type: "final", text: "tudo bem", languages: ["pt-BR"] });
924
+
925
+ expect(session.currentLanguage()).toBe("pt");
926
+
927
+ session.dispose();
928
+ });
929
+
930
+ test("empty and untagged finals keep the previous value", async () => {
931
+ const { session, fake } = await startStreamingSession();
932
+
933
+ // A silence final can carry container-level tags describing no
934
+ // emitted words; it must not update the language.
935
+ fake.emit({ type: "final", text: " ", languages: ["fr"] });
936
+ expect(session.currentLanguage()).toBeUndefined();
937
+
938
+ fake.emit({ type: "final", text: "hola", languages: ["es"] });
939
+ expect(session.currentLanguage()).toBe("es");
940
+
941
+ // A committed final without tags keeps the previous value.
942
+ fake.emit({ type: "final", text: "hello" });
943
+ expect(session.currentLanguage()).toBe("es");
944
+
945
+ // So does a tagged silence final after a language is established.
946
+ fake.emit({ type: "final", text: " ", languages: ["fr"] });
947
+ expect(session.currentLanguage()).toBe("es");
948
+
949
+ session.dispose();
950
+ });
951
+
952
+ test("falling back to batch mid-call clears the detected language", async () => {
953
+ const { session, fake } = await startStreamingSession();
954
+
955
+ fake.emit({ type: "final", text: "konnichiwa", languages: ["ja"] });
956
+ expect(session.currentLanguage()).toBe("ja");
957
+
958
+ // Provider closes the stream unexpectedly: the session settles on
959
+ // batch, whose transcripts carry no language metadata. The stale ja
960
+ // detection must not keep hinting synthesis for English speech.
961
+ fake.emit({ type: "closed" });
962
+ expect(session.currentLanguage()).toBeUndefined();
963
+
964
+ session.dispose();
965
+ });
966
+
900
967
  test("local VAD turn end never triggers batch transcription in streaming mode", async () => {
901
968
  const onTranscriptFinal = jest.fn();
902
969
  const { session } = await startStreamingSession(
@@ -98,6 +98,19 @@ describe("sanitizeForTts", () => {
98
98
  "use some_function_name here",
99
99
  );
100
100
  });
101
+
102
+ test("strips italic with accented content", () => {
103
+ expect(sanitizeForTts("Hello *café* there")).toBe("Hello café there");
104
+ });
105
+
106
+ test("strips underscore italic with CJK content", () => {
107
+ expect(sanitizeForTts("値は _変数_ です")).toBe("値は 変数 です");
108
+ });
109
+
110
+ test("does not strip a span whose marker touches a non-ASCII word char", () => {
111
+ expect(sanitizeForTts("*強調*です")).toBe("*強調*です");
112
+ expect(sanitizeForTts("café*note* x")).toBe("café*note* x");
113
+ });
101
114
  });
102
115
 
103
116
  describe("headers", () => {
@@ -76,6 +76,7 @@ import { assistantEventHub } from "../../runtime/assistant-event-hub.js";
76
76
  import { CALL_OPENING_MARKER } from "../voice-control-protocol.js";
77
77
  import {
78
78
  cutFrontDoorContentAtVerdict,
79
+ preSpeechLanguageRuleFragment,
79
80
  startVoiceTurn,
80
81
  TOOL_RESULT_PREVIEW_MAX_CHARS,
81
82
  type VoiceTurnOptions,
@@ -498,6 +499,18 @@ describe("startVoiceTurn hiddenSyntheticPrompt", () => {
498
499
  });
499
500
  });
500
501
 
502
+ // The turn installs its resolved control prompt, then cleanup resets it to
503
+ // null, so capture every applied value and read the installed (non-null) one.
504
+ function captureInstalledPrompt(): () => string | undefined {
505
+ const fake = makeFakeConversation({ processing: false });
506
+ fakeConversation = fake.conversation;
507
+ const applied: Array<string | null> = [];
508
+ fake.conversation.setVoiceCallControlPrompt = (prompt) => {
509
+ applied.push(prompt);
510
+ };
511
+ return () => applied.find((p): p is string => typeof p === "string");
512
+ }
513
+
501
514
  describe("startVoiceTurn triage-and-escalate control prompt", () => {
502
515
  // Live-voice supplies its own voiceControlPrompt, bypassing
503
516
  // buildVoiceCallControlPrompt where the routing-leg rule is normally injected.
@@ -505,18 +518,6 @@ describe("startVoiceTurn triage-and-escalate control prompt", () => {
505
518
  // verdict protocol and can't hold or hand off.
506
519
  const LIVE_VOICE_PROMPT = "You are speaking in a local live voice session.";
507
520
 
508
- // The turn installs its resolved control prompt, then cleanup resets it to
509
- // null — so capture every applied value and read the installed (non-null) one.
510
- function captureInstalledPrompt(): () => string | undefined {
511
- const fake = makeFakeConversation({ processing: false });
512
- fakeConversation = fake.conversation;
513
- const applied: Array<string | null> = [];
514
- fake.conversation.setVoiceCallControlPrompt = (prompt) => {
515
- applied.push(prompt);
516
- };
517
- return () => applied.find((p): p is string => typeof p === "string");
518
- }
519
-
520
521
  test("appends the front-door decision rule to a caller-supplied prompt", async () => {
521
522
  const installed = captureInstalledPrompt();
522
523
  await startVoiceTurn({
@@ -560,6 +561,74 @@ describe("startVoiceTurn triage-and-escalate control prompt", () => {
560
561
  });
561
562
  });
562
563
 
564
+ describe("default call protocol numbered rules", () => {
565
+ // With no caller-supplied voiceControlPrompt the bridge builds the numbered
566
+ // CALL PROTOCOL RULES itself. Pin the speak-the-caller's-language rule and
567
+ // keep the numbering gapless so no rule silently shadows another.
568
+ test("teaches speaking the caller's language as its own numbered rule", async () => {
569
+ const installed = captureInstalledPrompt();
570
+ await startVoiceTurn(makeTurnOptions());
571
+ expect(installed()).toContain(
572
+ "12. Speak the caller's language: reply in the language of the caller's most recent actual speech, and follow them if they switch languages mid-call. Synthetic user turns (parenthetical markers like the call-connected and verification-completed notices) are not caller speech and never set the language. Before the caller has spoken, such as on the opening greeting turn, use the language the Task context implies, if any; otherwise default to English.",
573
+ );
574
+ });
575
+
576
+ test("the language rule excludes synthetic turns and covers pre-speech turns", async () => {
577
+ // Outbound calls open with the English "(call connected ...)" sentinel as
578
+ // the latest user-role turn, and the verification-complete sentinel does
579
+ // the same mid-call. Neither is caller speech, so neither may pull a
580
+ // Spanish or Japanese Task into an English opener.
581
+ const installed = captureInstalledPrompt();
582
+ await startVoiceTurn(makeTurnOptions());
583
+ const prompt = installed()!;
584
+ expect(prompt).toContain("most recent actual speech");
585
+ expect(prompt).toContain("not caller speech and never set the language");
586
+ expect(prompt).toContain(
587
+ "use the language the Task context implies, if any; otherwise default to English",
588
+ );
589
+ });
590
+
591
+ test("a monolingual listening language becomes the pre-speech fallback", () => {
592
+ // An assistant pinned to services.stt.language = "es" on a provider that
593
+ // honors the pin (deepgram, vellum) is already transcribing Spanish, so
594
+ // the opener must not default to English. The default test config runs
595
+ // the auto-detect branch ("multi"), so the pinned branch is covered at
596
+ // the fragment level.
597
+ expect(preSpeechLanguageRuleFragment("es", "deepgram")).toContain(
598
+ 'configured listening language ("es")',
599
+ );
600
+ expect(preSpeechLanguageRuleFragment("es", "vellum")).toContain(
601
+ "default to English only when neither gives a language",
602
+ );
603
+ for (const autoDetect of ["multi", "", " ", undefined]) {
604
+ expect(preSpeechLanguageRuleFragment(autoDetect, "deepgram")).toBe(
605
+ "use the language the Task context implies, if any; otherwise default to English",
606
+ );
607
+ }
608
+ });
609
+
610
+ test("a language pin on an auto-detecting provider keeps the English fallback", () => {
611
+ // google-gemini and openai-whisper ignore services.stt.language entirely
612
+ // (languageSelection: "auto"), so a persisted "es" pin must not force a
613
+ // Spanish greeting the transcriber will not honor.
614
+ for (const provider of ["google-gemini", "openai-whisper", undefined]) {
615
+ expect(preSpeechLanguageRuleFragment("es", provider)).toBe(
616
+ "use the language the Task context implies, if any; otherwise default to English",
617
+ );
618
+ }
619
+ });
620
+
621
+ test("rule numbers stay sequential from 0, including the routing rule", async () => {
622
+ const installed = captureInstalledPrompt();
623
+ await startVoiceTurn({ ...makeTurnOptions(), routingLeg: "escalated" });
624
+ const numbers = [...installed()!.matchAll(/^(\d+)\. /gm)].map((match) =>
625
+ Number(match[1]),
626
+ );
627
+ expect(numbers.length).toBeGreaterThan(12);
628
+ expect(numbers).toEqual(numbers.map((_, index) => index));
629
+ });
630
+ });
631
+
563
632
  describe("startVoiceTurn channel capabilities", () => {
564
633
  // Whether a call can show a surface is a property of its channel, not of
565
634
  // calls in general, so the bridge applies no voice-specific override: a
@@ -1,5 +1,6 @@
1
1
  import { describe, expect, test } from "bun:test";
2
2
 
3
+ import { DEEPGRAM_MULTI_LANGUAGE_CODES } from "../../providers/speech-to-text/deepgram.js";
3
4
  import {
4
5
  capEscalationBridge,
5
6
  classifyFrontDoorLeading,
@@ -7,6 +8,8 @@ import {
7
8
  escalatedContinuationRule,
8
9
  ESCALATION_CONTINUATION_CONTENT,
9
10
  FALLBACK_ESCALATION_BRIDGE,
11
+ FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
12
+ fallbackEscalationBridgeFor,
10
13
  frontDoorCapabilityDigest,
11
14
  frontDoorDecisionRule,
12
15
  HOLD_VERDICT_TOKEN,
@@ -162,6 +165,33 @@ describe("front-door decision rule", () => {
162
165
  expect(rule.toLowerCase()).toContain("one short natural holding phrase");
163
166
  expect(rule.toLowerCase()).toContain("stop after that single sentence");
164
167
  });
168
+
169
+ test("hold completeness is judged in the caller's language", () => {
170
+ // Callers are not English-only: the hold branch's exemplars are English,
171
+ // so the rule must say completeness follows the grammar of the language
172
+ // being spoken, with the verb-final case called out (a missing final
173
+ // verb, not a missing conjunction, is the unfinished signal there).
174
+ const withHold = frontDoorDecisionRule({ includeHold: true });
175
+ expect(withHold.toLowerCase()).toContain("may speak any language");
176
+ expect(withHold.toLowerCase()).toContain(
177
+ "grammar of the language being spoken",
178
+ );
179
+ expect(withHold.toLowerCase()).toContain("verb-final");
180
+ expect(withHold.toLowerCase()).toContain("missing final verb");
181
+ });
182
+
183
+ test("the answer branch demands the caller's language", () => {
184
+ expect(rule).toContain("Answer in the language the caller is speaking.");
185
+ });
186
+
187
+ test("the escalation holding phrase demands the caller's language", () => {
188
+ // The bridge examples are English; without an explicit requirement a
189
+ // Spanish turn that needs a tool gets an English holding phrase before
190
+ // the localized answer. The examples stay, labeled as English only.
191
+ expect(rule).toContain("spoken in the language the caller is speaking");
192
+ expect(rule).toContain("those examples are English only");
193
+ expect(rule).toContain(`"${FALLBACK_ESCALATION_BRIDGE}"`);
194
+ });
165
195
  });
166
196
 
167
197
  describe("escalated continuation rule", () => {
@@ -190,6 +220,12 @@ describe("escalated continuation rule", () => {
190
220
  );
191
221
  });
192
222
 
223
+ test("demands the reply match the caller's language", () => {
224
+ expect(rule).toContain(
225
+ "Reply in the same language as the caller's question.",
226
+ );
227
+ });
228
+
193
229
  test("bans re-announcing the holding phrase (bridge-echo regression)", () => {
194
230
  // Regression: after the bridge "Let me check your calendar", the quality
195
231
  // model opened with "Let me check what calendar connections…" — a
@@ -201,6 +237,41 @@ describe("escalated continuation rule", () => {
201
237
  });
202
238
  });
203
239
 
240
+ describe("fallbackEscalationBridgeFor", () => {
241
+ test("covers every Deepgram code-switching language with a non-empty bridge", () => {
242
+ for (const code of DEEPGRAM_MULTI_LANGUAGE_CODES) {
243
+ const bridge = FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE[code];
244
+ expect(bridge).toBeDefined();
245
+ expect(bridge!.trim().length).toBeGreaterThan(0);
246
+ expect(fallbackEscalationBridgeFor(code)).toBe(bridge!);
247
+ }
248
+ });
249
+
250
+ test("every bridge fits the session-side cap", () => {
251
+ for (const bridge of Object.values(
252
+ FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
253
+ )) {
254
+ expect(bridge.length).toBeLessThanOrEqual(MAX_ESCALATION_BRIDGE_CHARS);
255
+ }
256
+ });
257
+
258
+ test("selects by lowercased base subtag", () => {
259
+ expect(fallbackEscalationBridgeFor("pt-BR")).toBe(
260
+ FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE.pt!,
261
+ );
262
+ expect(fallbackEscalationBridgeFor("JA")).toBe(
263
+ FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE.ja!,
264
+ );
265
+ });
266
+
267
+ test("falls back to English for unknown or absent languages", () => {
268
+ expect(fallbackEscalationBridgeFor()).toBe(FALLBACK_ESCALATION_BRIDGE);
269
+ expect(fallbackEscalationBridgeFor("ko")).toBe(FALLBACK_ESCALATION_BRIDGE);
270
+ expect(fallbackEscalationBridgeFor("")).toBe(FALLBACK_ESCALATION_BRIDGE);
271
+ expect(fallbackEscalationBridgeFor("en")).toBe(FALLBACK_ESCALATION_BRIDGE);
272
+ });
273
+ });
274
+
204
275
  describe("capEscalationBridge", () => {
205
276
  test("cuts just after the first sentence terminator", () => {
206
277
  expect(
@@ -218,6 +289,24 @@ describe("capEscalationBridge", () => {
218
289
  test("strips internal markers before capping", () => {
219
290
  expect(capEscalationBridge("[END_CALL] One moment.")).toBe("One moment.");
220
291
  });
292
+
293
+ test("the Unicode ellipsis still terminates a bridge", () => {
294
+ // Regression pin: widening the terminator class for non-Latin enders
295
+ // must not drop the ellipsis the original regex recognized.
296
+ expect(capEscalationBridge("One moment… and some rambling")).toBe(
297
+ "One moment…",
298
+ );
299
+ expect(isEscalationBridgeComplete("One moment…")).toBe(true);
300
+ });
301
+
302
+ test("cuts just after a non-Latin sentence terminator", () => {
303
+ expect(capEscalationBridge("少し考えさせてください。その間の余談")).toBe(
304
+ "少し考えさせてください。",
305
+ );
306
+ expect(capEscalationBridge("मुझे एक पल सोचने दीजिए। और कुछ बातें")).toBe(
307
+ "मुझे एक पल सोचने दीजिए।",
308
+ );
309
+ });
221
310
  });
222
311
 
223
312
  describe("isEscalationBridgeComplete", () => {
@@ -233,6 +322,23 @@ describe("isEscalationBridgeComplete", () => {
233
322
  isEscalationBridgeComplete("a".repeat(MAX_ESCALATION_BRIDGE_CHARS)),
234
323
  ).toBe(true);
235
324
  });
325
+
326
+ test("a Japanese bridge ending in 。 completes without waiting for the cap", () => {
327
+ expect(isEscalationBridgeComplete("少し考えさせてください")).toBe(false);
328
+ expect(isEscalationBridgeComplete("少し考えさせてください。")).toBe(true);
329
+ });
330
+
331
+ test("every localized fallback bridge ends in a recognized terminator", () => {
332
+ // A model-spoken bridge in any roster language must hand off at its
333
+ // terminator, never by buffering to the char cap; the canned bridges are
334
+ // the canonical sample of each language's ender.
335
+ for (const bridge of Object.values(
336
+ FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
337
+ )) {
338
+ expect(isEscalationBridgeComplete(bridge)).toBe(true);
339
+ expect(capEscalationBridge(bridge)).toBe(bridge);
340
+ }
341
+ });
236
342
  });
237
343
 
238
344
  describe("spokenBridgeText", () => {
@@ -65,6 +65,10 @@ import {
65
65
  resolveSynthesisFormats,
66
66
  } from "./resolve-call-tts-provider.js";
67
67
  import type { PromptSpeakerContext } from "./speaker-identification.js";
68
+ import {
69
+ resolveTelephonyLanguageVoice,
70
+ resolveTelephonySynthesisLanguage,
71
+ } from "./telephony-synthesis-language.js";
68
72
  import { sanitizeForTts } from "./tts-text-sanitizer.js";
69
73
  import {
70
74
  ASK_GUARDIAN_CAPTURE_REGEX,
@@ -189,6 +193,13 @@ export class CallController {
189
193
  private guardianUnavailableForCall = false;
190
194
  /** Active synthesized-TTS session — tracked so interrupt handling can close it. */
191
195
  private activeSynthesisAbort: AbortController | null = null;
196
+ /**
197
+ * Resolves the language hint for synthesized speech. The media-stream
198
+ * server supplies a resolver backed by the STT session's detected
199
+ * dominant language; the default falls back to the pin-based
200
+ * resolution only.
201
+ */
202
+ private resolveSynthesisLanguage: () => string | undefined;
192
203
 
193
204
  constructor(
194
205
  callSessionId: string,
@@ -198,6 +209,7 @@ export class CallController {
198
209
  broadcast?: (msg: AssistantEvent) => void;
199
210
  assistantId?: string;
200
211
  trustContext?: TrustContext;
212
+ resolveSynthesisLanguage?: () => string | undefined;
201
213
  },
202
214
  ) {
203
215
  this.callSessionId = callSessionId;
@@ -207,6 +219,9 @@ export class CallController {
207
219
  this.broadcast = opts?.broadcast;
208
220
  this.assistantId = opts?.assistantId ?? DAEMON_INTERNAL_ASSISTANT_ID;
209
221
  this.trustContext = opts?.trustContext ?? null;
222
+ this.resolveSynthesisLanguage =
223
+ opts?.resolveSynthesisLanguage ??
224
+ (() => resolveTelephonySynthesisLanguage());
210
225
 
211
226
  // Resolve the conversation ID and skipDisclosure from the call session
212
227
  const session = getCallSession(callSessionId);
@@ -663,7 +678,9 @@ export class CallController {
663
678
  // lock-hold wait budget, so surface a brief natural re-prompt (never a
664
679
  // technical-error message) and re-arm listening. last=true doubles as
665
680
  // the end-of-turn marker.
666
- this.transport.sendTextToken("Sorry, could you say that again?", true);
681
+ this.transport.sendTextToken("Sorry, could you say that again?", true, {
682
+ systemCopy: true,
683
+ });
667
684
  this.state = "idle";
668
685
  this.resetSilenceTimer();
669
686
  this.flushPendingInstructions();
@@ -673,6 +690,7 @@ export class CallController {
673
690
  this.transport.sendTextToken(
674
691
  "I'm sorry, I encountered a technical issue. Could you repeat that?",
675
692
  true,
693
+ { systemCopy: true },
676
694
  );
677
695
  this.state = "idle";
678
696
  this.resetSilenceTimer();
@@ -1080,11 +1098,17 @@ export class CallController {
1080
1098
 
1081
1099
  this.activeSynthesisAbort = abortController;
1082
1100
 
1101
+ const language = this.resolveSynthesisLanguage();
1102
+ // A language-known segment may select the synthesizing provider's
1103
+ // configured per-language voice; no entry keeps the provider default.
1104
+ const voiceId = resolveTelephonyLanguageVoice(provider.id, language);
1083
1105
  await synthesizeAndEmit({
1084
1106
  provider,
1085
1107
  text,
1086
1108
  useCase: "phone-call",
1087
1109
  outputFormat,
1110
+ ...(voiceId !== undefined ? { voiceId } : {}),
1111
+ ...(language !== undefined ? { language } : {}),
1088
1112
  signal: abortController.signal,
1089
1113
  isCurrent: () => this.isCurrentRun(runVersion),
1090
1114
  onChunk: sink.onChunk,
@@ -1804,6 +1828,7 @@ export class CallController {
1804
1828
  this.transport.sendTextToken(
1805
1829
  "Just to let you know, we're running low on time for this call.",
1806
1830
  true,
1831
+ { systemCopy: true },
1807
1832
  );
1808
1833
  }, warningMs);
1809
1834
  }
@@ -1816,6 +1841,7 @@ export class CallController {
1816
1841
  this.transport.sendTextToken(
1817
1842
  "I'm sorry, but we've reached the maximum time for this call. Thank you for your time. Goodbye!",
1818
1843
  true,
1844
+ { systemCopy: true },
1819
1845
  );
1820
1846
  // Give TTS a moment to play, then end
1821
1847
  this.durationEndTimer = setTimeout(() => {
@@ -1878,7 +1904,9 @@ export class CallController {
1878
1904
  { callSessionId: this.callSessionId },
1879
1905
  "Silence timeout triggered",
1880
1906
  );
1881
- this.transport.sendTextToken("Are you still there?", true);
1907
+ this.transport.sendTextToken("Are you still there?", true, {
1908
+ systemCopy: true,
1909
+ });
1882
1910
  }, getSilenceTimeoutMs());
1883
1911
  }
1884
1912
  }