@vellumai/assistant 0.11.3-staging.2 → 0.11.3-staging.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/__tests__/call-controller.test.ts +120 -0
- package/src/__tests__/config-loader-backfill.test.ts +19 -0
- package/src/__tests__/config-schema.test.ts +139 -5
- package/src/__tests__/events-dev-bypass-actor.test.ts +112 -1
- package/src/__tests__/media-stream-output.test.ts +175 -0
- package/src/__tests__/media-stream-stt-session.test.ts +67 -0
- package/src/calls/__tests__/tts-text-sanitizer.test.ts +13 -0
- package/src/calls/__tests__/voice-session-bridge.test.ts +81 -12
- package/src/calls/__tests__/voice-triage-escalate.test.ts +106 -0
- package/src/calls/call-controller.ts +30 -2
- package/src/calls/call-speech-output.ts +12 -4
- package/src/calls/call-transport.ts +18 -1
- package/src/calls/media-stream-output.ts +53 -5
- package/src/calls/media-stream-server.ts +12 -0
- package/src/calls/media-stream-stt-session.ts +34 -0
- package/src/calls/telephony-synthesis-language.ts +84 -0
- package/src/calls/tts-text-sanitizer.ts +12 -5
- package/src/calls/voice-session-bridge.ts +31 -3
- package/src/calls/voice-triage-escalate.ts +52 -5
- package/src/config/bundled-skills/phone-calls/references/CONFIG.md +15 -15
- package/src/config/loader.ts +5 -0
- package/src/config/schemas/calls.ts +0 -4
- package/src/config/schemas/tts.ts +63 -0
- package/src/live-voice/__tests__/front-decision.test.ts +120 -0
- package/src/live-voice/__tests__/live-voice-events.test.ts +1 -1
- package/src/live-voice/__tests__/live-voice-progress.test.ts +178 -11
- package/src/live-voice/__tests__/live-voice-stt.test.ts +304 -1
- package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +102 -2
- package/src/live-voice/__tests__/live-voice-tts.test.ts +148 -2
- package/src/live-voice/__tests__/progress-phrases.test.ts +167 -0
- package/src/live-voice/front-decision.ts +50 -3
- package/src/live-voice/live-voice-session.ts +202 -25
- package/src/live-voice/live-voice-tts.ts +18 -2
- package/src/live-voice/progress-phrases.ts +105 -2
- package/src/providers/speech-to-text/deepgram-realtime.test.ts +283 -1
- package/src/providers/speech-to-text/deepgram-realtime.ts +117 -3
- package/src/providers/speech-to-text/provider-catalog.ts +38 -0
- package/src/runtime/__tests__/local-actor-identity-force-refresh.test.ts +104 -0
- package/src/runtime/assistant-event-hub.ts +23 -0
- package/src/runtime/local-actor-identity.ts +18 -5
- package/src/runtime/routes/__tests__/sse-actor-principal-heal.test.ts +165 -0
- package/src/runtime/routes/__tests__/surface-action-routes.test.ts +2 -0
- package/src/runtime/routes/events-routes.ts +17 -16
- package/src/runtime/routes/sse-actor-principal-heal.ts +113 -0
- package/src/stt/__tests__/language-metadata.test.ts +85 -0
- package/src/stt/language-metadata.ts +65 -0
- package/src/stt/types.ts +16 -0
- package/src/tts/__tests__/provider-adapters.test.ts +147 -0
- package/src/tts/__tests__/speakable-segments.test.ts +470 -0
- package/src/tts/language-voices.ts +23 -0
- package/src/tts/providers/deepgram-provider.ts +3 -1
- package/src/tts/providers/elevenlabs-provider.ts +73 -1
- package/src/tts/providers/xai-provider.ts +28 -2
- package/src/tts/speakable-segments.ts +293 -23
- package/src/tts/synthesis-stream.ts +7 -0
- package/src/tts/types.ts +7 -0
- package/src/util/__tests__/language-subtag.test.ts +54 -0
- package/src/util/language-subtag.ts +43 -0
- package/src/util/unicode.ts +1 -1
|
@@ -26,6 +26,7 @@ import {
|
|
|
26
26
|
} from "../calls/media-stream-audio-transcode.js";
|
|
27
27
|
import { MediaStreamOutput } from "../calls/media-stream-output.js";
|
|
28
28
|
import { resolveCallTtsProvider } from "../calls/resolve-call-tts-provider.js";
|
|
29
|
+
import { setConfig } from "./helpers/set-config.js";
|
|
29
30
|
|
|
30
31
|
const mockResolveCallTtsProvider = resolveCallTtsProvider as ReturnType<
|
|
31
32
|
typeof jest.fn
|
|
@@ -1213,6 +1214,180 @@ describe("MediaStreamOutput", () => {
|
|
|
1213
1214
|
});
|
|
1214
1215
|
});
|
|
1215
1216
|
|
|
1217
|
+
// ---------------------------------------------------------------------------
|
|
1218
|
+
// Synthesis language hint
|
|
1219
|
+
// ---------------------------------------------------------------------------
|
|
1220
|
+
|
|
1221
|
+
describe("synthesis language hint", () => {
|
|
1222
|
+
/** Synthesize one turn and return the provider request it produced. */
|
|
1223
|
+
async function synthesizeOneTurn(
|
|
1224
|
+
output: MediaStreamOutput,
|
|
1225
|
+
): Promise<{ language?: string }> {
|
|
1226
|
+
mockSynthesize.mockResolvedValue({
|
|
1227
|
+
audio: makeWavBuffer([1000, 2000, 3000, 4000]),
|
|
1228
|
+
contentType: "audio/wav",
|
|
1229
|
+
});
|
|
1230
|
+
output.sendTextToken("Hello caller.", true);
|
|
1231
|
+
await drain(() => mockSynthesize.mock.calls.length > 0);
|
|
1232
|
+
return mockSynthesize.mock.calls[0][0] as { language?: string };
|
|
1233
|
+
}
|
|
1234
|
+
|
|
1235
|
+
test("the resolver's language rides the provider request", async () => {
|
|
1236
|
+
const { ws } = createMockWs();
|
|
1237
|
+
const output = makeOutput(ws, "stream-lang");
|
|
1238
|
+
output.setSynthesisLanguageResolver(() => "es");
|
|
1239
|
+
|
|
1240
|
+
const request = await synthesizeOneTurn(output);
|
|
1241
|
+
|
|
1242
|
+
expect(request.language).toBe("es");
|
|
1243
|
+
});
|
|
1244
|
+
|
|
1245
|
+
test("fixed system copy synthesizes without the caller-language hint while model text keeps it", async () => {
|
|
1246
|
+
const { ws } = createMockWs();
|
|
1247
|
+
const output = makeOutput(ws, "stream-lang");
|
|
1248
|
+
output.setSynthesisLanguageResolver(() => "ja");
|
|
1249
|
+
mockSynthesize.mockResolvedValue({
|
|
1250
|
+
audio: makeWavBuffer([1000, 2000, 3000, 4000]),
|
|
1251
|
+
contentType: "audio/wav",
|
|
1252
|
+
});
|
|
1253
|
+
|
|
1254
|
+
output.sendTextToken("Model reply.", true);
|
|
1255
|
+
output.sendTextToken("Are you still there?", true, { systemCopy: true });
|
|
1256
|
+
await drain(() => mockSynthesize.mock.calls.length >= 2);
|
|
1257
|
+
|
|
1258
|
+
const requests = mockSynthesize.mock.calls.map(
|
|
1259
|
+
(call) => call[0] as { text: string; language?: string },
|
|
1260
|
+
);
|
|
1261
|
+
expect(requests[0]?.text).toBe("Model reply.");
|
|
1262
|
+
expect(requests[0]?.language).toBe("ja");
|
|
1263
|
+
expect(requests[1]?.text).toBe("Are you still there?");
|
|
1264
|
+
expect(requests[1]?.language).toBeUndefined();
|
|
1265
|
+
});
|
|
1266
|
+
|
|
1267
|
+
test("no language when nothing resolves (default multilingual pin)", async () => {
|
|
1268
|
+
const { ws } = createMockWs();
|
|
1269
|
+
const output = makeOutput(ws, "stream-lang");
|
|
1270
|
+
|
|
1271
|
+
const request = await synthesizeOneTurn(output);
|
|
1272
|
+
|
|
1273
|
+
expect(request.language).toBeUndefined();
|
|
1274
|
+
});
|
|
1275
|
+
|
|
1276
|
+
test("a monolingual pin on a manual-selection provider resolves as the hint", async () => {
|
|
1277
|
+
setConfig("services", {
|
|
1278
|
+
stt: { provider: "deepgram", language: "es-ES" },
|
|
1279
|
+
});
|
|
1280
|
+
try {
|
|
1281
|
+
const { ws } = createMockWs();
|
|
1282
|
+
const output = makeOutput(ws, "stream-lang");
|
|
1283
|
+
|
|
1284
|
+
const request = await synthesizeOneTurn(output);
|
|
1285
|
+
|
|
1286
|
+
expect(request.language).toBe("es");
|
|
1287
|
+
} finally {
|
|
1288
|
+
setConfig("services", {});
|
|
1289
|
+
}
|
|
1290
|
+
});
|
|
1291
|
+
|
|
1292
|
+
test("the pin is ignored for auto-detecting providers", async () => {
|
|
1293
|
+
setConfig("services", {
|
|
1294
|
+
stt: { provider: "openai-whisper", language: "es" },
|
|
1295
|
+
});
|
|
1296
|
+
try {
|
|
1297
|
+
const { ws } = createMockWs();
|
|
1298
|
+
const output = makeOutput(ws, "stream-lang");
|
|
1299
|
+
|
|
1300
|
+
const request = await synthesizeOneTurn(output);
|
|
1301
|
+
|
|
1302
|
+
expect(request.language).toBeUndefined();
|
|
1303
|
+
} finally {
|
|
1304
|
+
setConfig("services", {});
|
|
1305
|
+
}
|
|
1306
|
+
});
|
|
1307
|
+
});
|
|
1308
|
+
|
|
1309
|
+
// ---------------------------------------------------------------------------
|
|
1310
|
+
// Per-language voice override (languageVoices)
|
|
1311
|
+
// ---------------------------------------------------------------------------
|
|
1312
|
+
|
|
1313
|
+
describe("per-language voice override", () => {
|
|
1314
|
+
afterEach(() => {
|
|
1315
|
+
setConfig("services", {});
|
|
1316
|
+
});
|
|
1317
|
+
|
|
1318
|
+
/** Seed a deepgram languageVoices entry and resolve deepgram as the provider. */
|
|
1319
|
+
function useDeepgramWithHindiVoice(): void {
|
|
1320
|
+
setConfig("services", {
|
|
1321
|
+
tts: {
|
|
1322
|
+
provider: "deepgram",
|
|
1323
|
+
providers: {
|
|
1324
|
+
deepgram: { languageVoices: { hi: "aura-2-hindi-voice" } },
|
|
1325
|
+
},
|
|
1326
|
+
},
|
|
1327
|
+
});
|
|
1328
|
+
useProvider({
|
|
1329
|
+
id: "deepgram",
|
|
1330
|
+
capabilities: { supportsStreaming: false, supportedFormats: ["wav"] },
|
|
1331
|
+
synthesize: mockSynthesize,
|
|
1332
|
+
});
|
|
1333
|
+
mockSynthesize.mockResolvedValue({
|
|
1334
|
+
audio: makeWavBuffer([1000, 2000, 3000, 4000]),
|
|
1335
|
+
contentType: "audio/wav",
|
|
1336
|
+
});
|
|
1337
|
+
}
|
|
1338
|
+
|
|
1339
|
+
test("a configured languageVoices entry for the resolved language rides the request as voiceId", async () => {
|
|
1340
|
+
useDeepgramWithHindiVoice();
|
|
1341
|
+
const { ws } = createMockWs();
|
|
1342
|
+
const output = makeOutput(ws, "stream-voice");
|
|
1343
|
+
output.setSynthesisLanguageResolver(() => "hi");
|
|
1344
|
+
|
|
1345
|
+
output.sendTextToken("Namaste.", true);
|
|
1346
|
+
await drain(() => mockSynthesize.mock.calls.length > 0);
|
|
1347
|
+
|
|
1348
|
+
const request = mockSynthesize.mock.calls[0][0] as {
|
|
1349
|
+
voiceId?: string;
|
|
1350
|
+
language?: string;
|
|
1351
|
+
};
|
|
1352
|
+
expect(request.language).toBe("hi");
|
|
1353
|
+
expect(request.voiceId).toBe("aura-2-hindi-voice");
|
|
1354
|
+
});
|
|
1355
|
+
|
|
1356
|
+
test("no voiceId when the map has no entry for the resolved language", async () => {
|
|
1357
|
+
useDeepgramWithHindiVoice();
|
|
1358
|
+
const { ws } = createMockWs();
|
|
1359
|
+
const output = makeOutput(ws, "stream-voice");
|
|
1360
|
+
output.setSynthesisLanguageResolver(() => "es");
|
|
1361
|
+
|
|
1362
|
+
output.sendTextToken("Hola.", true);
|
|
1363
|
+
await drain(() => mockSynthesize.mock.calls.length > 0);
|
|
1364
|
+
|
|
1365
|
+
const request = mockSynthesize.mock.calls[0][0] as {
|
|
1366
|
+
voiceId?: string;
|
|
1367
|
+
language?: string;
|
|
1368
|
+
};
|
|
1369
|
+
expect(request.language).toBe("es");
|
|
1370
|
+
expect(request.voiceId).toBeUndefined();
|
|
1371
|
+
});
|
|
1372
|
+
|
|
1373
|
+
test("system copy gets neither the language hint nor the language voice", async () => {
|
|
1374
|
+
useDeepgramWithHindiVoice();
|
|
1375
|
+
const { ws } = createMockWs();
|
|
1376
|
+
const output = makeOutput(ws, "stream-voice");
|
|
1377
|
+
output.setSynthesisLanguageResolver(() => "hi");
|
|
1378
|
+
|
|
1379
|
+
output.sendTextToken("Are you still there?", true, { systemCopy: true });
|
|
1380
|
+
await drain(() => mockSynthesize.mock.calls.length > 0);
|
|
1381
|
+
|
|
1382
|
+
const request = mockSynthesize.mock.calls[0][0] as {
|
|
1383
|
+
voiceId?: string;
|
|
1384
|
+
language?: string;
|
|
1385
|
+
};
|
|
1386
|
+
expect(request.language).toBeUndefined();
|
|
1387
|
+
expect(request.voiceId).toBeUndefined();
|
|
1388
|
+
});
|
|
1389
|
+
});
|
|
1390
|
+
|
|
1216
1391
|
// ---------------------------------------------------------------------------
|
|
1217
1392
|
// Streaming PCM synthesis (incremental transcode)
|
|
1218
1393
|
// ---------------------------------------------------------------------------
|
|
@@ -897,6 +897,73 @@ describe("MediaStreamSttSession", () => {
|
|
|
897
897
|
session.dispose();
|
|
898
898
|
});
|
|
899
899
|
|
|
900
|
+
// ── Detected language (latest tagged utterance) ─────────────────
|
|
901
|
+
|
|
902
|
+
test("a language switch retargets on the first final in the new language", async () => {
|
|
903
|
+
const { session, fake } = await startStreamingSession();
|
|
904
|
+
|
|
905
|
+
expect(session.currentLanguage()).toBeUndefined();
|
|
906
|
+
|
|
907
|
+
fake.emit({ type: "final", text: "hola", languages: ["es", "en"] });
|
|
908
|
+
fake.emit({ type: "final", text: "buenos dias", languages: ["es"] });
|
|
909
|
+
expect(session.currentLanguage()).toBe("es");
|
|
910
|
+
|
|
911
|
+
// The first ja-tagged utterance wins immediately: the caller does
|
|
912
|
+
// not have to outvote the session's es history. The secondary "es"
|
|
913
|
+
// tag does not dilute the utterance's dominant tag.
|
|
914
|
+
fake.emit({ type: "final", text: "konnichiwa", languages: ["ja", "es"] });
|
|
915
|
+
expect(session.currentLanguage()).toBe("ja");
|
|
916
|
+
|
|
917
|
+
session.dispose();
|
|
918
|
+
});
|
|
919
|
+
|
|
920
|
+
test("regional variants normalize to their base subtag", async () => {
|
|
921
|
+
const { session, fake } = await startStreamingSession();
|
|
922
|
+
|
|
923
|
+
fake.emit({ type: "final", text: "tudo bem", languages: ["pt-BR"] });
|
|
924
|
+
|
|
925
|
+
expect(session.currentLanguage()).toBe("pt");
|
|
926
|
+
|
|
927
|
+
session.dispose();
|
|
928
|
+
});
|
|
929
|
+
|
|
930
|
+
test("empty and untagged finals keep the previous value", async () => {
|
|
931
|
+
const { session, fake } = await startStreamingSession();
|
|
932
|
+
|
|
933
|
+
// A silence final can carry container-level tags describing no
|
|
934
|
+
// emitted words; it must not update the language.
|
|
935
|
+
fake.emit({ type: "final", text: " ", languages: ["fr"] });
|
|
936
|
+
expect(session.currentLanguage()).toBeUndefined();
|
|
937
|
+
|
|
938
|
+
fake.emit({ type: "final", text: "hola", languages: ["es"] });
|
|
939
|
+
expect(session.currentLanguage()).toBe("es");
|
|
940
|
+
|
|
941
|
+
// A committed final without tags keeps the previous value.
|
|
942
|
+
fake.emit({ type: "final", text: "hello" });
|
|
943
|
+
expect(session.currentLanguage()).toBe("es");
|
|
944
|
+
|
|
945
|
+
// So does a tagged silence final after a language is established.
|
|
946
|
+
fake.emit({ type: "final", text: " ", languages: ["fr"] });
|
|
947
|
+
expect(session.currentLanguage()).toBe("es");
|
|
948
|
+
|
|
949
|
+
session.dispose();
|
|
950
|
+
});
|
|
951
|
+
|
|
952
|
+
test("falling back to batch mid-call clears the detected language", async () => {
|
|
953
|
+
const { session, fake } = await startStreamingSession();
|
|
954
|
+
|
|
955
|
+
fake.emit({ type: "final", text: "konnichiwa", languages: ["ja"] });
|
|
956
|
+
expect(session.currentLanguage()).toBe("ja");
|
|
957
|
+
|
|
958
|
+
// Provider closes the stream unexpectedly: the session settles on
|
|
959
|
+
// batch, whose transcripts carry no language metadata. The stale ja
|
|
960
|
+
// detection must not keep hinting synthesis for English speech.
|
|
961
|
+
fake.emit({ type: "closed" });
|
|
962
|
+
expect(session.currentLanguage()).toBeUndefined();
|
|
963
|
+
|
|
964
|
+
session.dispose();
|
|
965
|
+
});
|
|
966
|
+
|
|
900
967
|
test("local VAD turn end never triggers batch transcription in streaming mode", async () => {
|
|
901
968
|
const onTranscriptFinal = jest.fn();
|
|
902
969
|
const { session } = await startStreamingSession(
|
|
@@ -98,6 +98,19 @@ describe("sanitizeForTts", () => {
|
|
|
98
98
|
"use some_function_name here",
|
|
99
99
|
);
|
|
100
100
|
});
|
|
101
|
+
|
|
102
|
+
test("strips italic with accented content", () => {
|
|
103
|
+
expect(sanitizeForTts("Hello *café* there")).toBe("Hello café there");
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
test("strips underscore italic with CJK content", () => {
|
|
107
|
+
expect(sanitizeForTts("値は _変数_ です")).toBe("値は 変数 です");
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
test("does not strip a span whose marker touches a non-ASCII word char", () => {
|
|
111
|
+
expect(sanitizeForTts("*強調*です")).toBe("*強調*です");
|
|
112
|
+
expect(sanitizeForTts("café*note* x")).toBe("café*note* x");
|
|
113
|
+
});
|
|
101
114
|
});
|
|
102
115
|
|
|
103
116
|
describe("headers", () => {
|
|
@@ -76,6 +76,7 @@ import { assistantEventHub } from "../../runtime/assistant-event-hub.js";
|
|
|
76
76
|
import { CALL_OPENING_MARKER } from "../voice-control-protocol.js";
|
|
77
77
|
import {
|
|
78
78
|
cutFrontDoorContentAtVerdict,
|
|
79
|
+
preSpeechLanguageRuleFragment,
|
|
79
80
|
startVoiceTurn,
|
|
80
81
|
TOOL_RESULT_PREVIEW_MAX_CHARS,
|
|
81
82
|
type VoiceTurnOptions,
|
|
@@ -498,6 +499,18 @@ describe("startVoiceTurn hiddenSyntheticPrompt", () => {
|
|
|
498
499
|
});
|
|
499
500
|
});
|
|
500
501
|
|
|
502
|
+
// The turn installs its resolved control prompt, then cleanup resets it to
|
|
503
|
+
// null, so capture every applied value and read the installed (non-null) one.
|
|
504
|
+
function captureInstalledPrompt(): () => string | undefined {
|
|
505
|
+
const fake = makeFakeConversation({ processing: false });
|
|
506
|
+
fakeConversation = fake.conversation;
|
|
507
|
+
const applied: Array<string | null> = [];
|
|
508
|
+
fake.conversation.setVoiceCallControlPrompt = (prompt) => {
|
|
509
|
+
applied.push(prompt);
|
|
510
|
+
};
|
|
511
|
+
return () => applied.find((p): p is string => typeof p === "string");
|
|
512
|
+
}
|
|
513
|
+
|
|
501
514
|
describe("startVoiceTurn triage-and-escalate control prompt", () => {
|
|
502
515
|
// Live-voice supplies its own voiceControlPrompt, bypassing
|
|
503
516
|
// buildVoiceCallControlPrompt where the routing-leg rule is normally injected.
|
|
@@ -505,18 +518,6 @@ describe("startVoiceTurn triage-and-escalate control prompt", () => {
|
|
|
505
518
|
// verdict protocol and can't hold or hand off.
|
|
506
519
|
const LIVE_VOICE_PROMPT = "You are speaking in a local live voice session.";
|
|
507
520
|
|
|
508
|
-
// The turn installs its resolved control prompt, then cleanup resets it to
|
|
509
|
-
// null — so capture every applied value and read the installed (non-null) one.
|
|
510
|
-
function captureInstalledPrompt(): () => string | undefined {
|
|
511
|
-
const fake = makeFakeConversation({ processing: false });
|
|
512
|
-
fakeConversation = fake.conversation;
|
|
513
|
-
const applied: Array<string | null> = [];
|
|
514
|
-
fake.conversation.setVoiceCallControlPrompt = (prompt) => {
|
|
515
|
-
applied.push(prompt);
|
|
516
|
-
};
|
|
517
|
-
return () => applied.find((p): p is string => typeof p === "string");
|
|
518
|
-
}
|
|
519
|
-
|
|
520
521
|
test("appends the front-door decision rule to a caller-supplied prompt", async () => {
|
|
521
522
|
const installed = captureInstalledPrompt();
|
|
522
523
|
await startVoiceTurn({
|
|
@@ -560,6 +561,74 @@ describe("startVoiceTurn triage-and-escalate control prompt", () => {
|
|
|
560
561
|
});
|
|
561
562
|
});
|
|
562
563
|
|
|
564
|
+
describe("default call protocol numbered rules", () => {
|
|
565
|
+
// With no caller-supplied voiceControlPrompt the bridge builds the numbered
|
|
566
|
+
// CALL PROTOCOL RULES itself. Pin the speak-the-caller's-language rule and
|
|
567
|
+
// keep the numbering gapless so no rule silently shadows another.
|
|
568
|
+
test("teaches speaking the caller's language as its own numbered rule", async () => {
|
|
569
|
+
const installed = captureInstalledPrompt();
|
|
570
|
+
await startVoiceTurn(makeTurnOptions());
|
|
571
|
+
expect(installed()).toContain(
|
|
572
|
+
"12. Speak the caller's language: reply in the language of the caller's most recent actual speech, and follow them if they switch languages mid-call. Synthetic user turns (parenthetical markers like the call-connected and verification-completed notices) are not caller speech and never set the language. Before the caller has spoken, such as on the opening greeting turn, use the language the Task context implies, if any; otherwise default to English.",
|
|
573
|
+
);
|
|
574
|
+
});
|
|
575
|
+
|
|
576
|
+
test("the language rule excludes synthetic turns and covers pre-speech turns", async () => {
|
|
577
|
+
// Outbound calls open with the English "(call connected ...)" sentinel as
|
|
578
|
+
// the latest user-role turn, and the verification-complete sentinel does
|
|
579
|
+
// the same mid-call. Neither is caller speech, so neither may pull a
|
|
580
|
+
// Spanish or Japanese Task into an English opener.
|
|
581
|
+
const installed = captureInstalledPrompt();
|
|
582
|
+
await startVoiceTurn(makeTurnOptions());
|
|
583
|
+
const prompt = installed()!;
|
|
584
|
+
expect(prompt).toContain("most recent actual speech");
|
|
585
|
+
expect(prompt).toContain("not caller speech and never set the language");
|
|
586
|
+
expect(prompt).toContain(
|
|
587
|
+
"use the language the Task context implies, if any; otherwise default to English",
|
|
588
|
+
);
|
|
589
|
+
});
|
|
590
|
+
|
|
591
|
+
test("a monolingual listening language becomes the pre-speech fallback", () => {
|
|
592
|
+
// An assistant pinned to services.stt.language = "es" on a provider that
|
|
593
|
+
// honors the pin (deepgram, vellum) is already transcribing Spanish, so
|
|
594
|
+
// the opener must not default to English. The default test config runs
|
|
595
|
+
// the auto-detect branch ("multi"), so the pinned branch is covered at
|
|
596
|
+
// the fragment level.
|
|
597
|
+
expect(preSpeechLanguageRuleFragment("es", "deepgram")).toContain(
|
|
598
|
+
'configured listening language ("es")',
|
|
599
|
+
);
|
|
600
|
+
expect(preSpeechLanguageRuleFragment("es", "vellum")).toContain(
|
|
601
|
+
"default to English only when neither gives a language",
|
|
602
|
+
);
|
|
603
|
+
for (const autoDetect of ["multi", "", " ", undefined]) {
|
|
604
|
+
expect(preSpeechLanguageRuleFragment(autoDetect, "deepgram")).toBe(
|
|
605
|
+
"use the language the Task context implies, if any; otherwise default to English",
|
|
606
|
+
);
|
|
607
|
+
}
|
|
608
|
+
});
|
|
609
|
+
|
|
610
|
+
test("a language pin on an auto-detecting provider keeps the English fallback", () => {
|
|
611
|
+
// google-gemini and openai-whisper ignore services.stt.language entirely
|
|
612
|
+
// (languageSelection: "auto"), so a persisted "es" pin must not force a
|
|
613
|
+
// Spanish greeting the transcriber will not honor.
|
|
614
|
+
for (const provider of ["google-gemini", "openai-whisper", undefined]) {
|
|
615
|
+
expect(preSpeechLanguageRuleFragment("es", provider)).toBe(
|
|
616
|
+
"use the language the Task context implies, if any; otherwise default to English",
|
|
617
|
+
);
|
|
618
|
+
}
|
|
619
|
+
});
|
|
620
|
+
|
|
621
|
+
test("rule numbers stay sequential from 0, including the routing rule", async () => {
|
|
622
|
+
const installed = captureInstalledPrompt();
|
|
623
|
+
await startVoiceTurn({ ...makeTurnOptions(), routingLeg: "escalated" });
|
|
624
|
+
const numbers = [...installed()!.matchAll(/^(\d+)\. /gm)].map((match) =>
|
|
625
|
+
Number(match[1]),
|
|
626
|
+
);
|
|
627
|
+
expect(numbers.length).toBeGreaterThan(12);
|
|
628
|
+
expect(numbers).toEqual(numbers.map((_, index) => index));
|
|
629
|
+
});
|
|
630
|
+
});
|
|
631
|
+
|
|
563
632
|
describe("startVoiceTurn channel capabilities", () => {
|
|
564
633
|
// Whether a call can show a surface is a property of its channel, not of
|
|
565
634
|
// calls in general, so the bridge applies no voice-specific override: a
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { describe, expect, test } from "bun:test";
|
|
2
2
|
|
|
3
|
+
import { DEEPGRAM_MULTI_LANGUAGE_CODES } from "../../providers/speech-to-text/deepgram.js";
|
|
3
4
|
import {
|
|
4
5
|
capEscalationBridge,
|
|
5
6
|
classifyFrontDoorLeading,
|
|
@@ -7,6 +8,8 @@ import {
|
|
|
7
8
|
escalatedContinuationRule,
|
|
8
9
|
ESCALATION_CONTINUATION_CONTENT,
|
|
9
10
|
FALLBACK_ESCALATION_BRIDGE,
|
|
11
|
+
FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
|
|
12
|
+
fallbackEscalationBridgeFor,
|
|
10
13
|
frontDoorCapabilityDigest,
|
|
11
14
|
frontDoorDecisionRule,
|
|
12
15
|
HOLD_VERDICT_TOKEN,
|
|
@@ -162,6 +165,33 @@ describe("front-door decision rule", () => {
|
|
|
162
165
|
expect(rule.toLowerCase()).toContain("one short natural holding phrase");
|
|
163
166
|
expect(rule.toLowerCase()).toContain("stop after that single sentence");
|
|
164
167
|
});
|
|
168
|
+
|
|
169
|
+
test("hold completeness is judged in the caller's language", () => {
|
|
170
|
+
// Callers are not English-only: the hold branch's exemplars are English,
|
|
171
|
+
// so the rule must say completeness follows the grammar of the language
|
|
172
|
+
// being spoken, with the verb-final case called out (a missing final
|
|
173
|
+
// verb, not a missing conjunction, is the unfinished signal there).
|
|
174
|
+
const withHold = frontDoorDecisionRule({ includeHold: true });
|
|
175
|
+
expect(withHold.toLowerCase()).toContain("may speak any language");
|
|
176
|
+
expect(withHold.toLowerCase()).toContain(
|
|
177
|
+
"grammar of the language being spoken",
|
|
178
|
+
);
|
|
179
|
+
expect(withHold.toLowerCase()).toContain("verb-final");
|
|
180
|
+
expect(withHold.toLowerCase()).toContain("missing final verb");
|
|
181
|
+
});
|
|
182
|
+
|
|
183
|
+
test("the answer branch demands the caller's language", () => {
|
|
184
|
+
expect(rule).toContain("Answer in the language the caller is speaking.");
|
|
185
|
+
});
|
|
186
|
+
|
|
187
|
+
test("the escalation holding phrase demands the caller's language", () => {
|
|
188
|
+
// The bridge examples are English; without an explicit requirement a
|
|
189
|
+
// Spanish turn that needs a tool gets an English holding phrase before
|
|
190
|
+
// the localized answer. The examples stay, labeled as English only.
|
|
191
|
+
expect(rule).toContain("spoken in the language the caller is speaking");
|
|
192
|
+
expect(rule).toContain("those examples are English only");
|
|
193
|
+
expect(rule).toContain(`"${FALLBACK_ESCALATION_BRIDGE}"`);
|
|
194
|
+
});
|
|
165
195
|
});
|
|
166
196
|
|
|
167
197
|
describe("escalated continuation rule", () => {
|
|
@@ -190,6 +220,12 @@ describe("escalated continuation rule", () => {
|
|
|
190
220
|
);
|
|
191
221
|
});
|
|
192
222
|
|
|
223
|
+
test("demands the reply match the caller's language", () => {
|
|
224
|
+
expect(rule).toContain(
|
|
225
|
+
"Reply in the same language as the caller's question.",
|
|
226
|
+
);
|
|
227
|
+
});
|
|
228
|
+
|
|
193
229
|
test("bans re-announcing the holding phrase (bridge-echo regression)", () => {
|
|
194
230
|
// Regression: after the bridge "Let me check your calendar", the quality
|
|
195
231
|
// model opened with "Let me check what calendar connections…" — a
|
|
@@ -201,6 +237,41 @@ describe("escalated continuation rule", () => {
|
|
|
201
237
|
});
|
|
202
238
|
});
|
|
203
239
|
|
|
240
|
+
describe("fallbackEscalationBridgeFor", () => {
|
|
241
|
+
test("covers every Deepgram code-switching language with a non-empty bridge", () => {
|
|
242
|
+
for (const code of DEEPGRAM_MULTI_LANGUAGE_CODES) {
|
|
243
|
+
const bridge = FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE[code];
|
|
244
|
+
expect(bridge).toBeDefined();
|
|
245
|
+
expect(bridge!.trim().length).toBeGreaterThan(0);
|
|
246
|
+
expect(fallbackEscalationBridgeFor(code)).toBe(bridge!);
|
|
247
|
+
}
|
|
248
|
+
});
|
|
249
|
+
|
|
250
|
+
test("every bridge fits the session-side cap", () => {
|
|
251
|
+
for (const bridge of Object.values(
|
|
252
|
+
FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
|
|
253
|
+
)) {
|
|
254
|
+
expect(bridge.length).toBeLessThanOrEqual(MAX_ESCALATION_BRIDGE_CHARS);
|
|
255
|
+
}
|
|
256
|
+
});
|
|
257
|
+
|
|
258
|
+
test("selects by lowercased base subtag", () => {
|
|
259
|
+
expect(fallbackEscalationBridgeFor("pt-BR")).toBe(
|
|
260
|
+
FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE.pt!,
|
|
261
|
+
);
|
|
262
|
+
expect(fallbackEscalationBridgeFor("JA")).toBe(
|
|
263
|
+
FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE.ja!,
|
|
264
|
+
);
|
|
265
|
+
});
|
|
266
|
+
|
|
267
|
+
test("falls back to English for unknown or absent languages", () => {
|
|
268
|
+
expect(fallbackEscalationBridgeFor()).toBe(FALLBACK_ESCALATION_BRIDGE);
|
|
269
|
+
expect(fallbackEscalationBridgeFor("ko")).toBe(FALLBACK_ESCALATION_BRIDGE);
|
|
270
|
+
expect(fallbackEscalationBridgeFor("")).toBe(FALLBACK_ESCALATION_BRIDGE);
|
|
271
|
+
expect(fallbackEscalationBridgeFor("en")).toBe(FALLBACK_ESCALATION_BRIDGE);
|
|
272
|
+
});
|
|
273
|
+
});
|
|
274
|
+
|
|
204
275
|
describe("capEscalationBridge", () => {
|
|
205
276
|
test("cuts just after the first sentence terminator", () => {
|
|
206
277
|
expect(
|
|
@@ -218,6 +289,24 @@ describe("capEscalationBridge", () => {
|
|
|
218
289
|
test("strips internal markers before capping", () => {
|
|
219
290
|
expect(capEscalationBridge("[END_CALL] One moment.")).toBe("One moment.");
|
|
220
291
|
});
|
|
292
|
+
|
|
293
|
+
test("the Unicode ellipsis still terminates a bridge", () => {
|
|
294
|
+
// Regression pin: widening the terminator class for non-Latin enders
|
|
295
|
+
// must not drop the ellipsis the original regex recognized.
|
|
296
|
+
expect(capEscalationBridge("One moment… and some rambling")).toBe(
|
|
297
|
+
"One moment…",
|
|
298
|
+
);
|
|
299
|
+
expect(isEscalationBridgeComplete("One moment…")).toBe(true);
|
|
300
|
+
});
|
|
301
|
+
|
|
302
|
+
test("cuts just after a non-Latin sentence terminator", () => {
|
|
303
|
+
expect(capEscalationBridge("少し考えさせてください。その間の余談")).toBe(
|
|
304
|
+
"少し考えさせてください。",
|
|
305
|
+
);
|
|
306
|
+
expect(capEscalationBridge("मुझे एक पल सोचने दीजिए। और कुछ बातें")).toBe(
|
|
307
|
+
"मुझे एक पल सोचने दीजिए।",
|
|
308
|
+
);
|
|
309
|
+
});
|
|
221
310
|
});
|
|
222
311
|
|
|
223
312
|
describe("isEscalationBridgeComplete", () => {
|
|
@@ -233,6 +322,23 @@ describe("isEscalationBridgeComplete", () => {
|
|
|
233
322
|
isEscalationBridgeComplete("a".repeat(MAX_ESCALATION_BRIDGE_CHARS)),
|
|
234
323
|
).toBe(true);
|
|
235
324
|
});
|
|
325
|
+
|
|
326
|
+
test("a Japanese bridge ending in 。 completes without waiting for the cap", () => {
|
|
327
|
+
expect(isEscalationBridgeComplete("少し考えさせてください")).toBe(false);
|
|
328
|
+
expect(isEscalationBridgeComplete("少し考えさせてください。")).toBe(true);
|
|
329
|
+
});
|
|
330
|
+
|
|
331
|
+
test("every localized fallback bridge ends in a recognized terminator", () => {
|
|
332
|
+
// A model-spoken bridge in any roster language must hand off at its
|
|
333
|
+
// terminator, never by buffering to the char cap; the canned bridges are
|
|
334
|
+
// the canonical sample of each language's ender.
|
|
335
|
+
for (const bridge of Object.values(
|
|
336
|
+
FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
|
|
337
|
+
)) {
|
|
338
|
+
expect(isEscalationBridgeComplete(bridge)).toBe(true);
|
|
339
|
+
expect(capEscalationBridge(bridge)).toBe(bridge);
|
|
340
|
+
}
|
|
341
|
+
});
|
|
236
342
|
});
|
|
237
343
|
|
|
238
344
|
describe("spokenBridgeText", () => {
|
|
@@ -65,6 +65,10 @@ import {
|
|
|
65
65
|
resolveSynthesisFormats,
|
|
66
66
|
} from "./resolve-call-tts-provider.js";
|
|
67
67
|
import type { PromptSpeakerContext } from "./speaker-identification.js";
|
|
68
|
+
import {
|
|
69
|
+
resolveTelephonyLanguageVoice,
|
|
70
|
+
resolveTelephonySynthesisLanguage,
|
|
71
|
+
} from "./telephony-synthesis-language.js";
|
|
68
72
|
import { sanitizeForTts } from "./tts-text-sanitizer.js";
|
|
69
73
|
import {
|
|
70
74
|
ASK_GUARDIAN_CAPTURE_REGEX,
|
|
@@ -189,6 +193,13 @@ export class CallController {
|
|
|
189
193
|
private guardianUnavailableForCall = false;
|
|
190
194
|
/** Active synthesized-TTS session — tracked so interrupt handling can close it. */
|
|
191
195
|
private activeSynthesisAbort: AbortController | null = null;
|
|
196
|
+
/**
|
|
197
|
+
* Resolves the language hint for synthesized speech. The media-stream
|
|
198
|
+
* server supplies a resolver backed by the STT session's detected
|
|
199
|
+
* dominant language; the default falls back to the pin-based
|
|
200
|
+
* resolution only.
|
|
201
|
+
*/
|
|
202
|
+
private resolveSynthesisLanguage: () => string | undefined;
|
|
192
203
|
|
|
193
204
|
constructor(
|
|
194
205
|
callSessionId: string,
|
|
@@ -198,6 +209,7 @@ export class CallController {
|
|
|
198
209
|
broadcast?: (msg: AssistantEvent) => void;
|
|
199
210
|
assistantId?: string;
|
|
200
211
|
trustContext?: TrustContext;
|
|
212
|
+
resolveSynthesisLanguage?: () => string | undefined;
|
|
201
213
|
},
|
|
202
214
|
) {
|
|
203
215
|
this.callSessionId = callSessionId;
|
|
@@ -207,6 +219,9 @@ export class CallController {
|
|
|
207
219
|
this.broadcast = opts?.broadcast;
|
|
208
220
|
this.assistantId = opts?.assistantId ?? DAEMON_INTERNAL_ASSISTANT_ID;
|
|
209
221
|
this.trustContext = opts?.trustContext ?? null;
|
|
222
|
+
this.resolveSynthesisLanguage =
|
|
223
|
+
opts?.resolveSynthesisLanguage ??
|
|
224
|
+
(() => resolveTelephonySynthesisLanguage());
|
|
210
225
|
|
|
211
226
|
// Resolve the conversation ID and skipDisclosure from the call session
|
|
212
227
|
const session = getCallSession(callSessionId);
|
|
@@ -663,7 +678,9 @@ export class CallController {
|
|
|
663
678
|
// lock-hold wait budget, so surface a brief natural re-prompt (never a
|
|
664
679
|
// technical-error message) and re-arm listening. last=true doubles as
|
|
665
680
|
// the end-of-turn marker.
|
|
666
|
-
this.transport.sendTextToken("Sorry, could you say that again?", true
|
|
681
|
+
this.transport.sendTextToken("Sorry, could you say that again?", true, {
|
|
682
|
+
systemCopy: true,
|
|
683
|
+
});
|
|
667
684
|
this.state = "idle";
|
|
668
685
|
this.resetSilenceTimer();
|
|
669
686
|
this.flushPendingInstructions();
|
|
@@ -673,6 +690,7 @@ export class CallController {
|
|
|
673
690
|
this.transport.sendTextToken(
|
|
674
691
|
"I'm sorry, I encountered a technical issue. Could you repeat that?",
|
|
675
692
|
true,
|
|
693
|
+
{ systemCopy: true },
|
|
676
694
|
);
|
|
677
695
|
this.state = "idle";
|
|
678
696
|
this.resetSilenceTimer();
|
|
@@ -1080,11 +1098,17 @@ export class CallController {
|
|
|
1080
1098
|
|
|
1081
1099
|
this.activeSynthesisAbort = abortController;
|
|
1082
1100
|
|
|
1101
|
+
const language = this.resolveSynthesisLanguage();
|
|
1102
|
+
// A language-known segment may select the synthesizing provider's
|
|
1103
|
+
// configured per-language voice; no entry keeps the provider default.
|
|
1104
|
+
const voiceId = resolveTelephonyLanguageVoice(provider.id, language);
|
|
1083
1105
|
await synthesizeAndEmit({
|
|
1084
1106
|
provider,
|
|
1085
1107
|
text,
|
|
1086
1108
|
useCase: "phone-call",
|
|
1087
1109
|
outputFormat,
|
|
1110
|
+
...(voiceId !== undefined ? { voiceId } : {}),
|
|
1111
|
+
...(language !== undefined ? { language } : {}),
|
|
1088
1112
|
signal: abortController.signal,
|
|
1089
1113
|
isCurrent: () => this.isCurrentRun(runVersion),
|
|
1090
1114
|
onChunk: sink.onChunk,
|
|
@@ -1804,6 +1828,7 @@ export class CallController {
|
|
|
1804
1828
|
this.transport.sendTextToken(
|
|
1805
1829
|
"Just to let you know, we're running low on time for this call.",
|
|
1806
1830
|
true,
|
|
1831
|
+
{ systemCopy: true },
|
|
1807
1832
|
);
|
|
1808
1833
|
}, warningMs);
|
|
1809
1834
|
}
|
|
@@ -1816,6 +1841,7 @@ export class CallController {
|
|
|
1816
1841
|
this.transport.sendTextToken(
|
|
1817
1842
|
"I'm sorry, but we've reached the maximum time for this call. Thank you for your time. Goodbye!",
|
|
1818
1843
|
true,
|
|
1844
|
+
{ systemCopy: true },
|
|
1819
1845
|
);
|
|
1820
1846
|
// Give TTS a moment to play, then end
|
|
1821
1847
|
this.durationEndTimer = setTimeout(() => {
|
|
@@ -1878,7 +1904,9 @@ export class CallController {
|
|
|
1878
1904
|
{ callSessionId: this.callSessionId },
|
|
1879
1905
|
"Silence timeout triggered",
|
|
1880
1906
|
);
|
|
1881
|
-
this.transport.sendTextToken("Are you still there?", true
|
|
1907
|
+
this.transport.sendTextToken("Are you still there?", true, {
|
|
1908
|
+
systemCopy: true,
|
|
1909
|
+
});
|
|
1882
1910
|
}, getSilenceTimeoutMs());
|
|
1883
1911
|
}
|
|
1884
1912
|
}
|