@vellumai/assistant 0.11.3-staging.2 → 0.11.3-staging.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/__tests__/call-controller.test.ts +120 -0
- package/src/__tests__/config-loader-backfill.test.ts +19 -0
- package/src/__tests__/config-schema.test.ts +139 -5
- package/src/__tests__/events-dev-bypass-actor.test.ts +112 -1
- package/src/__tests__/media-stream-output.test.ts +175 -0
- package/src/__tests__/media-stream-stt-session.test.ts +67 -0
- package/src/calls/__tests__/tts-text-sanitizer.test.ts +13 -0
- package/src/calls/__tests__/voice-session-bridge.test.ts +81 -12
- package/src/calls/__tests__/voice-triage-escalate.test.ts +106 -0
- package/src/calls/call-controller.ts +30 -2
- package/src/calls/call-speech-output.ts +12 -4
- package/src/calls/call-transport.ts +18 -1
- package/src/calls/media-stream-output.ts +53 -5
- package/src/calls/media-stream-server.ts +12 -0
- package/src/calls/media-stream-stt-session.ts +34 -0
- package/src/calls/telephony-synthesis-language.ts +84 -0
- package/src/calls/tts-text-sanitizer.ts +12 -5
- package/src/calls/voice-session-bridge.ts +31 -3
- package/src/calls/voice-triage-escalate.ts +52 -5
- package/src/config/bundled-skills/phone-calls/references/CONFIG.md +15 -15
- package/src/config/loader.ts +5 -0
- package/src/config/schemas/calls.ts +0 -4
- package/src/config/schemas/tts.ts +63 -0
- package/src/live-voice/__tests__/front-decision.test.ts +120 -0
- package/src/live-voice/__tests__/live-voice-events.test.ts +1 -1
- package/src/live-voice/__tests__/live-voice-progress.test.ts +178 -11
- package/src/live-voice/__tests__/live-voice-stt.test.ts +304 -1
- package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +102 -2
- package/src/live-voice/__tests__/live-voice-tts.test.ts +148 -2
- package/src/live-voice/__tests__/progress-phrases.test.ts +167 -0
- package/src/live-voice/front-decision.ts +50 -3
- package/src/live-voice/live-voice-session.ts +202 -25
- package/src/live-voice/live-voice-tts.ts +18 -2
- package/src/live-voice/progress-phrases.ts +105 -2
- package/src/providers/speech-to-text/deepgram-realtime.test.ts +283 -1
- package/src/providers/speech-to-text/deepgram-realtime.ts +117 -3
- package/src/providers/speech-to-text/provider-catalog.ts +38 -0
- package/src/runtime/__tests__/local-actor-identity-force-refresh.test.ts +104 -0
- package/src/runtime/assistant-event-hub.ts +23 -0
- package/src/runtime/local-actor-identity.ts +18 -5
- package/src/runtime/routes/__tests__/sse-actor-principal-heal.test.ts +165 -0
- package/src/runtime/routes/__tests__/surface-action-routes.test.ts +2 -0
- package/src/runtime/routes/events-routes.ts +17 -16
- package/src/runtime/routes/sse-actor-principal-heal.ts +113 -0
- package/src/stt/__tests__/language-metadata.test.ts +85 -0
- package/src/stt/language-metadata.ts +65 -0
- package/src/stt/types.ts +16 -0
- package/src/tts/__tests__/provider-adapters.test.ts +147 -0
- package/src/tts/__tests__/speakable-segments.test.ts +470 -0
- package/src/tts/language-voices.ts +23 -0
- package/src/tts/providers/deepgram-provider.ts +3 -1
- package/src/tts/providers/elevenlabs-provider.ts +73 -1
- package/src/tts/providers/xai-provider.ts +28 -2
- package/src/tts/speakable-segments.ts +293 -23
- package/src/tts/synthesis-stream.ts +7 -0
- package/src/tts/types.ts +7 -0
- package/src/util/__tests__/language-subtag.test.ts +54 -0
- package/src/util/language-subtag.ts +43 -0
- package/src/util/unicode.ts +1 -1
|
@@ -12,6 +12,7 @@ import {
|
|
|
12
12
|
type LiveVoiceStreamingTranscriberResolver,
|
|
13
13
|
} from "../live-voice-session.js";
|
|
14
14
|
import type { LiveVoiceSessionFactoryContext } from "../live-voice-session-manager.js";
|
|
15
|
+
import type { LiveVoiceTtsOptions } from "../live-voice-tts.js";
|
|
15
16
|
import {
|
|
16
17
|
createLiveVoiceServerFrameSequencer,
|
|
17
18
|
type LiveVoiceClientStartFrame,
|
|
@@ -29,7 +30,9 @@ const START_FRAME = {
|
|
|
29
30
|
} as const satisfies LiveVoiceClientStartFrame;
|
|
30
31
|
|
|
31
32
|
class MockStreamingTranscriber implements StreamingTranscriber {
|
|
32
|
-
|
|
33
|
+
// Configurable: the session gates the language-pin fallback on the DIALED
|
|
34
|
+
// transcriber's providerId (see turnLanguageFor).
|
|
35
|
+
readonly providerId: StreamingTranscriber["providerId"];
|
|
33
36
|
readonly boundaryId = "daemon-streaming" as const;
|
|
34
37
|
readonly audioChunks: Buffer[] = [];
|
|
35
38
|
readonly mimeTypes: string[] = [];
|
|
@@ -37,6 +40,10 @@ class MockStreamingTranscriber implements StreamingTranscriber {
|
|
|
37
40
|
stopped = false;
|
|
38
41
|
private onEvent: ((event: SttStreamServerEvent) => void) | null = null;
|
|
39
42
|
|
|
43
|
+
constructor(providerId: StreamingTranscriber["providerId"] = "deepgram") {
|
|
44
|
+
this.providerId = providerId;
|
|
45
|
+
}
|
|
46
|
+
|
|
40
47
|
async start(onEvent: (event: SttStreamServerEvent) => void): Promise<void> {
|
|
41
48
|
this.started = true;
|
|
42
49
|
this.onEvent = onEvent;
|
|
@@ -135,6 +142,39 @@ function completingVoiceTurnStarter() {
|
|
|
135
142
|
});
|
|
136
143
|
}
|
|
137
144
|
|
|
145
|
+
// A starter whose leg streams one spoken sentence so the turn reaches TTS.
|
|
146
|
+
function speakingVoiceTurnStarter(reply = "Okay.") {
|
|
147
|
+
return mock(async (options: VoiceTurnOptions) => {
|
|
148
|
+
options.callbacks?.assistant_text_delta?.({
|
|
149
|
+
type: "assistant_text_delta",
|
|
150
|
+
text: reply,
|
|
151
|
+
conversationId: options.conversationId,
|
|
152
|
+
});
|
|
153
|
+
options.callbacks?.message_complete?.({
|
|
154
|
+
type: "message_complete",
|
|
155
|
+
conversationId: options.conversationId,
|
|
156
|
+
messageId: "assistant-message-123",
|
|
157
|
+
});
|
|
158
|
+
return { turnId: "bridge-turn-1", abort: mock() };
|
|
159
|
+
});
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
// Records the exact options object of every TTS synthesis request.
|
|
163
|
+
function recordingTtsStreamer() {
|
|
164
|
+
const ttsCalls: LiveVoiceTtsOptions[] = [];
|
|
165
|
+
const streamTtsAudio = mock(async (options: LiveVoiceTtsOptions) => {
|
|
166
|
+
ttsCalls.push(options);
|
|
167
|
+
return {
|
|
168
|
+
provider: "fish-audio" as const,
|
|
169
|
+
contentType: "audio/pcm",
|
|
170
|
+
sampleRate: 24_000,
|
|
171
|
+
chunks: 0,
|
|
172
|
+
bytes: 0,
|
|
173
|
+
};
|
|
174
|
+
});
|
|
175
|
+
return { streamTtsAudio, ttsCalls };
|
|
176
|
+
}
|
|
177
|
+
|
|
138
178
|
function createContext(overrides: Partial<LiveVoiceClientStartFrame> = {}): {
|
|
139
179
|
context: LiveVoiceSessionFactoryContext;
|
|
140
180
|
frames: LiveVoiceServerFrame[];
|
|
@@ -1337,6 +1377,269 @@ describe("LiveVoiceSession STT", () => {
|
|
|
1337
1377
|
]);
|
|
1338
1378
|
});
|
|
1339
1379
|
|
|
1380
|
+
test("finals tagged with a detected language carry it to TTS and the control prompt", async () => {
|
|
1381
|
+
const transcriber = new MockStreamingTranscriber();
|
|
1382
|
+
const startVoiceTurn = speakingVoiceTurnStarter();
|
|
1383
|
+
const { streamTtsAudio, ttsCalls } = recordingTtsStreamer();
|
|
1384
|
+
const { context } = createContext();
|
|
1385
|
+
const session = new LiveVoiceSession(context, {
|
|
1386
|
+
resolveTranscriber: mock(async () => transcriber),
|
|
1387
|
+
startVoiceTurn,
|
|
1388
|
+
streamTtsAudio,
|
|
1389
|
+
});
|
|
1390
|
+
|
|
1391
|
+
await session.start();
|
|
1392
|
+
await session.handleBinaryAudio(new Uint8Array([1]));
|
|
1393
|
+
transcriber.emit({ type: "final", text: "नमस्ते", languages: ["hi"] });
|
|
1394
|
+
await session.handleClientFrame({ type: "ptt_release" });
|
|
1395
|
+
await waitFor(() => ttsCalls.length >= 1);
|
|
1396
|
+
|
|
1397
|
+
expect(ttsCalls[0]?.language).toBe("hi");
|
|
1398
|
+
expect(startVoiceTurn.mock.calls[0]?.[0]?.voiceControlPrompt).toContain(
|
|
1399
|
+
'speaking the language with code "hi"',
|
|
1400
|
+
);
|
|
1401
|
+
});
|
|
1402
|
+
|
|
1403
|
+
test("a tagged partial supplies the turn language when finals carry no tags", async () => {
|
|
1404
|
+
// Speculative turns dispatch from partials before the first tagged
|
|
1405
|
+
// final, so the latest tagged partial must resolve the language.
|
|
1406
|
+
const transcriber = new MockStreamingTranscriber();
|
|
1407
|
+
const startVoiceTurn = speakingVoiceTurnStarter();
|
|
1408
|
+
const { streamTtsAudio, ttsCalls } = recordingTtsStreamer();
|
|
1409
|
+
const { context } = createContext();
|
|
1410
|
+
const session = new LiveVoiceSession(context, {
|
|
1411
|
+
resolveTranscriber: mock(async () => transcriber),
|
|
1412
|
+
startVoiceTurn,
|
|
1413
|
+
streamTtsAudio,
|
|
1414
|
+
});
|
|
1415
|
+
|
|
1416
|
+
await session.start();
|
|
1417
|
+
await session.handleBinaryAudio(new Uint8Array([1]));
|
|
1418
|
+
transcriber.emit({ type: "partial", text: "hola", languages: ["es"] });
|
|
1419
|
+
transcriber.emit({ type: "final", text: "hola amigo" });
|
|
1420
|
+
await session.handleClientFrame({ type: "ptt_release" });
|
|
1421
|
+
await waitFor(() => ttsCalls.length >= 1);
|
|
1422
|
+
|
|
1423
|
+
expect(ttsCalls[0]?.language).toBe("es");
|
|
1424
|
+
});
|
|
1425
|
+
|
|
1426
|
+
test("empty finals carrying language tags do not vote", async () => {
|
|
1427
|
+
// Silence frames can carry container-level tags describing no emitted
|
|
1428
|
+
// words; they must not outvote the language of real speech.
|
|
1429
|
+
const transcriber = new MockStreamingTranscriber();
|
|
1430
|
+
const startVoiceTurn = speakingVoiceTurnStarter();
|
|
1431
|
+
const { streamTtsAudio, ttsCalls } = recordingTtsStreamer();
|
|
1432
|
+
const { context } = createContext();
|
|
1433
|
+
const session = new LiveVoiceSession(context, {
|
|
1434
|
+
resolveTranscriber: mock(async () => transcriber),
|
|
1435
|
+
startVoiceTurn,
|
|
1436
|
+
streamTtsAudio,
|
|
1437
|
+
});
|
|
1438
|
+
|
|
1439
|
+
await session.start();
|
|
1440
|
+
await session.handleBinaryAudio(new Uint8Array([1]));
|
|
1441
|
+
transcriber.emit({ type: "final", text: "", languages: ["es"] });
|
|
1442
|
+
transcriber.emit({ type: "final", text: "", languages: ["es"] });
|
|
1443
|
+
transcriber.emit({ type: "final", text: "hello there", languages: ["en"] });
|
|
1444
|
+
await session.handleClientFrame({ type: "ptt_release" });
|
|
1445
|
+
await waitFor(() => ttsCalls.length >= 1);
|
|
1446
|
+
|
|
1447
|
+
expect(ttsCalls[0]?.language).toBe("en");
|
|
1448
|
+
});
|
|
1449
|
+
|
|
1450
|
+
test("only each final's dominant language votes", async () => {
|
|
1451
|
+
// `languages` is dominance-ranked per event: a Spanish-dominant segment
|
|
1452
|
+
// tagged ["es", "en"] plus a short English segment tagged ["en"] must
|
|
1453
|
+
// resolve to Spanish (one vote each, tie broken by first appearance),
|
|
1454
|
+
// not English by counting the secondary tag as a full vote.
|
|
1455
|
+
const transcriber = new MockStreamingTranscriber();
|
|
1456
|
+
const startVoiceTurn = speakingVoiceTurnStarter();
|
|
1457
|
+
const { streamTtsAudio, ttsCalls } = recordingTtsStreamer();
|
|
1458
|
+
const { context } = createContext();
|
|
1459
|
+
const session = new LiveVoiceSession(context, {
|
|
1460
|
+
resolveTranscriber: mock(async () => transcriber),
|
|
1461
|
+
startVoiceTurn,
|
|
1462
|
+
streamTtsAudio,
|
|
1463
|
+
});
|
|
1464
|
+
|
|
1465
|
+
await session.start();
|
|
1466
|
+
await session.handleBinaryAudio(new Uint8Array([1]));
|
|
1467
|
+
transcriber.emit({
|
|
1468
|
+
type: "final",
|
|
1469
|
+
text: "hola amigo como estas hoy",
|
|
1470
|
+
languages: ["es", "en"],
|
|
1471
|
+
});
|
|
1472
|
+
transcriber.emit({ type: "final", text: "ok", languages: ["en"] });
|
|
1473
|
+
await session.handleClientFrame({ type: "ptt_release" });
|
|
1474
|
+
await waitFor(() => ttsCalls.length >= 1);
|
|
1475
|
+
|
|
1476
|
+
expect(ttsCalls[0]?.language).toBe("es");
|
|
1477
|
+
});
|
|
1478
|
+
|
|
1479
|
+
test("a tagged final outranks an earlier tagged partial", async () => {
|
|
1480
|
+
const transcriber = new MockStreamingTranscriber();
|
|
1481
|
+
const startVoiceTurn = speakingVoiceTurnStarter();
|
|
1482
|
+
const { streamTtsAudio, ttsCalls } = recordingTtsStreamer();
|
|
1483
|
+
const { context } = createContext();
|
|
1484
|
+
const session = new LiveVoiceSession(context, {
|
|
1485
|
+
resolveTranscriber: mock(async () => transcriber),
|
|
1486
|
+
startVoiceTurn,
|
|
1487
|
+
streamTtsAudio,
|
|
1488
|
+
});
|
|
1489
|
+
|
|
1490
|
+
await session.start();
|
|
1491
|
+
await session.handleBinaryAudio(new Uint8Array([1]));
|
|
1492
|
+
transcriber.emit({ type: "partial", text: "hola", languages: ["es"] });
|
|
1493
|
+
transcriber.emit({ type: "final", text: "नमस्ते", languages: ["hi"] });
|
|
1494
|
+
await session.handleClientFrame({ type: "ptt_release" });
|
|
1495
|
+
await waitFor(() => ttsCalls.length >= 1);
|
|
1496
|
+
|
|
1497
|
+
expect(ttsCalls[0]?.language).toBe("hi");
|
|
1498
|
+
});
|
|
1499
|
+
|
|
1500
|
+
test("a monolingual services.stt.language pin is the turn language when finals carry no tags", async () => {
|
|
1501
|
+
const originalRaw = loadRawConfig();
|
|
1502
|
+
const rawServices = (originalRaw.services ?? {}) as Record<string, unknown>;
|
|
1503
|
+
saveRawConfig({
|
|
1504
|
+
...originalRaw,
|
|
1505
|
+
services: {
|
|
1506
|
+
...rawServices,
|
|
1507
|
+
stt: {
|
|
1508
|
+
...((rawServices.stt ?? {}) as Record<string, unknown>),
|
|
1509
|
+
language: "ja",
|
|
1510
|
+
},
|
|
1511
|
+
},
|
|
1512
|
+
});
|
|
1513
|
+
const transcriber = new MockStreamingTranscriber();
|
|
1514
|
+
const startVoiceTurn = speakingVoiceTurnStarter();
|
|
1515
|
+
const { streamTtsAudio, ttsCalls } = recordingTtsStreamer();
|
|
1516
|
+
const { context } = createContext();
|
|
1517
|
+
const session = new LiveVoiceSession(context, {
|
|
1518
|
+
resolveTranscriber: mock(async () => transcriber),
|
|
1519
|
+
startVoiceTurn,
|
|
1520
|
+
streamTtsAudio,
|
|
1521
|
+
});
|
|
1522
|
+
|
|
1523
|
+
try {
|
|
1524
|
+
await session.start();
|
|
1525
|
+
await session.handleBinaryAudio(new Uint8Array([1]));
|
|
1526
|
+
await session.handleClientFrame({ type: "ptt_release" });
|
|
1527
|
+
await waitFor(() => ttsCalls.length >= 1);
|
|
1528
|
+
|
|
1529
|
+
expect(ttsCalls[0]?.language).toBe("ja");
|
|
1530
|
+
expect(startVoiceTurn.mock.calls[0]?.[0]?.voiceControlPrompt).toContain(
|
|
1531
|
+
'speaking the language with code "ja"',
|
|
1532
|
+
);
|
|
1533
|
+
} finally {
|
|
1534
|
+
await session.close("websocket_close");
|
|
1535
|
+
saveRawConfig(originalRaw);
|
|
1536
|
+
}
|
|
1537
|
+
});
|
|
1538
|
+
|
|
1539
|
+
test("a pin on an auto-detecting provider does not become the turn language", async () => {
|
|
1540
|
+
// google-gemini and openai-whisper ignore services.stt.language, so a
|
|
1541
|
+
// stale persisted pin must not force every turn into that language.
|
|
1542
|
+
const originalRaw = loadRawConfig();
|
|
1543
|
+
const rawServices = (originalRaw.services ?? {}) as Record<string, unknown>;
|
|
1544
|
+
saveRawConfig({
|
|
1545
|
+
...originalRaw,
|
|
1546
|
+
services: {
|
|
1547
|
+
...rawServices,
|
|
1548
|
+
stt: {
|
|
1549
|
+
...((rawServices.stt ?? {}) as Record<string, unknown>),
|
|
1550
|
+
provider: "google-gemini",
|
|
1551
|
+
language: "es",
|
|
1552
|
+
},
|
|
1553
|
+
},
|
|
1554
|
+
});
|
|
1555
|
+
const transcriber = new MockStreamingTranscriber("google-gemini");
|
|
1556
|
+
const startVoiceTurn = speakingVoiceTurnStarter();
|
|
1557
|
+
const { streamTtsAudio, ttsCalls } = recordingTtsStreamer();
|
|
1558
|
+
const { context } = createContext();
|
|
1559
|
+
const session = new LiveVoiceSession(context, {
|
|
1560
|
+
resolveTranscriber: mock(async () => transcriber),
|
|
1561
|
+
startVoiceTurn,
|
|
1562
|
+
streamTtsAudio,
|
|
1563
|
+
});
|
|
1564
|
+
|
|
1565
|
+
try {
|
|
1566
|
+
await session.start();
|
|
1567
|
+
await session.handleBinaryAudio(new Uint8Array([1]));
|
|
1568
|
+
await session.handleClientFrame({ type: "ptt_release" });
|
|
1569
|
+
await waitFor(() => ttsCalls.length >= 1);
|
|
1570
|
+
|
|
1571
|
+
expect(ttsCalls[0] && "language" in ttsCalls[0]).toBe(false);
|
|
1572
|
+
} finally {
|
|
1573
|
+
await session.close("websocket_close");
|
|
1574
|
+
saveRawConfig(originalRaw);
|
|
1575
|
+
}
|
|
1576
|
+
});
|
|
1577
|
+
|
|
1578
|
+
test("the pin gate follows the dialed provider, not the configured one", async () => {
|
|
1579
|
+
// A BYOK gemini config without credentials silently dials the managed
|
|
1580
|
+
// vellum transcriber, which DOES honor the pin; the gate must follow
|
|
1581
|
+
// what was dialed, not what was configured.
|
|
1582
|
+
const originalRaw = loadRawConfig();
|
|
1583
|
+
const rawServices = (originalRaw.services ?? {}) as Record<string, unknown>;
|
|
1584
|
+
saveRawConfig({
|
|
1585
|
+
...originalRaw,
|
|
1586
|
+
services: {
|
|
1587
|
+
...rawServices,
|
|
1588
|
+
stt: {
|
|
1589
|
+
...((rawServices.stt ?? {}) as Record<string, unknown>),
|
|
1590
|
+
provider: "google-gemini",
|
|
1591
|
+
language: "es",
|
|
1592
|
+
},
|
|
1593
|
+
},
|
|
1594
|
+
});
|
|
1595
|
+
const transcriber = new MockStreamingTranscriber("vellum");
|
|
1596
|
+
const startVoiceTurn = speakingVoiceTurnStarter();
|
|
1597
|
+
const { streamTtsAudio, ttsCalls } = recordingTtsStreamer();
|
|
1598
|
+
const { context } = createContext();
|
|
1599
|
+
const session = new LiveVoiceSession(context, {
|
|
1600
|
+
resolveTranscriber: mock(async () => transcriber),
|
|
1601
|
+
startVoiceTurn,
|
|
1602
|
+
streamTtsAudio,
|
|
1603
|
+
});
|
|
1604
|
+
|
|
1605
|
+
try {
|
|
1606
|
+
await session.start();
|
|
1607
|
+
await session.handleBinaryAudio(new Uint8Array([1]));
|
|
1608
|
+
await session.handleClientFrame({ type: "ptt_release" });
|
|
1609
|
+
await waitFor(() => ttsCalls.length >= 1);
|
|
1610
|
+
|
|
1611
|
+
expect(ttsCalls[0]?.language).toBe("es");
|
|
1612
|
+
} finally {
|
|
1613
|
+
await session.close("websocket_close");
|
|
1614
|
+
saveRawConfig(originalRaw);
|
|
1615
|
+
}
|
|
1616
|
+
});
|
|
1617
|
+
|
|
1618
|
+
test('the default "multi" with tag-less finals leaves every language-aware path untouched', async () => {
|
|
1619
|
+
const transcriber = new MockStreamingTranscriber();
|
|
1620
|
+
const startVoiceTurn = speakingVoiceTurnStarter();
|
|
1621
|
+
const { streamTtsAudio, ttsCalls } = recordingTtsStreamer();
|
|
1622
|
+
const { context } = createContext();
|
|
1623
|
+
const session = new LiveVoiceSession(context, {
|
|
1624
|
+
resolveTranscriber: mock(async () => transcriber),
|
|
1625
|
+
startVoiceTurn,
|
|
1626
|
+
streamTtsAudio,
|
|
1627
|
+
});
|
|
1628
|
+
|
|
1629
|
+
await session.start();
|
|
1630
|
+
await session.handleBinaryAudio(new Uint8Array([1]));
|
|
1631
|
+
await session.handleClientFrame({ type: "ptt_release" });
|
|
1632
|
+
await waitFor(() => ttsCalls.length >= 1);
|
|
1633
|
+
|
|
1634
|
+
// With the language unknown the TTS request carries no language key at
|
|
1635
|
+
// all and the control prompt has no per-turn language note: byte-for-byte
|
|
1636
|
+
// the language-blind behavior.
|
|
1637
|
+
expect("language" in ttsCalls[0]!).toBe(false);
|
|
1638
|
+
expect(startVoiceTurn.mock.calls[0]?.[0]?.voiceControlPrompt).not.toContain(
|
|
1639
|
+
"language with code",
|
|
1640
|
+
);
|
|
1641
|
+
});
|
|
1642
|
+
|
|
1340
1643
|
test("uses the production streaming transcriber resolver by default", () => {
|
|
1341
1644
|
const source = readFileSync(
|
|
1342
1645
|
new URL("../live-voice-session.ts", import.meta.url),
|
|
@@ -1,9 +1,11 @@
|
|
|
1
1
|
import { describe, expect, mock, test } from "bun:test";
|
|
2
2
|
|
|
3
|
+
import { sanitizeForTts } from "../../calls/tts-text-sanitizer.js";
|
|
3
4
|
import type { VoiceTurnOptions } from "../../calls/voice-session-bridge.js";
|
|
4
5
|
import {
|
|
5
6
|
ESCALATION_CONTINUATION_CONTENT,
|
|
6
7
|
FALLBACK_ESCALATION_BRIDGE,
|
|
8
|
+
FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
|
|
7
9
|
} from "../../calls/voice-triage-escalate.js";
|
|
8
10
|
import type {
|
|
9
11
|
StreamingTranscriber,
|
|
@@ -11,9 +13,11 @@ import type {
|
|
|
11
13
|
} from "../../stt/types.js";
|
|
12
14
|
import {
|
|
13
15
|
LiveVoiceSession,
|
|
16
|
+
type LiveVoiceTtsStreamer,
|
|
14
17
|
type LiveVoiceTurnStarter,
|
|
15
18
|
} from "../live-voice-session.js";
|
|
16
19
|
import type { LiveVoiceSessionFactoryContext } from "../live-voice-session-manager.js";
|
|
20
|
+
import type { LiveVoiceTtsOptions } from "../live-voice-tts.js";
|
|
17
21
|
import {
|
|
18
22
|
createLiveVoiceServerFrameSequencer,
|
|
19
23
|
type LiveVoiceClientStartFrame,
|
|
@@ -57,7 +61,13 @@ class MockStreamingTranscriber implements StreamingTranscriber {
|
|
|
57
61
|
}
|
|
58
62
|
}
|
|
59
63
|
|
|
60
|
-
function createHarness(
|
|
64
|
+
function createHarness(
|
|
65
|
+
startVoiceTurn: LiveVoiceTurnStarter,
|
|
66
|
+
opts: {
|
|
67
|
+
transcriber?: MockStreamingTranscriber;
|
|
68
|
+
streamTtsAudio?: LiveVoiceTtsStreamer;
|
|
69
|
+
} = {},
|
|
70
|
+
) {
|
|
61
71
|
const sequencer = createLiveVoiceServerFrameSequencer();
|
|
62
72
|
const frames: LiveVoiceServerFrame[] = [];
|
|
63
73
|
const context: LiveVoiceSessionFactoryContext = {
|
|
@@ -69,10 +79,11 @@ function createHarness(startVoiceTurn: LiveVoiceTurnStarter) {
|
|
|
69
79
|
return frame;
|
|
70
80
|
}),
|
|
71
81
|
};
|
|
72
|
-
const transcriber = new MockStreamingTranscriber();
|
|
82
|
+
const transcriber = opts.transcriber ?? new MockStreamingTranscriber();
|
|
73
83
|
const session = new LiveVoiceSession(context, {
|
|
74
84
|
resolveTranscriber: mock(async () => transcriber),
|
|
75
85
|
startVoiceTurn,
|
|
86
|
+
...(opts.streamTtsAudio ? { streamTtsAudio: opts.streamTtsAudio } : {}),
|
|
76
87
|
createTurnId: () => "live-turn-1",
|
|
77
88
|
emitMetrics: false,
|
|
78
89
|
});
|
|
@@ -313,6 +324,95 @@ describe("live-voice triage-and-escalate routing", () => {
|
|
|
313
324
|
expect(spokenText(frames)).not.toContain(FALLBACK_ESCALATION_BRIDGE);
|
|
314
325
|
});
|
|
315
326
|
|
|
327
|
+
test("escalating a Hindi turn speaks the Hindi fallback bridge and quotes it to the escalated leg", async () => {
|
|
328
|
+
const { starter } = scriptedStartVoiceTurn({
|
|
329
|
+
frontDoor: ["[1]"],
|
|
330
|
+
escalated: ["तैयार उत्तर।"],
|
|
331
|
+
});
|
|
332
|
+
const ttsCalls: LiveVoiceTtsOptions[] = [];
|
|
333
|
+
const streamTtsAudio = mock(async (options: LiveVoiceTtsOptions) => {
|
|
334
|
+
ttsCalls.push(options);
|
|
335
|
+
return {
|
|
336
|
+
provider: "fish-audio" as const,
|
|
337
|
+
contentType: "audio/pcm",
|
|
338
|
+
sampleRate: 24_000,
|
|
339
|
+
chunks: 0,
|
|
340
|
+
bytes: 0,
|
|
341
|
+
};
|
|
342
|
+
});
|
|
343
|
+
const { frames, session } = createHarness(starter, {
|
|
344
|
+
transcriber: new MockStreamingTranscriber([
|
|
345
|
+
{ type: "final", text: "नमस्ते", languages: ["hi"] },
|
|
346
|
+
{ type: "closed" },
|
|
347
|
+
]),
|
|
348
|
+
streamTtsAudio,
|
|
349
|
+
});
|
|
350
|
+
|
|
351
|
+
await driveTurn(session);
|
|
352
|
+
await waitFor(() => starter.mock.calls.length >= 2);
|
|
353
|
+
await waitFor(() => frames.some((frame) => frame.type === "tts_done"));
|
|
354
|
+
|
|
355
|
+
const hindiBridge = FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE.hi!;
|
|
356
|
+
// The escalated leg is told the exact phrase the caller heard, so its
|
|
357
|
+
// continuation rule can quote it and ban a re-announcing echo.
|
|
358
|
+
expect(starter.mock.calls[1]?.[0]?.spokenEscalationBridge).toBe(
|
|
359
|
+
hindiBridge,
|
|
360
|
+
);
|
|
361
|
+
// The caller hears the Hindi fallback; like every canned bridge it is
|
|
362
|
+
// audio-only, so it reaches TTS but never a caption frame.
|
|
363
|
+
expect(ttsCalls.map((call) => call.text)).toContain(
|
|
364
|
+
sanitizeForTts(hindiBridge).trim(),
|
|
365
|
+
);
|
|
366
|
+
expect(spokenText(frames)).not.toContain(hindiBridge);
|
|
367
|
+
// The detected language rides every synthesis request of the turn.
|
|
368
|
+
expect(ttsCalls.every((call) => call.language === "hi")).toBe(true);
|
|
369
|
+
});
|
|
370
|
+
|
|
371
|
+
test("escalating an out-of-roster-language turn hints the English fallback bridge as 'en'", async () => {
|
|
372
|
+
// "ar" is outside the localized bridge table, so the canned fallback is
|
|
373
|
+
// English text; its synthesis request must carry an "en" hint rather
|
|
374
|
+
// than the turn's "ar", while the escalated leg's model speech keeps
|
|
375
|
+
// the turn language.
|
|
376
|
+
const { starter } = scriptedStartVoiceTurn({
|
|
377
|
+
frontDoor: ["[1]"],
|
|
378
|
+
escalated: ["The thorough answer."],
|
|
379
|
+
});
|
|
380
|
+
const ttsCalls: LiveVoiceTtsOptions[] = [];
|
|
381
|
+
const streamTtsAudio = mock(async (options: LiveVoiceTtsOptions) => {
|
|
382
|
+
ttsCalls.push(options);
|
|
383
|
+
return {
|
|
384
|
+
provider: "fish-audio" as const,
|
|
385
|
+
contentType: "audio/pcm",
|
|
386
|
+
sampleRate: 24_000,
|
|
387
|
+
chunks: 0,
|
|
388
|
+
bytes: 0,
|
|
389
|
+
};
|
|
390
|
+
});
|
|
391
|
+
const { frames, session } = createHarness(starter, {
|
|
392
|
+
transcriber: new MockStreamingTranscriber([
|
|
393
|
+
{ type: "final", text: "مرحبا", languages: ["ar"] },
|
|
394
|
+
{ type: "closed" },
|
|
395
|
+
]),
|
|
396
|
+
streamTtsAudio,
|
|
397
|
+
});
|
|
398
|
+
|
|
399
|
+
await driveTurn(session);
|
|
400
|
+
await waitFor(() => starter.mock.calls.length >= 2);
|
|
401
|
+
await waitFor(() => frames.some((frame) => frame.type === "tts_done"));
|
|
402
|
+
|
|
403
|
+
expect(starter.mock.calls[1]?.[0]?.spokenEscalationBridge).toBe(
|
|
404
|
+
FALLBACK_ESCALATION_BRIDGE,
|
|
405
|
+
);
|
|
406
|
+
const bridgeCall = ttsCalls.find(
|
|
407
|
+
(call) => call.text === sanitizeForTts(FALLBACK_ESCALATION_BRIDGE).trim(),
|
|
408
|
+
);
|
|
409
|
+
expect(bridgeCall?.language).toBe("en");
|
|
410
|
+
const answerCall = ttsCalls.find((call) =>
|
|
411
|
+
call.text.includes("The thorough answer"),
|
|
412
|
+
);
|
|
413
|
+
expect(answerCall?.language).toBe("ar");
|
|
414
|
+
});
|
|
415
|
+
|
|
316
416
|
test("barge-in during the escalated leg aborts it", async () => {
|
|
317
417
|
const { starter, escalatedAbort } = scriptedStartVoiceTurn({
|
|
318
418
|
frontDoor: ["[1] ", "Let me think about that."],
|
|
@@ -24,8 +24,11 @@ mock.module("../../security/secure-keys.js", () => ({
|
|
|
24
24
|
getProviderKeyAsync: async () => "test-api-key",
|
|
25
25
|
}));
|
|
26
26
|
|
|
27
|
-
const {
|
|
28
|
-
|
|
27
|
+
const {
|
|
28
|
+
LiveVoiceTtsError,
|
|
29
|
+
resolveLanguageVoiceOverride,
|
|
30
|
+
streamLiveVoiceTtsAudio,
|
|
31
|
+
} = await import("../live-voice-tts.js");
|
|
29
32
|
|
|
30
33
|
beforeEach(() => {
|
|
31
34
|
config = makeConfig();
|
|
@@ -494,6 +497,76 @@ describe("streamLiveVoiceTtsAudio", () => {
|
|
|
494
497
|
}
|
|
495
498
|
});
|
|
496
499
|
|
|
500
|
+
test("applies the configured per-language voice for a language-known turn", async () => {
|
|
501
|
+
config = makeConfig({
|
|
502
|
+
provider: "elevenlabs",
|
|
503
|
+
elevenlabsLanguageVoices: { hi: "voice-hindi", ja: "voice-japanese" },
|
|
504
|
+
});
|
|
505
|
+
const { requests, probeRequests } = installCapturingElevenLabsStub();
|
|
506
|
+
|
|
507
|
+
await streamLiveVoiceTtsAudio({
|
|
508
|
+
config,
|
|
509
|
+
text: "namaste",
|
|
510
|
+
language: "hi",
|
|
511
|
+
onAudioChunk: () => {},
|
|
512
|
+
});
|
|
513
|
+
|
|
514
|
+
expect(requests[0]?.voiceId).toBe("voice-hindi");
|
|
515
|
+
expect(probeRequests[0]?.voiceId).toBe("voice-hindi");
|
|
516
|
+
});
|
|
517
|
+
|
|
518
|
+
test("skips the override when the map has no entry for the turn language", async () => {
|
|
519
|
+
config = makeConfig({
|
|
520
|
+
provider: "elevenlabs",
|
|
521
|
+
elevenlabsLanguageVoices: { hi: "voice-hindi" },
|
|
522
|
+
});
|
|
523
|
+
const { requests } = installCapturingElevenLabsStub();
|
|
524
|
+
|
|
525
|
+
await streamLiveVoiceTtsAudio({
|
|
526
|
+
config,
|
|
527
|
+
text: "hola",
|
|
528
|
+
language: "es",
|
|
529
|
+
onAudioChunk: () => {},
|
|
530
|
+
});
|
|
531
|
+
|
|
532
|
+
expect(requests[0]?.voiceId).toBeUndefined();
|
|
533
|
+
});
|
|
534
|
+
|
|
535
|
+
test("an explicit request voiceId outranks the per-language map", async () => {
|
|
536
|
+
config = makeConfig({
|
|
537
|
+
provider: "elevenlabs",
|
|
538
|
+
elevenlabsLanguageVoices: { hi: "voice-hindi" },
|
|
539
|
+
});
|
|
540
|
+
const { requests, probeRequests } = installCapturingElevenLabsStub();
|
|
541
|
+
|
|
542
|
+
await streamLiveVoiceTtsAudio({
|
|
543
|
+
config,
|
|
544
|
+
text: "namaste",
|
|
545
|
+
language: "hi",
|
|
546
|
+
voiceId: "voice-explicit",
|
|
547
|
+
onAudioChunk: () => {},
|
|
548
|
+
});
|
|
549
|
+
|
|
550
|
+
expect(requests[0]?.voiceId).toBe("voice-explicit");
|
|
551
|
+
expect(probeRequests[0]?.voiceId).toBe("voice-explicit");
|
|
552
|
+
});
|
|
553
|
+
|
|
554
|
+
test("performs no lookup when the turn has no language", async () => {
|
|
555
|
+
config = makeConfig({
|
|
556
|
+
provider: "elevenlabs",
|
|
557
|
+
elevenlabsLanguageVoices: { hi: "voice-hindi" },
|
|
558
|
+
});
|
|
559
|
+
const { requests } = installCapturingElevenLabsStub();
|
|
560
|
+
|
|
561
|
+
await streamLiveVoiceTtsAudio({
|
|
562
|
+
config,
|
|
563
|
+
text: "hello",
|
|
564
|
+
onAudioChunk: () => {},
|
|
565
|
+
});
|
|
566
|
+
|
|
567
|
+
expect(requests[0]?.voiceId).toBeUndefined();
|
|
568
|
+
});
|
|
569
|
+
|
|
497
570
|
test("keeps live voice TTS behind the registry instead of direct provider SDKs", () => {
|
|
498
571
|
const source = readFileSync(
|
|
499
572
|
new URL("../live-voice-tts.ts", import.meta.url),
|
|
@@ -508,11 +581,83 @@ describe("streamLiveVoiceTtsAudio", () => {
|
|
|
508
581
|
});
|
|
509
582
|
});
|
|
510
583
|
|
|
584
|
+
describe("resolveLanguageVoiceOverride", () => {
|
|
585
|
+
test("keys the lookup by the language's lowercase base subtag", () => {
|
|
586
|
+
const map = { hi: "voice-hindi" };
|
|
587
|
+
expect(resolveLanguageVoiceOverride(map, "hi")).toBe("voice-hindi");
|
|
588
|
+
expect(resolveLanguageVoiceOverride(map, "hi-IN")).toBe("voice-hindi");
|
|
589
|
+
expect(resolveLanguageVoiceOverride(map, "HI")).toBe("voice-hindi");
|
|
590
|
+
});
|
|
591
|
+
|
|
592
|
+
test("returns undefined without a language", () => {
|
|
593
|
+
expect(
|
|
594
|
+
resolveLanguageVoiceOverride({ hi: "voice-hindi" }, undefined),
|
|
595
|
+
).toBeUndefined();
|
|
596
|
+
expect(
|
|
597
|
+
resolveLanguageVoiceOverride({ hi: "voice-hindi" }, ""),
|
|
598
|
+
).toBeUndefined();
|
|
599
|
+
});
|
|
600
|
+
|
|
601
|
+
test("returns undefined for a missing map, a missing entry, or a blank voice", () => {
|
|
602
|
+
expect(resolveLanguageVoiceOverride(undefined, "hi")).toBeUndefined();
|
|
603
|
+
expect(
|
|
604
|
+
resolveLanguageVoiceOverride({ ja: "voice-japanese" }, "hi"),
|
|
605
|
+
).toBeUndefined();
|
|
606
|
+
expect(resolveLanguageVoiceOverride({ hi: " " }, "hi")).toBeUndefined();
|
|
607
|
+
expect(resolveLanguageVoiceOverride({}, "constructor")).toBeUndefined();
|
|
608
|
+
});
|
|
609
|
+
|
|
610
|
+
test("trims the resolved voice identifier", () => {
|
|
611
|
+
expect(resolveLanguageVoiceOverride({ hi: " voice-hindi " }, "hi")).toBe(
|
|
612
|
+
"voice-hindi",
|
|
613
|
+
);
|
|
614
|
+
});
|
|
615
|
+
});
|
|
616
|
+
|
|
617
|
+
/**
|
|
618
|
+
* Install an ElevenLabs-id streaming stub that records the sample-rate probe
|
|
619
|
+
* and synthesis requests.
|
|
620
|
+
*/
|
|
621
|
+
function installCapturingElevenLabsStub(): {
|
|
622
|
+
requests: TtsSynthesisRequest[];
|
|
623
|
+
probeRequests: TtsSynthesisRequest[];
|
|
624
|
+
} {
|
|
625
|
+
const requests: TtsSynthesisRequest[] = [];
|
|
626
|
+
const probeRequests: TtsSynthesisRequest[] = [];
|
|
627
|
+
_setTtsProviderForTests({
|
|
628
|
+
id: "elevenlabs",
|
|
629
|
+
capabilities: {
|
|
630
|
+
supportsStreaming: true,
|
|
631
|
+
supportedFormats: ["mp3", "pcm"],
|
|
632
|
+
},
|
|
633
|
+
resolveOutputSampleRateHz: (request) => {
|
|
634
|
+
probeRequests.push(request);
|
|
635
|
+
return 24_000;
|
|
636
|
+
},
|
|
637
|
+
async synthesize(): Promise<TtsSynthesisResult> {
|
|
638
|
+
throw new Error("buffered synthesis should not be used");
|
|
639
|
+
},
|
|
640
|
+
async synthesizeStream(
|
|
641
|
+
request: TtsSynthesisRequest,
|
|
642
|
+
onChunk: (chunk: Uint8Array) => void,
|
|
643
|
+
): Promise<TtsSynthesisResult> {
|
|
644
|
+
requests.push(request);
|
|
645
|
+
onChunk(Buffer.from("pcm-one!"));
|
|
646
|
+
return {
|
|
647
|
+
audio: Buffer.from("pcm-one!"),
|
|
648
|
+
contentType: "audio/pcm",
|
|
649
|
+
};
|
|
650
|
+
},
|
|
651
|
+
});
|
|
652
|
+
return { requests, probeRequests };
|
|
653
|
+
}
|
|
654
|
+
|
|
511
655
|
function makeConfig(
|
|
512
656
|
overrides: {
|
|
513
657
|
provider?: string;
|
|
514
658
|
format?: "mp3" | "wav" | "opus";
|
|
515
659
|
sampleRate?: number;
|
|
660
|
+
elevenlabsLanguageVoices?: Record<string, string>;
|
|
516
661
|
} = {},
|
|
517
662
|
): LiveVoiceTtsConfig {
|
|
518
663
|
return {
|
|
@@ -535,6 +680,7 @@ function makeConfig(
|
|
|
535
680
|
stability: 0.5,
|
|
536
681
|
similarityBoost: 0.75,
|
|
537
682
|
conversationTimeoutSeconds: 30,
|
|
683
|
+
languageVoices: overrides.elevenlabsLanguageVoices,
|
|
538
684
|
},
|
|
539
685
|
deepgram: {
|
|
540
686
|
model: "aura-asteria-en",
|