@vellumai/assistant 0.11.3-staging.2 → 0.11.3-staging.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/__tests__/call-controller.test.ts +120 -0
- package/src/__tests__/config-loader-backfill.test.ts +19 -0
- package/src/__tests__/config-schema.test.ts +142 -5
- package/src/__tests__/events-dev-bypass-actor.test.ts +112 -1
- package/src/__tests__/media-stream-output.test.ts +175 -0
- package/src/__tests__/media-stream-stt-session.test.ts +67 -0
- package/src/calls/__tests__/tts-text-sanitizer.test.ts +13 -0
- package/src/calls/__tests__/voice-session-bridge.test.ts +81 -12
- package/src/calls/__tests__/voice-triage-escalate.test.ts +106 -0
- package/src/calls/call-controller.ts +30 -2
- package/src/calls/call-speech-output.ts +12 -4
- package/src/calls/call-transport.ts +18 -1
- package/src/calls/media-stream-output.ts +53 -5
- package/src/calls/media-stream-server.ts +12 -0
- package/src/calls/media-stream-stt-session.ts +34 -0
- package/src/calls/telephony-synthesis-language.ts +84 -0
- package/src/calls/tts-text-sanitizer.ts +12 -5
- package/src/calls/voice-session-bridge.ts +31 -3
- package/src/calls/voice-triage-escalate.ts +52 -5
- package/src/config/bundled-skills/phone-calls/references/CONFIG.md +15 -15
- package/src/config/loader.ts +5 -0
- package/src/config/schemas/__tests__/live-voice.test.ts +35 -0
- package/src/config/schemas/calls.ts +0 -4
- package/src/config/schemas/live-voice.ts +25 -0
- package/src/config/schemas/tts.ts +63 -0
- package/src/live-voice/__tests__/front-decision.test.ts +120 -0
- package/src/live-voice/__tests__/live-voice-events.test.ts +1 -1
- package/src/live-voice/__tests__/live-voice-integration.test.ts +5 -0
- package/src/live-voice/__tests__/live-voice-progress.test.ts +178 -11
- package/src/live-voice/__tests__/live-voice-stt.test.ts +304 -1
- package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +102 -2
- package/src/live-voice/__tests__/live-voice-tts.test.ts +148 -2
- package/src/live-voice/__tests__/live-voice-vad.test.ts +278 -1
- package/src/live-voice/__tests__/progress-phrases.test.ts +167 -0
- package/src/live-voice/front-decision.ts +50 -3
- package/src/live-voice/live-voice-session.ts +517 -45
- package/src/live-voice/live-voice-tts.ts +18 -2
- package/src/live-voice/progress-phrases.ts +105 -2
- package/src/providers/speech-to-text/deepgram-realtime.test.ts +283 -1
- package/src/providers/speech-to-text/deepgram-realtime.ts +117 -3
- package/src/providers/speech-to-text/provider-catalog.ts +38 -0
- package/src/runtime/__tests__/local-actor-identity-force-refresh.test.ts +104 -0
- package/src/runtime/assistant-event-hub.ts +23 -0
- package/src/runtime/local-actor-identity.ts +18 -5
- package/src/runtime/routes/__tests__/sse-actor-principal-heal.test.ts +165 -0
- package/src/runtime/routes/__tests__/surface-action-routes.test.ts +2 -0
- package/src/runtime/routes/events-routes.ts +17 -16
- package/src/runtime/routes/sse-actor-principal-heal.ts +113 -0
- package/src/stt/__tests__/language-metadata.test.ts +85 -0
- package/src/stt/__tests__/speech-energy.test.ts +79 -0
- package/src/stt/language-metadata.ts +65 -0
- package/src/stt/speech-energy.ts +115 -12
- package/src/stt/types.ts +16 -0
- package/src/tts/__tests__/provider-adapters.test.ts +147 -0
- package/src/tts/__tests__/speakable-segments.test.ts +470 -0
- package/src/tts/language-voices.ts +23 -0
- package/src/tts/providers/deepgram-provider.ts +3 -1
- package/src/tts/providers/elevenlabs-provider.ts +73 -1
- package/src/tts/providers/xai-provider.ts +28 -2
- package/src/tts/speakable-segments.ts +293 -23
- package/src/tts/synthesis-stream.ts +7 -0
- package/src/tts/types.ts +7 -0
- package/src/util/__tests__/language-subtag.test.ts +54 -0
- package/src/util/language-subtag.ts +43 -0
- package/src/util/unicode.ts +1 -1
|
@@ -68,8 +68,10 @@ export async function speakSystemPrompt(
|
|
|
68
68
|
});
|
|
69
69
|
|
|
70
70
|
if (!useSynthesizedPath || !provider) {
|
|
71
|
-
// Native path
|
|
72
|
-
|
|
71
|
+
// Native path: send text tokens through the transport. Marked as
|
|
72
|
+
// system copy so transports that re-synthesize tokens (media-stream)
|
|
73
|
+
// never attach the caller-language hint to these fixed English prompts.
|
|
74
|
+
relay.sendTextToken(text, true, { systemCopy: true });
|
|
73
75
|
return;
|
|
74
76
|
}
|
|
75
77
|
|
|
@@ -128,6 +130,11 @@ async function synthesizeAndPlay(
|
|
|
128
130
|
},
|
|
129
131
|
});
|
|
130
132
|
|
|
133
|
+
// No language hint: this path speaks fixed English copy (verification
|
|
134
|
+
// codes, guardian prompts), so a pin-based hint would tell an
|
|
135
|
+
// enforcing provider to render English text in the pinned language.
|
|
136
|
+
// Conversational synthesis (call-controller, media-stream-output)
|
|
137
|
+
// carries the hint instead.
|
|
131
138
|
const result = await synthesizeAndEmit({
|
|
132
139
|
provider,
|
|
133
140
|
text,
|
|
@@ -258,8 +265,9 @@ async function synthesizeAndPlay(
|
|
|
258
265
|
// Fallback: send text via native TTS so the caller still hears the message.
|
|
259
266
|
// sendTextToken with last:true includes the end-of-turn signal inherently.
|
|
260
267
|
// This fallback is only used for providers whose catalog entry allows
|
|
261
|
-
// native fallback.
|
|
262
|
-
|
|
268
|
+
// native fallback. Marked as system copy for the same reason as the
|
|
269
|
+
// native path above.
|
|
270
|
+
relay.sendTextToken(text, true, { systemCopy: true });
|
|
263
271
|
} finally {
|
|
264
272
|
sink?.finalize();
|
|
265
273
|
}
|
|
@@ -9,6 +9,19 @@
|
|
|
9
9
|
|
|
10
10
|
// ── Transport interface ──────────────────────────────────────────────
|
|
11
11
|
|
|
12
|
+
/** Options for {@link CallTransport.sendTextToken}. */
|
|
13
|
+
export interface SendTextTokenOptions {
|
|
14
|
+
/**
|
|
15
|
+
* The token is fixed, known-English system copy (error recovery, silence
|
|
16
|
+
* checks, duration warnings, deterministic prompts) rather than
|
|
17
|
+
* model-generated turn text. Transports that synthesize text themselves
|
|
18
|
+
* must not attach the caller-language hint to these tokens: a provider
|
|
19
|
+
* that enforces the hint would render the English words as though they
|
|
20
|
+
* were the caller's language. Model text keeps the hint.
|
|
21
|
+
*/
|
|
22
|
+
systemCopy?: boolean;
|
|
23
|
+
}
|
|
24
|
+
|
|
12
25
|
/**
|
|
13
26
|
* Minimal output surface that CallController uses to send speech,
|
|
14
27
|
* audio, and lifecycle signals to the caller.
|
|
@@ -18,7 +31,11 @@ export interface CallTransport {
|
|
|
18
31
|
* Send a text token for TTS playback. When `last` is true the
|
|
19
32
|
* transport should signal end-of-turn to the caller.
|
|
20
33
|
*/
|
|
21
|
-
sendTextToken(
|
|
34
|
+
sendTextToken(
|
|
35
|
+
token: string,
|
|
36
|
+
last: boolean,
|
|
37
|
+
opts?: SendTextTokenOptions,
|
|
38
|
+
): void;
|
|
22
39
|
|
|
23
40
|
/**
|
|
24
41
|
* Send a pre-synthesized audio URL for playback.
|
|
@@ -41,7 +41,7 @@ import { extractSpeakableSegments } from "../tts/speakable-segments.js";
|
|
|
41
41
|
import { synthesizeAndEmit } from "../tts/synthesis-stream.js";
|
|
42
42
|
import { getLogger } from "../util/logger.js";
|
|
43
43
|
import type { CallAudioFormat } from "./audio-store.js";
|
|
44
|
-
import type { CallTransport } from "./call-transport.js";
|
|
44
|
+
import type { CallTransport, SendTextTokenOptions } from "./call-transport.js";
|
|
45
45
|
import {
|
|
46
46
|
chunkMulawToBase64Frames,
|
|
47
47
|
MULAW_FRAME_SIZE,
|
|
@@ -54,6 +54,10 @@ import type {
|
|
|
54
54
|
MediaStreamSendMediaCommand,
|
|
55
55
|
} from "./media-stream-protocol.js";
|
|
56
56
|
import { resolveCallTtsProvider } from "./resolve-call-tts-provider.js";
|
|
57
|
+
import {
|
|
58
|
+
resolveTelephonyLanguageVoice,
|
|
59
|
+
resolveTelephonySynthesisLanguage,
|
|
60
|
+
} from "./telephony-synthesis-language.js";
|
|
57
61
|
|
|
58
62
|
const log = getLogger("media-stream-output");
|
|
59
63
|
|
|
@@ -289,7 +293,7 @@ export type MediaStreamOutputState = "connected" | "closed";
|
|
|
289
293
|
*/
|
|
290
294
|
type PlaybackItem =
|
|
291
295
|
| { type: "frames"; frames: string[] }
|
|
292
|
-
| { type: "synthesize"; text: string }
|
|
296
|
+
| { type: "synthesize"; text: string; systemCopy: boolean }
|
|
293
297
|
| { type: "fetch-url"; url: string }
|
|
294
298
|
| { type: "mark"; name: string };
|
|
295
299
|
|
|
@@ -338,6 +342,15 @@ export class MediaStreamOutput implements CallTransport {
|
|
|
338
342
|
*/
|
|
339
343
|
private audioStartCallback: (() => void) | null = null;
|
|
340
344
|
|
|
345
|
+
/**
|
|
346
|
+
* Resolves the language hint passed on synthesis requests. Defaults to
|
|
347
|
+
* the pin-based resolution; the media-stream server overrides it with
|
|
348
|
+
* a resolver that consults the STT session's detected dominant
|
|
349
|
+
* language.
|
|
350
|
+
*/
|
|
351
|
+
private resolveSynthesisLanguage: () => string | undefined = () =>
|
|
352
|
+
resolveTelephonySynthesisLanguage();
|
|
353
|
+
|
|
341
354
|
/** Incremented per end-of-turn mark enqueued. */
|
|
342
355
|
private enqueuedEndOfTurnSeq = 0;
|
|
343
356
|
|
|
@@ -373,8 +386,16 @@ export class MediaStreamOutput implements CallTransport {
|
|
|
373
386
|
* An empty token with `last: true` signals end-of-turn without TTS:
|
|
374
387
|
* a mark is sent so the session transitions from "assistant speaking"
|
|
375
388
|
* to "caller speaking".
|
|
389
|
+
*
|
|
390
|
+
* `opts.systemCopy` marks fixed English system copy: its segments
|
|
391
|
+
* synthesize without the caller-language hint (see
|
|
392
|
+
* {@link processSynthesizeItem}).
|
|
376
393
|
*/
|
|
377
|
-
sendTextToken(
|
|
394
|
+
sendTextToken(
|
|
395
|
+
token: string,
|
|
396
|
+
last: boolean,
|
|
397
|
+
opts?: SendTextTokenOptions,
|
|
398
|
+
): void {
|
|
378
399
|
if (this.state === "closed") {
|
|
379
400
|
return;
|
|
380
401
|
}
|
|
@@ -388,7 +409,11 @@ export class MediaStreamOutput implements CallTransport {
|
|
|
388
409
|
);
|
|
389
410
|
this.textBuffer = remainder;
|
|
390
411
|
for (const segment of segments) {
|
|
391
|
-
this.enqueuePlayback({
|
|
412
|
+
this.enqueuePlayback({
|
|
413
|
+
type: "synthesize",
|
|
414
|
+
text: segment,
|
|
415
|
+
systemCopy: opts?.systemCopy === true,
|
|
416
|
+
});
|
|
392
417
|
this.turnSegmentEnqueued = true;
|
|
393
418
|
}
|
|
394
419
|
|
|
@@ -427,6 +452,14 @@ export class MediaStreamOutput implements CallTransport {
|
|
|
427
452
|
this.audioStartCallback = cb;
|
|
428
453
|
}
|
|
429
454
|
|
|
455
|
+
/**
|
|
456
|
+
* Override the synthesis-language resolver (see
|
|
457
|
+
* {@link resolveSynthesisLanguage}).
|
|
458
|
+
*/
|
|
459
|
+
setSynthesisLanguageResolver(resolver: () => string | undefined): void {
|
|
460
|
+
this.resolveSynthesisLanguage = resolver;
|
|
461
|
+
}
|
|
462
|
+
|
|
430
463
|
/**
|
|
431
464
|
* Discard accumulated text that has not yet been queued for synthesis.
|
|
432
465
|
* The call controller invokes this when it aborts an in-flight turn so
|
|
@@ -778,7 +811,9 @@ export class MediaStreamOutput implements CallTransport {
|
|
|
778
811
|
break;
|
|
779
812
|
|
|
780
813
|
case "synthesize":
|
|
781
|
-
await this.processSynthesizeItem(item.text, version
|
|
814
|
+
await this.processSynthesizeItem(item.text, version, {
|
|
815
|
+
systemCopy: item.systemCopy,
|
|
816
|
+
});
|
|
782
817
|
break;
|
|
783
818
|
|
|
784
819
|
case "fetch-url":
|
|
@@ -827,10 +862,17 @@ export class MediaStreamOutput implements CallTransport {
|
|
|
827
862
|
* each streamed chunk becomes frames as it arrives — while other
|
|
828
863
|
* providers accumulate into the whole-buffer conversion path. Falls
|
|
829
864
|
* back to a silent frame if synthesis fails.
|
|
865
|
+
*
|
|
866
|
+
* `systemCopy` items are fixed English copy: they synthesize without a
|
|
867
|
+
* language hint (and therefore without a per-language voice override),
|
|
868
|
+
* so an enforcing provider never renders English text in the caller's
|
|
869
|
+
* language (same exemption as call-speech-output's synthesized path).
|
|
870
|
+
* Model text keeps the resolver's hint.
|
|
830
871
|
*/
|
|
831
872
|
private async processSynthesizeItem(
|
|
832
873
|
text: string,
|
|
833
874
|
version: number,
|
|
875
|
+
{ systemCopy }: { systemCopy: boolean },
|
|
834
876
|
): Promise<void> {
|
|
835
877
|
const abortController = new AbortController();
|
|
836
878
|
this.activePlaybackAbort = abortController;
|
|
@@ -876,12 +918,18 @@ export class MediaStreamOutput implements CallTransport {
|
|
|
876
918
|
// Providers that support it (e.g. ElevenLabs pcm_16000) will
|
|
877
919
|
// return raw PCM; others fall back to their default format and
|
|
878
920
|
// the content-type sniffing below handles the mismatch.
|
|
921
|
+
const language = systemCopy ? undefined : this.resolveSynthesisLanguage();
|
|
922
|
+
// A language-known segment may select the synthesizing provider's
|
|
923
|
+
// configured per-language voice; no entry keeps the provider default.
|
|
924
|
+
const voiceId = resolveTelephonyLanguageVoice(provider.id, language);
|
|
879
925
|
const result = await synthesizeAndEmit({
|
|
880
926
|
provider,
|
|
881
927
|
text,
|
|
882
928
|
useCase: "phone-call",
|
|
883
929
|
outputFormat: "pcm",
|
|
884
930
|
sampleRateHz: STREAMING_PCM_SAMPLE_RATE_HZ,
|
|
931
|
+
...(voiceId !== undefined ? { voiceId } : {}),
|
|
932
|
+
...(language !== undefined ? { language } : {}),
|
|
885
933
|
signal: abortController.signal,
|
|
886
934
|
isCurrent,
|
|
887
935
|
onChunk: (chunk) => {
|
|
@@ -81,6 +81,7 @@ import {
|
|
|
81
81
|
type MediaStreamSttSessionCallbacks,
|
|
82
82
|
type MediaStreamSttSessionConfig,
|
|
83
83
|
} from "./media-stream-stt-session.js";
|
|
84
|
+
import { resolveTelephonySynthesisLanguage } from "./telephony-synthesis-language.js";
|
|
84
85
|
import {
|
|
85
86
|
TRUST_UNAVAILABLE_DENY_MESSAGE,
|
|
86
87
|
unresolvedActorTrust,
|
|
@@ -192,6 +193,14 @@ export class MediaStreamCallSession {
|
|
|
192
193
|
/** Number of transcript finals produced (non-empty). */
|
|
193
194
|
private transcriptFinalsProduced = 0;
|
|
194
195
|
|
|
196
|
+
/**
|
|
197
|
+
* Synthesized speech follows the caller's latest detected language when
|
|
198
|
+
* the transcriber tags finals, falling back to the pin-based
|
|
199
|
+
* resolution. Shared by the output adapter and the call controller.
|
|
200
|
+
*/
|
|
201
|
+
private readonly resolveSynthesisLanguage = (): string | undefined =>
|
|
202
|
+
resolveTelephonySynthesisLanguage(this.sttSession.currentLanguage());
|
|
203
|
+
|
|
195
204
|
constructor(
|
|
196
205
|
ws: ServerWebSocket<unknown>,
|
|
197
206
|
callSessionId: string,
|
|
@@ -216,6 +225,8 @@ export class MediaStreamCallSession {
|
|
|
216
225
|
|
|
217
226
|
this.sttSession = new MediaStreamSttSession(sttConfig ?? {}, callbacks);
|
|
218
227
|
|
|
228
|
+
this.output.setSynthesisLanguageResolver(this.resolveSynthesisLanguage);
|
|
229
|
+
|
|
219
230
|
log.info({ callSessionId }, "Media stream call session created");
|
|
220
231
|
}
|
|
221
232
|
|
|
@@ -681,6 +692,7 @@ export class MediaStreamCallSession {
|
|
|
681
692
|
{
|
|
682
693
|
assistantId: result.assistantId,
|
|
683
694
|
trustContext: result.trustContext,
|
|
695
|
+
resolveSynthesisLanguage: this.resolveSynthesisLanguage,
|
|
684
696
|
},
|
|
685
697
|
);
|
|
686
698
|
this.controller = controller;
|
|
@@ -48,6 +48,7 @@ import type {
|
|
|
48
48
|
SttCallContextHints,
|
|
49
49
|
SttStreamServerEvent,
|
|
50
50
|
} from "../stt/types.js";
|
|
51
|
+
import { baseLanguageSubtag } from "../util/language-subtag.js";
|
|
51
52
|
import { getLogger } from "../util/logger.js";
|
|
52
53
|
import {
|
|
53
54
|
mulawToLinear,
|
|
@@ -220,6 +221,20 @@ export class MediaStreamSttSession {
|
|
|
220
221
|
/** Speech-bearing audio milliseconds since the last streaming final. */
|
|
221
222
|
private utteranceAudioMs = 0;
|
|
222
223
|
|
|
224
|
+
/**
|
|
225
|
+
* The dominant detected-language base subtag of the latest committed
|
|
226
|
+
* streaming final that carried language tags. Overwritten per
|
|
227
|
+
* utterance, matching live voice's per-utterance resolution, so a
|
|
228
|
+
* caller who switches languages retargets synthesis on their next
|
|
229
|
+
* utterance instead of having to outvote the session's history. Only
|
|
230
|
+
* finals that carry transcript text update it: silence finals can
|
|
231
|
+
* carry container-level tags describing no emitted words. Untagged
|
|
232
|
+
* finals keep the previous value; batch transcription reports no
|
|
233
|
+
* language tags, so it stays unset there and any streaming-detected
|
|
234
|
+
* value is cleared when the session settles on batch mode.
|
|
235
|
+
*/
|
|
236
|
+
private latestUtteranceLanguage: string | undefined;
|
|
237
|
+
|
|
223
238
|
constructor(
|
|
224
239
|
config: MediaStreamSttSessionConfig = {},
|
|
225
240
|
callbacks: MediaStreamSttSessionCallbacks = {},
|
|
@@ -304,6 +319,15 @@ export class MediaStreamSttSession {
|
|
|
304
319
|
return this.startupFramesDroppedCount;
|
|
305
320
|
}
|
|
306
321
|
|
|
322
|
+
/**
|
|
323
|
+
* The caller's detected language as of the latest tagged streaming
|
|
324
|
+
* final, as a lowercase base subtag. Undefined until a tagged final
|
|
325
|
+
* commits (batch mode, non-tagging providers, silence).
|
|
326
|
+
*/
|
|
327
|
+
currentLanguage(): string | undefined {
|
|
328
|
+
return this.latestUtteranceLanguage;
|
|
329
|
+
}
|
|
330
|
+
|
|
307
331
|
// ── Event handlers ─────────────────────────────────────────────────
|
|
308
332
|
|
|
309
333
|
private handleStart(event: MediaStreamStartEvent): void {
|
|
@@ -477,6 +501,10 @@ export class MediaStreamSttSession {
|
|
|
477
501
|
*/
|
|
478
502
|
private enterBatchMode(): void {
|
|
479
503
|
this.mode = "batch";
|
|
504
|
+
// Batch transcripts carry no language metadata, so a language detected
|
|
505
|
+
// while streaming would otherwise hint synthesis for the rest of the
|
|
506
|
+
// call. Clear it so the configured pin (or no hint) takes over.
|
|
507
|
+
this.latestUtteranceLanguage = undefined;
|
|
480
508
|
this.startupFrames = [];
|
|
481
509
|
this.capabilityPromise ??= resolveTelephonySttCapability();
|
|
482
510
|
|
|
@@ -515,6 +543,12 @@ export class MediaStreamSttSession {
|
|
|
515
543
|
this.utteranceAudioMs = 0;
|
|
516
544
|
const text = event.text.trim();
|
|
517
545
|
if (text.length > 0) {
|
|
546
|
+
// `languages` is dominance-ranked, so the first entry is the
|
|
547
|
+
// utterance's dominant tag.
|
|
548
|
+
const dominant = baseLanguageSubtag(event.languages?.[0]);
|
|
549
|
+
if (dominant !== undefined) {
|
|
550
|
+
this.latestUtteranceLanguage = dominant;
|
|
551
|
+
}
|
|
518
552
|
this.callbacks.onTranscriptFinal?.(text, durationMs);
|
|
519
553
|
}
|
|
520
554
|
return;
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Synthesis-language resolution for telephony TTS.
|
|
3
|
+
*
|
|
4
|
+
* The telephony call-control prompt instructs the model to speak the
|
|
5
|
+
* caller's language; this resolver produces the matching TTS hint so
|
|
6
|
+
* providers that can enforce a language render the reply in it (the hint
|
|
7
|
+
* is a no-op for providers that cannot). Resolution mirrors live voice's
|
|
8
|
+
* per-utterance turn language (live-voice-session.ts): the caller's
|
|
9
|
+
* latest STT-detected language when the transcriber tags finals, else a
|
|
10
|
+
* monolingual `services.stt.language` pin when the configured provider
|
|
11
|
+
* honors manual language selection, else undefined (no hint, provider
|
|
12
|
+
* default behavior).
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import { getConfig } from "../config/loader.js";
|
|
16
|
+
import type { AssistantConfig } from "../config/types.js";
|
|
17
|
+
import { pinnedListeningLanguage } from "../providers/speech-to-text/provider-catalog.js";
|
|
18
|
+
import { resolveLanguageVoiceOverride } from "../tts/language-voices.js";
|
|
19
|
+
import { resolveTtsConfig } from "../tts/tts-config-resolver.js";
|
|
20
|
+
import type { TtsProviderId } from "../tts/types.js";
|
|
21
|
+
import { baseLanguageSubtag } from "../util/language-subtag.js";
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Resolve the language hint for telephony synthesis as a lowercase base
|
|
25
|
+
* subtag, or undefined when no signal resolves.
|
|
26
|
+
*
|
|
27
|
+
* @param detectedLanguage - The caller language the STT session detected
|
|
28
|
+
* on its latest tagged utterance, when a session is reachable from the
|
|
29
|
+
* call site. Wins over the pin whenever it normalizes to a non-empty
|
|
30
|
+
* tag.
|
|
31
|
+
*/
|
|
32
|
+
export function resolveTelephonySynthesisLanguage(
|
|
33
|
+
detectedLanguage?: string,
|
|
34
|
+
): string | undefined {
|
|
35
|
+
const detected = baseLanguageSubtag(detectedLanguage);
|
|
36
|
+
if (detected !== undefined) {
|
|
37
|
+
return detected;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
let stt: { provider: string; language?: string };
|
|
41
|
+
try {
|
|
42
|
+
stt = getConfig().services.stt;
|
|
43
|
+
} catch {
|
|
44
|
+
// Config unavailable (early startup, minimal test workspaces): no hint.
|
|
45
|
+
return undefined;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
return pinnedListeningLanguage(stt.provider, stt.language);
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Resolve the per-language voice for a telephony synthesis request from
|
|
53
|
+
* the synthesizing provider's `services.tts.providers.<id>.languageVoices`
|
|
54
|
+
* map. Keyed by the provider actually synthesizing, which on the
|
|
55
|
+
* media-stream transport may be a playability fallback rather than the
|
|
56
|
+
* configured provider, because voice identifiers are provider-specific.
|
|
57
|
+
* Returns undefined (provider default voice) when the request carries no
|
|
58
|
+
* language, config is unavailable, or the provider's map has no entry.
|
|
59
|
+
*/
|
|
60
|
+
export function resolveTelephonyLanguageVoice(
|
|
61
|
+
providerId: string,
|
|
62
|
+
language: string | undefined,
|
|
63
|
+
): string | undefined {
|
|
64
|
+
if (language === undefined) {
|
|
65
|
+
return undefined;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
let config: AssistantConfig;
|
|
69
|
+
try {
|
|
70
|
+
config = getConfig();
|
|
71
|
+
} catch {
|
|
72
|
+
// Config unavailable (early startup, minimal test workspaces): no override.
|
|
73
|
+
return undefined;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const { providerConfig } = resolveTtsConfig(
|
|
77
|
+
config,
|
|
78
|
+
providerId as TtsProviderId,
|
|
79
|
+
);
|
|
80
|
+
return resolveLanguageVoiceOverride(
|
|
81
|
+
providerConfig.languageVoices as Record<string, string> | undefined,
|
|
82
|
+
language,
|
|
83
|
+
);
|
|
84
|
+
}
|
|
@@ -49,11 +49,18 @@ export function sanitizeForTts(text: string): string {
|
|
|
49
49
|
result = result.replace(/^[-*]\s+/gm, "");
|
|
50
50
|
|
|
51
51
|
// 7. Italic: *text* or _text_ → text
|
|
52
|
-
// Word-boundary-aware
|
|
53
|
-
//
|
|
54
|
-
//
|
|
55
|
-
|
|
56
|
-
result = result.replace(
|
|
52
|
+
// Word-boundary-aware (Unicode letters/digits, so café and 変数 count as
|
|
53
|
+
// word chars) with non-whitespace content edges, so arithmetic like
|
|
54
|
+
// `5 * 3 and 4 * 2` and identifiers like `my_var` survive. Mirrors the
|
|
55
|
+
// open-span rule in tts/speakable-segments.ts.
|
|
56
|
+
result = result.replace(
|
|
57
|
+
/(?<![\p{L}\p{N}_])\*([^\s*](?:[^*]*[^\s*])?)\*(?![\p{L}\p{N}_])/gu,
|
|
58
|
+
"$1",
|
|
59
|
+
);
|
|
60
|
+
result = result.replace(
|
|
61
|
+
/(?<![\p{L}\p{N}_])_([^\s_](?:[^_]*[^\s_])?)_(?![\p{L}\p{N}_])/gu,
|
|
62
|
+
"$1",
|
|
63
|
+
);
|
|
57
64
|
|
|
58
65
|
// 8. Emojis: strip extended pictographic characters, variation selectors,
|
|
59
66
|
// zero-width joiners, skin tone modifiers, and regional indicator symbols (flags).
|
|
@@ -34,6 +34,7 @@ import {
|
|
|
34
34
|
recordConversationPersistedSeq,
|
|
35
35
|
updateMessageContent,
|
|
36
36
|
} from "../persistence/conversation-crud.js";
|
|
37
|
+
import { pinnedListeningLanguage } from "../providers/speech-to-text/provider-catalog.js";
|
|
37
38
|
import type { ContentBlock } from "../providers/types.js";
|
|
38
39
|
import { broadcastMessage } from "../runtime/assistant-event-hub.js";
|
|
39
40
|
import { DAEMON_INTERNAL_ASSISTANT_ID } from "../runtime/assistant-scope.js";
|
|
@@ -431,6 +432,32 @@ const VOICE_APPROVAL_TIMEOUT_MS = 45_000;
|
|
|
431
432
|
export const VOICE_NO_SETUP_FLOWS_RULE =
|
|
432
433
|
"Never start account connections, OAuth or sign-in flows, or any other action that opens a browser window or needs the user's screen during this call — not even through shell or CLI tools. If the task needs one, say so briefly and offer to finish it in text chat after the call.";
|
|
433
434
|
|
|
435
|
+
/**
|
|
436
|
+
* The pre-speech tail of the speak-the-caller's-language rule. A monolingual
|
|
437
|
+
* `services.stt.language` pin is the strongest pre-speech signal of the
|
|
438
|
+
* caller's language (the transcriber is already listening in it, see
|
|
439
|
+
* media-stream-stt-session.ts and providers/speech-to-text/resolve.ts), so it
|
|
440
|
+
* outranks the English default; "multi" and unset mean auto-detect, where
|
|
441
|
+
* English remains the fallback. The pin only counts when the active provider
|
|
442
|
+
* honors manual language selection (see pinnedListeningLanguage):
|
|
443
|
+
* auto-detecting providers (gemini, whisper) ignore a persisted language
|
|
444
|
+
* entirely, so greeting in it would contradict what the transcriber
|
|
445
|
+
* actually hears. Exported for tests: the default test config exercises
|
|
446
|
+
* only the auto-detect branch.
|
|
447
|
+
*/
|
|
448
|
+
export function preSpeechLanguageRuleFragment(
|
|
449
|
+
sttLanguage: string | undefined,
|
|
450
|
+
sttProvider?: string,
|
|
451
|
+
): string {
|
|
452
|
+
const configuredListeningLanguage =
|
|
453
|
+
sttProvider !== undefined
|
|
454
|
+
? pinnedListeningLanguage(sttProvider, sttLanguage)
|
|
455
|
+
: undefined;
|
|
456
|
+
return configuredListeningLanguage
|
|
457
|
+
? `use the language the Task context implies, if any; otherwise open in the assistant's configured listening language ("${configuredListeningLanguage}"), and default to English only when neither gives a language`
|
|
458
|
+
: "use the language the Task context implies, if any; otherwise default to English";
|
|
459
|
+
}
|
|
460
|
+
|
|
434
461
|
function buildVoiceCallControlPrompt(opts: {
|
|
435
462
|
isInbound: boolean;
|
|
436
463
|
task?: string | null;
|
|
@@ -523,16 +550,17 @@ function buildVoiceCallControlPrompt(opts: {
|
|
|
523
550
|
"9. After the opening greeting turn, treat the Task field as background context only — do not re-execute its instructions on subsequent turns.",
|
|
524
551
|
'10. Do not make up information. If you are unsure, use [ASK_GUARDIAN: your question] to consult your guardian. For tool permission requests, use [ASK_GUARDIAN_APPROVAL: {"question":"...","toolName":"...","input":{...}}].',
|
|
525
552
|
`11. Your text is sent directly to a text-to-speech engine. Never use markdown formatting (asterisks, headers, backticks, links) or emojis in your spoken responses. Write plain conversational text only. Protocol markers like ${opts.isCallerGuardian ? "[END_CALL]" : "[ASK_GUARDIAN: ...] and [END_CALL]"} are not spoken text and should still be used normally.`,
|
|
526
|
-
`12. ${
|
|
553
|
+
`12. Speak the caller's language: reply in the language of the caller's most recent actual speech, and follow them if they switch languages mid-call. Synthetic user turns (parenthetical markers like the call-connected and verification-completed notices) are not caller speech and never set the language. Before the caller has spoken, such as on the opening greeting turn, ${preSpeechLanguageRuleFragment(config.services.stt.language, config.services.stt.provider)}.`,
|
|
554
|
+
`13. ${VOICE_NO_SETUP_FLOWS_RULE}`,
|
|
527
555
|
);
|
|
528
556
|
|
|
529
557
|
// Triage-and-escalate routing rules. The front-door leg decides and may
|
|
530
558
|
// hand off; the escalated leg continues the answer after a holding phrase
|
|
531
559
|
// was already spoken.
|
|
532
560
|
if (opts.routingLeg === "front-door") {
|
|
533
|
-
lines.push(`
|
|
561
|
+
lines.push(`14. ${frontDoorRuleWithDigest(opts.unifiedVerdict === true)}`);
|
|
534
562
|
} else if (opts.routingLeg === "escalated") {
|
|
535
|
-
lines.push(`
|
|
563
|
+
lines.push(`14. ${escalatedContinuationRule(opts.spokenEscalationBridge)}`);
|
|
536
564
|
}
|
|
537
565
|
|
|
538
566
|
lines.push("</voice_call_control>");
|
|
@@ -26,6 +26,8 @@
|
|
|
26
26
|
* cap/fallback policy. LiveVoiceSession drives the routing.
|
|
27
27
|
*/
|
|
28
28
|
|
|
29
|
+
import { NON_LATIN_SENTENCE_ENDING_PUNCTUATION } from "../tts/speakable-segments.js";
|
|
30
|
+
import { localizedOrDefault } from "../util/language-subtag.js";
|
|
29
31
|
import {
|
|
30
32
|
ESCALATE_VERDICT_TOKEN,
|
|
31
33
|
HOLD_VERDICT_TOKEN,
|
|
@@ -57,6 +59,41 @@ export type VoiceRoutingLeg = "front-door" | "escalated";
|
|
|
57
59
|
export const FALLBACK_ESCALATION_BRIDGE =
|
|
58
60
|
"Let me think about that for a second.";
|
|
59
61
|
|
|
62
|
+
/**
|
|
63
|
+
* Per-language spellings of the fallback escalation bridge, keyed by
|
|
64
|
+
* lowercased BCP 47 base subtag, covering the Deepgram code-switching roster
|
|
65
|
+
* (DEEPGRAM_MULTI_LANGUAGE_CODES in providers/speech-to-text/deepgram.ts).
|
|
66
|
+
* These are spoken audio; the English constant above also serves as the
|
|
67
|
+
* prompt exemplar and stays the default.
|
|
68
|
+
*/
|
|
69
|
+
export const FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE: Readonly<
|
|
70
|
+
Record<string, string>
|
|
71
|
+
> = {
|
|
72
|
+
en: FALLBACK_ESCALATION_BRIDGE,
|
|
73
|
+
es: "Déjame pensarlo un segundo.",
|
|
74
|
+
fr: "Laissez-moi y réfléchir un instant.",
|
|
75
|
+
de: "Lass mich kurz darüber nachdenken.",
|
|
76
|
+
hi: "मुझे एक पल सोचने दीजिए।",
|
|
77
|
+
ru: "Дайте мне секунду подумать.",
|
|
78
|
+
pt: "Deixe-me pensar nisso um segundo.",
|
|
79
|
+
ja: "少し考えさせてください。",
|
|
80
|
+
it: "Fammi pensare un attimo.",
|
|
81
|
+
nl: "Laat me daar even over nadenken.",
|
|
82
|
+
};
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* The fallback escalation bridge in the caller's language: selected by the
|
|
86
|
+
* lowercased base subtag of `language` (e.g. "pt-BR" -> "pt"), defaulting to
|
|
87
|
+
* {@link FALLBACK_ESCALATION_BRIDGE} for unknown or absent languages.
|
|
88
|
+
*/
|
|
89
|
+
export function fallbackEscalationBridgeFor(language?: string): string {
|
|
90
|
+
return localizedOrDefault(
|
|
91
|
+
FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
|
|
92
|
+
language,
|
|
93
|
+
FALLBACK_ESCALATION_BRIDGE,
|
|
94
|
+
);
|
|
95
|
+
}
|
|
96
|
+
|
|
60
97
|
/**
|
|
61
98
|
* Minimum length (after capping, trimmed) of the front-door leg's spoken
|
|
62
99
|
* bridge for it to count as a real bridge. Below this, the fallback bridge
|
|
@@ -71,8 +108,17 @@ export const MIN_SPOKEN_BRIDGE_CHARS = 3;
|
|
|
71
108
|
*/
|
|
72
109
|
export const MAX_ESCALATION_BRIDGE_CHARS = 140;
|
|
73
110
|
|
|
74
|
-
/**
|
|
75
|
-
|
|
111
|
+
/**
|
|
112
|
+
* Sentence terminators that end an escalation bridge: the segmenter's
|
|
113
|
+
* non-Latin ender roster (tts/speakable-segments.ts) plus the ASCII enders
|
|
114
|
+
* and the ellipsis. Built from the shared set so the rosters cannot
|
|
115
|
+
* diverge: without the non-Latin enders a Japanese or Hindi bridge never
|
|
116
|
+
* hits a terminator and buffers to the char cap before hand-off. Exported
|
|
117
|
+
* for tests that assert spoken phrases end in a recognized terminator.
|
|
118
|
+
*/
|
|
119
|
+
export const BRIDGE_SENTENCE_END_REGEX = new RegExp(
|
|
120
|
+
`[.!?…${[...NON_LATIN_SENTENCE_ENDING_PUNCTUATION].join("")}]`,
|
|
121
|
+
);
|
|
76
122
|
|
|
77
123
|
/**
|
|
78
124
|
* Normalize a raw post-`[1]` stream into the bridge that is actually
|
|
@@ -230,7 +276,7 @@ export function frontDoorDecisionRule(opts?: {
|
|
|
230
276
|
// replay dispatch all elapse with the thinking frame and ack
|
|
231
277
|
// deferred until commit, roughly tripling felt latency. A false
|
|
232
278
|
// release only answers a beat early, which barge-in absorbs.
|
|
233
|
-
`- If the caller's words are visibly unfinished
|
|
279
|
+
`- If the caller's words are visibly unfinished (a trailing conjunction, a dangling clause, a list still being dictated) output ONLY ${HOLD_VERDICT_TOKEN} and stop, no other text. Judge the words themselves: a complete question or statement means they are done, even when it is short or leans on earlier context ("What do you think?", "Why?", "And then?"). Callers may speak any language: those examples are English exemplars only, and completeness is judged by the grammar of the language being spoken. In verb-final languages such as Hindi, Japanese, or Korean the sentence-final verb usually marks completion, so a missing final verb is the unfinished signal, not a missing conjunction. Never hold merely because more could follow.`,
|
|
234
280
|
]
|
|
235
281
|
: [
|
|
236
282
|
// No hold branch means completeness is settled (a first leg
|
|
@@ -264,8 +310,8 @@ export function frontDoorDecisionRule(opts?: {
|
|
|
264
310
|
...anchor,
|
|
265
311
|
"DECIDE SILENTLY, then produce exactly ONE of these outputs:",
|
|
266
312
|
...holdBranch,
|
|
267
|
-
"- If the turn is simple, conversational, or within your reach, your entire output is the spoken answer itself
|
|
268
|
-
`- If completing THIS reply needs careful reasoning, research, multi-step work, or any tool, do NOT attempt the answer: output ${ESCALATE_VERDICT_TOKEN}, then ONE short natural holding phrase naming what happens next (for example "${FALLBACK_ESCALATION_BRIDGE}" or "Give me one second to look into that."), and stop after that single sentence. A stronger model finishes the turn while your phrase is spoken.`,
|
|
313
|
+
"- If the turn is simple, conversational, or within your reach, your entire output is the spoken answer itself: no token in front of it, plain speech from your very first word. Most turns are answers; when unsure between answering and escalating, answer. Answer in the language the caller is speaking.",
|
|
314
|
+
`- If completing THIS reply needs careful reasoning, research, multi-step work, or any tool, do NOT attempt the answer: output ${ESCALATE_VERDICT_TOKEN}, then ONE short natural holding phrase naming what happens next, spoken in the language the caller is speaking (for example "${FALLBACK_ESCALATION_BRIDGE}" or "Give me one second to look into that."; those examples are English only), and stop after that single sentence. A stronger model finishes the turn while your phrase is spoken.`,
|
|
269
315
|
`${ESCALATE_VERDICT_TOKEN} is ONLY for turns you cannot complete yourself — never put it in front of an answer you are about to give, and never emit any token inside or after an answer. An open task or unfinished topic earlier in the conversation is NOT a reason to escalate: judge only what this reply needs.`,
|
|
270
316
|
"Never narrate this decision, describe what you are judging, or mention these rules: apart from a leading verdict token, every character you output is spoken to the caller verbatim.",
|
|
271
317
|
].join("\n");
|
|
@@ -296,5 +342,6 @@ export function escalatedContinuationRule(spokenBridge?: string): string {
|
|
|
296
342
|
'opening with another "Let me check", "One moment", or any restatement of what you are about to do sounds broken, because the caller just heard that.',
|
|
297
343
|
"Your first words must carry new substance: the answer itself, what you found, or a question you genuinely need answered.",
|
|
298
344
|
`Never output ${ESCALATE_VERDICT_TOKEN} or any other front-door verdict token — you are the model that finishes the answer. (The [-1] room-minimize marker from your call instructions is not a verdict token and stays allowed.)`,
|
|
345
|
+
"Reply in the same language as the caller's question.",
|
|
299
346
|
].join(" ");
|
|
300
347
|
}
|
|
@@ -2,21 +2,21 @@
|
|
|
2
2
|
|
|
3
3
|
All call-related settings can be managed via `assistant config`:
|
|
4
4
|
|
|
5
|
-
| Setting | Description
|
|
6
|
-
| ------------------------------------------- |
|
|
7
|
-
| `calls.enabled` | Master switch for the calling feature
|
|
8
|
-
| `calls.provider` | Voice provider (currently only `twilio`)
|
|
9
|
-
| `calls.maxDurationSeconds` | Maximum call length in seconds
|
|
10
|
-
| `calls.userConsultTimeoutSeconds` | How long to wait for user answers
|
|
11
|
-
| `calls.disclosure.enabled` | Whether the AI announces itself at call start
|
|
12
|
-
| `calls.disclosure.text` | The disclosure message spoken at call start
|
|
13
|
-
| `llm.callSites.callAgent.model` | Override LLM model for call orchestration
|
|
14
|
-
| `calls.callerIdentity.allowPerCallOverride` | Allow per-call caller identity selection
|
|
15
|
-
| `calls.callerIdentity.userNumber` | E.164 phone number for user-number mode
|
|
16
|
-
| `
|
|
17
|
-
| `services.stt.provider` | STT provider for transcription and telephony. The assistant transcribes call audio itself over the Twilio media stream (streaming when the provider supports it, batch otherwise), so calls require a working API key for this provider.
|
|
18
|
-
| `services.tts.provider` | Active TTS provider for speech synthesis. Must be a provider ID from the catalog (`elevenlabs`, `fish-audio`, `deepgram`, `xai`). New providers can be added via the catalog without code changes to call routing.
|
|
19
|
-
| `services.tts.providers.<id>.*` | Provider-specific settings block. Each catalog provider has its own settings namespace under `services.tts.providers.<id>`. See voice settings in the desktop/iOS app or run `assistant config list` for available settings per provider.
|
|
5
|
+
| Setting | Description | Default |
|
|
6
|
+
| ------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- |
|
|
7
|
+
| `calls.enabled` | Master switch for the calling feature | `true` |
|
|
8
|
+
| `calls.provider` | Voice provider (currently only `twilio`) | `twilio` |
|
|
9
|
+
| `calls.maxDurationSeconds` | Maximum call length in seconds | `3600` (1 hour) |
|
|
10
|
+
| `calls.userConsultTimeoutSeconds` | How long to wait for user answers | `120` (2 min) |
|
|
11
|
+
| `calls.disclosure.enabled` | Whether the AI announces itself at call start | `true` |
|
|
12
|
+
| `calls.disclosure.text` | The disclosure message spoken at call start | `"At the very beginning of the call, introduce yourself as an assistant calling on behalf of my human."` |
|
|
13
|
+
| `llm.callSites.callAgent.model` | Override LLM model for call orchestration | _(unset — falls back to the resolved call-site default)_ |
|
|
14
|
+
| `calls.callerIdentity.allowPerCallOverride` | Allow per-call caller identity selection | `true` |
|
|
15
|
+
| `calls.callerIdentity.userNumber` | E.164 phone number for user-number mode | _(empty)_ |
|
|
16
|
+
| `services.stt.language` | Spoken language for transcription (BCP-47 code, or unset for the provider's multilingual default). Per-language TTS voices are configured via `services.tts.providers.<id>.languageVoices` (supported on the elevenlabs, deepgram, and vellum provider blocks). | _(unset)_ |
|
|
17
|
+
| `services.stt.provider` | STT provider for transcription and telephony. The assistant transcribes call audio itself over the Twilio media stream (streaming when the provider supports it, batch otherwise), so calls require a working API key for this provider. | `deepgram` |
|
|
18
|
+
| `services.tts.provider` | Active TTS provider for speech synthesis. Must be a provider ID from the catalog (`elevenlabs`, `fish-audio`, `deepgram`, `xai`). New providers can be added via the catalog without code changes to call routing. | `elevenlabs` |
|
|
19
|
+
| `services.tts.providers.<id>.*` | Provider-specific settings block. Each catalog provider has its own settings namespace under `services.tts.providers.<id>`. See voice settings in the desktop/iOS app or run `assistant config list` for available settings per provider. | _(per-provider defaults)_ |
|
|
20
20
|
|
|
21
21
|
## TTS provider call-path behavior
|
|
22
22
|
|
package/src/config/loader.ts
CHANGED
|
@@ -767,6 +767,11 @@ const DEPRECATED_FIELDS: Record<string, string> = {
|
|
|
767
767
|
"daemon.reapOrphanedSubprocesses has been removed. The daemon now reaps " +
|
|
768
768
|
"orphaned subprocesses automatically whenever it runs as PID 1 on Linux. " +
|
|
769
769
|
"The field will be removed from your config file.",
|
|
770
|
+
"calls.voice.language":
|
|
771
|
+
"calls.voice.language has been removed; it had no effect. Use " +
|
|
772
|
+
"services.stt.language for the spoken language and " +
|
|
773
|
+
"services.tts.providers.<id>.languageVoices for per-language voices. " +
|
|
774
|
+
"The field will be removed from your config file.",
|
|
770
775
|
};
|
|
771
776
|
|
|
772
777
|
/**
|