@vellumai/assistant 0.11.3-staging.2 → 0.11.3-staging.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. package/package.json +1 -1
  2. package/src/__tests__/call-controller.test.ts +120 -0
  3. package/src/__tests__/config-loader-backfill.test.ts +19 -0
  4. package/src/__tests__/config-schema.test.ts +142 -5
  5. package/src/__tests__/events-dev-bypass-actor.test.ts +112 -1
  6. package/src/__tests__/media-stream-output.test.ts +175 -0
  7. package/src/__tests__/media-stream-stt-session.test.ts +67 -0
  8. package/src/calls/__tests__/tts-text-sanitizer.test.ts +13 -0
  9. package/src/calls/__tests__/voice-session-bridge.test.ts +81 -12
  10. package/src/calls/__tests__/voice-triage-escalate.test.ts +106 -0
  11. package/src/calls/call-controller.ts +30 -2
  12. package/src/calls/call-speech-output.ts +12 -4
  13. package/src/calls/call-transport.ts +18 -1
  14. package/src/calls/media-stream-output.ts +53 -5
  15. package/src/calls/media-stream-server.ts +12 -0
  16. package/src/calls/media-stream-stt-session.ts +34 -0
  17. package/src/calls/telephony-synthesis-language.ts +84 -0
  18. package/src/calls/tts-text-sanitizer.ts +12 -5
  19. package/src/calls/voice-session-bridge.ts +31 -3
  20. package/src/calls/voice-triage-escalate.ts +52 -5
  21. package/src/config/bundled-skills/phone-calls/references/CONFIG.md +15 -15
  22. package/src/config/loader.ts +5 -0
  23. package/src/config/schemas/__tests__/live-voice.test.ts +35 -0
  24. package/src/config/schemas/calls.ts +0 -4
  25. package/src/config/schemas/live-voice.ts +25 -0
  26. package/src/config/schemas/tts.ts +63 -0
  27. package/src/live-voice/__tests__/front-decision.test.ts +120 -0
  28. package/src/live-voice/__tests__/live-voice-events.test.ts +1 -1
  29. package/src/live-voice/__tests__/live-voice-integration.test.ts +5 -0
  30. package/src/live-voice/__tests__/live-voice-progress.test.ts +178 -11
  31. package/src/live-voice/__tests__/live-voice-stt.test.ts +304 -1
  32. package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +102 -2
  33. package/src/live-voice/__tests__/live-voice-tts.test.ts +148 -2
  34. package/src/live-voice/__tests__/live-voice-vad.test.ts +278 -1
  35. package/src/live-voice/__tests__/progress-phrases.test.ts +167 -0
  36. package/src/live-voice/front-decision.ts +50 -3
  37. package/src/live-voice/live-voice-session.ts +517 -45
  38. package/src/live-voice/live-voice-tts.ts +18 -2
  39. package/src/live-voice/progress-phrases.ts +105 -2
  40. package/src/providers/speech-to-text/deepgram-realtime.test.ts +283 -1
  41. package/src/providers/speech-to-text/deepgram-realtime.ts +117 -3
  42. package/src/providers/speech-to-text/provider-catalog.ts +38 -0
  43. package/src/runtime/__tests__/local-actor-identity-force-refresh.test.ts +104 -0
  44. package/src/runtime/assistant-event-hub.ts +23 -0
  45. package/src/runtime/local-actor-identity.ts +18 -5
  46. package/src/runtime/routes/__tests__/sse-actor-principal-heal.test.ts +165 -0
  47. package/src/runtime/routes/__tests__/surface-action-routes.test.ts +2 -0
  48. package/src/runtime/routes/events-routes.ts +17 -16
  49. package/src/runtime/routes/sse-actor-principal-heal.ts +113 -0
  50. package/src/stt/__tests__/language-metadata.test.ts +85 -0
  51. package/src/stt/__tests__/speech-energy.test.ts +79 -0
  52. package/src/stt/language-metadata.ts +65 -0
  53. package/src/stt/speech-energy.ts +115 -12
  54. package/src/stt/types.ts +16 -0
  55. package/src/tts/__tests__/provider-adapters.test.ts +147 -0
  56. package/src/tts/__tests__/speakable-segments.test.ts +470 -0
  57. package/src/tts/language-voices.ts +23 -0
  58. package/src/tts/providers/deepgram-provider.ts +3 -1
  59. package/src/tts/providers/elevenlabs-provider.ts +73 -1
  60. package/src/tts/providers/xai-provider.ts +28 -2
  61. package/src/tts/speakable-segments.ts +293 -23
  62. package/src/tts/synthesis-stream.ts +7 -0
  63. package/src/tts/types.ts +7 -0
  64. package/src/util/__tests__/language-subtag.test.ts +54 -0
  65. package/src/util/language-subtag.ts +43 -0
  66. package/src/util/unicode.ts +1 -1
@@ -68,8 +68,10 @@ export async function speakSystemPrompt(
68
68
  });
69
69
 
70
70
  if (!useSynthesizedPath || !provider) {
71
- // Native path — send text tokens through the transport.
72
- relay.sendTextToken(text, true);
71
+ // Native path: send text tokens through the transport. Marked as
72
+ // system copy so transports that re-synthesize tokens (media-stream)
73
+ // never attach the caller-language hint to these fixed English prompts.
74
+ relay.sendTextToken(text, true, { systemCopy: true });
73
75
  return;
74
76
  }
75
77
 
@@ -128,6 +130,11 @@ async function synthesizeAndPlay(
128
130
  },
129
131
  });
130
132
 
133
+ // No language hint: this path speaks fixed English copy (verification
134
+ // codes, guardian prompts), so a pin-based hint would tell an
135
+ // enforcing provider to render English text in the pinned language.
136
+ // Conversational synthesis (call-controller, media-stream-output)
137
+ // carries the hint instead.
131
138
  const result = await synthesizeAndEmit({
132
139
  provider,
133
140
  text,
@@ -258,8 +265,9 @@ async function synthesizeAndPlay(
258
265
  // Fallback: send text via native TTS so the caller still hears the message.
259
266
  // sendTextToken with last:true includes the end-of-turn signal inherently.
260
267
  // This fallback is only used for providers whose catalog entry allows
261
- // native fallback.
262
- relay.sendTextToken(text, true);
268
+ // native fallback. Marked as system copy for the same reason as the
269
+ // native path above.
270
+ relay.sendTextToken(text, true, { systemCopy: true });
263
271
  } finally {
264
272
  sink?.finalize();
265
273
  }
@@ -9,6 +9,19 @@
9
9
 
10
10
  // ── Transport interface ──────────────────────────────────────────────
11
11
 
12
+ /** Options for {@link CallTransport.sendTextToken}. */
13
+ export interface SendTextTokenOptions {
14
+ /**
15
+ * The token is fixed, known-English system copy (error recovery, silence
16
+ * checks, duration warnings, deterministic prompts) rather than
17
+ * model-generated turn text. Transports that synthesize text themselves
18
+ * must not attach the caller-language hint to these tokens: a provider
19
+ * that enforces the hint would render the English words as though they
20
+ * were the caller's language. Model text keeps the hint.
21
+ */
22
+ systemCopy?: boolean;
23
+ }
24
+
12
25
  /**
13
26
  * Minimal output surface that CallController uses to send speech,
14
27
  * audio, and lifecycle signals to the caller.
@@ -18,7 +31,11 @@ export interface CallTransport {
18
31
  * Send a text token for TTS playback. When `last` is true the
19
32
  * transport should signal end-of-turn to the caller.
20
33
  */
21
- sendTextToken(token: string, last: boolean): void;
34
+ sendTextToken(
35
+ token: string,
36
+ last: boolean,
37
+ opts?: SendTextTokenOptions,
38
+ ): void;
22
39
 
23
40
  /**
24
41
  * Send a pre-synthesized audio URL for playback.
@@ -41,7 +41,7 @@ import { extractSpeakableSegments } from "../tts/speakable-segments.js";
41
41
  import { synthesizeAndEmit } from "../tts/synthesis-stream.js";
42
42
  import { getLogger } from "../util/logger.js";
43
43
  import type { CallAudioFormat } from "./audio-store.js";
44
- import type { CallTransport } from "./call-transport.js";
44
+ import type { CallTransport, SendTextTokenOptions } from "./call-transport.js";
45
45
  import {
46
46
  chunkMulawToBase64Frames,
47
47
  MULAW_FRAME_SIZE,
@@ -54,6 +54,10 @@ import type {
54
54
  MediaStreamSendMediaCommand,
55
55
  } from "./media-stream-protocol.js";
56
56
  import { resolveCallTtsProvider } from "./resolve-call-tts-provider.js";
57
+ import {
58
+ resolveTelephonyLanguageVoice,
59
+ resolveTelephonySynthesisLanguage,
60
+ } from "./telephony-synthesis-language.js";
57
61
 
58
62
  const log = getLogger("media-stream-output");
59
63
 
@@ -289,7 +293,7 @@ export type MediaStreamOutputState = "connected" | "closed";
289
293
  */
290
294
  type PlaybackItem =
291
295
  | { type: "frames"; frames: string[] }
292
- | { type: "synthesize"; text: string }
296
+ | { type: "synthesize"; text: string; systemCopy: boolean }
293
297
  | { type: "fetch-url"; url: string }
294
298
  | { type: "mark"; name: string };
295
299
 
@@ -338,6 +342,15 @@ export class MediaStreamOutput implements CallTransport {
338
342
  */
339
343
  private audioStartCallback: (() => void) | null = null;
340
344
 
345
+ /**
346
+ * Resolves the language hint passed on synthesis requests. Defaults to
347
+ * the pin-based resolution; the media-stream server overrides it with
348
+ * a resolver that consults the STT session's detected dominant
349
+ * language.
350
+ */
351
+ private resolveSynthesisLanguage: () => string | undefined = () =>
352
+ resolveTelephonySynthesisLanguage();
353
+
341
354
  /** Incremented per end-of-turn mark enqueued. */
342
355
  private enqueuedEndOfTurnSeq = 0;
343
356
 
@@ -373,8 +386,16 @@ export class MediaStreamOutput implements CallTransport {
373
386
  * An empty token with `last: true` signals end-of-turn without TTS:
374
387
  * a mark is sent so the session transitions from "assistant speaking"
375
388
  * to "caller speaking".
389
+ *
390
+ * `opts.systemCopy` marks fixed English system copy: its segments
391
+ * synthesize without the caller-language hint (see
392
+ * {@link processSynthesizeItem}).
376
393
  */
377
- sendTextToken(token: string, last: boolean): void {
394
+ sendTextToken(
395
+ token: string,
396
+ last: boolean,
397
+ opts?: SendTextTokenOptions,
398
+ ): void {
378
399
  if (this.state === "closed") {
379
400
  return;
380
401
  }
@@ -388,7 +409,11 @@ export class MediaStreamOutput implements CallTransport {
388
409
  );
389
410
  this.textBuffer = remainder;
390
411
  for (const segment of segments) {
391
- this.enqueuePlayback({ type: "synthesize", text: segment });
412
+ this.enqueuePlayback({
413
+ type: "synthesize",
414
+ text: segment,
415
+ systemCopy: opts?.systemCopy === true,
416
+ });
392
417
  this.turnSegmentEnqueued = true;
393
418
  }
394
419
 
@@ -427,6 +452,14 @@ export class MediaStreamOutput implements CallTransport {
427
452
  this.audioStartCallback = cb;
428
453
  }
429
454
 
455
+ /**
456
+ * Override the synthesis-language resolver (see
457
+ * {@link resolveSynthesisLanguage}).
458
+ */
459
+ setSynthesisLanguageResolver(resolver: () => string | undefined): void {
460
+ this.resolveSynthesisLanguage = resolver;
461
+ }
462
+
430
463
  /**
431
464
  * Discard accumulated text that has not yet been queued for synthesis.
432
465
  * The call controller invokes this when it aborts an in-flight turn so
@@ -778,7 +811,9 @@ export class MediaStreamOutput implements CallTransport {
778
811
  break;
779
812
 
780
813
  case "synthesize":
781
- await this.processSynthesizeItem(item.text, version);
814
+ await this.processSynthesizeItem(item.text, version, {
815
+ systemCopy: item.systemCopy,
816
+ });
782
817
  break;
783
818
 
784
819
  case "fetch-url":
@@ -827,10 +862,17 @@ export class MediaStreamOutput implements CallTransport {
827
862
  * each streamed chunk becomes frames as it arrives — while other
828
863
  * providers accumulate into the whole-buffer conversion path. Falls
829
864
  * back to a silent frame if synthesis fails.
865
+ *
866
+ * `systemCopy` items are fixed English copy: they synthesize without a
867
+ * language hint (and therefore without a per-language voice override),
868
+ * so an enforcing provider never renders English text in the caller's
869
+ * language (same exemption as call-speech-output's synthesized path).
870
+ * Model text keeps the resolver's hint.
830
871
  */
831
872
  private async processSynthesizeItem(
832
873
  text: string,
833
874
  version: number,
875
+ { systemCopy }: { systemCopy: boolean },
834
876
  ): Promise<void> {
835
877
  const abortController = new AbortController();
836
878
  this.activePlaybackAbort = abortController;
@@ -876,12 +918,18 @@ export class MediaStreamOutput implements CallTransport {
876
918
  // Providers that support it (e.g. ElevenLabs pcm_16000) will
877
919
  // return raw PCM; others fall back to their default format and
878
920
  // the content-type sniffing below handles the mismatch.
921
+ const language = systemCopy ? undefined : this.resolveSynthesisLanguage();
922
+ // A language-known segment may select the synthesizing provider's
923
+ // configured per-language voice; no entry keeps the provider default.
924
+ const voiceId = resolveTelephonyLanguageVoice(provider.id, language);
879
925
  const result = await synthesizeAndEmit({
880
926
  provider,
881
927
  text,
882
928
  useCase: "phone-call",
883
929
  outputFormat: "pcm",
884
930
  sampleRateHz: STREAMING_PCM_SAMPLE_RATE_HZ,
931
+ ...(voiceId !== undefined ? { voiceId } : {}),
932
+ ...(language !== undefined ? { language } : {}),
885
933
  signal: abortController.signal,
886
934
  isCurrent,
887
935
  onChunk: (chunk) => {
@@ -81,6 +81,7 @@ import {
81
81
  type MediaStreamSttSessionCallbacks,
82
82
  type MediaStreamSttSessionConfig,
83
83
  } from "./media-stream-stt-session.js";
84
+ import { resolveTelephonySynthesisLanguage } from "./telephony-synthesis-language.js";
84
85
  import {
85
86
  TRUST_UNAVAILABLE_DENY_MESSAGE,
86
87
  unresolvedActorTrust,
@@ -192,6 +193,14 @@ export class MediaStreamCallSession {
192
193
  /** Number of transcript finals produced (non-empty). */
193
194
  private transcriptFinalsProduced = 0;
194
195
 
196
+ /**
197
+ * Synthesized speech follows the caller's latest detected language when
198
+ * the transcriber tags finals, falling back to the pin-based
199
+ * resolution. Shared by the output adapter and the call controller.
200
+ */
201
+ private readonly resolveSynthesisLanguage = (): string | undefined =>
202
+ resolveTelephonySynthesisLanguage(this.sttSession.currentLanguage());
203
+
195
204
  constructor(
196
205
  ws: ServerWebSocket<unknown>,
197
206
  callSessionId: string,
@@ -216,6 +225,8 @@ export class MediaStreamCallSession {
216
225
 
217
226
  this.sttSession = new MediaStreamSttSession(sttConfig ?? {}, callbacks);
218
227
 
228
+ this.output.setSynthesisLanguageResolver(this.resolveSynthesisLanguage);
229
+
219
230
  log.info({ callSessionId }, "Media stream call session created");
220
231
  }
221
232
 
@@ -681,6 +692,7 @@ export class MediaStreamCallSession {
681
692
  {
682
693
  assistantId: result.assistantId,
683
694
  trustContext: result.trustContext,
695
+ resolveSynthesisLanguage: this.resolveSynthesisLanguage,
684
696
  },
685
697
  );
686
698
  this.controller = controller;
@@ -48,6 +48,7 @@ import type {
48
48
  SttCallContextHints,
49
49
  SttStreamServerEvent,
50
50
  } from "../stt/types.js";
51
+ import { baseLanguageSubtag } from "../util/language-subtag.js";
51
52
  import { getLogger } from "../util/logger.js";
52
53
  import {
53
54
  mulawToLinear,
@@ -220,6 +221,20 @@ export class MediaStreamSttSession {
220
221
  /** Speech-bearing audio milliseconds since the last streaming final. */
221
222
  private utteranceAudioMs = 0;
222
223
 
224
+ /**
225
+ * The dominant detected-language base subtag of the latest committed
226
+ * streaming final that carried language tags. Overwritten per
227
+ * utterance, matching live voice's per-utterance resolution, so a
228
+ * caller who switches languages retargets synthesis on their next
229
+ * utterance instead of having to outvote the session's history. Only
230
+ * finals that carry transcript text update it: silence finals can
231
+ * carry container-level tags describing no emitted words. Untagged
232
+ * finals keep the previous value; batch transcription reports no
233
+ * language tags, so it stays unset there and any streaming-detected
234
+ * value is cleared when the session settles on batch mode.
235
+ */
236
+ private latestUtteranceLanguage: string | undefined;
237
+
223
238
  constructor(
224
239
  config: MediaStreamSttSessionConfig = {},
225
240
  callbacks: MediaStreamSttSessionCallbacks = {},
@@ -304,6 +319,15 @@ export class MediaStreamSttSession {
304
319
  return this.startupFramesDroppedCount;
305
320
  }
306
321
 
322
+ /**
323
+ * The caller's detected language as of the latest tagged streaming
324
+ * final, as a lowercase base subtag. Undefined until a tagged final
325
+ * commits (batch mode, non-tagging providers, silence).
326
+ */
327
+ currentLanguage(): string | undefined {
328
+ return this.latestUtteranceLanguage;
329
+ }
330
+
307
331
  // ── Event handlers ─────────────────────────────────────────────────
308
332
 
309
333
  private handleStart(event: MediaStreamStartEvent): void {
@@ -477,6 +501,10 @@ export class MediaStreamSttSession {
477
501
  */
478
502
  private enterBatchMode(): void {
479
503
  this.mode = "batch";
504
+ // Batch transcripts carry no language metadata, so a language detected
505
+ // while streaming would otherwise hint synthesis for the rest of the
506
+ // call. Clear it so the configured pin (or no hint) takes over.
507
+ this.latestUtteranceLanguage = undefined;
480
508
  this.startupFrames = [];
481
509
  this.capabilityPromise ??= resolveTelephonySttCapability();
482
510
 
@@ -515,6 +543,12 @@ export class MediaStreamSttSession {
515
543
  this.utteranceAudioMs = 0;
516
544
  const text = event.text.trim();
517
545
  if (text.length > 0) {
546
+ // `languages` is dominance-ranked, so the first entry is the
547
+ // utterance's dominant tag.
548
+ const dominant = baseLanguageSubtag(event.languages?.[0]);
549
+ if (dominant !== undefined) {
550
+ this.latestUtteranceLanguage = dominant;
551
+ }
518
552
  this.callbacks.onTranscriptFinal?.(text, durationMs);
519
553
  }
520
554
  return;
@@ -0,0 +1,84 @@
1
+ /**
2
+ * Synthesis-language resolution for telephony TTS.
3
+ *
4
+ * The telephony call-control prompt instructs the model to speak the
5
+ * caller's language; this resolver produces the matching TTS hint so
6
+ * providers that can enforce a language render the reply in it (the hint
7
+ * is a no-op for providers that cannot). Resolution mirrors live voice's
8
+ * per-utterance turn language (live-voice-session.ts): the caller's
9
+ * latest STT-detected language when the transcriber tags finals, else a
10
+ * monolingual `services.stt.language` pin when the configured provider
11
+ * honors manual language selection, else undefined (no hint, provider
12
+ * default behavior).
13
+ */
14
+
15
+ import { getConfig } from "../config/loader.js";
16
+ import type { AssistantConfig } from "../config/types.js";
17
+ import { pinnedListeningLanguage } from "../providers/speech-to-text/provider-catalog.js";
18
+ import { resolveLanguageVoiceOverride } from "../tts/language-voices.js";
19
+ import { resolveTtsConfig } from "../tts/tts-config-resolver.js";
20
+ import type { TtsProviderId } from "../tts/types.js";
21
+ import { baseLanguageSubtag } from "../util/language-subtag.js";
22
+
23
+ /**
24
+ * Resolve the language hint for telephony synthesis as a lowercase base
25
+ * subtag, or undefined when no signal resolves.
26
+ *
27
+ * @param detectedLanguage - The caller language the STT session detected
28
+ * on its latest tagged utterance, when a session is reachable from the
29
+ * call site. Wins over the pin whenever it normalizes to a non-empty
30
+ * tag.
31
+ */
32
+ export function resolveTelephonySynthesisLanguage(
33
+ detectedLanguage?: string,
34
+ ): string | undefined {
35
+ const detected = baseLanguageSubtag(detectedLanguage);
36
+ if (detected !== undefined) {
37
+ return detected;
38
+ }
39
+
40
+ let stt: { provider: string; language?: string };
41
+ try {
42
+ stt = getConfig().services.stt;
43
+ } catch {
44
+ // Config unavailable (early startup, minimal test workspaces): no hint.
45
+ return undefined;
46
+ }
47
+
48
+ return pinnedListeningLanguage(stt.provider, stt.language);
49
+ }
50
+
51
+ /**
52
+ * Resolve the per-language voice for a telephony synthesis request from
53
+ * the synthesizing provider's `services.tts.providers.<id>.languageVoices`
54
+ * map. Keyed by the provider actually synthesizing, which on the
55
+ * media-stream transport may be a playability fallback rather than the
56
+ * configured provider, because voice identifiers are provider-specific.
57
+ * Returns undefined (provider default voice) when the request carries no
58
+ * language, config is unavailable, or the provider's map has no entry.
59
+ */
60
+ export function resolveTelephonyLanguageVoice(
61
+ providerId: string,
62
+ language: string | undefined,
63
+ ): string | undefined {
64
+ if (language === undefined) {
65
+ return undefined;
66
+ }
67
+
68
+ let config: AssistantConfig;
69
+ try {
70
+ config = getConfig();
71
+ } catch {
72
+ // Config unavailable (early startup, minimal test workspaces): no override.
73
+ return undefined;
74
+ }
75
+
76
+ const { providerConfig } = resolveTtsConfig(
77
+ config,
78
+ providerId as TtsProviderId,
79
+ );
80
+ return resolveLanguageVoiceOverride(
81
+ providerConfig.languageVoices as Record<string, string> | undefined,
82
+ language,
83
+ );
84
+ }
@@ -49,11 +49,18 @@ export function sanitizeForTts(text: string): string {
49
49
  result = result.replace(/^[-*]\s+/gm, "");
50
50
 
51
51
  // 7. Italic: *text* or _text_ → text
52
- // Word-boundary-aware with non-whitespace content edges, so arithmetic
53
- // like `5 * 3 and 4 * 2` and identifiers like `my_var` survive. Mirrors
54
- // the open-span rule in tts/speakable-segments.ts.
55
- result = result.replace(/(?<!\w)\*([^\s*](?:[^*]*[^\s*])?)\*(?!\w)/g, "$1");
56
- result = result.replace(/(?<!\w)_([^\s_](?:[^_]*[^\s_])?)_(?!\w)/g, "$1");
52
+ // Word-boundary-aware (Unicode letters/digits, so café and 変数 count as
53
+ // word chars) with non-whitespace content edges, so arithmetic like
54
+ // `5 * 3 and 4 * 2` and identifiers like `my_var` survive. Mirrors the
55
+ // open-span rule in tts/speakable-segments.ts.
56
+ result = result.replace(
57
+ /(?<![\p{L}\p{N}_])\*([^\s*](?:[^*]*[^\s*])?)\*(?![\p{L}\p{N}_])/gu,
58
+ "$1",
59
+ );
60
+ result = result.replace(
61
+ /(?<![\p{L}\p{N}_])_([^\s_](?:[^_]*[^\s_])?)_(?![\p{L}\p{N}_])/gu,
62
+ "$1",
63
+ );
57
64
 
58
65
  // 8. Emojis: strip extended pictographic characters, variation selectors,
59
66
  // zero-width joiners, skin tone modifiers, and regional indicator symbols (flags).
@@ -34,6 +34,7 @@ import {
34
34
  recordConversationPersistedSeq,
35
35
  updateMessageContent,
36
36
  } from "../persistence/conversation-crud.js";
37
+ import { pinnedListeningLanguage } from "../providers/speech-to-text/provider-catalog.js";
37
38
  import type { ContentBlock } from "../providers/types.js";
38
39
  import { broadcastMessage } from "../runtime/assistant-event-hub.js";
39
40
  import { DAEMON_INTERNAL_ASSISTANT_ID } from "../runtime/assistant-scope.js";
@@ -431,6 +432,32 @@ const VOICE_APPROVAL_TIMEOUT_MS = 45_000;
431
432
  export const VOICE_NO_SETUP_FLOWS_RULE =
432
433
  "Never start account connections, OAuth or sign-in flows, or any other action that opens a browser window or needs the user's screen during this call — not even through shell or CLI tools. If the task needs one, say so briefly and offer to finish it in text chat after the call.";
433
434
 
435
+ /**
436
+ * The pre-speech tail of the speak-the-caller's-language rule. A monolingual
437
+ * `services.stt.language` pin is the strongest pre-speech signal of the
438
+ * caller's language (the transcriber is already listening in it, see
439
+ * media-stream-stt-session.ts and providers/speech-to-text/resolve.ts), so it
440
+ * outranks the English default; "multi" and unset mean auto-detect, where
441
+ * English remains the fallback. The pin only counts when the active provider
442
+ * honors manual language selection (see pinnedListeningLanguage):
443
+ * auto-detecting providers (gemini, whisper) ignore a persisted language
444
+ * entirely, so greeting in it would contradict what the transcriber
445
+ * actually hears. Exported for tests: the default test config exercises
446
+ * only the auto-detect branch.
447
+ */
448
+ export function preSpeechLanguageRuleFragment(
449
+ sttLanguage: string | undefined,
450
+ sttProvider?: string,
451
+ ): string {
452
+ const configuredListeningLanguage =
453
+ sttProvider !== undefined
454
+ ? pinnedListeningLanguage(sttProvider, sttLanguage)
455
+ : undefined;
456
+ return configuredListeningLanguage
457
+ ? `use the language the Task context implies, if any; otherwise open in the assistant's configured listening language ("${configuredListeningLanguage}"), and default to English only when neither gives a language`
458
+ : "use the language the Task context implies, if any; otherwise default to English";
459
+ }
460
+
434
461
  function buildVoiceCallControlPrompt(opts: {
435
462
  isInbound: boolean;
436
463
  task?: string | null;
@@ -523,16 +550,17 @@ function buildVoiceCallControlPrompt(opts: {
523
550
  "9. After the opening greeting turn, treat the Task field as background context only — do not re-execute its instructions on subsequent turns.",
524
551
  '10. Do not make up information. If you are unsure, use [ASK_GUARDIAN: your question] to consult your guardian. For tool permission requests, use [ASK_GUARDIAN_APPROVAL: {"question":"...","toolName":"...","input":{...}}].',
525
552
  `11. Your text is sent directly to a text-to-speech engine. Never use markdown formatting (asterisks, headers, backticks, links) or emojis in your spoken responses. Write plain conversational text only. Protocol markers like ${opts.isCallerGuardian ? "[END_CALL]" : "[ASK_GUARDIAN: ...] and [END_CALL]"} are not spoken text and should still be used normally.`,
526
- `12. ${VOICE_NO_SETUP_FLOWS_RULE}`,
553
+ `12. Speak the caller's language: reply in the language of the caller's most recent actual speech, and follow them if they switch languages mid-call. Synthetic user turns (parenthetical markers like the call-connected and verification-completed notices) are not caller speech and never set the language. Before the caller has spoken, such as on the opening greeting turn, ${preSpeechLanguageRuleFragment(config.services.stt.language, config.services.stt.provider)}.`,
554
+ `13. ${VOICE_NO_SETUP_FLOWS_RULE}`,
527
555
  );
528
556
 
529
557
  // Triage-and-escalate routing rules. The front-door leg decides and may
530
558
  // hand off; the escalated leg continues the answer after a holding phrase
531
559
  // was already spoken.
532
560
  if (opts.routingLeg === "front-door") {
533
- lines.push(`13. ${frontDoorRuleWithDigest(opts.unifiedVerdict === true)}`);
561
+ lines.push(`14. ${frontDoorRuleWithDigest(opts.unifiedVerdict === true)}`);
534
562
  } else if (opts.routingLeg === "escalated") {
535
- lines.push(`13. ${escalatedContinuationRule(opts.spokenEscalationBridge)}`);
563
+ lines.push(`14. ${escalatedContinuationRule(opts.spokenEscalationBridge)}`);
536
564
  }
537
565
 
538
566
  lines.push("</voice_call_control>");
@@ -26,6 +26,8 @@
26
26
  * cap/fallback policy. LiveVoiceSession drives the routing.
27
27
  */
28
28
 
29
+ import { NON_LATIN_SENTENCE_ENDING_PUNCTUATION } from "../tts/speakable-segments.js";
30
+ import { localizedOrDefault } from "../util/language-subtag.js";
29
31
  import {
30
32
  ESCALATE_VERDICT_TOKEN,
31
33
  HOLD_VERDICT_TOKEN,
@@ -57,6 +59,41 @@ export type VoiceRoutingLeg = "front-door" | "escalated";
57
59
  export const FALLBACK_ESCALATION_BRIDGE =
58
60
  "Let me think about that for a second.";
59
61
 
62
+ /**
63
+ * Per-language spellings of the fallback escalation bridge, keyed by
64
+ * lowercased BCP 47 base subtag, covering the Deepgram code-switching roster
65
+ * (DEEPGRAM_MULTI_LANGUAGE_CODES in providers/speech-to-text/deepgram.ts).
66
+ * These are spoken audio; the English constant above also serves as the
67
+ * prompt exemplar and stays the default.
68
+ */
69
+ export const FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE: Readonly<
70
+ Record<string, string>
71
+ > = {
72
+ en: FALLBACK_ESCALATION_BRIDGE,
73
+ es: "Déjame pensarlo un segundo.",
74
+ fr: "Laissez-moi y réfléchir un instant.",
75
+ de: "Lass mich kurz darüber nachdenken.",
76
+ hi: "मुझे एक पल सोचने दीजिए।",
77
+ ru: "Дайте мне секунду подумать.",
78
+ pt: "Deixe-me pensar nisso um segundo.",
79
+ ja: "少し考えさせてください。",
80
+ it: "Fammi pensare un attimo.",
81
+ nl: "Laat me daar even over nadenken.",
82
+ };
83
+
84
+ /**
85
+ * The fallback escalation bridge in the caller's language: selected by the
86
+ * lowercased base subtag of `language` (e.g. "pt-BR" -> "pt"), defaulting to
87
+ * {@link FALLBACK_ESCALATION_BRIDGE} for unknown or absent languages.
88
+ */
89
+ export function fallbackEscalationBridgeFor(language?: string): string {
90
+ return localizedOrDefault(
91
+ FALLBACK_ESCALATION_BRIDGE_BY_LANGUAGE,
92
+ language,
93
+ FALLBACK_ESCALATION_BRIDGE,
94
+ );
95
+ }
96
+
60
97
  /**
61
98
  * Minimum length (after capping, trimmed) of the front-door leg's spoken
62
99
  * bridge for it to count as a real bridge. Below this, the fallback bridge
@@ -71,8 +108,17 @@ export const MIN_SPOKEN_BRIDGE_CHARS = 3;
71
108
  */
72
109
  export const MAX_ESCALATION_BRIDGE_CHARS = 140;
73
110
 
74
- /** Sentence terminators that end an escalation bridge. */
75
- const BRIDGE_SENTENCE_END_REGEX = /[.!?…]/;
111
+ /**
112
+ * Sentence terminators that end an escalation bridge: the segmenter's
113
+ * non-Latin ender roster (tts/speakable-segments.ts) plus the ASCII enders
114
+ * and the ellipsis. Built from the shared set so the rosters cannot
115
+ * diverge: without the non-Latin enders a Japanese or Hindi bridge never
116
+ * hits a terminator and buffers to the char cap before hand-off. Exported
117
+ * for tests that assert spoken phrases end in a recognized terminator.
118
+ */
119
+ export const BRIDGE_SENTENCE_END_REGEX = new RegExp(
120
+ `[.!?…${[...NON_LATIN_SENTENCE_ENDING_PUNCTUATION].join("")}]`,
121
+ );
76
122
 
77
123
  /**
78
124
  * Normalize a raw post-`[1]` stream into the bridge that is actually
@@ -230,7 +276,7 @@ export function frontDoorDecisionRule(opts?: {
230
276
  // replay dispatch all elapse with the thinking frame and ack
231
277
  // deferred until commit, roughly tripling felt latency. A false
232
278
  // release only answers a beat early, which barge-in absorbs.
233
- `- If the caller's words are visibly unfinished — a trailing conjunction, a dangling clause, a list still being dictated — output ONLY ${HOLD_VERDICT_TOKEN} and stop, no other text. Judge the words themselves: a complete question or statement means they are done, even when it is short or leans on earlier context ("What do you think?", "Why?", "And then?"). Never hold merely because more could follow.`,
279
+ `- If the caller's words are visibly unfinished (a trailing conjunction, a dangling clause, a list still being dictated) output ONLY ${HOLD_VERDICT_TOKEN} and stop, no other text. Judge the words themselves: a complete question or statement means they are done, even when it is short or leans on earlier context ("What do you think?", "Why?", "And then?"). Callers may speak any language: those examples are English exemplars only, and completeness is judged by the grammar of the language being spoken. In verb-final languages such as Hindi, Japanese, or Korean the sentence-final verb usually marks completion, so a missing final verb is the unfinished signal, not a missing conjunction. Never hold merely because more could follow.`,
234
280
  ]
235
281
  : [
236
282
  // No hold branch means completeness is settled (a first leg
@@ -264,8 +310,8 @@ export function frontDoorDecisionRule(opts?: {
264
310
  ...anchor,
265
311
  "DECIDE SILENTLY, then produce exactly ONE of these outputs:",
266
312
  ...holdBranch,
267
- "- If the turn is simple, conversational, or within your reach, your entire output is the spoken answer itself — no token in front of it, plain speech from your very first word. Most turns are answers; when unsure between answering and escalating, answer.",
268
- `- If completing THIS reply needs careful reasoning, research, multi-step work, or any tool, do NOT attempt the answer: output ${ESCALATE_VERDICT_TOKEN}, then ONE short natural holding phrase naming what happens next (for example "${FALLBACK_ESCALATION_BRIDGE}" or "Give me one second to look into that."), and stop after that single sentence. A stronger model finishes the turn while your phrase is spoken.`,
313
+ "- If the turn is simple, conversational, or within your reach, your entire output is the spoken answer itself: no token in front of it, plain speech from your very first word. Most turns are answers; when unsure between answering and escalating, answer. Answer in the language the caller is speaking.",
314
+ `- If completing THIS reply needs careful reasoning, research, multi-step work, or any tool, do NOT attempt the answer: output ${ESCALATE_VERDICT_TOKEN}, then ONE short natural holding phrase naming what happens next, spoken in the language the caller is speaking (for example "${FALLBACK_ESCALATION_BRIDGE}" or "Give me one second to look into that."; those examples are English only), and stop after that single sentence. A stronger model finishes the turn while your phrase is spoken.`,
269
315
  `${ESCALATE_VERDICT_TOKEN} is ONLY for turns you cannot complete yourself — never put it in front of an answer you are about to give, and never emit any token inside or after an answer. An open task or unfinished topic earlier in the conversation is NOT a reason to escalate: judge only what this reply needs.`,
270
316
  "Never narrate this decision, describe what you are judging, or mention these rules: apart from a leading verdict token, every character you output is spoken to the caller verbatim.",
271
317
  ].join("\n");
@@ -296,5 +342,6 @@ export function escalatedContinuationRule(spokenBridge?: string): string {
296
342
  'opening with another "Let me check", "One moment", or any restatement of what you are about to do sounds broken, because the caller just heard that.',
297
343
  "Your first words must carry new substance: the answer itself, what you found, or a question you genuinely need answered.",
298
344
  `Never output ${ESCALATE_VERDICT_TOKEN} or any other front-door verdict token — you are the model that finishes the answer. (The [-1] room-minimize marker from your call instructions is not a verdict token and stays allowed.)`,
345
+ "Reply in the same language as the caller's question.",
299
346
  ].join(" ");
300
347
  }
@@ -2,21 +2,21 @@
2
2
 
3
3
  All call-related settings can be managed via `assistant config`:
4
4
 
5
- | Setting | Description | Default |
6
- | ------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- |
7
- | `calls.enabled` | Master switch for the calling feature | `true` |
8
- | `calls.provider` | Voice provider (currently only `twilio`) | `twilio` |
9
- | `calls.maxDurationSeconds` | Maximum call length in seconds | `3600` (1 hour) |
10
- | `calls.userConsultTimeoutSeconds` | How long to wait for user answers | `120` (2 min) |
11
- | `calls.disclosure.enabled` | Whether the AI announces itself at call start | `true` |
12
- | `calls.disclosure.text` | The disclosure message spoken at call start | `"At the very beginning of the call, introduce yourself as an assistant calling on behalf of my human."` |
13
- | `llm.callSites.callAgent.model` | Override LLM model for call orchestration | _(unset — falls back to the resolved call-site default)_ |
14
- | `calls.callerIdentity.allowPerCallOverride` | Allow per-call caller identity selection | `true` |
15
- | `calls.callerIdentity.userNumber` | E.164 phone number for user-number mode | _(empty)_ |
16
- | `calls.voice.language` | Language code for TTS and transcription | `en-US` |
17
- | `services.stt.provider` | STT provider for transcription and telephony. The assistant transcribes call audio itself over the Twilio media stream (streaming when the provider supports it, batch otherwise), so calls require a working API key for this provider. | `deepgram` |
18
- | `services.tts.provider` | Active TTS provider for speech synthesis. Must be a provider ID from the catalog (`elevenlabs`, `fish-audio`, `deepgram`, `xai`). New providers can be added via the catalog without code changes to call routing. | `elevenlabs` |
19
- | `services.tts.providers.<id>.*` | Provider-specific settings block. Each catalog provider has its own settings namespace under `services.tts.providers.<id>`. See voice settings in the desktop/iOS app or run `assistant config list` for available settings per provider. | _(per-provider defaults)_ |
5
+ | Setting | Description | Default |
6
+ | ------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- |
7
+ | `calls.enabled` | Master switch for the calling feature | `true` |
8
+ | `calls.provider` | Voice provider (currently only `twilio`) | `twilio` |
9
+ | `calls.maxDurationSeconds` | Maximum call length in seconds | `3600` (1 hour) |
10
+ | `calls.userConsultTimeoutSeconds` | How long to wait for user answers | `120` (2 min) |
11
+ | `calls.disclosure.enabled` | Whether the AI announces itself at call start | `true` |
12
+ | `calls.disclosure.text` | The disclosure message spoken at call start | `"At the very beginning of the call, introduce yourself as an assistant calling on behalf of my human."` |
13
+ | `llm.callSites.callAgent.model` | Override LLM model for call orchestration | _(unset — falls back to the resolved call-site default)_ |
14
+ | `calls.callerIdentity.allowPerCallOverride` | Allow per-call caller identity selection | `true` |
15
+ | `calls.callerIdentity.userNumber` | E.164 phone number for user-number mode | _(empty)_ |
16
+ | `services.stt.language` | Spoken language for transcription (BCP-47 code, or unset for the provider's multilingual default). Per-language TTS voices are configured via `services.tts.providers.<id>.languageVoices` (supported on the elevenlabs, deepgram, and vellum provider blocks). | _(unset)_ |
17
+ | `services.stt.provider` | STT provider for transcription and telephony. The assistant transcribes call audio itself over the Twilio media stream (streaming when the provider supports it, batch otherwise), so calls require a working API key for this provider. | `deepgram` |
18
+ | `services.tts.provider` | Active TTS provider for speech synthesis. Must be a provider ID from the catalog (`elevenlabs`, `fish-audio`, `deepgram`, `xai`). New providers can be added via the catalog without code changes to call routing. | `elevenlabs` |
19
+ | `services.tts.providers.<id>.*` | Provider-specific settings block. Each catalog provider has its own settings namespace under `services.tts.providers.<id>`. See voice settings in the desktop/iOS app or run `assistant config list` for available settings per provider. | _(per-provider defaults)_ |
20
20
 
21
21
  ## TTS provider call-path behavior
22
22
 
@@ -767,6 +767,11 @@ const DEPRECATED_FIELDS: Record<string, string> = {
767
767
  "daemon.reapOrphanedSubprocesses has been removed. The daemon now reaps " +
768
768
  "orphaned subprocesses automatically whenever it runs as PID 1 on Linux. " +
769
769
  "The field will be removed from your config file.",
770
+ "calls.voice.language":
771
+ "calls.voice.language has been removed; it had no effect. Use " +
772
+ "services.stt.language for the spoken language and " +
773
+ "services.tts.providers.<id>.languageVoices for per-language voices. " +
774
+ "The field will be removed from your config file.",
770
775
  };
771
776
 
772
777
  /**