@vellumai/assistant 0.11.3-staging.2 → 0.11.3-staging.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/package.json +1 -1
  2. package/src/__tests__/call-controller.test.ts +120 -0
  3. package/src/__tests__/config-loader-backfill.test.ts +19 -0
  4. package/src/__tests__/config-schema.test.ts +139 -5
  5. package/src/__tests__/events-dev-bypass-actor.test.ts +112 -1
  6. package/src/__tests__/media-stream-output.test.ts +175 -0
  7. package/src/__tests__/media-stream-stt-session.test.ts +67 -0
  8. package/src/calls/__tests__/tts-text-sanitizer.test.ts +13 -0
  9. package/src/calls/__tests__/voice-session-bridge.test.ts +81 -12
  10. package/src/calls/__tests__/voice-triage-escalate.test.ts +106 -0
  11. package/src/calls/call-controller.ts +30 -2
  12. package/src/calls/call-speech-output.ts +12 -4
  13. package/src/calls/call-transport.ts +18 -1
  14. package/src/calls/media-stream-output.ts +53 -5
  15. package/src/calls/media-stream-server.ts +12 -0
  16. package/src/calls/media-stream-stt-session.ts +34 -0
  17. package/src/calls/telephony-synthesis-language.ts +84 -0
  18. package/src/calls/tts-text-sanitizer.ts +12 -5
  19. package/src/calls/voice-session-bridge.ts +31 -3
  20. package/src/calls/voice-triage-escalate.ts +52 -5
  21. package/src/config/bundled-skills/phone-calls/references/CONFIG.md +15 -15
  22. package/src/config/loader.ts +5 -0
  23. package/src/config/schemas/calls.ts +0 -4
  24. package/src/config/schemas/tts.ts +63 -0
  25. package/src/live-voice/__tests__/front-decision.test.ts +120 -0
  26. package/src/live-voice/__tests__/live-voice-events.test.ts +1 -1
  27. package/src/live-voice/__tests__/live-voice-progress.test.ts +178 -11
  28. package/src/live-voice/__tests__/live-voice-stt.test.ts +304 -1
  29. package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +102 -2
  30. package/src/live-voice/__tests__/live-voice-tts.test.ts +148 -2
  31. package/src/live-voice/__tests__/progress-phrases.test.ts +167 -0
  32. package/src/live-voice/front-decision.ts +50 -3
  33. package/src/live-voice/live-voice-session.ts +202 -25
  34. package/src/live-voice/live-voice-tts.ts +18 -2
  35. package/src/live-voice/progress-phrases.ts +105 -2
  36. package/src/providers/speech-to-text/deepgram-realtime.test.ts +283 -1
  37. package/src/providers/speech-to-text/deepgram-realtime.ts +117 -3
  38. package/src/providers/speech-to-text/provider-catalog.ts +38 -0
  39. package/src/runtime/__tests__/local-actor-identity-force-refresh.test.ts +104 -0
  40. package/src/runtime/assistant-event-hub.ts +23 -0
  41. package/src/runtime/local-actor-identity.ts +18 -5
  42. package/src/runtime/routes/__tests__/sse-actor-principal-heal.test.ts +165 -0
  43. package/src/runtime/routes/__tests__/surface-action-routes.test.ts +2 -0
  44. package/src/runtime/routes/events-routes.ts +17 -16
  45. package/src/runtime/routes/sse-actor-principal-heal.ts +113 -0
  46. package/src/stt/__tests__/language-metadata.test.ts +85 -0
  47. package/src/stt/language-metadata.ts +65 -0
  48. package/src/stt/types.ts +16 -0
  49. package/src/tts/__tests__/provider-adapters.test.ts +147 -0
  50. package/src/tts/__tests__/speakable-segments.test.ts +470 -0
  51. package/src/tts/language-voices.ts +23 -0
  52. package/src/tts/providers/deepgram-provider.ts +3 -1
  53. package/src/tts/providers/elevenlabs-provider.ts +73 -1
  54. package/src/tts/providers/xai-provider.ts +28 -2
  55. package/src/tts/speakable-segments.ts +293 -23
  56. package/src/tts/synthesis-stream.ts +7 -0
  57. package/src/tts/types.ts +7 -0
  58. package/src/util/__tests__/language-subtag.test.ts +54 -0
  59. package/src/util/language-subtag.ts +43 -0
  60. package/src/util/unicode.ts +1 -1
@@ -102,7 +102,11 @@ function resultsFrame(
102
102
  is_final?: boolean;
103
103
  speech_final?: boolean;
104
104
  from_finalize?: boolean;
105
- words?: { word: string; speaker?: number }[];
105
+ words?: { word: string; speaker?: number; language?: string }[];
106
+ /** Container-level detected languages on the alternative. */
107
+ alternativeLanguages?: string[];
108
+ /** Container-level detected languages on the channel. */
109
+ channelLanguages?: string[];
106
110
  } = {},
107
111
  ): string {
108
112
  return JSON.stringify({
@@ -121,8 +125,14 @@ function resultsFrame(
121
125
  transcript,
122
126
  confidence: 0.95,
123
127
  ...(options.words ? { words: options.words } : {}),
128
+ ...(options.alternativeLanguages
129
+ ? { languages: options.alternativeLanguages }
130
+ : {}),
124
131
  },
125
132
  ],
133
+ ...(options.channelLanguages
134
+ ? { languages: options.channelLanguages }
135
+ : {}),
126
136
  },
127
137
  });
128
138
  }
@@ -542,6 +552,154 @@ describe("DeepgramRealtimeTranscriber", () => {
542
552
  (globalThis as Record<string, unknown>).WebSocket = origWs;
543
553
  });
544
554
 
555
+ // ─────────────────────────────────────────────────────────────────
556
+ // Language metadata (nova-3 multi code-switching)
557
+ // ─────────────────────────────────────────────────────────────────
558
+
559
+ describe("language metadata", () => {
560
+ test("ranks per-word language tags by dominance on final events", async () => {
561
+ const { events } = await startSession();
562
+
563
+ mockWs.simulateMessage(
564
+ resultsFrame("hello world hola", {
565
+ is_final: true,
566
+ words: [
567
+ { word: "hello", language: "en" },
568
+ { word: "world", language: "en" },
569
+ { word: "hola", language: "es" },
570
+ ],
571
+ }),
572
+ );
573
+
574
+ expect(events).toHaveLength(1);
575
+ expect(events[0]).toEqual({
576
+ type: "final",
577
+ text: "hello world hola",
578
+ confidence: 0.95,
579
+ languages: ["en", "es"],
580
+ });
581
+ });
582
+
583
+ test("omits the languages field entirely when no language metadata is present", async () => {
584
+ const { events } = await startSession();
585
+
586
+ mockWs.simulateMessage(
587
+ resultsFrame("hello world", {
588
+ is_final: true,
589
+ words: [{ word: "hello" }, { word: "world" }],
590
+ }),
591
+ );
592
+ mockWs.simulateMessage(resultsFrame("still typing", { is_final: false }));
593
+
594
+ expect(events).toHaveLength(2);
595
+ for (const event of events) {
596
+ // The keys must not exist at all, not just be undefined-valued.
597
+ expect("language" in event).toBe(false);
598
+ expect("languages" in event).toBe(false);
599
+ }
600
+ });
601
+
602
+ test("normalizes regional tags to their base subtag", async () => {
603
+ const { events } = await startSession();
604
+
605
+ mockWs.simulateMessage(
606
+ resultsFrame("hello", {
607
+ is_final: true,
608
+ words: [{ word: "hello", language: "en-US" }],
609
+ }),
610
+ );
611
+
612
+ expect(events[0]).toEqual({
613
+ type: "final",
614
+ text: "hello",
615
+ confidence: 0.95,
616
+ languages: ["en"],
617
+ });
618
+ });
619
+
620
+ test("partial events carry the fields when interim results are enabled", async () => {
621
+ const { events } = await startSession();
622
+
623
+ mockWs.simulateMessage(
624
+ resultsFrame("namaste hello", {
625
+ is_final: false,
626
+ words: [
627
+ { word: "namaste", language: "hi" },
628
+ { word: "hello", language: "en" },
629
+ { word: "there", language: "en" },
630
+ ],
631
+ }),
632
+ );
633
+
634
+ expect(events).toHaveLength(1);
635
+ expect(events[0]).toEqual({
636
+ type: "partial",
637
+ text: "namaste hello",
638
+ confidence: 0.95,
639
+ languages: ["en", "hi"],
640
+ });
641
+ });
642
+
643
+ test("falls back to the alternative-level languages array when words carry no tags", async () => {
644
+ const { events } = await startSession();
645
+
646
+ mockWs.simulateMessage(
647
+ resultsFrame("mixed speech", {
648
+ is_final: true,
649
+ words: [{ word: "mixed" }, { word: "speech" }],
650
+ alternativeLanguages: ["es-419", "en", "ES"],
651
+ }),
652
+ );
653
+
654
+ expect(events[0]).toEqual({
655
+ type: "final",
656
+ text: "mixed speech",
657
+ confidence: 0.95,
658
+ languages: ["es", "en"],
659
+ });
660
+ });
661
+
662
+ test("falls back to the channel-level languages array when the alternative has none", async () => {
663
+ const { events } = await startSession();
664
+
665
+ mockWs.simulateMessage(
666
+ resultsFrame("bonjour", {
667
+ is_final: true,
668
+ channelLanguages: ["fr", "en"],
669
+ }),
670
+ );
671
+
672
+ expect(events[0]).toEqual({
673
+ type: "final",
674
+ text: "bonjour",
675
+ confidence: 0.95,
676
+ languages: ["fr", "en"],
677
+ });
678
+ });
679
+
680
+ test("per-word tags take precedence over container arrays", async () => {
681
+ const { events } = await startSession();
682
+
683
+ mockWs.simulateMessage(
684
+ resultsFrame("hola amigo", {
685
+ is_final: true,
686
+ words: [
687
+ { word: "hola", language: "es" },
688
+ { word: "amigo", language: "es" },
689
+ ],
690
+ alternativeLanguages: ["en", "es"],
691
+ }),
692
+ );
693
+
694
+ expect(events[0]).toEqual({
695
+ type: "final",
696
+ text: "hola amigo",
697
+ confidence: 0.95,
698
+ languages: ["es"],
699
+ });
700
+ });
701
+ });
702
+
545
703
  // ─────────────────────────────────────────────────────────────────
546
704
  // Multi-event sequence
547
705
  // ─────────────────────────────────────────────────────────────────
@@ -671,6 +829,130 @@ describe("DeepgramRealtimeTranscriber", () => {
671
829
  "second utterance",
672
830
  ]);
673
831
  });
832
+
833
+ test("aggregated final ranks language tags across all withheld frames", async () => {
834
+ const { events } = await startSession({ utteranceBoundaryFinals: true });
835
+
836
+ // Raw per-word tags are accumulated, so "es" (2 words) outranks
837
+ // "en" (1 word) even though "en" arrived in the earlier frame.
838
+ mockWs.simulateMessage(
839
+ resultsFrame("hello", {
840
+ is_final: true,
841
+ words: [{ word: "hello", language: "en" }],
842
+ }),
843
+ );
844
+ mockWs.simulateMessage(
845
+ resultsFrame("hola amigo", {
846
+ is_final: true,
847
+ speech_final: true,
848
+ words: [
849
+ { word: "hola", language: "es" },
850
+ { word: "amigo", language: "es" },
851
+ ],
852
+ }),
853
+ );
854
+
855
+ const finals = events.filter((e) => e.type === "final");
856
+ expect(finals).toHaveLength(1);
857
+ expect(finals[0]).toEqual({
858
+ type: "final",
859
+ text: "hello hola amigo",
860
+ languages: ["es", "en"],
861
+ });
862
+ });
863
+
864
+ test("UtteranceEnd flush carries the accumulated language metadata", async () => {
865
+ const { events } = await startSession({ utteranceBoundaryFinals: true });
866
+
867
+ mockWs.simulateMessage(
868
+ resultsFrame("bonjour", {
869
+ is_final: true,
870
+ words: [{ word: "bonjour", language: "fr" }],
871
+ }),
872
+ );
873
+ mockWs.simulateMessage(
874
+ resultsFrame("hello there", {
875
+ is_final: true,
876
+ alternativeLanguages: ["en"],
877
+ }),
878
+ );
879
+ mockWs.simulateMessage(utteranceEndFrame());
880
+
881
+ const finals = events.filter((e) => e.type === "final");
882
+ expect(finals).toHaveLength(1);
883
+ expect(finals[0]).toEqual({
884
+ type: "final",
885
+ text: "bonjour hello there",
886
+ languages: ["fr", "en"],
887
+ });
888
+ });
889
+
890
+ test("aggregated final omits language fields when no withheld frame carried tags", async () => {
891
+ const { events } = await startSession({ utteranceBoundaryFinals: true });
892
+
893
+ mockWs.simulateMessage(resultsFrame("no tags", { is_final: true }));
894
+ mockWs.simulateMessage(
895
+ resultsFrame("at all", { is_final: true, speech_final: true }),
896
+ );
897
+
898
+ const finals = events.filter((e) => e.type === "final");
899
+ expect(finals).toHaveLength(1);
900
+ expect(finals[0]).toEqual({ type: "final", text: "no tags at all" });
901
+ expect("language" in finals[0]!).toBe(false);
902
+ expect("languages" in finals[0]!).toBe(false);
903
+ });
904
+
905
+ test("tags on empty-text frames do not leak into the aggregated final", async () => {
906
+ const { events } = await startSession({ utteranceBoundaryFinals: true });
907
+
908
+ // A silence segment may still carry container-level tags; only
909
+ // frames that contributed transcript text feed the metadata.
910
+ mockWs.simulateMessage(
911
+ resultsFrame("", { is_final: true, alternativeLanguages: ["fr"] }),
912
+ );
913
+ mockWs.simulateMessage(
914
+ resultsFrame("hello", {
915
+ is_final: true,
916
+ speech_final: true,
917
+ words: [{ word: "hello", language: "en" }],
918
+ }),
919
+ );
920
+
921
+ const finals = events.filter((e) => e.type === "final");
922
+ expect(finals).toHaveLength(1);
923
+ expect(finals[0]).toEqual({
924
+ type: "final",
925
+ text: "hello",
926
+ languages: ["en"],
927
+ });
928
+ });
929
+
930
+ test("language tags reset between aggregated utterances", async () => {
931
+ const { events } = await startSession({ utteranceBoundaryFinals: true });
932
+
933
+ mockWs.simulateMessage(
934
+ resultsFrame("hola", {
935
+ is_final: true,
936
+ speech_final: true,
937
+ words: [{ word: "hola", language: "es" }],
938
+ }),
939
+ );
940
+ mockWs.simulateMessage(
941
+ resultsFrame("hello", {
942
+ is_final: true,
943
+ speech_final: true,
944
+ words: [{ word: "hello", language: "en" }],
945
+ }),
946
+ );
947
+
948
+ const finals = events.filter((e) => e.type === "final");
949
+ expect(finals).toHaveLength(2);
950
+ expect(finals[1]).toEqual({
951
+ type: "final",
952
+ text: "hello",
953
+ languages: ["en"],
954
+ });
955
+ });
674
956
  });
675
957
 
676
958
  // ─────────────────────────────────────────────────────────────────
@@ -29,10 +29,12 @@
29
29
  * - All timers and listeners are cleaned up on close to prevent leaks.
30
30
  */
31
31
 
32
+ import { rankLanguages } from "../../stt/language-metadata.js";
32
33
  import type {
33
34
  StreamingTranscriber,
34
35
  SttStreamServerEvent,
35
36
  } from "../../stt/types.js";
37
+ import { baseLanguageSubtag } from "../../util/language-subtag.js";
36
38
  import { getLogger } from "../../util/logger.js";
37
39
 
38
40
  const log = getLogger("deepgram-realtime");
@@ -208,6 +210,11 @@ interface DeepgramStreamWord {
208
210
  confidence?: number;
209
211
  start?: number;
210
212
  end?: number;
213
+ /**
214
+ * BCP-47 tag of the language this word was spoken in. Present only on
215
+ * code-switching models (nova-3 with `language=multi`).
216
+ */
217
+ language?: string;
211
218
  }
212
219
 
213
220
  /**
@@ -225,11 +232,23 @@ interface DeepgramStreamAlternative {
225
232
  speaker?: number;
226
233
  /** Per-word speaker tags when diarization is enabled. */
227
234
  words?: DeepgramStreamWord[];
235
+ /**
236
+ * Detected languages for the chunk in dominance order. Emitted by
237
+ * code-switching models; the container varies by API version, so
238
+ * {@link DeepgramStreamChannel.languages} is checked as well.
239
+ */
240
+ languages?: string[];
228
241
  }
229
242
 
230
243
  /** A channel within a Deepgram streaming response. */
231
244
  interface DeepgramStreamChannel {
232
245
  alternatives?: DeepgramStreamAlternative[];
246
+ /**
247
+ * Detected languages for the chunk in dominance order. Alternate
248
+ * container for {@link DeepgramStreamAlternative.languages} on some
249
+ * API versions.
250
+ */
251
+ languages?: string[];
233
252
  }
234
253
 
235
254
  /**
@@ -343,6 +362,15 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
343
362
  */
344
363
  private pendingFinalSegments: string[] = [];
345
364
 
365
+ /**
366
+ * Raw detected-language tags for the withheld segments, accumulated
367
+ * alongside {@link pendingFinalSegments} and ranked into the event's
368
+ * `languages` when the utterance flushes. Cleared wherever the pending
369
+ * segments are cleared. Only populated when
370
+ * {@link utteranceBoundaryFinals} is enabled.
371
+ */
372
+ private pendingLanguageTags: string[] = [];
373
+
346
374
  /** The live WebSocket connection, set during start(). */
347
375
  private ws: WsLike | null = null;
348
376
 
@@ -748,6 +776,12 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
748
776
  * words — see {@link extractSpeakerLabel}. Confidence is taken from
749
777
  * the top alternative when present.
750
778
  *
779
+ * Code-switching models (nova-3 with `language=multi`) tag detected
780
+ * languages per word and per container. When present, these become the
781
+ * dominance-ranked `languages` field on the emitted events (see
782
+ * {@link extractLanguages}). The field is omitted when the frame
783
+ * carries no language metadata.
784
+ *
751
785
  * We emit:
752
786
  * - `partial` for `is_final: false` frames (if interim results enabled).
753
787
  * - `final` for `is_final: true` frames.
@@ -788,6 +822,7 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
788
822
  typeof alternative?.confidence === "number"
789
823
  ? alternative.confidence
790
824
  : undefined;
825
+ const languages = extractLanguages(frame.channel, alternative);
791
826
 
792
827
  if (frame.is_final) {
793
828
  if (this.utteranceBoundaryFinals) {
@@ -795,6 +830,16 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
795
830
  // Finalize flush is a forced boundary — flush what is pending.
796
831
  if (text.length > 0) {
797
832
  this.pendingFinalSegments.push(text);
833
+ // Collect language tags only from frames that contributed text
834
+ // so the flushed metadata stays aligned with the emitted
835
+ // transcript (empty frames may still carry tags, but they
836
+ // describe no emitted words). Raw per-word tags are preferred
837
+ // over the frame's ranked list so cross-frame frequency
838
+ // weighting survives until the flush ranks the whole utterance.
839
+ const wordTags = collectWordLanguageTags(alternative);
840
+ this.pendingLanguageTags.push(
841
+ ...(wordTags.length > 0 ? wordTags : languages),
842
+ );
798
843
  }
799
844
  if (frame.speech_final || fromFinalize) {
800
845
  this.flushPendingUtterance();
@@ -807,6 +852,7 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
807
852
  text,
808
853
  ...(speakerLabel !== undefined ? { speakerLabel } : {}),
809
854
  ...(confidence !== undefined ? { confidence } : {}),
855
+ ...(languages.length > 0 ? { languages } : {}),
810
856
  // Mark the finalize flush so consumers can attribute it to the
811
857
  // utterance that requested the flush rather than new speech.
812
858
  ...(fromFinalize ? { fromFinalize: true } : {}),
@@ -819,6 +865,7 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
819
865
  text,
820
866
  ...(speakerLabel !== undefined ? { speakerLabel } : {}),
821
867
  ...(confidence !== undefined ? { confidence } : {}),
868
+ ...(languages.length > 0 ? { languages } : {}),
822
869
  });
823
870
  }
824
871
 
@@ -917,16 +964,27 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
917
964
 
918
965
  /**
919
966
  * Emit a single aggregated `final` for the withheld `is_final` segments
920
- * of the current utterance. No-op when nothing is pending, so boundary
921
- * signals over silence emit nothing.
967
+ * of the current utterance, carrying the dominance-ranked detected
968
+ * languages accumulated alongside them (field omitted when no segment
969
+ * carried language metadata). No-op when nothing is pending, so
970
+ * boundary signals over silence emit nothing.
922
971
  */
923
972
  private flushPendingUtterance(): void {
924
973
  if (this.pendingFinalSegments.length === 0) {
974
+ // Tags accumulate only alongside text, but clear defensively so a
975
+ // future drift cannot leak one utterance's tags into the next.
976
+ this.pendingLanguageTags = [];
925
977
  return;
926
978
  }
927
979
  const text = this.pendingFinalSegments.join(" ");
980
+ const languages = rankLanguages(this.pendingLanguageTags);
928
981
  this.pendingFinalSegments = [];
929
- this.emitEvent({ type: "final", text });
982
+ this.pendingLanguageTags = [];
983
+ this.emitEvent({
984
+ type: "final",
985
+ text,
986
+ ...(languages.length > 0 ? { languages } : {}),
987
+ });
930
988
  }
931
989
 
932
990
  /**
@@ -1222,6 +1280,62 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
1222
1280
  * contract on {@link SttStreamServerPartialEvent} /
1223
1281
  * {@link SttStreamServerFinalEvent}.
1224
1282
  */
1283
+ /**
1284
+ * Derive the detected languages for a chunk, most dominant first.
1285
+ *
1286
+ * Code-switching models tag languages in two shapes:
1287
+ * 1. Per-word `language` tags on `alternatives[0].words[]`, the richest
1288
+ * signal; ranked by frequency via {@link rankLanguages} (ties broken
1289
+ * by first appearance).
1290
+ * 2. A container-level `languages` array in dominance order, attached to
1291
+ * the alternative or (on some API versions) the channel. Used as the
1292
+ * fallback when no word carries a tag; normalized and deduped with
1293
+ * the provider's order preserved.
1294
+ *
1295
+ * Returns `[]` when the frame carries no language metadata: callers omit
1296
+ * the event fields entirely so absence stays distinguishable from a
1297
+ * detected language.
1298
+ */
1299
+ function extractLanguages(
1300
+ channel: DeepgramStreamChannel | undefined,
1301
+ alternative: DeepgramStreamAlternative | undefined,
1302
+ ): string[] {
1303
+ const wordTags = collectWordLanguageTags(alternative);
1304
+ if (wordTags.length > 0) {
1305
+ return rankLanguages(wordTags);
1306
+ }
1307
+
1308
+ const container = Array.isArray(alternative?.languages)
1309
+ ? alternative.languages
1310
+ : Array.isArray(channel?.languages)
1311
+ ? channel.languages
1312
+ : [];
1313
+ const deduped = new Set(
1314
+ container
1315
+ .filter((tag): tag is string => typeof tag === "string")
1316
+ .flatMap((tag) => {
1317
+ const base = baseLanguageSubtag(tag);
1318
+ return base !== undefined ? [base] : [];
1319
+ }),
1320
+ );
1321
+ return [...deduped];
1322
+ }
1323
+
1324
+ /**
1325
+ * Collect the raw per-word `language` tags of a chunk, in word order and
1326
+ * without ranking or normalization. Used both for per-frame ranking in
1327
+ * {@link extractLanguages} and for cross-frame accumulation in
1328
+ * utterance-boundary mode, where ranking is deferred to the flush so
1329
+ * frequency weighting spans the whole utterance.
1330
+ */
1331
+ function collectWordLanguageTags(
1332
+ alternative: DeepgramStreamAlternative | undefined,
1333
+ ): string[] {
1334
+ return (alternative?.words ?? []).flatMap((word) =>
1335
+ typeof word.language === "string" ? [word.language] : [],
1336
+ );
1337
+ }
1338
+
1225
1339
  function extractSpeakerLabel(
1226
1340
  alternative: DeepgramStreamAlternative | undefined,
1227
1341
  ): string | undefined {
@@ -16,6 +16,7 @@ import type {
16
16
  SttProviderId,
17
17
  TelephonySttMode,
18
18
  } from "../../stt/types.js";
19
+ import { baseLanguageSubtag } from "../../util/language-subtag.js";
19
20
 
20
21
  // ---------------------------------------------------------------------------
21
22
  // Client display metadata
@@ -282,6 +283,43 @@ export function listProviderEntries(): readonly SttProviderEntry[] {
282
283
  return [...CATALOG.values()];
283
284
  }
284
285
 
286
+ /**
287
+ * A base-subtag regex over the pinned listening language. The pin is
288
+ * free-form workspace config, and it flows into prompt interpolation and
289
+ * per-language table lookups, so only a plausible ISO 639 base subtag
290
+ * passes; anything else (junk strings, prototype keys like "constructor")
291
+ * resolves as no pin.
292
+ */
293
+ const PINNED_LANGUAGE_SUBTAG_REGEX = /^[a-z]{2,3}$/;
294
+
295
+ /**
296
+ * The configured `services.stt.language` pin as the caller's listening
297
+ * language, or undefined when the pin carries no signal.
298
+ *
299
+ * A persisted pin only counts when the provider honors manual language
300
+ * selection: auto-detecting providers (gemini, whisper) ignore the setting
301
+ * entirely, so treating it as the caller's language would force every
302
+ * turn into a stale pin. "multi" and blank mean auto-detect (no pin), and
303
+ * the value must normalize to a plausible base subtag. Shared by the
304
+ * telephony pre-speech prompt rule (voice-session-bridge.ts), live
305
+ * voice's turn language (live-voice-session.ts), and telephony synthesis
306
+ * (telephony-synthesis-language.ts) so the gate cannot drift.
307
+ */
308
+ export function pinnedListeningLanguage(
309
+ provider: string,
310
+ configuredLanguage: string | undefined,
311
+ ): string | undefined {
312
+ const providerHonorsLanguagePin =
313
+ getProviderEntry(provider as SttProviderId)?.languageSelection === "manual";
314
+ if (!providerHonorsLanguagePin || configuredLanguage?.trim() === "multi") {
315
+ return undefined;
316
+ }
317
+ const base = baseLanguageSubtag(configuredLanguage);
318
+ return base !== undefined && PINNED_LANGUAGE_SUBTAG_REGEX.test(base)
319
+ ? base
320
+ : undefined;
321
+ }
322
+
285
323
  /**
286
324
  * Look up the credential-provider name for a given STT provider.
287
325
  *
@@ -0,0 +1,104 @@
1
+ /**
2
+ * Tests for the cache-bypass option on the local guardian principal lookup.
3
+ *
4
+ * The guardian-delivery reader caches a successful read that finds no binding,
5
+ * and gateway-side binding writes do not invalidate the daemon's cache. A
6
+ * caller polling for a binding it expects to appear (the SSE actor-principal
7
+ * heal) therefore has to force the read, or every attempt re-reads the same
8
+ * empty answer until the TTL lapses.
9
+ */
10
+ import { afterAll, beforeEach, describe, expect, mock, test } from "bun:test";
11
+
12
+ import type { GuardianDelivery } from "@vellumai/gateway-client";
13
+
14
+ let cachedResult: GuardianDelivery[] | null = null;
15
+ let freshResult: GuardianDelivery[] | null = null;
16
+ let cachedCalls = 0;
17
+ let freshCalls = 0;
18
+
19
+ mock.module("../../config/env.js", () => ({
20
+ isHttpAuthDisabled: () => true,
21
+ hasUngatedHttpAuthDisabled: () => false,
22
+ }));
23
+
24
+ mock.module("../../contacts/guardian-delivery-reader.js", () => ({
25
+ getGuardianDelivery: () => {
26
+ cachedCalls++;
27
+ return Promise.resolve(cachedResult);
28
+ },
29
+ getGuardianDeliveryFresh: () => {
30
+ freshCalls++;
31
+ return Promise.resolve(freshResult);
32
+ },
33
+ peekCachedGuardianDelivery: () => undefined,
34
+ guardianForChannel: (list: GuardianDelivery[]) => list[0],
35
+ }));
36
+
37
+ import {
38
+ findLocalGuardianPrincipalId,
39
+ resolveActorPrincipalIdForLocalGuardian,
40
+ } from "../local-actor-identity.js";
41
+
42
+ afterAll(() => {
43
+ mock.restore();
44
+ });
45
+
46
+ /** Minimal guardian row: only `principalId` is read here. */
47
+ function guardian(principalId: string): GuardianDelivery {
48
+ return { principalId } as unknown as GuardianDelivery;
49
+ }
50
+
51
+ describe("local guardian principal lookup — cache bypass", () => {
52
+ beforeEach(() => {
53
+ cachedCalls = 0;
54
+ freshCalls = 0;
55
+ cachedResult = [];
56
+ freshResult = [];
57
+ });
58
+
59
+ test("reads the cache by default", async () => {
60
+ cachedResult = [guardian("guardian-cached")];
61
+ freshResult = [guardian("guardian-fresh")];
62
+
63
+ expect(await findLocalGuardianPrincipalId()).toBe("guardian-cached");
64
+ expect(cachedCalls).toBe(1);
65
+ expect(freshCalls).toBe(0);
66
+ });
67
+
68
+ test("forceRefresh bypasses the cache and sees a binding the cache would miss", async () => {
69
+ // Cached read holds the empty result from before the binding existed.
70
+ cachedResult = [];
71
+ freshResult = [guardian("guardian-fresh")];
72
+
73
+ expect(await findLocalGuardianPrincipalId()).toBeUndefined();
74
+ expect(await findLocalGuardianPrincipalId({ forceRefresh: true })).toBe(
75
+ "guardian-fresh",
76
+ );
77
+ expect(freshCalls).toBe(1);
78
+ });
79
+
80
+ test("resolveActorPrincipalIdForLocalGuardian threads forceRefresh through", async () => {
81
+ cachedResult = [];
82
+ freshResult = [guardian("guardian-fresh")];
83
+
84
+ expect(
85
+ await resolveActorPrincipalIdForLocalGuardian("dev-bypass"),
86
+ ).toBeUndefined();
87
+ expect(
88
+ await resolveActorPrincipalIdForLocalGuardian("dev-bypass", {
89
+ forceRefresh: true,
90
+ }),
91
+ ).toBe("guardian-fresh");
92
+ expect(freshCalls).toBe(1);
93
+ });
94
+
95
+ test("a non-dev-bypass principal is passed through without any lookup", async () => {
96
+ expect(
97
+ await resolveActorPrincipalIdForLocalGuardian("actor-123", {
98
+ forceRefresh: true,
99
+ }),
100
+ ).toBe("actor-123");
101
+ expect(cachedCalls).toBe(0);
102
+ expect(freshCalls).toBe(0);
103
+ });
104
+ });