@vellumai/assistant 0.11.3-staging.2 → 0.11.3-staging.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. package/package.json +1 -1
  2. package/src/__tests__/call-controller.test.ts +120 -0
  3. package/src/__tests__/config-loader-backfill.test.ts +19 -0
  4. package/src/__tests__/config-schema.test.ts +142 -5
  5. package/src/__tests__/events-dev-bypass-actor.test.ts +112 -1
  6. package/src/__tests__/media-stream-output.test.ts +175 -0
  7. package/src/__tests__/media-stream-stt-session.test.ts +67 -0
  8. package/src/calls/__tests__/tts-text-sanitizer.test.ts +13 -0
  9. package/src/calls/__tests__/voice-session-bridge.test.ts +81 -12
  10. package/src/calls/__tests__/voice-triage-escalate.test.ts +106 -0
  11. package/src/calls/call-controller.ts +30 -2
  12. package/src/calls/call-speech-output.ts +12 -4
  13. package/src/calls/call-transport.ts +18 -1
  14. package/src/calls/media-stream-output.ts +53 -5
  15. package/src/calls/media-stream-server.ts +12 -0
  16. package/src/calls/media-stream-stt-session.ts +34 -0
  17. package/src/calls/telephony-synthesis-language.ts +84 -0
  18. package/src/calls/tts-text-sanitizer.ts +12 -5
  19. package/src/calls/voice-session-bridge.ts +31 -3
  20. package/src/calls/voice-triage-escalate.ts +52 -5
  21. package/src/config/bundled-skills/phone-calls/references/CONFIG.md +15 -15
  22. package/src/config/loader.ts +5 -0
  23. package/src/config/schemas/__tests__/live-voice.test.ts +35 -0
  24. package/src/config/schemas/calls.ts +0 -4
  25. package/src/config/schemas/live-voice.ts +25 -0
  26. package/src/config/schemas/tts.ts +63 -0
  27. package/src/live-voice/__tests__/front-decision.test.ts +120 -0
  28. package/src/live-voice/__tests__/live-voice-events.test.ts +1 -1
  29. package/src/live-voice/__tests__/live-voice-integration.test.ts +5 -0
  30. package/src/live-voice/__tests__/live-voice-progress.test.ts +178 -11
  31. package/src/live-voice/__tests__/live-voice-stt.test.ts +304 -1
  32. package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +102 -2
  33. package/src/live-voice/__tests__/live-voice-tts.test.ts +148 -2
  34. package/src/live-voice/__tests__/live-voice-vad.test.ts +278 -1
  35. package/src/live-voice/__tests__/progress-phrases.test.ts +167 -0
  36. package/src/live-voice/front-decision.ts +50 -3
  37. package/src/live-voice/live-voice-session.ts +517 -45
  38. package/src/live-voice/live-voice-tts.ts +18 -2
  39. package/src/live-voice/progress-phrases.ts +105 -2
  40. package/src/providers/speech-to-text/deepgram-realtime.test.ts +283 -1
  41. package/src/providers/speech-to-text/deepgram-realtime.ts +117 -3
  42. package/src/providers/speech-to-text/provider-catalog.ts +38 -0
  43. package/src/runtime/__tests__/local-actor-identity-force-refresh.test.ts +104 -0
  44. package/src/runtime/assistant-event-hub.ts +23 -0
  45. package/src/runtime/local-actor-identity.ts +18 -5
  46. package/src/runtime/routes/__tests__/sse-actor-principal-heal.test.ts +165 -0
  47. package/src/runtime/routes/__tests__/surface-action-routes.test.ts +2 -0
  48. package/src/runtime/routes/events-routes.ts +17 -16
  49. package/src/runtime/routes/sse-actor-principal-heal.ts +113 -0
  50. package/src/stt/__tests__/language-metadata.test.ts +85 -0
  51. package/src/stt/__tests__/speech-energy.test.ts +79 -0
  52. package/src/stt/language-metadata.ts +65 -0
  53. package/src/stt/speech-energy.ts +115 -12
  54. package/src/stt/types.ts +16 -0
  55. package/src/tts/__tests__/provider-adapters.test.ts +147 -0
  56. package/src/tts/__tests__/speakable-segments.test.ts +470 -0
  57. package/src/tts/language-voices.ts +23 -0
  58. package/src/tts/providers/deepgram-provider.ts +3 -1
  59. package/src/tts/providers/elevenlabs-provider.ts +73 -1
  60. package/src/tts/providers/xai-provider.ts +28 -2
  61. package/src/tts/speakable-segments.ts +293 -23
  62. package/src/tts/synthesis-stream.ts +7 -0
  63. package/src/tts/types.ts +7 -0
  64. package/src/util/__tests__/language-subtag.test.ts +54 -0
  65. package/src/util/language-subtag.ts +43 -0
  66. package/src/util/unicode.ts +1 -1
@@ -1,3 +1,4 @@
1
+ import { resolveLanguageVoiceOverride } from "../tts/language-voices.js";
1
2
  import { createPcmChunkAligner } from "../tts/pcm-chunk-aligner.js";
2
3
  import { getTtsProvider } from "../tts/provider-catalog.js";
3
4
  import { synthesizeAndEmit } from "../tts/synthesis-stream.js";
@@ -30,6 +31,7 @@ export interface LiveVoiceTtsOptions {
30
31
  useCase?: TtsUseCase;
31
32
  outputFormat?: TtsSynthesisRequest["outputFormat"];
32
33
  sampleRate?: number;
34
+ language?: string;
33
35
  config?: LiveVoiceTtsConfig;
34
36
  onAudioChunk: (chunk: LiveVoiceTtsAudioChunk) => void;
35
37
  }
@@ -72,11 +74,23 @@ interface ResolvedStreamingTtsProvider {
72
74
  providerConfig: Record<string, unknown>;
73
75
  }
74
76
 
77
+ export { resolveLanguageVoiceOverride };
78
+
75
79
  export async function streamLiveVoiceTtsAudio(
76
80
  options: LiveVoiceTtsOptions,
77
81
  ): Promise<LiveVoiceTtsResult> {
78
82
  const { provider, providerId, providerConfig } =
79
83
  await resolveLiveVoiceStreamingTtsProvider(options.config);
84
+ // An explicit request voice wins outright; otherwise a language-known
85
+ // turn may select the provider's configured per-language voice. The cast
86
+ // recovers the schema-typed map that resolveTtsConfig's generic
87
+ // provider-block lookup erases.
88
+ const voiceId =
89
+ options.voiceId ??
90
+ resolveLanguageVoiceOverride(
91
+ providerConfig.languageVoices as Record<string, string> | undefined,
92
+ options.language,
93
+ );
80
94
  const useCase = options.useCase ?? "phone-call";
81
95
  const requestedSampleRate = resolveSampleRate(
82
96
  options.sampleRate,
@@ -89,9 +103,10 @@ export async function streamLiveVoiceTtsAudio(
89
103
  const providerSampleRate = provider.resolveOutputSampleRateHz?.({
90
104
  text: options.text,
91
105
  useCase,
92
- voiceId: options.voiceId,
106
+ voiceId,
93
107
  outputFormat: options.outputFormat,
94
108
  sampleRateHz: requestedSampleRate,
109
+ language: options.language,
95
110
  signal: options.signal,
96
111
  });
97
112
  const sampleRate = providerSampleRate ?? requestedSampleRate;
@@ -138,9 +153,10 @@ export async function streamLiveVoiceTtsAudio(
138
153
  provider,
139
154
  text: options.text,
140
155
  useCase,
141
- voiceId: options.voiceId,
156
+ voiceId,
142
157
  outputFormat: options.outputFormat,
143
158
  sampleRateHz: requestedSampleRate,
159
+ language: options.language,
144
160
  signal: options.signal,
145
161
  onChunk: (chunk) => {
146
162
  if (canStreamChunks) {
@@ -1,3 +1,5 @@
1
+ import { localizedOrDefault } from "../util/language-subtag.js";
2
+
1
3
  // Static fallbacks for an idle-triggered progress narration whose LLM
2
4
  // phrasing failed — the one case where prolonged silence is actively harmful.
3
5
  // The idle trigger can fire on a slow turn with zero tool activity, so every
@@ -10,8 +12,109 @@ export const PROGRESS_FALLBACK_PHRASES: readonly string[] = [
10
12
  "Almost there — thanks for waiting.",
11
13
  ];
12
14
 
15
+ // Per-language fallback phrases, keyed by lowercased BCP 47 base subtag,
16
+ // covering the Deepgram code-switching roster (DEEPGRAM_MULTI_LANGUAGE_CODES
17
+ // in providers/speech-to-text/deepgram.ts). Every list carries the same
18
+ // invariants as the English one above: persona-neutral, no claims about
19
+ // running tools or tasks, at most 8 words per phrase.
20
+ export const PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE: Readonly<
21
+ Record<string, readonly string[]>
22
+ > = {
23
+ en: PROGRESS_FALLBACK_PHRASES,
24
+ es: [
25
+ "Sigo en ello, un momento.",
26
+ "Todavía lo estoy pensando.",
27
+ "Casi listo, gracias por esperar.",
28
+ ],
29
+ fr: [
30
+ "J'y suis encore, un instant.",
31
+ "J'y réfléchis encore.",
32
+ "Presque fini, merci de patienter.",
33
+ ],
34
+ de: [
35
+ "Bin noch dabei, einen Moment.",
36
+ "Ich denke noch darüber nach.",
37
+ "Fast fertig, danke fürs Warten.",
38
+ ],
39
+ hi: [
40
+ "बस एक पल रुकिए।",
41
+ "अभी इस पर विचार चल रहा है।",
42
+ "बस थोड़ा और इंतज़ार कीजिए, धन्यवाद।",
43
+ ],
44
+ ru: [
45
+ "Секундочку, я ещё здесь.",
46
+ "Я всё ещё думаю над этим.",
47
+ "Почти готово, спасибо за ожидание.",
48
+ ],
49
+ pt: [
50
+ "Ainda estou nisso, um momento.",
51
+ "Ainda estou pensando nisso.",
52
+ "Quase lá, agradeço a espera.",
53
+ ],
54
+ ja: [
55
+ "まだ対応中です。少々お待ちください。",
56
+ "まだ考えているところです。",
57
+ "もうすぐです。お待ちいただきありがとうございます。",
58
+ ],
59
+ it: [
60
+ "Ancora un attimo, per favore.",
61
+ "Ci sto ancora pensando.",
62
+ "Quasi fatto, grazie per l'attesa.",
63
+ ],
64
+ nl: [
65
+ "Ik ben er nog mee bezig.",
66
+ "Ik denk er nog over na.",
67
+ "Bijna klaar, bedankt voor het wachten.",
68
+ ],
69
+ };
70
+
13
71
  // Deterministic rotation through the phrase list: callers hold a nonnegative
14
72
  // monotonic counter, so consecutive picks vary while tests stay reproducible.
15
- export function pickProgressPhrase(counter: number): string {
16
- return PROGRESS_FALLBACK_PHRASES[counter % PROGRESS_FALLBACK_PHRASES.length];
73
+ // `language` selects the per-language list by its lowercased base subtag
74
+ // (e.g. "pt-BR" -> "pt"); unknown or absent languages fall back to English.
75
+ export function pickProgressPhrase(counter: number, language?: string): string {
76
+ const phrases = localizedOrDefault(
77
+ PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE,
78
+ language,
79
+ PROGRESS_FALLBACK_PHRASES,
80
+ );
81
+ return phrases[counter % phrases.length];
82
+ }
83
+
84
+ // Spoken once when a turn starts waiting on the user's approval decision.
85
+ // Fixed rather than generated: this is a statement about the system's
86
+ // state, not about the work, and it has to be true every time. Kept in the
87
+ // shape of the progress phrases it displaces (short, neutral, no claim
88
+ // about tools).
89
+ export const APPROVAL_PENDING_PHRASE =
90
+ "I need your okay for that one. Take a look.";
91
+
92
+ // Per-language spellings of the approval-pending phrase, keyed by lowercased
93
+ // BCP 47 base subtag, covering the Deepgram code-switching roster
94
+ // (DEEPGRAM_MULTI_LANGUAGE_CODES in providers/speech-to-text/deepgram.ts).
95
+ // Same invariants as the progress phrases: persona-neutral, no claims about
96
+ // running tools or tasks.
97
+ export const APPROVAL_PENDING_PHRASE_BY_LANGUAGE: Readonly<
98
+ Record<string, string>
99
+ > = {
100
+ en: APPROVAL_PENDING_PHRASE,
101
+ es: "Necesito tu visto bueno para eso. Échale un vistazo.",
102
+ fr: "J'ai besoin de ton accord pour ça. Jette un œil.",
103
+ de: "Dafür brauche ich dein Okay. Schau mal drauf.",
104
+ hi: "इसके लिए मुझे आपकी मंज़ूरी चाहिए। एक नज़र डाल लीजिए।",
105
+ ru: "Для этого мне нужно твоё согласие. Взгляни, пожалуйста.",
106
+ pt: "Preciso do seu ok para isso. Dê uma olhada.",
107
+ ja: "これには許可が必要です。ご確認ください。",
108
+ it: "Mi serve il tuo via libera per questo. Dai un'occhiata.",
109
+ nl: "Hiervoor heb ik je akkoord nodig. Kijk even mee.",
110
+ };
111
+
112
+ // The approval-pending phrase in the turn's spoken language, defaulting to
113
+ // English for unknown or absent languages.
114
+ export function approvalPendingPhraseFor(language?: string): string {
115
+ return localizedOrDefault(
116
+ APPROVAL_PENDING_PHRASE_BY_LANGUAGE,
117
+ language,
118
+ APPROVAL_PENDING_PHRASE,
119
+ );
17
120
  }
@@ -102,7 +102,11 @@ function resultsFrame(
102
102
  is_final?: boolean;
103
103
  speech_final?: boolean;
104
104
  from_finalize?: boolean;
105
- words?: { word: string; speaker?: number }[];
105
+ words?: { word: string; speaker?: number; language?: string }[];
106
+ /** Container-level detected languages on the alternative. */
107
+ alternativeLanguages?: string[];
108
+ /** Container-level detected languages on the channel. */
109
+ channelLanguages?: string[];
106
110
  } = {},
107
111
  ): string {
108
112
  return JSON.stringify({
@@ -121,8 +125,14 @@ function resultsFrame(
121
125
  transcript,
122
126
  confidence: 0.95,
123
127
  ...(options.words ? { words: options.words } : {}),
128
+ ...(options.alternativeLanguages
129
+ ? { languages: options.alternativeLanguages }
130
+ : {}),
124
131
  },
125
132
  ],
133
+ ...(options.channelLanguages
134
+ ? { languages: options.channelLanguages }
135
+ : {}),
126
136
  },
127
137
  });
128
138
  }
@@ -542,6 +552,154 @@ describe("DeepgramRealtimeTranscriber", () => {
542
552
  (globalThis as Record<string, unknown>).WebSocket = origWs;
543
553
  });
544
554
 
555
+ // ─────────────────────────────────────────────────────────────────
556
+ // Language metadata (nova-3 multi code-switching)
557
+ // ─────────────────────────────────────────────────────────────────
558
+
559
+ describe("language metadata", () => {
560
+ test("ranks per-word language tags by dominance on final events", async () => {
561
+ const { events } = await startSession();
562
+
563
+ mockWs.simulateMessage(
564
+ resultsFrame("hello world hola", {
565
+ is_final: true,
566
+ words: [
567
+ { word: "hello", language: "en" },
568
+ { word: "world", language: "en" },
569
+ { word: "hola", language: "es" },
570
+ ],
571
+ }),
572
+ );
573
+
574
+ expect(events).toHaveLength(1);
575
+ expect(events[0]).toEqual({
576
+ type: "final",
577
+ text: "hello world hola",
578
+ confidence: 0.95,
579
+ languages: ["en", "es"],
580
+ });
581
+ });
582
+
583
+ test("omits the languages field entirely when no language metadata is present", async () => {
584
+ const { events } = await startSession();
585
+
586
+ mockWs.simulateMessage(
587
+ resultsFrame("hello world", {
588
+ is_final: true,
589
+ words: [{ word: "hello" }, { word: "world" }],
590
+ }),
591
+ );
592
+ mockWs.simulateMessage(resultsFrame("still typing", { is_final: false }));
593
+
594
+ expect(events).toHaveLength(2);
595
+ for (const event of events) {
596
+ // The keys must not exist at all, not just be undefined-valued.
597
+ expect("language" in event).toBe(false);
598
+ expect("languages" in event).toBe(false);
599
+ }
600
+ });
601
+
602
+ test("normalizes regional tags to their base subtag", async () => {
603
+ const { events } = await startSession();
604
+
605
+ mockWs.simulateMessage(
606
+ resultsFrame("hello", {
607
+ is_final: true,
608
+ words: [{ word: "hello", language: "en-US" }],
609
+ }),
610
+ );
611
+
612
+ expect(events[0]).toEqual({
613
+ type: "final",
614
+ text: "hello",
615
+ confidence: 0.95,
616
+ languages: ["en"],
617
+ });
618
+ });
619
+
620
+ test("partial events carry the fields when interim results are enabled", async () => {
621
+ const { events } = await startSession();
622
+
623
+ mockWs.simulateMessage(
624
+ resultsFrame("namaste hello", {
625
+ is_final: false,
626
+ words: [
627
+ { word: "namaste", language: "hi" },
628
+ { word: "hello", language: "en" },
629
+ { word: "there", language: "en" },
630
+ ],
631
+ }),
632
+ );
633
+
634
+ expect(events).toHaveLength(1);
635
+ expect(events[0]).toEqual({
636
+ type: "partial",
637
+ text: "namaste hello",
638
+ confidence: 0.95,
639
+ languages: ["en", "hi"],
640
+ });
641
+ });
642
+
643
+ test("falls back to the alternative-level languages array when words carry no tags", async () => {
644
+ const { events } = await startSession();
645
+
646
+ mockWs.simulateMessage(
647
+ resultsFrame("mixed speech", {
648
+ is_final: true,
649
+ words: [{ word: "mixed" }, { word: "speech" }],
650
+ alternativeLanguages: ["es-419", "en", "ES"],
651
+ }),
652
+ );
653
+
654
+ expect(events[0]).toEqual({
655
+ type: "final",
656
+ text: "mixed speech",
657
+ confidence: 0.95,
658
+ languages: ["es", "en"],
659
+ });
660
+ });
661
+
662
+ test("falls back to the channel-level languages array when the alternative has none", async () => {
663
+ const { events } = await startSession();
664
+
665
+ mockWs.simulateMessage(
666
+ resultsFrame("bonjour", {
667
+ is_final: true,
668
+ channelLanguages: ["fr", "en"],
669
+ }),
670
+ );
671
+
672
+ expect(events[0]).toEqual({
673
+ type: "final",
674
+ text: "bonjour",
675
+ confidence: 0.95,
676
+ languages: ["fr", "en"],
677
+ });
678
+ });
679
+
680
+ test("per-word tags take precedence over container arrays", async () => {
681
+ const { events } = await startSession();
682
+
683
+ mockWs.simulateMessage(
684
+ resultsFrame("hola amigo", {
685
+ is_final: true,
686
+ words: [
687
+ { word: "hola", language: "es" },
688
+ { word: "amigo", language: "es" },
689
+ ],
690
+ alternativeLanguages: ["en", "es"],
691
+ }),
692
+ );
693
+
694
+ expect(events[0]).toEqual({
695
+ type: "final",
696
+ text: "hola amigo",
697
+ confidence: 0.95,
698
+ languages: ["es"],
699
+ });
700
+ });
701
+ });
702
+
545
703
  // ─────────────────────────────────────────────────────────────────
546
704
  // Multi-event sequence
547
705
  // ─────────────────────────────────────────────────────────────────
@@ -671,6 +829,130 @@ describe("DeepgramRealtimeTranscriber", () => {
671
829
  "second utterance",
672
830
  ]);
673
831
  });
832
+
833
+ test("aggregated final ranks language tags across all withheld frames", async () => {
834
+ const { events } = await startSession({ utteranceBoundaryFinals: true });
835
+
836
+ // Raw per-word tags are accumulated, so "es" (2 words) outranks
837
+ // "en" (1 word) even though "en" arrived in the earlier frame.
838
+ mockWs.simulateMessage(
839
+ resultsFrame("hello", {
840
+ is_final: true,
841
+ words: [{ word: "hello", language: "en" }],
842
+ }),
843
+ );
844
+ mockWs.simulateMessage(
845
+ resultsFrame("hola amigo", {
846
+ is_final: true,
847
+ speech_final: true,
848
+ words: [
849
+ { word: "hola", language: "es" },
850
+ { word: "amigo", language: "es" },
851
+ ],
852
+ }),
853
+ );
854
+
855
+ const finals = events.filter((e) => e.type === "final");
856
+ expect(finals).toHaveLength(1);
857
+ expect(finals[0]).toEqual({
858
+ type: "final",
859
+ text: "hello hola amigo",
860
+ languages: ["es", "en"],
861
+ });
862
+ });
863
+
864
+ test("UtteranceEnd flush carries the accumulated language metadata", async () => {
865
+ const { events } = await startSession({ utteranceBoundaryFinals: true });
866
+
867
+ mockWs.simulateMessage(
868
+ resultsFrame("bonjour", {
869
+ is_final: true,
870
+ words: [{ word: "bonjour", language: "fr" }],
871
+ }),
872
+ );
873
+ mockWs.simulateMessage(
874
+ resultsFrame("hello there", {
875
+ is_final: true,
876
+ alternativeLanguages: ["en"],
877
+ }),
878
+ );
879
+ mockWs.simulateMessage(utteranceEndFrame());
880
+
881
+ const finals = events.filter((e) => e.type === "final");
882
+ expect(finals).toHaveLength(1);
883
+ expect(finals[0]).toEqual({
884
+ type: "final",
885
+ text: "bonjour hello there",
886
+ languages: ["fr", "en"],
887
+ });
888
+ });
889
+
890
+ test("aggregated final omits language fields when no withheld frame carried tags", async () => {
891
+ const { events } = await startSession({ utteranceBoundaryFinals: true });
892
+
893
+ mockWs.simulateMessage(resultsFrame("no tags", { is_final: true }));
894
+ mockWs.simulateMessage(
895
+ resultsFrame("at all", { is_final: true, speech_final: true }),
896
+ );
897
+
898
+ const finals = events.filter((e) => e.type === "final");
899
+ expect(finals).toHaveLength(1);
900
+ expect(finals[0]).toEqual({ type: "final", text: "no tags at all" });
901
+ expect("language" in finals[0]!).toBe(false);
902
+ expect("languages" in finals[0]!).toBe(false);
903
+ });
904
+
905
+ test("tags on empty-text frames do not leak into the aggregated final", async () => {
906
+ const { events } = await startSession({ utteranceBoundaryFinals: true });
907
+
908
+ // A silence segment may still carry container-level tags; only
909
+ // frames that contributed transcript text feed the metadata.
910
+ mockWs.simulateMessage(
911
+ resultsFrame("", { is_final: true, alternativeLanguages: ["fr"] }),
912
+ );
913
+ mockWs.simulateMessage(
914
+ resultsFrame("hello", {
915
+ is_final: true,
916
+ speech_final: true,
917
+ words: [{ word: "hello", language: "en" }],
918
+ }),
919
+ );
920
+
921
+ const finals = events.filter((e) => e.type === "final");
922
+ expect(finals).toHaveLength(1);
923
+ expect(finals[0]).toEqual({
924
+ type: "final",
925
+ text: "hello",
926
+ languages: ["en"],
927
+ });
928
+ });
929
+
930
+ test("language tags reset between aggregated utterances", async () => {
931
+ const { events } = await startSession({ utteranceBoundaryFinals: true });
932
+
933
+ mockWs.simulateMessage(
934
+ resultsFrame("hola", {
935
+ is_final: true,
936
+ speech_final: true,
937
+ words: [{ word: "hola", language: "es" }],
938
+ }),
939
+ );
940
+ mockWs.simulateMessage(
941
+ resultsFrame("hello", {
942
+ is_final: true,
943
+ speech_final: true,
944
+ words: [{ word: "hello", language: "en" }],
945
+ }),
946
+ );
947
+
948
+ const finals = events.filter((e) => e.type === "final");
949
+ expect(finals).toHaveLength(2);
950
+ expect(finals[1]).toEqual({
951
+ type: "final",
952
+ text: "hello",
953
+ languages: ["en"],
954
+ });
955
+ });
674
956
  });
675
957
 
676
958
  // ─────────────────────────────────────────────────────────────────
@@ -29,10 +29,12 @@
29
29
  * - All timers and listeners are cleaned up on close to prevent leaks.
30
30
  */
31
31
 
32
+ import { rankLanguages } from "../../stt/language-metadata.js";
32
33
  import type {
33
34
  StreamingTranscriber,
34
35
  SttStreamServerEvent,
35
36
  } from "../../stt/types.js";
37
+ import { baseLanguageSubtag } from "../../util/language-subtag.js";
36
38
  import { getLogger } from "../../util/logger.js";
37
39
 
38
40
  const log = getLogger("deepgram-realtime");
@@ -208,6 +210,11 @@ interface DeepgramStreamWord {
208
210
  confidence?: number;
209
211
  start?: number;
210
212
  end?: number;
213
+ /**
214
+ * BCP-47 tag of the language this word was spoken in. Present only on
215
+ * code-switching models (nova-3 with `language=multi`).
216
+ */
217
+ language?: string;
211
218
  }
212
219
 
213
220
  /**
@@ -225,11 +232,23 @@ interface DeepgramStreamAlternative {
225
232
  speaker?: number;
226
233
  /** Per-word speaker tags when diarization is enabled. */
227
234
  words?: DeepgramStreamWord[];
235
+ /**
236
+ * Detected languages for the chunk in dominance order. Emitted by
237
+ * code-switching models; the container varies by API version, so
238
+ * {@link DeepgramStreamChannel.languages} is checked as well.
239
+ */
240
+ languages?: string[];
228
241
  }
229
242
 
230
243
  /** A channel within a Deepgram streaming response. */
231
244
  interface DeepgramStreamChannel {
232
245
  alternatives?: DeepgramStreamAlternative[];
246
+ /**
247
+ * Detected languages for the chunk in dominance order. Alternate
248
+ * container for {@link DeepgramStreamAlternative.languages} on some
249
+ * API versions.
250
+ */
251
+ languages?: string[];
233
252
  }
234
253
 
235
254
  /**
@@ -343,6 +362,15 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
343
362
  */
344
363
  private pendingFinalSegments: string[] = [];
345
364
 
365
+ /**
366
+ * Raw detected-language tags for the withheld segments, accumulated
367
+ * alongside {@link pendingFinalSegments} and ranked into the event's
368
+ * `languages` when the utterance flushes. Cleared wherever the pending
369
+ * segments are cleared. Only populated when
370
+ * {@link utteranceBoundaryFinals} is enabled.
371
+ */
372
+ private pendingLanguageTags: string[] = [];
373
+
346
374
  /** The live WebSocket connection, set during start(). */
347
375
  private ws: WsLike | null = null;
348
376
 
@@ -748,6 +776,12 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
748
776
  * words — see {@link extractSpeakerLabel}. Confidence is taken from
749
777
  * the top alternative when present.
750
778
  *
779
+ * Code-switching models (nova-3 with `language=multi`) tag detected
780
+ * languages per word and per container. When present, these become the
781
+ * dominance-ranked `languages` field on the emitted events (see
782
+ * {@link extractLanguages}). The field is omitted when the frame
783
+ * carries no language metadata.
784
+ *
751
785
  * We emit:
752
786
  * - `partial` for `is_final: false` frames (if interim results enabled).
753
787
  * - `final` for `is_final: true` frames.
@@ -788,6 +822,7 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
788
822
  typeof alternative?.confidence === "number"
789
823
  ? alternative.confidence
790
824
  : undefined;
825
+ const languages = extractLanguages(frame.channel, alternative);
791
826
 
792
827
  if (frame.is_final) {
793
828
  if (this.utteranceBoundaryFinals) {
@@ -795,6 +830,16 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
795
830
  // Finalize flush is a forced boundary — flush what is pending.
796
831
  if (text.length > 0) {
797
832
  this.pendingFinalSegments.push(text);
833
+ // Collect language tags only from frames that contributed text
834
+ // so the flushed metadata stays aligned with the emitted
835
+ // transcript (empty frames may still carry tags, but they
836
+ // describe no emitted words). Raw per-word tags are preferred
837
+ // over the frame's ranked list so cross-frame frequency
838
+ // weighting survives until the flush ranks the whole utterance.
839
+ const wordTags = collectWordLanguageTags(alternative);
840
+ this.pendingLanguageTags.push(
841
+ ...(wordTags.length > 0 ? wordTags : languages),
842
+ );
798
843
  }
799
844
  if (frame.speech_final || fromFinalize) {
800
845
  this.flushPendingUtterance();
@@ -807,6 +852,7 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
807
852
  text,
808
853
  ...(speakerLabel !== undefined ? { speakerLabel } : {}),
809
854
  ...(confidence !== undefined ? { confidence } : {}),
855
+ ...(languages.length > 0 ? { languages } : {}),
810
856
  // Mark the finalize flush so consumers can attribute it to the
811
857
  // utterance that requested the flush rather than new speech.
812
858
  ...(fromFinalize ? { fromFinalize: true } : {}),
@@ -819,6 +865,7 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
819
865
  text,
820
866
  ...(speakerLabel !== undefined ? { speakerLabel } : {}),
821
867
  ...(confidence !== undefined ? { confidence } : {}),
868
+ ...(languages.length > 0 ? { languages } : {}),
822
869
  });
823
870
  }
824
871
 
@@ -917,16 +964,27 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
917
964
 
918
965
  /**
919
966
  * Emit a single aggregated `final` for the withheld `is_final` segments
920
- * of the current utterance. No-op when nothing is pending, so boundary
921
- * signals over silence emit nothing.
967
+ * of the current utterance, carrying the dominance-ranked detected
968
+ * languages accumulated alongside them (field omitted when no segment
969
+ * carried language metadata). No-op when nothing is pending, so
970
+ * boundary signals over silence emit nothing.
922
971
  */
923
972
  private flushPendingUtterance(): void {
924
973
  if (this.pendingFinalSegments.length === 0) {
974
+ // Tags accumulate only alongside text, but clear defensively so a
975
+ // future drift cannot leak one utterance's tags into the next.
976
+ this.pendingLanguageTags = [];
925
977
  return;
926
978
  }
927
979
  const text = this.pendingFinalSegments.join(" ");
980
+ const languages = rankLanguages(this.pendingLanguageTags);
928
981
  this.pendingFinalSegments = [];
929
- this.emitEvent({ type: "final", text });
982
+ this.pendingLanguageTags = [];
983
+ this.emitEvent({
984
+ type: "final",
985
+ text,
986
+ ...(languages.length > 0 ? { languages } : {}),
987
+ });
930
988
  }
931
989
 
932
990
  /**
@@ -1222,6 +1280,62 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
1222
1280
  * contract on {@link SttStreamServerPartialEvent} /
1223
1281
  * {@link SttStreamServerFinalEvent}.
1224
1282
  */
1283
+ /**
1284
+ * Derive the detected languages for a chunk, most dominant first.
1285
+ *
1286
+ * Code-switching models tag languages in two shapes:
1287
+ * 1. Per-word `language` tags on `alternatives[0].words[]`, the richest
1288
+ * signal; ranked by frequency via {@link rankLanguages} (ties broken
1289
+ * by first appearance).
1290
+ * 2. A container-level `languages` array in dominance order, attached to
1291
+ * the alternative or (on some API versions) the channel. Used as the
1292
+ * fallback when no word carries a tag; normalized and deduped with
1293
+ * the provider's order preserved.
1294
+ *
1295
+ * Returns `[]` when the frame carries no language metadata: callers omit
1296
+ * the event fields entirely so absence stays distinguishable from a
1297
+ * detected language.
1298
+ */
1299
+ function extractLanguages(
1300
+ channel: DeepgramStreamChannel | undefined,
1301
+ alternative: DeepgramStreamAlternative | undefined,
1302
+ ): string[] {
1303
+ const wordTags = collectWordLanguageTags(alternative);
1304
+ if (wordTags.length > 0) {
1305
+ return rankLanguages(wordTags);
1306
+ }
1307
+
1308
+ const container = Array.isArray(alternative?.languages)
1309
+ ? alternative.languages
1310
+ : Array.isArray(channel?.languages)
1311
+ ? channel.languages
1312
+ : [];
1313
+ const deduped = new Set(
1314
+ container
1315
+ .filter((tag): tag is string => typeof tag === "string")
1316
+ .flatMap((tag) => {
1317
+ const base = baseLanguageSubtag(tag);
1318
+ return base !== undefined ? [base] : [];
1319
+ }),
1320
+ );
1321
+ return [...deduped];
1322
+ }
1323
+
1324
+ /**
1325
+ * Collect the raw per-word `language` tags of a chunk, in word order and
1326
+ * without ranking or normalization. Used both for per-frame ranking in
1327
+ * {@link extractLanguages} and for cross-frame accumulation in
1328
+ * utterance-boundary mode, where ranking is deferred to the flush so
1329
+ * frequency weighting spans the whole utterance.
1330
+ */
1331
+ function collectWordLanguageTags(
1332
+ alternative: DeepgramStreamAlternative | undefined,
1333
+ ): string[] {
1334
+ return (alternative?.words ?? []).flatMap((word) =>
1335
+ typeof word.language === "string" ? [word.language] : [],
1336
+ );
1337
+ }
1338
+
1225
1339
  function extractSpeakerLabel(
1226
1340
  alternative: DeepgramStreamAlternative | undefined,
1227
1341
  ): string | undefined {