@vellumai/assistant 0.11.3-staging.2 → 0.11.3-staging.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/__tests__/call-controller.test.ts +120 -0
- package/src/__tests__/config-loader-backfill.test.ts +19 -0
- package/src/__tests__/config-schema.test.ts +142 -5
- package/src/__tests__/events-dev-bypass-actor.test.ts +112 -1
- package/src/__tests__/media-stream-output.test.ts +175 -0
- package/src/__tests__/media-stream-stt-session.test.ts +67 -0
- package/src/calls/__tests__/tts-text-sanitizer.test.ts +13 -0
- package/src/calls/__tests__/voice-session-bridge.test.ts +81 -12
- package/src/calls/__tests__/voice-triage-escalate.test.ts +106 -0
- package/src/calls/call-controller.ts +30 -2
- package/src/calls/call-speech-output.ts +12 -4
- package/src/calls/call-transport.ts +18 -1
- package/src/calls/media-stream-output.ts +53 -5
- package/src/calls/media-stream-server.ts +12 -0
- package/src/calls/media-stream-stt-session.ts +34 -0
- package/src/calls/telephony-synthesis-language.ts +84 -0
- package/src/calls/tts-text-sanitizer.ts +12 -5
- package/src/calls/voice-session-bridge.ts +31 -3
- package/src/calls/voice-triage-escalate.ts +52 -5
- package/src/config/bundled-skills/phone-calls/references/CONFIG.md +15 -15
- package/src/config/loader.ts +5 -0
- package/src/config/schemas/__tests__/live-voice.test.ts +35 -0
- package/src/config/schemas/calls.ts +0 -4
- package/src/config/schemas/live-voice.ts +25 -0
- package/src/config/schemas/tts.ts +63 -0
- package/src/live-voice/__tests__/front-decision.test.ts +120 -0
- package/src/live-voice/__tests__/live-voice-events.test.ts +1 -1
- package/src/live-voice/__tests__/live-voice-integration.test.ts +5 -0
- package/src/live-voice/__tests__/live-voice-progress.test.ts +178 -11
- package/src/live-voice/__tests__/live-voice-stt.test.ts +304 -1
- package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +102 -2
- package/src/live-voice/__tests__/live-voice-tts.test.ts +148 -2
- package/src/live-voice/__tests__/live-voice-vad.test.ts +278 -1
- package/src/live-voice/__tests__/progress-phrases.test.ts +167 -0
- package/src/live-voice/front-decision.ts +50 -3
- package/src/live-voice/live-voice-session.ts +517 -45
- package/src/live-voice/live-voice-tts.ts +18 -2
- package/src/live-voice/progress-phrases.ts +105 -2
- package/src/providers/speech-to-text/deepgram-realtime.test.ts +283 -1
- package/src/providers/speech-to-text/deepgram-realtime.ts +117 -3
- package/src/providers/speech-to-text/provider-catalog.ts +38 -0
- package/src/runtime/__tests__/local-actor-identity-force-refresh.test.ts +104 -0
- package/src/runtime/assistant-event-hub.ts +23 -0
- package/src/runtime/local-actor-identity.ts +18 -5
- package/src/runtime/routes/__tests__/sse-actor-principal-heal.test.ts +165 -0
- package/src/runtime/routes/__tests__/surface-action-routes.test.ts +2 -0
- package/src/runtime/routes/events-routes.ts +17 -16
- package/src/runtime/routes/sse-actor-principal-heal.ts +113 -0
- package/src/stt/__tests__/language-metadata.test.ts +85 -0
- package/src/stt/__tests__/speech-energy.test.ts +79 -0
- package/src/stt/language-metadata.ts +65 -0
- package/src/stt/speech-energy.ts +115 -12
- package/src/stt/types.ts +16 -0
- package/src/tts/__tests__/provider-adapters.test.ts +147 -0
- package/src/tts/__tests__/speakable-segments.test.ts +470 -0
- package/src/tts/language-voices.ts +23 -0
- package/src/tts/providers/deepgram-provider.ts +3 -1
- package/src/tts/providers/elevenlabs-provider.ts +73 -1
- package/src/tts/providers/xai-provider.ts +28 -2
- package/src/tts/speakable-segments.ts +293 -23
- package/src/tts/synthesis-stream.ts +7 -0
- package/src/tts/types.ts +7 -0
- package/src/util/__tests__/language-subtag.test.ts +54 -0
- package/src/util/language-subtag.ts +43 -0
- package/src/util/unicode.ts +1 -1
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { resolveLanguageVoiceOverride } from "../tts/language-voices.js";
|
|
1
2
|
import { createPcmChunkAligner } from "../tts/pcm-chunk-aligner.js";
|
|
2
3
|
import { getTtsProvider } from "../tts/provider-catalog.js";
|
|
3
4
|
import { synthesizeAndEmit } from "../tts/synthesis-stream.js";
|
|
@@ -30,6 +31,7 @@ export interface LiveVoiceTtsOptions {
|
|
|
30
31
|
useCase?: TtsUseCase;
|
|
31
32
|
outputFormat?: TtsSynthesisRequest["outputFormat"];
|
|
32
33
|
sampleRate?: number;
|
|
34
|
+
language?: string;
|
|
33
35
|
config?: LiveVoiceTtsConfig;
|
|
34
36
|
onAudioChunk: (chunk: LiveVoiceTtsAudioChunk) => void;
|
|
35
37
|
}
|
|
@@ -72,11 +74,23 @@ interface ResolvedStreamingTtsProvider {
|
|
|
72
74
|
providerConfig: Record<string, unknown>;
|
|
73
75
|
}
|
|
74
76
|
|
|
77
|
+
export { resolveLanguageVoiceOverride };
|
|
78
|
+
|
|
75
79
|
export async function streamLiveVoiceTtsAudio(
|
|
76
80
|
options: LiveVoiceTtsOptions,
|
|
77
81
|
): Promise<LiveVoiceTtsResult> {
|
|
78
82
|
const { provider, providerId, providerConfig } =
|
|
79
83
|
await resolveLiveVoiceStreamingTtsProvider(options.config);
|
|
84
|
+
// An explicit request voice wins outright; otherwise a language-known
|
|
85
|
+
// turn may select the provider's configured per-language voice. The cast
|
|
86
|
+
// recovers the schema-typed map that resolveTtsConfig's generic
|
|
87
|
+
// provider-block lookup erases.
|
|
88
|
+
const voiceId =
|
|
89
|
+
options.voiceId ??
|
|
90
|
+
resolveLanguageVoiceOverride(
|
|
91
|
+
providerConfig.languageVoices as Record<string, string> | undefined,
|
|
92
|
+
options.language,
|
|
93
|
+
);
|
|
80
94
|
const useCase = options.useCase ?? "phone-call";
|
|
81
95
|
const requestedSampleRate = resolveSampleRate(
|
|
82
96
|
options.sampleRate,
|
|
@@ -89,9 +103,10 @@ export async function streamLiveVoiceTtsAudio(
|
|
|
89
103
|
const providerSampleRate = provider.resolveOutputSampleRateHz?.({
|
|
90
104
|
text: options.text,
|
|
91
105
|
useCase,
|
|
92
|
-
voiceId
|
|
106
|
+
voiceId,
|
|
93
107
|
outputFormat: options.outputFormat,
|
|
94
108
|
sampleRateHz: requestedSampleRate,
|
|
109
|
+
language: options.language,
|
|
95
110
|
signal: options.signal,
|
|
96
111
|
});
|
|
97
112
|
const sampleRate = providerSampleRate ?? requestedSampleRate;
|
|
@@ -138,9 +153,10 @@ export async function streamLiveVoiceTtsAudio(
|
|
|
138
153
|
provider,
|
|
139
154
|
text: options.text,
|
|
140
155
|
useCase,
|
|
141
|
-
voiceId
|
|
156
|
+
voiceId,
|
|
142
157
|
outputFormat: options.outputFormat,
|
|
143
158
|
sampleRateHz: requestedSampleRate,
|
|
159
|
+
language: options.language,
|
|
144
160
|
signal: options.signal,
|
|
145
161
|
onChunk: (chunk) => {
|
|
146
162
|
if (canStreamChunks) {
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { localizedOrDefault } from "../util/language-subtag.js";
|
|
2
|
+
|
|
1
3
|
// Static fallbacks for an idle-triggered progress narration whose LLM
|
|
2
4
|
// phrasing failed — the one case where prolonged silence is actively harmful.
|
|
3
5
|
// The idle trigger can fire on a slow turn with zero tool activity, so every
|
|
@@ -10,8 +12,109 @@ export const PROGRESS_FALLBACK_PHRASES: readonly string[] = [
|
|
|
10
12
|
"Almost there — thanks for waiting.",
|
|
11
13
|
];
|
|
12
14
|
|
|
15
|
+
// Per-language fallback phrases, keyed by lowercased BCP 47 base subtag,
|
|
16
|
+
// covering the Deepgram code-switching roster (DEEPGRAM_MULTI_LANGUAGE_CODES
|
|
17
|
+
// in providers/speech-to-text/deepgram.ts). Every list carries the same
|
|
18
|
+
// invariants as the English one above: persona-neutral, no claims about
|
|
19
|
+
// running tools or tasks, at most 8 words per phrase.
|
|
20
|
+
export const PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE: Readonly<
|
|
21
|
+
Record<string, readonly string[]>
|
|
22
|
+
> = {
|
|
23
|
+
en: PROGRESS_FALLBACK_PHRASES,
|
|
24
|
+
es: [
|
|
25
|
+
"Sigo en ello, un momento.",
|
|
26
|
+
"Todavía lo estoy pensando.",
|
|
27
|
+
"Casi listo, gracias por esperar.",
|
|
28
|
+
],
|
|
29
|
+
fr: [
|
|
30
|
+
"J'y suis encore, un instant.",
|
|
31
|
+
"J'y réfléchis encore.",
|
|
32
|
+
"Presque fini, merci de patienter.",
|
|
33
|
+
],
|
|
34
|
+
de: [
|
|
35
|
+
"Bin noch dabei, einen Moment.",
|
|
36
|
+
"Ich denke noch darüber nach.",
|
|
37
|
+
"Fast fertig, danke fürs Warten.",
|
|
38
|
+
],
|
|
39
|
+
hi: [
|
|
40
|
+
"बस एक पल रुकिए।",
|
|
41
|
+
"अभी इस पर विचार चल रहा है।",
|
|
42
|
+
"बस थोड़ा और इंतज़ार कीजिए, धन्यवाद।",
|
|
43
|
+
],
|
|
44
|
+
ru: [
|
|
45
|
+
"Секундочку, я ещё здесь.",
|
|
46
|
+
"Я всё ещё думаю над этим.",
|
|
47
|
+
"Почти готово, спасибо за ожидание.",
|
|
48
|
+
],
|
|
49
|
+
pt: [
|
|
50
|
+
"Ainda estou nisso, um momento.",
|
|
51
|
+
"Ainda estou pensando nisso.",
|
|
52
|
+
"Quase lá, agradeço a espera.",
|
|
53
|
+
],
|
|
54
|
+
ja: [
|
|
55
|
+
"まだ対応中です。少々お待ちください。",
|
|
56
|
+
"まだ考えているところです。",
|
|
57
|
+
"もうすぐです。お待ちいただきありがとうございます。",
|
|
58
|
+
],
|
|
59
|
+
it: [
|
|
60
|
+
"Ancora un attimo, per favore.",
|
|
61
|
+
"Ci sto ancora pensando.",
|
|
62
|
+
"Quasi fatto, grazie per l'attesa.",
|
|
63
|
+
],
|
|
64
|
+
nl: [
|
|
65
|
+
"Ik ben er nog mee bezig.",
|
|
66
|
+
"Ik denk er nog over na.",
|
|
67
|
+
"Bijna klaar, bedankt voor het wachten.",
|
|
68
|
+
],
|
|
69
|
+
};
|
|
70
|
+
|
|
13
71
|
// Deterministic rotation through the phrase list: callers hold a nonnegative
|
|
14
72
|
// monotonic counter, so consecutive picks vary while tests stay reproducible.
|
|
15
|
-
|
|
16
|
-
|
|
73
|
+
// `language` selects the per-language list by its lowercased base subtag
|
|
74
|
+
// (e.g. "pt-BR" -> "pt"); unknown or absent languages fall back to English.
|
|
75
|
+
export function pickProgressPhrase(counter: number, language?: string): string {
|
|
76
|
+
const phrases = localizedOrDefault(
|
|
77
|
+
PROGRESS_FALLBACK_PHRASES_BY_LANGUAGE,
|
|
78
|
+
language,
|
|
79
|
+
PROGRESS_FALLBACK_PHRASES,
|
|
80
|
+
);
|
|
81
|
+
return phrases[counter % phrases.length];
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// Spoken once when a turn starts waiting on the user's approval decision.
|
|
85
|
+
// Fixed rather than generated: this is a statement about the system's
|
|
86
|
+
// state, not about the work, and it has to be true every time. Kept in the
|
|
87
|
+
// shape of the progress phrases it displaces (short, neutral, no claim
|
|
88
|
+
// about tools).
|
|
89
|
+
export const APPROVAL_PENDING_PHRASE =
|
|
90
|
+
"I need your okay for that one. Take a look.";
|
|
91
|
+
|
|
92
|
+
// Per-language spellings of the approval-pending phrase, keyed by lowercased
|
|
93
|
+
// BCP 47 base subtag, covering the Deepgram code-switching roster
|
|
94
|
+
// (DEEPGRAM_MULTI_LANGUAGE_CODES in providers/speech-to-text/deepgram.ts).
|
|
95
|
+
// Same invariants as the progress phrases: persona-neutral, no claims about
|
|
96
|
+
// running tools or tasks.
|
|
97
|
+
export const APPROVAL_PENDING_PHRASE_BY_LANGUAGE: Readonly<
|
|
98
|
+
Record<string, string>
|
|
99
|
+
> = {
|
|
100
|
+
en: APPROVAL_PENDING_PHRASE,
|
|
101
|
+
es: "Necesito tu visto bueno para eso. Échale un vistazo.",
|
|
102
|
+
fr: "J'ai besoin de ton accord pour ça. Jette un œil.",
|
|
103
|
+
de: "Dafür brauche ich dein Okay. Schau mal drauf.",
|
|
104
|
+
hi: "इसके लिए मुझे आपकी मंज़ूरी चाहिए। एक नज़र डाल लीजिए।",
|
|
105
|
+
ru: "Для этого мне нужно твоё согласие. Взгляни, пожалуйста.",
|
|
106
|
+
pt: "Preciso do seu ok para isso. Dê uma olhada.",
|
|
107
|
+
ja: "これには許可が必要です。ご確認ください。",
|
|
108
|
+
it: "Mi serve il tuo via libera per questo. Dai un'occhiata.",
|
|
109
|
+
nl: "Hiervoor heb ik je akkoord nodig. Kijk even mee.",
|
|
110
|
+
};
|
|
111
|
+
|
|
112
|
+
// The approval-pending phrase in the turn's spoken language, defaulting to
|
|
113
|
+
// English for unknown or absent languages.
|
|
114
|
+
export function approvalPendingPhraseFor(language?: string): string {
|
|
115
|
+
return localizedOrDefault(
|
|
116
|
+
APPROVAL_PENDING_PHRASE_BY_LANGUAGE,
|
|
117
|
+
language,
|
|
118
|
+
APPROVAL_PENDING_PHRASE,
|
|
119
|
+
);
|
|
17
120
|
}
|
|
@@ -102,7 +102,11 @@ function resultsFrame(
|
|
|
102
102
|
is_final?: boolean;
|
|
103
103
|
speech_final?: boolean;
|
|
104
104
|
from_finalize?: boolean;
|
|
105
|
-
words?: { word: string; speaker?: number }[];
|
|
105
|
+
words?: { word: string; speaker?: number; language?: string }[];
|
|
106
|
+
/** Container-level detected languages on the alternative. */
|
|
107
|
+
alternativeLanguages?: string[];
|
|
108
|
+
/** Container-level detected languages on the channel. */
|
|
109
|
+
channelLanguages?: string[];
|
|
106
110
|
} = {},
|
|
107
111
|
): string {
|
|
108
112
|
return JSON.stringify({
|
|
@@ -121,8 +125,14 @@ function resultsFrame(
|
|
|
121
125
|
transcript,
|
|
122
126
|
confidence: 0.95,
|
|
123
127
|
...(options.words ? { words: options.words } : {}),
|
|
128
|
+
...(options.alternativeLanguages
|
|
129
|
+
? { languages: options.alternativeLanguages }
|
|
130
|
+
: {}),
|
|
124
131
|
},
|
|
125
132
|
],
|
|
133
|
+
...(options.channelLanguages
|
|
134
|
+
? { languages: options.channelLanguages }
|
|
135
|
+
: {}),
|
|
126
136
|
},
|
|
127
137
|
});
|
|
128
138
|
}
|
|
@@ -542,6 +552,154 @@ describe("DeepgramRealtimeTranscriber", () => {
|
|
|
542
552
|
(globalThis as Record<string, unknown>).WebSocket = origWs;
|
|
543
553
|
});
|
|
544
554
|
|
|
555
|
+
// ─────────────────────────────────────────────────────────────────
|
|
556
|
+
// Language metadata (nova-3 multi code-switching)
|
|
557
|
+
// ─────────────────────────────────────────────────────────────────
|
|
558
|
+
|
|
559
|
+
describe("language metadata", () => {
|
|
560
|
+
test("ranks per-word language tags by dominance on final events", async () => {
|
|
561
|
+
const { events } = await startSession();
|
|
562
|
+
|
|
563
|
+
mockWs.simulateMessage(
|
|
564
|
+
resultsFrame("hello world hola", {
|
|
565
|
+
is_final: true,
|
|
566
|
+
words: [
|
|
567
|
+
{ word: "hello", language: "en" },
|
|
568
|
+
{ word: "world", language: "en" },
|
|
569
|
+
{ word: "hola", language: "es" },
|
|
570
|
+
],
|
|
571
|
+
}),
|
|
572
|
+
);
|
|
573
|
+
|
|
574
|
+
expect(events).toHaveLength(1);
|
|
575
|
+
expect(events[0]).toEqual({
|
|
576
|
+
type: "final",
|
|
577
|
+
text: "hello world hola",
|
|
578
|
+
confidence: 0.95,
|
|
579
|
+
languages: ["en", "es"],
|
|
580
|
+
});
|
|
581
|
+
});
|
|
582
|
+
|
|
583
|
+
test("omits the languages field entirely when no language metadata is present", async () => {
|
|
584
|
+
const { events } = await startSession();
|
|
585
|
+
|
|
586
|
+
mockWs.simulateMessage(
|
|
587
|
+
resultsFrame("hello world", {
|
|
588
|
+
is_final: true,
|
|
589
|
+
words: [{ word: "hello" }, { word: "world" }],
|
|
590
|
+
}),
|
|
591
|
+
);
|
|
592
|
+
mockWs.simulateMessage(resultsFrame("still typing", { is_final: false }));
|
|
593
|
+
|
|
594
|
+
expect(events).toHaveLength(2);
|
|
595
|
+
for (const event of events) {
|
|
596
|
+
// The keys must not exist at all, not just be undefined-valued.
|
|
597
|
+
expect("language" in event).toBe(false);
|
|
598
|
+
expect("languages" in event).toBe(false);
|
|
599
|
+
}
|
|
600
|
+
});
|
|
601
|
+
|
|
602
|
+
test("normalizes regional tags to their base subtag", async () => {
|
|
603
|
+
const { events } = await startSession();
|
|
604
|
+
|
|
605
|
+
mockWs.simulateMessage(
|
|
606
|
+
resultsFrame("hello", {
|
|
607
|
+
is_final: true,
|
|
608
|
+
words: [{ word: "hello", language: "en-US" }],
|
|
609
|
+
}),
|
|
610
|
+
);
|
|
611
|
+
|
|
612
|
+
expect(events[0]).toEqual({
|
|
613
|
+
type: "final",
|
|
614
|
+
text: "hello",
|
|
615
|
+
confidence: 0.95,
|
|
616
|
+
languages: ["en"],
|
|
617
|
+
});
|
|
618
|
+
});
|
|
619
|
+
|
|
620
|
+
test("partial events carry the fields when interim results are enabled", async () => {
|
|
621
|
+
const { events } = await startSession();
|
|
622
|
+
|
|
623
|
+
mockWs.simulateMessage(
|
|
624
|
+
resultsFrame("namaste hello", {
|
|
625
|
+
is_final: false,
|
|
626
|
+
words: [
|
|
627
|
+
{ word: "namaste", language: "hi" },
|
|
628
|
+
{ word: "hello", language: "en" },
|
|
629
|
+
{ word: "there", language: "en" },
|
|
630
|
+
],
|
|
631
|
+
}),
|
|
632
|
+
);
|
|
633
|
+
|
|
634
|
+
expect(events).toHaveLength(1);
|
|
635
|
+
expect(events[0]).toEqual({
|
|
636
|
+
type: "partial",
|
|
637
|
+
text: "namaste hello",
|
|
638
|
+
confidence: 0.95,
|
|
639
|
+
languages: ["en", "hi"],
|
|
640
|
+
});
|
|
641
|
+
});
|
|
642
|
+
|
|
643
|
+
test("falls back to the alternative-level languages array when words carry no tags", async () => {
|
|
644
|
+
const { events } = await startSession();
|
|
645
|
+
|
|
646
|
+
mockWs.simulateMessage(
|
|
647
|
+
resultsFrame("mixed speech", {
|
|
648
|
+
is_final: true,
|
|
649
|
+
words: [{ word: "mixed" }, { word: "speech" }],
|
|
650
|
+
alternativeLanguages: ["es-419", "en", "ES"],
|
|
651
|
+
}),
|
|
652
|
+
);
|
|
653
|
+
|
|
654
|
+
expect(events[0]).toEqual({
|
|
655
|
+
type: "final",
|
|
656
|
+
text: "mixed speech",
|
|
657
|
+
confidence: 0.95,
|
|
658
|
+
languages: ["es", "en"],
|
|
659
|
+
});
|
|
660
|
+
});
|
|
661
|
+
|
|
662
|
+
test("falls back to the channel-level languages array when the alternative has none", async () => {
|
|
663
|
+
const { events } = await startSession();
|
|
664
|
+
|
|
665
|
+
mockWs.simulateMessage(
|
|
666
|
+
resultsFrame("bonjour", {
|
|
667
|
+
is_final: true,
|
|
668
|
+
channelLanguages: ["fr", "en"],
|
|
669
|
+
}),
|
|
670
|
+
);
|
|
671
|
+
|
|
672
|
+
expect(events[0]).toEqual({
|
|
673
|
+
type: "final",
|
|
674
|
+
text: "bonjour",
|
|
675
|
+
confidence: 0.95,
|
|
676
|
+
languages: ["fr", "en"],
|
|
677
|
+
});
|
|
678
|
+
});
|
|
679
|
+
|
|
680
|
+
test("per-word tags take precedence over container arrays", async () => {
|
|
681
|
+
const { events } = await startSession();
|
|
682
|
+
|
|
683
|
+
mockWs.simulateMessage(
|
|
684
|
+
resultsFrame("hola amigo", {
|
|
685
|
+
is_final: true,
|
|
686
|
+
words: [
|
|
687
|
+
{ word: "hola", language: "es" },
|
|
688
|
+
{ word: "amigo", language: "es" },
|
|
689
|
+
],
|
|
690
|
+
alternativeLanguages: ["en", "es"],
|
|
691
|
+
}),
|
|
692
|
+
);
|
|
693
|
+
|
|
694
|
+
expect(events[0]).toEqual({
|
|
695
|
+
type: "final",
|
|
696
|
+
text: "hola amigo",
|
|
697
|
+
confidence: 0.95,
|
|
698
|
+
languages: ["es"],
|
|
699
|
+
});
|
|
700
|
+
});
|
|
701
|
+
});
|
|
702
|
+
|
|
545
703
|
// ─────────────────────────────────────────────────────────────────
|
|
546
704
|
// Multi-event sequence
|
|
547
705
|
// ─────────────────────────────────────────────────────────────────
|
|
@@ -671,6 +829,130 @@ describe("DeepgramRealtimeTranscriber", () => {
|
|
|
671
829
|
"second utterance",
|
|
672
830
|
]);
|
|
673
831
|
});
|
|
832
|
+
|
|
833
|
+
test("aggregated final ranks language tags across all withheld frames", async () => {
|
|
834
|
+
const { events } = await startSession({ utteranceBoundaryFinals: true });
|
|
835
|
+
|
|
836
|
+
// Raw per-word tags are accumulated, so "es" (2 words) outranks
|
|
837
|
+
// "en" (1 word) even though "en" arrived in the earlier frame.
|
|
838
|
+
mockWs.simulateMessage(
|
|
839
|
+
resultsFrame("hello", {
|
|
840
|
+
is_final: true,
|
|
841
|
+
words: [{ word: "hello", language: "en" }],
|
|
842
|
+
}),
|
|
843
|
+
);
|
|
844
|
+
mockWs.simulateMessage(
|
|
845
|
+
resultsFrame("hola amigo", {
|
|
846
|
+
is_final: true,
|
|
847
|
+
speech_final: true,
|
|
848
|
+
words: [
|
|
849
|
+
{ word: "hola", language: "es" },
|
|
850
|
+
{ word: "amigo", language: "es" },
|
|
851
|
+
],
|
|
852
|
+
}),
|
|
853
|
+
);
|
|
854
|
+
|
|
855
|
+
const finals = events.filter((e) => e.type === "final");
|
|
856
|
+
expect(finals).toHaveLength(1);
|
|
857
|
+
expect(finals[0]).toEqual({
|
|
858
|
+
type: "final",
|
|
859
|
+
text: "hello hola amigo",
|
|
860
|
+
languages: ["es", "en"],
|
|
861
|
+
});
|
|
862
|
+
});
|
|
863
|
+
|
|
864
|
+
test("UtteranceEnd flush carries the accumulated language metadata", async () => {
|
|
865
|
+
const { events } = await startSession({ utteranceBoundaryFinals: true });
|
|
866
|
+
|
|
867
|
+
mockWs.simulateMessage(
|
|
868
|
+
resultsFrame("bonjour", {
|
|
869
|
+
is_final: true,
|
|
870
|
+
words: [{ word: "bonjour", language: "fr" }],
|
|
871
|
+
}),
|
|
872
|
+
);
|
|
873
|
+
mockWs.simulateMessage(
|
|
874
|
+
resultsFrame("hello there", {
|
|
875
|
+
is_final: true,
|
|
876
|
+
alternativeLanguages: ["en"],
|
|
877
|
+
}),
|
|
878
|
+
);
|
|
879
|
+
mockWs.simulateMessage(utteranceEndFrame());
|
|
880
|
+
|
|
881
|
+
const finals = events.filter((e) => e.type === "final");
|
|
882
|
+
expect(finals).toHaveLength(1);
|
|
883
|
+
expect(finals[0]).toEqual({
|
|
884
|
+
type: "final",
|
|
885
|
+
text: "bonjour hello there",
|
|
886
|
+
languages: ["fr", "en"],
|
|
887
|
+
});
|
|
888
|
+
});
|
|
889
|
+
|
|
890
|
+
test("aggregated final omits language fields when no withheld frame carried tags", async () => {
|
|
891
|
+
const { events } = await startSession({ utteranceBoundaryFinals: true });
|
|
892
|
+
|
|
893
|
+
mockWs.simulateMessage(resultsFrame("no tags", { is_final: true }));
|
|
894
|
+
mockWs.simulateMessage(
|
|
895
|
+
resultsFrame("at all", { is_final: true, speech_final: true }),
|
|
896
|
+
);
|
|
897
|
+
|
|
898
|
+
const finals = events.filter((e) => e.type === "final");
|
|
899
|
+
expect(finals).toHaveLength(1);
|
|
900
|
+
expect(finals[0]).toEqual({ type: "final", text: "no tags at all" });
|
|
901
|
+
expect("language" in finals[0]!).toBe(false);
|
|
902
|
+
expect("languages" in finals[0]!).toBe(false);
|
|
903
|
+
});
|
|
904
|
+
|
|
905
|
+
test("tags on empty-text frames do not leak into the aggregated final", async () => {
|
|
906
|
+
const { events } = await startSession({ utteranceBoundaryFinals: true });
|
|
907
|
+
|
|
908
|
+
// A silence segment may still carry container-level tags; only
|
|
909
|
+
// frames that contributed transcript text feed the metadata.
|
|
910
|
+
mockWs.simulateMessage(
|
|
911
|
+
resultsFrame("", { is_final: true, alternativeLanguages: ["fr"] }),
|
|
912
|
+
);
|
|
913
|
+
mockWs.simulateMessage(
|
|
914
|
+
resultsFrame("hello", {
|
|
915
|
+
is_final: true,
|
|
916
|
+
speech_final: true,
|
|
917
|
+
words: [{ word: "hello", language: "en" }],
|
|
918
|
+
}),
|
|
919
|
+
);
|
|
920
|
+
|
|
921
|
+
const finals = events.filter((e) => e.type === "final");
|
|
922
|
+
expect(finals).toHaveLength(1);
|
|
923
|
+
expect(finals[0]).toEqual({
|
|
924
|
+
type: "final",
|
|
925
|
+
text: "hello",
|
|
926
|
+
languages: ["en"],
|
|
927
|
+
});
|
|
928
|
+
});
|
|
929
|
+
|
|
930
|
+
test("language tags reset between aggregated utterances", async () => {
|
|
931
|
+
const { events } = await startSession({ utteranceBoundaryFinals: true });
|
|
932
|
+
|
|
933
|
+
mockWs.simulateMessage(
|
|
934
|
+
resultsFrame("hola", {
|
|
935
|
+
is_final: true,
|
|
936
|
+
speech_final: true,
|
|
937
|
+
words: [{ word: "hola", language: "es" }],
|
|
938
|
+
}),
|
|
939
|
+
);
|
|
940
|
+
mockWs.simulateMessage(
|
|
941
|
+
resultsFrame("hello", {
|
|
942
|
+
is_final: true,
|
|
943
|
+
speech_final: true,
|
|
944
|
+
words: [{ word: "hello", language: "en" }],
|
|
945
|
+
}),
|
|
946
|
+
);
|
|
947
|
+
|
|
948
|
+
const finals = events.filter((e) => e.type === "final");
|
|
949
|
+
expect(finals).toHaveLength(2);
|
|
950
|
+
expect(finals[1]).toEqual({
|
|
951
|
+
type: "final",
|
|
952
|
+
text: "hello",
|
|
953
|
+
languages: ["en"],
|
|
954
|
+
});
|
|
955
|
+
});
|
|
674
956
|
});
|
|
675
957
|
|
|
676
958
|
// ─────────────────────────────────────────────────────────────────
|
|
@@ -29,10 +29,12 @@
|
|
|
29
29
|
* - All timers and listeners are cleaned up on close to prevent leaks.
|
|
30
30
|
*/
|
|
31
31
|
|
|
32
|
+
import { rankLanguages } from "../../stt/language-metadata.js";
|
|
32
33
|
import type {
|
|
33
34
|
StreamingTranscriber,
|
|
34
35
|
SttStreamServerEvent,
|
|
35
36
|
} from "../../stt/types.js";
|
|
37
|
+
import { baseLanguageSubtag } from "../../util/language-subtag.js";
|
|
36
38
|
import { getLogger } from "../../util/logger.js";
|
|
37
39
|
|
|
38
40
|
const log = getLogger("deepgram-realtime");
|
|
@@ -208,6 +210,11 @@ interface DeepgramStreamWord {
|
|
|
208
210
|
confidence?: number;
|
|
209
211
|
start?: number;
|
|
210
212
|
end?: number;
|
|
213
|
+
/**
|
|
214
|
+
* BCP-47 tag of the language this word was spoken in. Present only on
|
|
215
|
+
* code-switching models (nova-3 with `language=multi`).
|
|
216
|
+
*/
|
|
217
|
+
language?: string;
|
|
211
218
|
}
|
|
212
219
|
|
|
213
220
|
/**
|
|
@@ -225,11 +232,23 @@ interface DeepgramStreamAlternative {
|
|
|
225
232
|
speaker?: number;
|
|
226
233
|
/** Per-word speaker tags when diarization is enabled. */
|
|
227
234
|
words?: DeepgramStreamWord[];
|
|
235
|
+
/**
|
|
236
|
+
* Detected languages for the chunk in dominance order. Emitted by
|
|
237
|
+
* code-switching models; the container varies by API version, so
|
|
238
|
+
* {@link DeepgramStreamChannel.languages} is checked as well.
|
|
239
|
+
*/
|
|
240
|
+
languages?: string[];
|
|
228
241
|
}
|
|
229
242
|
|
|
230
243
|
/** A channel within a Deepgram streaming response. */
|
|
231
244
|
interface DeepgramStreamChannel {
|
|
232
245
|
alternatives?: DeepgramStreamAlternative[];
|
|
246
|
+
/**
|
|
247
|
+
* Detected languages for the chunk in dominance order. Alternate
|
|
248
|
+
* container for {@link DeepgramStreamAlternative.languages} on some
|
|
249
|
+
* API versions.
|
|
250
|
+
*/
|
|
251
|
+
languages?: string[];
|
|
233
252
|
}
|
|
234
253
|
|
|
235
254
|
/**
|
|
@@ -343,6 +362,15 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
343
362
|
*/
|
|
344
363
|
private pendingFinalSegments: string[] = [];
|
|
345
364
|
|
|
365
|
+
/**
|
|
366
|
+
* Raw detected-language tags for the withheld segments, accumulated
|
|
367
|
+
* alongside {@link pendingFinalSegments} and ranked into the event's
|
|
368
|
+
* `languages` when the utterance flushes. Cleared wherever the pending
|
|
369
|
+
* segments are cleared. Only populated when
|
|
370
|
+
* {@link utteranceBoundaryFinals} is enabled.
|
|
371
|
+
*/
|
|
372
|
+
private pendingLanguageTags: string[] = [];
|
|
373
|
+
|
|
346
374
|
/** The live WebSocket connection, set during start(). */
|
|
347
375
|
private ws: WsLike | null = null;
|
|
348
376
|
|
|
@@ -748,6 +776,12 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
748
776
|
* words — see {@link extractSpeakerLabel}. Confidence is taken from
|
|
749
777
|
* the top alternative when present.
|
|
750
778
|
*
|
|
779
|
+
* Code-switching models (nova-3 with `language=multi`) tag detected
|
|
780
|
+
* languages per word and per container. When present, these become the
|
|
781
|
+
* dominance-ranked `languages` field on the emitted events (see
|
|
782
|
+
* {@link extractLanguages}). The field is omitted when the frame
|
|
783
|
+
* carries no language metadata.
|
|
784
|
+
*
|
|
751
785
|
* We emit:
|
|
752
786
|
* - `partial` for `is_final: false` frames (if interim results enabled).
|
|
753
787
|
* - `final` for `is_final: true` frames.
|
|
@@ -788,6 +822,7 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
788
822
|
typeof alternative?.confidence === "number"
|
|
789
823
|
? alternative.confidence
|
|
790
824
|
: undefined;
|
|
825
|
+
const languages = extractLanguages(frame.channel, alternative);
|
|
791
826
|
|
|
792
827
|
if (frame.is_final) {
|
|
793
828
|
if (this.utteranceBoundaryFinals) {
|
|
@@ -795,6 +830,16 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
795
830
|
// Finalize flush is a forced boundary — flush what is pending.
|
|
796
831
|
if (text.length > 0) {
|
|
797
832
|
this.pendingFinalSegments.push(text);
|
|
833
|
+
// Collect language tags only from frames that contributed text
|
|
834
|
+
// so the flushed metadata stays aligned with the emitted
|
|
835
|
+
// transcript (empty frames may still carry tags, but they
|
|
836
|
+
// describe no emitted words). Raw per-word tags are preferred
|
|
837
|
+
// over the frame's ranked list so cross-frame frequency
|
|
838
|
+
// weighting survives until the flush ranks the whole utterance.
|
|
839
|
+
const wordTags = collectWordLanguageTags(alternative);
|
|
840
|
+
this.pendingLanguageTags.push(
|
|
841
|
+
...(wordTags.length > 0 ? wordTags : languages),
|
|
842
|
+
);
|
|
798
843
|
}
|
|
799
844
|
if (frame.speech_final || fromFinalize) {
|
|
800
845
|
this.flushPendingUtterance();
|
|
@@ -807,6 +852,7 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
807
852
|
text,
|
|
808
853
|
...(speakerLabel !== undefined ? { speakerLabel } : {}),
|
|
809
854
|
...(confidence !== undefined ? { confidence } : {}),
|
|
855
|
+
...(languages.length > 0 ? { languages } : {}),
|
|
810
856
|
// Mark the finalize flush so consumers can attribute it to the
|
|
811
857
|
// utterance that requested the flush rather than new speech.
|
|
812
858
|
...(fromFinalize ? { fromFinalize: true } : {}),
|
|
@@ -819,6 +865,7 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
819
865
|
text,
|
|
820
866
|
...(speakerLabel !== undefined ? { speakerLabel } : {}),
|
|
821
867
|
...(confidence !== undefined ? { confidence } : {}),
|
|
868
|
+
...(languages.length > 0 ? { languages } : {}),
|
|
822
869
|
});
|
|
823
870
|
}
|
|
824
871
|
|
|
@@ -917,16 +964,27 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
917
964
|
|
|
918
965
|
/**
|
|
919
966
|
* Emit a single aggregated `final` for the withheld `is_final` segments
|
|
920
|
-
* of the current utterance
|
|
921
|
-
*
|
|
967
|
+
* of the current utterance, carrying the dominance-ranked detected
|
|
968
|
+
* languages accumulated alongside them (field omitted when no segment
|
|
969
|
+
* carried language metadata). No-op when nothing is pending, so
|
|
970
|
+
* boundary signals over silence emit nothing.
|
|
922
971
|
*/
|
|
923
972
|
private flushPendingUtterance(): void {
|
|
924
973
|
if (this.pendingFinalSegments.length === 0) {
|
|
974
|
+
// Tags accumulate only alongside text, but clear defensively so a
|
|
975
|
+
// future drift cannot leak one utterance's tags into the next.
|
|
976
|
+
this.pendingLanguageTags = [];
|
|
925
977
|
return;
|
|
926
978
|
}
|
|
927
979
|
const text = this.pendingFinalSegments.join(" ");
|
|
980
|
+
const languages = rankLanguages(this.pendingLanguageTags);
|
|
928
981
|
this.pendingFinalSegments = [];
|
|
929
|
-
this.
|
|
982
|
+
this.pendingLanguageTags = [];
|
|
983
|
+
this.emitEvent({
|
|
984
|
+
type: "final",
|
|
985
|
+
text,
|
|
986
|
+
...(languages.length > 0 ? { languages } : {}),
|
|
987
|
+
});
|
|
930
988
|
}
|
|
931
989
|
|
|
932
990
|
/**
|
|
@@ -1222,6 +1280,62 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
1222
1280
|
* contract on {@link SttStreamServerPartialEvent} /
|
|
1223
1281
|
* {@link SttStreamServerFinalEvent}.
|
|
1224
1282
|
*/
|
|
1283
|
+
/**
|
|
1284
|
+
* Derive the detected languages for a chunk, most dominant first.
|
|
1285
|
+
*
|
|
1286
|
+
* Code-switching models tag languages in two shapes:
|
|
1287
|
+
* 1. Per-word `language` tags on `alternatives[0].words[]`, the richest
|
|
1288
|
+
* signal; ranked by frequency via {@link rankLanguages} (ties broken
|
|
1289
|
+
* by first appearance).
|
|
1290
|
+
* 2. A container-level `languages` array in dominance order, attached to
|
|
1291
|
+
* the alternative or (on some API versions) the channel. Used as the
|
|
1292
|
+
* fallback when no word carries a tag; normalized and deduped with
|
|
1293
|
+
* the provider's order preserved.
|
|
1294
|
+
*
|
|
1295
|
+
* Returns `[]` when the frame carries no language metadata: callers omit
|
|
1296
|
+
* the event fields entirely so absence stays distinguishable from a
|
|
1297
|
+
* detected language.
|
|
1298
|
+
*/
|
|
1299
|
+
function extractLanguages(
|
|
1300
|
+
channel: DeepgramStreamChannel | undefined,
|
|
1301
|
+
alternative: DeepgramStreamAlternative | undefined,
|
|
1302
|
+
): string[] {
|
|
1303
|
+
const wordTags = collectWordLanguageTags(alternative);
|
|
1304
|
+
if (wordTags.length > 0) {
|
|
1305
|
+
return rankLanguages(wordTags);
|
|
1306
|
+
}
|
|
1307
|
+
|
|
1308
|
+
const container = Array.isArray(alternative?.languages)
|
|
1309
|
+
? alternative.languages
|
|
1310
|
+
: Array.isArray(channel?.languages)
|
|
1311
|
+
? channel.languages
|
|
1312
|
+
: [];
|
|
1313
|
+
const deduped = new Set(
|
|
1314
|
+
container
|
|
1315
|
+
.filter((tag): tag is string => typeof tag === "string")
|
|
1316
|
+
.flatMap((tag) => {
|
|
1317
|
+
const base = baseLanguageSubtag(tag);
|
|
1318
|
+
return base !== undefined ? [base] : [];
|
|
1319
|
+
}),
|
|
1320
|
+
);
|
|
1321
|
+
return [...deduped];
|
|
1322
|
+
}
|
|
1323
|
+
|
|
1324
|
+
/**
|
|
1325
|
+
* Collect the raw per-word `language` tags of a chunk, in word order and
|
|
1326
|
+
* without ranking or normalization. Used both for per-frame ranking in
|
|
1327
|
+
* {@link extractLanguages} and for cross-frame accumulation in
|
|
1328
|
+
* utterance-boundary mode, where ranking is deferred to the flush so
|
|
1329
|
+
* frequency weighting spans the whole utterance.
|
|
1330
|
+
*/
|
|
1331
|
+
function collectWordLanguageTags(
|
|
1332
|
+
alternative: DeepgramStreamAlternative | undefined,
|
|
1333
|
+
): string[] {
|
|
1334
|
+
return (alternative?.words ?? []).flatMap((word) =>
|
|
1335
|
+
typeof word.language === "string" ? [word.language] : [],
|
|
1336
|
+
);
|
|
1337
|
+
}
|
|
1338
|
+
|
|
1225
1339
|
function extractSpeakerLabel(
|
|
1226
1340
|
alternative: DeepgramStreamAlternative | undefined,
|
|
1227
1341
|
): string | undefined {
|