@vellumai/assistant 0.11.3-staging.2 → 0.11.3-staging.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/__tests__/call-controller.test.ts +120 -0
- package/src/__tests__/config-loader-backfill.test.ts +19 -0
- package/src/__tests__/config-schema.test.ts +139 -5
- package/src/__tests__/events-dev-bypass-actor.test.ts +112 -1
- package/src/__tests__/media-stream-output.test.ts +175 -0
- package/src/__tests__/media-stream-stt-session.test.ts +67 -0
- package/src/calls/__tests__/tts-text-sanitizer.test.ts +13 -0
- package/src/calls/__tests__/voice-session-bridge.test.ts +81 -12
- package/src/calls/__tests__/voice-triage-escalate.test.ts +106 -0
- package/src/calls/call-controller.ts +30 -2
- package/src/calls/call-speech-output.ts +12 -4
- package/src/calls/call-transport.ts +18 -1
- package/src/calls/media-stream-output.ts +53 -5
- package/src/calls/media-stream-server.ts +12 -0
- package/src/calls/media-stream-stt-session.ts +34 -0
- package/src/calls/telephony-synthesis-language.ts +84 -0
- package/src/calls/tts-text-sanitizer.ts +12 -5
- package/src/calls/voice-session-bridge.ts +31 -3
- package/src/calls/voice-triage-escalate.ts +52 -5
- package/src/config/bundled-skills/phone-calls/references/CONFIG.md +15 -15
- package/src/config/loader.ts +5 -0
- package/src/config/schemas/calls.ts +0 -4
- package/src/config/schemas/tts.ts +63 -0
- package/src/live-voice/__tests__/front-decision.test.ts +120 -0
- package/src/live-voice/__tests__/live-voice-events.test.ts +1 -1
- package/src/live-voice/__tests__/live-voice-progress.test.ts +178 -11
- package/src/live-voice/__tests__/live-voice-stt.test.ts +304 -1
- package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +102 -2
- package/src/live-voice/__tests__/live-voice-tts.test.ts +148 -2
- package/src/live-voice/__tests__/progress-phrases.test.ts +167 -0
- package/src/live-voice/front-decision.ts +50 -3
- package/src/live-voice/live-voice-session.ts +202 -25
- package/src/live-voice/live-voice-tts.ts +18 -2
- package/src/live-voice/progress-phrases.ts +105 -2
- package/src/providers/speech-to-text/deepgram-realtime.test.ts +283 -1
- package/src/providers/speech-to-text/deepgram-realtime.ts +117 -3
- package/src/providers/speech-to-text/provider-catalog.ts +38 -0
- package/src/runtime/__tests__/local-actor-identity-force-refresh.test.ts +104 -0
- package/src/runtime/assistant-event-hub.ts +23 -0
- package/src/runtime/local-actor-identity.ts +18 -5
- package/src/runtime/routes/__tests__/sse-actor-principal-heal.test.ts +165 -0
- package/src/runtime/routes/__tests__/surface-action-routes.test.ts +2 -0
- package/src/runtime/routes/events-routes.ts +17 -16
- package/src/runtime/routes/sse-actor-principal-heal.ts +113 -0
- package/src/stt/__tests__/language-metadata.test.ts +85 -0
- package/src/stt/language-metadata.ts +65 -0
- package/src/stt/types.ts +16 -0
- package/src/tts/__tests__/provider-adapters.test.ts +147 -0
- package/src/tts/__tests__/speakable-segments.test.ts +470 -0
- package/src/tts/language-voices.ts +23 -0
- package/src/tts/providers/deepgram-provider.ts +3 -1
- package/src/tts/providers/elevenlabs-provider.ts +73 -1
- package/src/tts/providers/xai-provider.ts +28 -2
- package/src/tts/speakable-segments.ts +293 -23
- package/src/tts/synthesis-stream.ts +7 -0
- package/src/tts/types.ts +7 -0
- package/src/util/__tests__/language-subtag.test.ts +54 -0
- package/src/util/language-subtag.ts +43 -0
- package/src/util/unicode.ts +1 -1
|
@@ -102,7 +102,11 @@ function resultsFrame(
|
|
|
102
102
|
is_final?: boolean;
|
|
103
103
|
speech_final?: boolean;
|
|
104
104
|
from_finalize?: boolean;
|
|
105
|
-
words?: { word: string; speaker?: number }[];
|
|
105
|
+
words?: { word: string; speaker?: number; language?: string }[];
|
|
106
|
+
/** Container-level detected languages on the alternative. */
|
|
107
|
+
alternativeLanguages?: string[];
|
|
108
|
+
/** Container-level detected languages on the channel. */
|
|
109
|
+
channelLanguages?: string[];
|
|
106
110
|
} = {},
|
|
107
111
|
): string {
|
|
108
112
|
return JSON.stringify({
|
|
@@ -121,8 +125,14 @@ function resultsFrame(
|
|
|
121
125
|
transcript,
|
|
122
126
|
confidence: 0.95,
|
|
123
127
|
...(options.words ? { words: options.words } : {}),
|
|
128
|
+
...(options.alternativeLanguages
|
|
129
|
+
? { languages: options.alternativeLanguages }
|
|
130
|
+
: {}),
|
|
124
131
|
},
|
|
125
132
|
],
|
|
133
|
+
...(options.channelLanguages
|
|
134
|
+
? { languages: options.channelLanguages }
|
|
135
|
+
: {}),
|
|
126
136
|
},
|
|
127
137
|
});
|
|
128
138
|
}
|
|
@@ -542,6 +552,154 @@ describe("DeepgramRealtimeTranscriber", () => {
|
|
|
542
552
|
(globalThis as Record<string, unknown>).WebSocket = origWs;
|
|
543
553
|
});
|
|
544
554
|
|
|
555
|
+
// ─────────────────────────────────────────────────────────────────
|
|
556
|
+
// Language metadata (nova-3 multi code-switching)
|
|
557
|
+
// ─────────────────────────────────────────────────────────────────
|
|
558
|
+
|
|
559
|
+
describe("language metadata", () => {
|
|
560
|
+
test("ranks per-word language tags by dominance on final events", async () => {
|
|
561
|
+
const { events } = await startSession();
|
|
562
|
+
|
|
563
|
+
mockWs.simulateMessage(
|
|
564
|
+
resultsFrame("hello world hola", {
|
|
565
|
+
is_final: true,
|
|
566
|
+
words: [
|
|
567
|
+
{ word: "hello", language: "en" },
|
|
568
|
+
{ word: "world", language: "en" },
|
|
569
|
+
{ word: "hola", language: "es" },
|
|
570
|
+
],
|
|
571
|
+
}),
|
|
572
|
+
);
|
|
573
|
+
|
|
574
|
+
expect(events).toHaveLength(1);
|
|
575
|
+
expect(events[0]).toEqual({
|
|
576
|
+
type: "final",
|
|
577
|
+
text: "hello world hola",
|
|
578
|
+
confidence: 0.95,
|
|
579
|
+
languages: ["en", "es"],
|
|
580
|
+
});
|
|
581
|
+
});
|
|
582
|
+
|
|
583
|
+
test("omits the languages field entirely when no language metadata is present", async () => {
|
|
584
|
+
const { events } = await startSession();
|
|
585
|
+
|
|
586
|
+
mockWs.simulateMessage(
|
|
587
|
+
resultsFrame("hello world", {
|
|
588
|
+
is_final: true,
|
|
589
|
+
words: [{ word: "hello" }, { word: "world" }],
|
|
590
|
+
}),
|
|
591
|
+
);
|
|
592
|
+
mockWs.simulateMessage(resultsFrame("still typing", { is_final: false }));
|
|
593
|
+
|
|
594
|
+
expect(events).toHaveLength(2);
|
|
595
|
+
for (const event of events) {
|
|
596
|
+
// The keys must not exist at all, not just be undefined-valued.
|
|
597
|
+
expect("language" in event).toBe(false);
|
|
598
|
+
expect("languages" in event).toBe(false);
|
|
599
|
+
}
|
|
600
|
+
});
|
|
601
|
+
|
|
602
|
+
test("normalizes regional tags to their base subtag", async () => {
|
|
603
|
+
const { events } = await startSession();
|
|
604
|
+
|
|
605
|
+
mockWs.simulateMessage(
|
|
606
|
+
resultsFrame("hello", {
|
|
607
|
+
is_final: true,
|
|
608
|
+
words: [{ word: "hello", language: "en-US" }],
|
|
609
|
+
}),
|
|
610
|
+
);
|
|
611
|
+
|
|
612
|
+
expect(events[0]).toEqual({
|
|
613
|
+
type: "final",
|
|
614
|
+
text: "hello",
|
|
615
|
+
confidence: 0.95,
|
|
616
|
+
languages: ["en"],
|
|
617
|
+
});
|
|
618
|
+
});
|
|
619
|
+
|
|
620
|
+
test("partial events carry the fields when interim results are enabled", async () => {
|
|
621
|
+
const { events } = await startSession();
|
|
622
|
+
|
|
623
|
+
mockWs.simulateMessage(
|
|
624
|
+
resultsFrame("namaste hello", {
|
|
625
|
+
is_final: false,
|
|
626
|
+
words: [
|
|
627
|
+
{ word: "namaste", language: "hi" },
|
|
628
|
+
{ word: "hello", language: "en" },
|
|
629
|
+
{ word: "there", language: "en" },
|
|
630
|
+
],
|
|
631
|
+
}),
|
|
632
|
+
);
|
|
633
|
+
|
|
634
|
+
expect(events).toHaveLength(1);
|
|
635
|
+
expect(events[0]).toEqual({
|
|
636
|
+
type: "partial",
|
|
637
|
+
text: "namaste hello",
|
|
638
|
+
confidence: 0.95,
|
|
639
|
+
languages: ["en", "hi"],
|
|
640
|
+
});
|
|
641
|
+
});
|
|
642
|
+
|
|
643
|
+
test("falls back to the alternative-level languages array when words carry no tags", async () => {
|
|
644
|
+
const { events } = await startSession();
|
|
645
|
+
|
|
646
|
+
mockWs.simulateMessage(
|
|
647
|
+
resultsFrame("mixed speech", {
|
|
648
|
+
is_final: true,
|
|
649
|
+
words: [{ word: "mixed" }, { word: "speech" }],
|
|
650
|
+
alternativeLanguages: ["es-419", "en", "ES"],
|
|
651
|
+
}),
|
|
652
|
+
);
|
|
653
|
+
|
|
654
|
+
expect(events[0]).toEqual({
|
|
655
|
+
type: "final",
|
|
656
|
+
text: "mixed speech",
|
|
657
|
+
confidence: 0.95,
|
|
658
|
+
languages: ["es", "en"],
|
|
659
|
+
});
|
|
660
|
+
});
|
|
661
|
+
|
|
662
|
+
test("falls back to the channel-level languages array when the alternative has none", async () => {
|
|
663
|
+
const { events } = await startSession();
|
|
664
|
+
|
|
665
|
+
mockWs.simulateMessage(
|
|
666
|
+
resultsFrame("bonjour", {
|
|
667
|
+
is_final: true,
|
|
668
|
+
channelLanguages: ["fr", "en"],
|
|
669
|
+
}),
|
|
670
|
+
);
|
|
671
|
+
|
|
672
|
+
expect(events[0]).toEqual({
|
|
673
|
+
type: "final",
|
|
674
|
+
text: "bonjour",
|
|
675
|
+
confidence: 0.95,
|
|
676
|
+
languages: ["fr", "en"],
|
|
677
|
+
});
|
|
678
|
+
});
|
|
679
|
+
|
|
680
|
+
test("per-word tags take precedence over container arrays", async () => {
|
|
681
|
+
const { events } = await startSession();
|
|
682
|
+
|
|
683
|
+
mockWs.simulateMessage(
|
|
684
|
+
resultsFrame("hola amigo", {
|
|
685
|
+
is_final: true,
|
|
686
|
+
words: [
|
|
687
|
+
{ word: "hola", language: "es" },
|
|
688
|
+
{ word: "amigo", language: "es" },
|
|
689
|
+
],
|
|
690
|
+
alternativeLanguages: ["en", "es"],
|
|
691
|
+
}),
|
|
692
|
+
);
|
|
693
|
+
|
|
694
|
+
expect(events[0]).toEqual({
|
|
695
|
+
type: "final",
|
|
696
|
+
text: "hola amigo",
|
|
697
|
+
confidence: 0.95,
|
|
698
|
+
languages: ["es"],
|
|
699
|
+
});
|
|
700
|
+
});
|
|
701
|
+
});
|
|
702
|
+
|
|
545
703
|
// ─────────────────────────────────────────────────────────────────
|
|
546
704
|
// Multi-event sequence
|
|
547
705
|
// ─────────────────────────────────────────────────────────────────
|
|
@@ -671,6 +829,130 @@ describe("DeepgramRealtimeTranscriber", () => {
|
|
|
671
829
|
"second utterance",
|
|
672
830
|
]);
|
|
673
831
|
});
|
|
832
|
+
|
|
833
|
+
test("aggregated final ranks language tags across all withheld frames", async () => {
|
|
834
|
+
const { events } = await startSession({ utteranceBoundaryFinals: true });
|
|
835
|
+
|
|
836
|
+
// Raw per-word tags are accumulated, so "es" (2 words) outranks
|
|
837
|
+
// "en" (1 word) even though "en" arrived in the earlier frame.
|
|
838
|
+
mockWs.simulateMessage(
|
|
839
|
+
resultsFrame("hello", {
|
|
840
|
+
is_final: true,
|
|
841
|
+
words: [{ word: "hello", language: "en" }],
|
|
842
|
+
}),
|
|
843
|
+
);
|
|
844
|
+
mockWs.simulateMessage(
|
|
845
|
+
resultsFrame("hola amigo", {
|
|
846
|
+
is_final: true,
|
|
847
|
+
speech_final: true,
|
|
848
|
+
words: [
|
|
849
|
+
{ word: "hola", language: "es" },
|
|
850
|
+
{ word: "amigo", language: "es" },
|
|
851
|
+
],
|
|
852
|
+
}),
|
|
853
|
+
);
|
|
854
|
+
|
|
855
|
+
const finals = events.filter((e) => e.type === "final");
|
|
856
|
+
expect(finals).toHaveLength(1);
|
|
857
|
+
expect(finals[0]).toEqual({
|
|
858
|
+
type: "final",
|
|
859
|
+
text: "hello hola amigo",
|
|
860
|
+
languages: ["es", "en"],
|
|
861
|
+
});
|
|
862
|
+
});
|
|
863
|
+
|
|
864
|
+
test("UtteranceEnd flush carries the accumulated language metadata", async () => {
|
|
865
|
+
const { events } = await startSession({ utteranceBoundaryFinals: true });
|
|
866
|
+
|
|
867
|
+
mockWs.simulateMessage(
|
|
868
|
+
resultsFrame("bonjour", {
|
|
869
|
+
is_final: true,
|
|
870
|
+
words: [{ word: "bonjour", language: "fr" }],
|
|
871
|
+
}),
|
|
872
|
+
);
|
|
873
|
+
mockWs.simulateMessage(
|
|
874
|
+
resultsFrame("hello there", {
|
|
875
|
+
is_final: true,
|
|
876
|
+
alternativeLanguages: ["en"],
|
|
877
|
+
}),
|
|
878
|
+
);
|
|
879
|
+
mockWs.simulateMessage(utteranceEndFrame());
|
|
880
|
+
|
|
881
|
+
const finals = events.filter((e) => e.type === "final");
|
|
882
|
+
expect(finals).toHaveLength(1);
|
|
883
|
+
expect(finals[0]).toEqual({
|
|
884
|
+
type: "final",
|
|
885
|
+
text: "bonjour hello there",
|
|
886
|
+
languages: ["fr", "en"],
|
|
887
|
+
});
|
|
888
|
+
});
|
|
889
|
+
|
|
890
|
+
test("aggregated final omits language fields when no withheld frame carried tags", async () => {
|
|
891
|
+
const { events } = await startSession({ utteranceBoundaryFinals: true });
|
|
892
|
+
|
|
893
|
+
mockWs.simulateMessage(resultsFrame("no tags", { is_final: true }));
|
|
894
|
+
mockWs.simulateMessage(
|
|
895
|
+
resultsFrame("at all", { is_final: true, speech_final: true }),
|
|
896
|
+
);
|
|
897
|
+
|
|
898
|
+
const finals = events.filter((e) => e.type === "final");
|
|
899
|
+
expect(finals).toHaveLength(1);
|
|
900
|
+
expect(finals[0]).toEqual({ type: "final", text: "no tags at all" });
|
|
901
|
+
expect("language" in finals[0]!).toBe(false);
|
|
902
|
+
expect("languages" in finals[0]!).toBe(false);
|
|
903
|
+
});
|
|
904
|
+
|
|
905
|
+
test("tags on empty-text frames do not leak into the aggregated final", async () => {
|
|
906
|
+
const { events } = await startSession({ utteranceBoundaryFinals: true });
|
|
907
|
+
|
|
908
|
+
// A silence segment may still carry container-level tags; only
|
|
909
|
+
// frames that contributed transcript text feed the metadata.
|
|
910
|
+
mockWs.simulateMessage(
|
|
911
|
+
resultsFrame("", { is_final: true, alternativeLanguages: ["fr"] }),
|
|
912
|
+
);
|
|
913
|
+
mockWs.simulateMessage(
|
|
914
|
+
resultsFrame("hello", {
|
|
915
|
+
is_final: true,
|
|
916
|
+
speech_final: true,
|
|
917
|
+
words: [{ word: "hello", language: "en" }],
|
|
918
|
+
}),
|
|
919
|
+
);
|
|
920
|
+
|
|
921
|
+
const finals = events.filter((e) => e.type === "final");
|
|
922
|
+
expect(finals).toHaveLength(1);
|
|
923
|
+
expect(finals[0]).toEqual({
|
|
924
|
+
type: "final",
|
|
925
|
+
text: "hello",
|
|
926
|
+
languages: ["en"],
|
|
927
|
+
});
|
|
928
|
+
});
|
|
929
|
+
|
|
930
|
+
test("language tags reset between aggregated utterances", async () => {
|
|
931
|
+
const { events } = await startSession({ utteranceBoundaryFinals: true });
|
|
932
|
+
|
|
933
|
+
mockWs.simulateMessage(
|
|
934
|
+
resultsFrame("hola", {
|
|
935
|
+
is_final: true,
|
|
936
|
+
speech_final: true,
|
|
937
|
+
words: [{ word: "hola", language: "es" }],
|
|
938
|
+
}),
|
|
939
|
+
);
|
|
940
|
+
mockWs.simulateMessage(
|
|
941
|
+
resultsFrame("hello", {
|
|
942
|
+
is_final: true,
|
|
943
|
+
speech_final: true,
|
|
944
|
+
words: [{ word: "hello", language: "en" }],
|
|
945
|
+
}),
|
|
946
|
+
);
|
|
947
|
+
|
|
948
|
+
const finals = events.filter((e) => e.type === "final");
|
|
949
|
+
expect(finals).toHaveLength(2);
|
|
950
|
+
expect(finals[1]).toEqual({
|
|
951
|
+
type: "final",
|
|
952
|
+
text: "hello",
|
|
953
|
+
languages: ["en"],
|
|
954
|
+
});
|
|
955
|
+
});
|
|
674
956
|
});
|
|
675
957
|
|
|
676
958
|
// ─────────────────────────────────────────────────────────────────
|
|
@@ -29,10 +29,12 @@
|
|
|
29
29
|
* - All timers and listeners are cleaned up on close to prevent leaks.
|
|
30
30
|
*/
|
|
31
31
|
|
|
32
|
+
import { rankLanguages } from "../../stt/language-metadata.js";
|
|
32
33
|
import type {
|
|
33
34
|
StreamingTranscriber,
|
|
34
35
|
SttStreamServerEvent,
|
|
35
36
|
} from "../../stt/types.js";
|
|
37
|
+
import { baseLanguageSubtag } from "../../util/language-subtag.js";
|
|
36
38
|
import { getLogger } from "../../util/logger.js";
|
|
37
39
|
|
|
38
40
|
const log = getLogger("deepgram-realtime");
|
|
@@ -208,6 +210,11 @@ interface DeepgramStreamWord {
|
|
|
208
210
|
confidence?: number;
|
|
209
211
|
start?: number;
|
|
210
212
|
end?: number;
|
|
213
|
+
/**
|
|
214
|
+
* BCP-47 tag of the language this word was spoken in. Present only on
|
|
215
|
+
* code-switching models (nova-3 with `language=multi`).
|
|
216
|
+
*/
|
|
217
|
+
language?: string;
|
|
211
218
|
}
|
|
212
219
|
|
|
213
220
|
/**
|
|
@@ -225,11 +232,23 @@ interface DeepgramStreamAlternative {
|
|
|
225
232
|
speaker?: number;
|
|
226
233
|
/** Per-word speaker tags when diarization is enabled. */
|
|
227
234
|
words?: DeepgramStreamWord[];
|
|
235
|
+
/**
|
|
236
|
+
* Detected languages for the chunk in dominance order. Emitted by
|
|
237
|
+
* code-switching models; the container varies by API version, so
|
|
238
|
+
* {@link DeepgramStreamChannel.languages} is checked as well.
|
|
239
|
+
*/
|
|
240
|
+
languages?: string[];
|
|
228
241
|
}
|
|
229
242
|
|
|
230
243
|
/** A channel within a Deepgram streaming response. */
|
|
231
244
|
interface DeepgramStreamChannel {
|
|
232
245
|
alternatives?: DeepgramStreamAlternative[];
|
|
246
|
+
/**
|
|
247
|
+
* Detected languages for the chunk in dominance order. Alternate
|
|
248
|
+
* container for {@link DeepgramStreamAlternative.languages} on some
|
|
249
|
+
* API versions.
|
|
250
|
+
*/
|
|
251
|
+
languages?: string[];
|
|
233
252
|
}
|
|
234
253
|
|
|
235
254
|
/**
|
|
@@ -343,6 +362,15 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
343
362
|
*/
|
|
344
363
|
private pendingFinalSegments: string[] = [];
|
|
345
364
|
|
|
365
|
+
/**
|
|
366
|
+
* Raw detected-language tags for the withheld segments, accumulated
|
|
367
|
+
* alongside {@link pendingFinalSegments} and ranked into the event's
|
|
368
|
+
* `languages` when the utterance flushes. Cleared wherever the pending
|
|
369
|
+
* segments are cleared. Only populated when
|
|
370
|
+
* {@link utteranceBoundaryFinals} is enabled.
|
|
371
|
+
*/
|
|
372
|
+
private pendingLanguageTags: string[] = [];
|
|
373
|
+
|
|
346
374
|
/** The live WebSocket connection, set during start(). */
|
|
347
375
|
private ws: WsLike | null = null;
|
|
348
376
|
|
|
@@ -748,6 +776,12 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
748
776
|
* words — see {@link extractSpeakerLabel}. Confidence is taken from
|
|
749
777
|
* the top alternative when present.
|
|
750
778
|
*
|
|
779
|
+
* Code-switching models (nova-3 with `language=multi`) tag detected
|
|
780
|
+
* languages per word and per container. When present, these become the
|
|
781
|
+
* dominance-ranked `languages` field on the emitted events (see
|
|
782
|
+
* {@link extractLanguages}). The field is omitted when the frame
|
|
783
|
+
* carries no language metadata.
|
|
784
|
+
*
|
|
751
785
|
* We emit:
|
|
752
786
|
* - `partial` for `is_final: false` frames (if interim results enabled).
|
|
753
787
|
* - `final` for `is_final: true` frames.
|
|
@@ -788,6 +822,7 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
788
822
|
typeof alternative?.confidence === "number"
|
|
789
823
|
? alternative.confidence
|
|
790
824
|
: undefined;
|
|
825
|
+
const languages = extractLanguages(frame.channel, alternative);
|
|
791
826
|
|
|
792
827
|
if (frame.is_final) {
|
|
793
828
|
if (this.utteranceBoundaryFinals) {
|
|
@@ -795,6 +830,16 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
795
830
|
// Finalize flush is a forced boundary — flush what is pending.
|
|
796
831
|
if (text.length > 0) {
|
|
797
832
|
this.pendingFinalSegments.push(text);
|
|
833
|
+
// Collect language tags only from frames that contributed text
|
|
834
|
+
// so the flushed metadata stays aligned with the emitted
|
|
835
|
+
// transcript (empty frames may still carry tags, but they
|
|
836
|
+
// describe no emitted words). Raw per-word tags are preferred
|
|
837
|
+
// over the frame's ranked list so cross-frame frequency
|
|
838
|
+
// weighting survives until the flush ranks the whole utterance.
|
|
839
|
+
const wordTags = collectWordLanguageTags(alternative);
|
|
840
|
+
this.pendingLanguageTags.push(
|
|
841
|
+
...(wordTags.length > 0 ? wordTags : languages),
|
|
842
|
+
);
|
|
798
843
|
}
|
|
799
844
|
if (frame.speech_final || fromFinalize) {
|
|
800
845
|
this.flushPendingUtterance();
|
|
@@ -807,6 +852,7 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
807
852
|
text,
|
|
808
853
|
...(speakerLabel !== undefined ? { speakerLabel } : {}),
|
|
809
854
|
...(confidence !== undefined ? { confidence } : {}),
|
|
855
|
+
...(languages.length > 0 ? { languages } : {}),
|
|
810
856
|
// Mark the finalize flush so consumers can attribute it to the
|
|
811
857
|
// utterance that requested the flush rather than new speech.
|
|
812
858
|
...(fromFinalize ? { fromFinalize: true } : {}),
|
|
@@ -819,6 +865,7 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
819
865
|
text,
|
|
820
866
|
...(speakerLabel !== undefined ? { speakerLabel } : {}),
|
|
821
867
|
...(confidence !== undefined ? { confidence } : {}),
|
|
868
|
+
...(languages.length > 0 ? { languages } : {}),
|
|
822
869
|
});
|
|
823
870
|
}
|
|
824
871
|
|
|
@@ -917,16 +964,27 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
917
964
|
|
|
918
965
|
/**
|
|
919
966
|
* Emit a single aggregated `final` for the withheld `is_final` segments
|
|
920
|
-
* of the current utterance
|
|
921
|
-
*
|
|
967
|
+
* of the current utterance, carrying the dominance-ranked detected
|
|
968
|
+
* languages accumulated alongside them (field omitted when no segment
|
|
969
|
+
* carried language metadata). No-op when nothing is pending, so
|
|
970
|
+
* boundary signals over silence emit nothing.
|
|
922
971
|
*/
|
|
923
972
|
private flushPendingUtterance(): void {
|
|
924
973
|
if (this.pendingFinalSegments.length === 0) {
|
|
974
|
+
// Tags accumulate only alongside text, but clear defensively so a
|
|
975
|
+
// future drift cannot leak one utterance's tags into the next.
|
|
976
|
+
this.pendingLanguageTags = [];
|
|
925
977
|
return;
|
|
926
978
|
}
|
|
927
979
|
const text = this.pendingFinalSegments.join(" ");
|
|
980
|
+
const languages = rankLanguages(this.pendingLanguageTags);
|
|
928
981
|
this.pendingFinalSegments = [];
|
|
929
|
-
this.
|
|
982
|
+
this.pendingLanguageTags = [];
|
|
983
|
+
this.emitEvent({
|
|
984
|
+
type: "final",
|
|
985
|
+
text,
|
|
986
|
+
...(languages.length > 0 ? { languages } : {}),
|
|
987
|
+
});
|
|
930
988
|
}
|
|
931
989
|
|
|
932
990
|
/**
|
|
@@ -1222,6 +1280,62 @@ export class DeepgramRealtimeTranscriber implements StreamingTranscriber {
|
|
|
1222
1280
|
* contract on {@link SttStreamServerPartialEvent} /
|
|
1223
1281
|
* {@link SttStreamServerFinalEvent}.
|
|
1224
1282
|
*/
|
|
1283
|
+
/**
|
|
1284
|
+
* Derive the detected languages for a chunk, most dominant first.
|
|
1285
|
+
*
|
|
1286
|
+
* Code-switching models tag languages in two shapes:
|
|
1287
|
+
* 1. Per-word `language` tags on `alternatives[0].words[]`, the richest
|
|
1288
|
+
* signal; ranked by frequency via {@link rankLanguages} (ties broken
|
|
1289
|
+
* by first appearance).
|
|
1290
|
+
* 2. A container-level `languages` array in dominance order, attached to
|
|
1291
|
+
* the alternative or (on some API versions) the channel. Used as the
|
|
1292
|
+
* fallback when no word carries a tag; normalized and deduped with
|
|
1293
|
+
* the provider's order preserved.
|
|
1294
|
+
*
|
|
1295
|
+
* Returns `[]` when the frame carries no language metadata: callers omit
|
|
1296
|
+
* the event fields entirely so absence stays distinguishable from a
|
|
1297
|
+
* detected language.
|
|
1298
|
+
*/
|
|
1299
|
+
function extractLanguages(
|
|
1300
|
+
channel: DeepgramStreamChannel | undefined,
|
|
1301
|
+
alternative: DeepgramStreamAlternative | undefined,
|
|
1302
|
+
): string[] {
|
|
1303
|
+
const wordTags = collectWordLanguageTags(alternative);
|
|
1304
|
+
if (wordTags.length > 0) {
|
|
1305
|
+
return rankLanguages(wordTags);
|
|
1306
|
+
}
|
|
1307
|
+
|
|
1308
|
+
const container = Array.isArray(alternative?.languages)
|
|
1309
|
+
? alternative.languages
|
|
1310
|
+
: Array.isArray(channel?.languages)
|
|
1311
|
+
? channel.languages
|
|
1312
|
+
: [];
|
|
1313
|
+
const deduped = new Set(
|
|
1314
|
+
container
|
|
1315
|
+
.filter((tag): tag is string => typeof tag === "string")
|
|
1316
|
+
.flatMap((tag) => {
|
|
1317
|
+
const base = baseLanguageSubtag(tag);
|
|
1318
|
+
return base !== undefined ? [base] : [];
|
|
1319
|
+
}),
|
|
1320
|
+
);
|
|
1321
|
+
return [...deduped];
|
|
1322
|
+
}
|
|
1323
|
+
|
|
1324
|
+
/**
|
|
1325
|
+
* Collect the raw per-word `language` tags of a chunk, in word order and
|
|
1326
|
+
* without ranking or normalization. Used both for per-frame ranking in
|
|
1327
|
+
* {@link extractLanguages} and for cross-frame accumulation in
|
|
1328
|
+
* utterance-boundary mode, where ranking is deferred to the flush so
|
|
1329
|
+
* frequency weighting spans the whole utterance.
|
|
1330
|
+
*/
|
|
1331
|
+
function collectWordLanguageTags(
|
|
1332
|
+
alternative: DeepgramStreamAlternative | undefined,
|
|
1333
|
+
): string[] {
|
|
1334
|
+
return (alternative?.words ?? []).flatMap((word) =>
|
|
1335
|
+
typeof word.language === "string" ? [word.language] : [],
|
|
1336
|
+
);
|
|
1337
|
+
}
|
|
1338
|
+
|
|
1225
1339
|
function extractSpeakerLabel(
|
|
1226
1340
|
alternative: DeepgramStreamAlternative | undefined,
|
|
1227
1341
|
): string | undefined {
|
|
@@ -16,6 +16,7 @@ import type {
|
|
|
16
16
|
SttProviderId,
|
|
17
17
|
TelephonySttMode,
|
|
18
18
|
} from "../../stt/types.js";
|
|
19
|
+
import { baseLanguageSubtag } from "../../util/language-subtag.js";
|
|
19
20
|
|
|
20
21
|
// ---------------------------------------------------------------------------
|
|
21
22
|
// Client display metadata
|
|
@@ -282,6 +283,43 @@ export function listProviderEntries(): readonly SttProviderEntry[] {
|
|
|
282
283
|
return [...CATALOG.values()];
|
|
283
284
|
}
|
|
284
285
|
|
|
286
|
+
/**
|
|
287
|
+
* A base-subtag regex over the pinned listening language. The pin is
|
|
288
|
+
* free-form workspace config, and it flows into prompt interpolation and
|
|
289
|
+
* per-language table lookups, so only a plausible ISO 639 base subtag
|
|
290
|
+
* passes; anything else (junk strings, prototype keys like "constructor")
|
|
291
|
+
* resolves as no pin.
|
|
292
|
+
*/
|
|
293
|
+
const PINNED_LANGUAGE_SUBTAG_REGEX = /^[a-z]{2,3}$/;
|
|
294
|
+
|
|
295
|
+
/**
|
|
296
|
+
* The configured `services.stt.language` pin as the caller's listening
|
|
297
|
+
* language, or undefined when the pin carries no signal.
|
|
298
|
+
*
|
|
299
|
+
* A persisted pin only counts when the provider honors manual language
|
|
300
|
+
* selection: auto-detecting providers (gemini, whisper) ignore the setting
|
|
301
|
+
* entirely, so treating it as the caller's language would force every
|
|
302
|
+
* turn into a stale pin. "multi" and blank mean auto-detect (no pin), and
|
|
303
|
+
* the value must normalize to a plausible base subtag. Shared by the
|
|
304
|
+
* telephony pre-speech prompt rule (voice-session-bridge.ts), live
|
|
305
|
+
* voice's turn language (live-voice-session.ts), and telephony synthesis
|
|
306
|
+
* (telephony-synthesis-language.ts) so the gate cannot drift.
|
|
307
|
+
*/
|
|
308
|
+
export function pinnedListeningLanguage(
|
|
309
|
+
provider: string,
|
|
310
|
+
configuredLanguage: string | undefined,
|
|
311
|
+
): string | undefined {
|
|
312
|
+
const providerHonorsLanguagePin =
|
|
313
|
+
getProviderEntry(provider as SttProviderId)?.languageSelection === "manual";
|
|
314
|
+
if (!providerHonorsLanguagePin || configuredLanguage?.trim() === "multi") {
|
|
315
|
+
return undefined;
|
|
316
|
+
}
|
|
317
|
+
const base = baseLanguageSubtag(configuredLanguage);
|
|
318
|
+
return base !== undefined && PINNED_LANGUAGE_SUBTAG_REGEX.test(base)
|
|
319
|
+
? base
|
|
320
|
+
: undefined;
|
|
321
|
+
}
|
|
322
|
+
|
|
285
323
|
/**
|
|
286
324
|
* Look up the credential-provider name for a given STT provider.
|
|
287
325
|
*
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tests for the cache-bypass option on the local guardian principal lookup.
|
|
3
|
+
*
|
|
4
|
+
* The guardian-delivery reader caches a successful read that finds no binding,
|
|
5
|
+
* and gateway-side binding writes do not invalidate the daemon's cache. A
|
|
6
|
+
* caller polling for a binding it expects to appear (the SSE actor-principal
|
|
7
|
+
* heal) therefore has to force the read, or every attempt re-reads the same
|
|
8
|
+
* empty answer until the TTL lapses.
|
|
9
|
+
*/
|
|
10
|
+
import { afterAll, beforeEach, describe, expect, mock, test } from "bun:test";
|
|
11
|
+
|
|
12
|
+
import type { GuardianDelivery } from "@vellumai/gateway-client";
|
|
13
|
+
|
|
14
|
+
let cachedResult: GuardianDelivery[] | null = null;
|
|
15
|
+
let freshResult: GuardianDelivery[] | null = null;
|
|
16
|
+
let cachedCalls = 0;
|
|
17
|
+
let freshCalls = 0;
|
|
18
|
+
|
|
19
|
+
mock.module("../../config/env.js", () => ({
|
|
20
|
+
isHttpAuthDisabled: () => true,
|
|
21
|
+
hasUngatedHttpAuthDisabled: () => false,
|
|
22
|
+
}));
|
|
23
|
+
|
|
24
|
+
mock.module("../../contacts/guardian-delivery-reader.js", () => ({
|
|
25
|
+
getGuardianDelivery: () => {
|
|
26
|
+
cachedCalls++;
|
|
27
|
+
return Promise.resolve(cachedResult);
|
|
28
|
+
},
|
|
29
|
+
getGuardianDeliveryFresh: () => {
|
|
30
|
+
freshCalls++;
|
|
31
|
+
return Promise.resolve(freshResult);
|
|
32
|
+
},
|
|
33
|
+
peekCachedGuardianDelivery: () => undefined,
|
|
34
|
+
guardianForChannel: (list: GuardianDelivery[]) => list[0],
|
|
35
|
+
}));
|
|
36
|
+
|
|
37
|
+
import {
|
|
38
|
+
findLocalGuardianPrincipalId,
|
|
39
|
+
resolveActorPrincipalIdForLocalGuardian,
|
|
40
|
+
} from "../local-actor-identity.js";
|
|
41
|
+
|
|
42
|
+
afterAll(() => {
|
|
43
|
+
mock.restore();
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
/** Minimal guardian row: only `principalId` is read here. */
|
|
47
|
+
function guardian(principalId: string): GuardianDelivery {
|
|
48
|
+
return { principalId } as unknown as GuardianDelivery;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
describe("local guardian principal lookup — cache bypass", () => {
|
|
52
|
+
beforeEach(() => {
|
|
53
|
+
cachedCalls = 0;
|
|
54
|
+
freshCalls = 0;
|
|
55
|
+
cachedResult = [];
|
|
56
|
+
freshResult = [];
|
|
57
|
+
});
|
|
58
|
+
|
|
59
|
+
test("reads the cache by default", async () => {
|
|
60
|
+
cachedResult = [guardian("guardian-cached")];
|
|
61
|
+
freshResult = [guardian("guardian-fresh")];
|
|
62
|
+
|
|
63
|
+
expect(await findLocalGuardianPrincipalId()).toBe("guardian-cached");
|
|
64
|
+
expect(cachedCalls).toBe(1);
|
|
65
|
+
expect(freshCalls).toBe(0);
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
test("forceRefresh bypasses the cache and sees a binding the cache would miss", async () => {
|
|
69
|
+
// Cached read holds the empty result from before the binding existed.
|
|
70
|
+
cachedResult = [];
|
|
71
|
+
freshResult = [guardian("guardian-fresh")];
|
|
72
|
+
|
|
73
|
+
expect(await findLocalGuardianPrincipalId()).toBeUndefined();
|
|
74
|
+
expect(await findLocalGuardianPrincipalId({ forceRefresh: true })).toBe(
|
|
75
|
+
"guardian-fresh",
|
|
76
|
+
);
|
|
77
|
+
expect(freshCalls).toBe(1);
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
test("resolveActorPrincipalIdForLocalGuardian threads forceRefresh through", async () => {
|
|
81
|
+
cachedResult = [];
|
|
82
|
+
freshResult = [guardian("guardian-fresh")];
|
|
83
|
+
|
|
84
|
+
expect(
|
|
85
|
+
await resolveActorPrincipalIdForLocalGuardian("dev-bypass"),
|
|
86
|
+
).toBeUndefined();
|
|
87
|
+
expect(
|
|
88
|
+
await resolveActorPrincipalIdForLocalGuardian("dev-bypass", {
|
|
89
|
+
forceRefresh: true,
|
|
90
|
+
}),
|
|
91
|
+
).toBe("guardian-fresh");
|
|
92
|
+
expect(freshCalls).toBe(1);
|
|
93
|
+
});
|
|
94
|
+
|
|
95
|
+
test("a non-dev-bypass principal is passed through without any lookup", async () => {
|
|
96
|
+
expect(
|
|
97
|
+
await resolveActorPrincipalIdForLocalGuardian("actor-123", {
|
|
98
|
+
forceRefresh: true,
|
|
99
|
+
}),
|
|
100
|
+
).toBe("actor-123");
|
|
101
|
+
expect(cachedCalls).toBe(0);
|
|
102
|
+
expect(freshCalls).toBe(0);
|
|
103
|
+
});
|
|
104
|
+
});
|