@absolutejs/voice 0.0.22-beta.635 → 0.0.22-beta.637
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -0
- package/dist/core/types.d.ts +23 -0
- package/dist/index.js +104 -1
- package/dist/testing/benchmark.d.ts +7 -0
- package/dist/testing/confidenceCalibration.d.ts +19 -0
- package/dist/testing/criticalFields.d.ts +22 -0
- package/dist/testing/fixtures.d.ts +2 -0
- package/dist/testing/index.d.ts +3 -0
- package/dist/testing/index.js +316 -195
- package/dist/testing/routingBenchmark.d.ts +16 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -5015,6 +5015,14 @@ That keeps HTMX pages declarative without inventing custom fragment endpoints fo
|
|
|
5015
5015
|
|
|
5016
5016
|
Performance & accuracy benchmarks (STT, TTS, duplex, telephony, sessions) and head-to-head comparisons against Vapi live in a dedicated repo: **[absolutejs/benchmarks](https://github.com/absolutejs/benchmarks)**. They consume the published `@absolutejs/voice` package and provider adapters.
|
|
5017
5017
|
|
|
5018
|
+
Reusable eval contracts stay in `@absolutejs/voice/testing`: fixture manifests
|
|
5019
|
+
can label critical names, organizations, currency, percentages, phone numbers,
|
|
5020
|
+
and other exact fields; benchmark reports score those fields independently from
|
|
5021
|
+
WER; confidence calibration reports ECE/Brier scores; and routing reports expose
|
|
5022
|
+
fallback improvement and harm rates. Audio corpora and executable provider runs
|
|
5023
|
+
remain separate so applications can consume the same contracts without shipping
|
|
5024
|
+
benchmark media in the runtime package.
|
|
5025
|
+
|
|
5018
5026
|
## Adapter Contract
|
|
5019
5027
|
|
|
5020
5028
|
Adapters normalize vendor behavior into a core event model so the plugin never branches on vendor names.
|
|
@@ -5035,6 +5043,7 @@ type STTAdapterSession = {
|
|
|
5035
5043
|
handler: (payload: STTSessionEventMap[K]) => void | Promise<void>,
|
|
5036
5044
|
) => () => void;
|
|
5037
5045
|
send: (audio: AudioChunk) => Promise<void>;
|
|
5046
|
+
configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
|
|
5038
5047
|
close: (reason?: string) => Promise<void>;
|
|
5039
5048
|
};
|
|
5040
5049
|
```
|
package/dist/core/types.d.ts
CHANGED
|
@@ -90,6 +90,15 @@ export type TranscriptWord = {
|
|
|
90
90
|
startedAtMs?: number;
|
|
91
91
|
text: string;
|
|
92
92
|
};
|
|
93
|
+
/** Provider token evidence. Tokens are intentionally kept separate from words:
|
|
94
|
+
* subword log probabilities are useful for calibration and routing, but are not
|
|
95
|
+
* word-level timestamps or confidence scores. */
|
|
96
|
+
export type TranscriptToken = {
|
|
97
|
+
bytes?: number[];
|
|
98
|
+
confidence?: number;
|
|
99
|
+
logProbability?: number;
|
|
100
|
+
text: string;
|
|
101
|
+
};
|
|
93
102
|
export type Transcript = {
|
|
94
103
|
id: string;
|
|
95
104
|
text: string;
|
|
@@ -101,8 +110,15 @@ export type Transcript = {
|
|
|
101
110
|
startedAtMs?: number;
|
|
102
111
|
endedAtMs?: number;
|
|
103
112
|
vendor?: string;
|
|
113
|
+
tokens?: TranscriptToken[];
|
|
104
114
|
words?: TranscriptWord[];
|
|
105
115
|
};
|
|
116
|
+
export type VoiceSTTSessionConfiguration = {
|
|
117
|
+
languageHints?: string[];
|
|
118
|
+
lexicon?: VoiceLexiconEntry[];
|
|
119
|
+
phraseHints?: VoicePhraseHint[];
|
|
120
|
+
turnDetection?: Partial<VoiceTurnDetectionConfig>;
|
|
121
|
+
};
|
|
106
122
|
export type VoiceTranscriptQuality = {
|
|
107
123
|
averageConfidence?: number;
|
|
108
124
|
confidenceSampleCount: number;
|
|
@@ -184,6 +200,8 @@ export type STTSessionEventMap = {
|
|
|
184
200
|
export type STTAdapterSession = {
|
|
185
201
|
on: <K extends keyof STTSessionEventMap>(event: K, handler: (payload: STTSessionEventMap[K]) => void | Promise<void>) => () => void;
|
|
186
202
|
send: (audio: AudioChunk) => Promise<void>;
|
|
203
|
+
/** Update provider-supported STT context without restarting the stream. */
|
|
204
|
+
configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
|
|
187
205
|
close: (reason?: string) => Promise<void>;
|
|
188
206
|
};
|
|
189
207
|
export type STTAdapterOpenOptions = {
|
|
@@ -240,6 +258,8 @@ export type RealtimeSessionEventMap = STTSessionEventMap & {
|
|
|
240
258
|
export type RealtimeAdapterSession = {
|
|
241
259
|
on: <K extends keyof RealtimeSessionEventMap>(event: K, handler: (payload: RealtimeSessionEventMap[K]) => void | Promise<void>) => () => void;
|
|
242
260
|
send: (input: AudioChunk | string) => Promise<void>;
|
|
261
|
+
/** Update provider-supported input transcription context in place. */
|
|
262
|
+
configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
|
|
243
263
|
close: (reason?: string) => Promise<void>;
|
|
244
264
|
};
|
|
245
265
|
export type RealtimeAdapterOpenOptions = {
|
|
@@ -584,6 +604,9 @@ export type VoiceSessionHandle<TContext = unknown, TSession extends VoiceSession
|
|
|
584
604
|
speechThreshold: number;
|
|
585
605
|
transcriptStabilityMs: number;
|
|
586
606
|
}>;
|
|
607
|
+
/** Refresh vocabulary, language hints, or provider turn settings while a
|
|
608
|
+
* call is active. Unsupported fields are ignored by the active adapter. */
|
|
609
|
+
configureSTT: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
|
|
587
610
|
};
|
|
588
611
|
export type VoiceLLMUsage = {
|
|
589
612
|
provider?: string;
|
package/dist/index.js
CHANGED
|
@@ -7041,6 +7041,10 @@ var createVoiceSession = (options) => {
|
|
|
7041
7041
|
commitTurn: async (reason = "manual") => runSerial("api.commitTurn", async () => {
|
|
7042
7042
|
await commitTurnInternal(reason);
|
|
7043
7043
|
}),
|
|
7044
|
+
configureSTT: async (configuration) => runSerial("api.configureSTT", async () => {
|
|
7045
|
+
const adapter = await ensureAdapter();
|
|
7046
|
+
await adapter.configure?.(configuration);
|
|
7047
|
+
}),
|
|
7044
7048
|
complete: async (result) => runSerial("api.complete", async () => {
|
|
7045
7049
|
await completeInternal(result);
|
|
7046
7050
|
}),
|
|
@@ -44493,6 +44497,7 @@ var createContractApi = (session) => ({
|
|
|
44493
44497
|
id: session.id,
|
|
44494
44498
|
attachUserMedia: async () => {},
|
|
44495
44499
|
close: async () => {},
|
|
44500
|
+
configureSTT: async () => {},
|
|
44496
44501
|
commitTurn: async () => {},
|
|
44497
44502
|
complete: async () => {},
|
|
44498
44503
|
connect: async () => {},
|
|
@@ -49558,6 +49563,84 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
|
|
|
49558
49563
|
};
|
|
49559
49564
|
};
|
|
49560
49565
|
|
|
49566
|
+
// src/testing/confidenceCalibration.ts
|
|
49567
|
+
var clampConfidence2 = (value) => Math.max(0, Math.min(1, value));
|
|
49568
|
+
var calibrateVoiceConfidence = (samples, binCount = 10) => {
|
|
49569
|
+
const safeBinCount = Math.max(1, Math.round(binCount));
|
|
49570
|
+
const bins = Array.from({ length: safeBinCount }, (_, index) => {
|
|
49571
|
+
const lowerBound = index / safeBinCount;
|
|
49572
|
+
return {
|
|
49573
|
+
accuracy: 0,
|
|
49574
|
+
averageConfidence: 0,
|
|
49575
|
+
count: 0,
|
|
49576
|
+
lowerBound,
|
|
49577
|
+
upperBound: (index + 1) / safeBinCount
|
|
49578
|
+
};
|
|
49579
|
+
});
|
|
49580
|
+
let brierTotal = 0;
|
|
49581
|
+
for (const sample of samples) {
|
|
49582
|
+
const confidence = clampConfidence2(sample.confidence);
|
|
49583
|
+
const binIndex = Math.min(safeBinCount - 1, Math.floor(confidence * safeBinCount));
|
|
49584
|
+
const bin = bins[binIndex];
|
|
49585
|
+
bin.count += 1;
|
|
49586
|
+
bin.averageConfidence += confidence;
|
|
49587
|
+
bin.accuracy += sample.correct ? 1 : 0;
|
|
49588
|
+
brierTotal += (confidence - (sample.correct ? 1 : 0)) ** 2;
|
|
49589
|
+
}
|
|
49590
|
+
let expectedCalibrationError = 0;
|
|
49591
|
+
for (const bin of bins) {
|
|
49592
|
+
if (bin.count === 0)
|
|
49593
|
+
continue;
|
|
49594
|
+
bin.averageConfidence /= bin.count;
|
|
49595
|
+
bin.accuracy /= bin.count;
|
|
49596
|
+
expectedCalibrationError += bin.count / Math.max(1, samples.length) * Math.abs(bin.accuracy - bin.averageConfidence);
|
|
49597
|
+
}
|
|
49598
|
+
return {
|
|
49599
|
+
bins,
|
|
49600
|
+
brierScore: samples.length > 0 ? brierTotal / samples.length : 0,
|
|
49601
|
+
expectedCalibrationError,
|
|
49602
|
+
sampleCount: samples.length
|
|
49603
|
+
};
|
|
49604
|
+
};
|
|
49605
|
+
|
|
49606
|
+
// src/testing/criticalFields.ts
|
|
49607
|
+
var normalizeText4 = (value) => value.normalize("NFKC").toLowerCase().replace(/[^\p{L}\p{N}@.%+$'-]+/gu, " ").replace(/\s+/g, " ").trim();
|
|
49608
|
+
var normalizeDigits = (value) => normalizeSpokenNumbers(value).replace(/\D/g, "");
|
|
49609
|
+
var normalizeSemanticNumber = (value) => normalizeSpokenNumbers(value).toLowerCase().replace(/\bdollars?\b|\busd\b|\bpercent(age)?\b|[%,$]/g, "").replace(/\s+/g, "").trim();
|
|
49610
|
+
var matchesCandidate = (actual, candidate, kind) => {
|
|
49611
|
+
if (kind === "phone") {
|
|
49612
|
+
const expectedDigits = normalizeDigits(candidate);
|
|
49613
|
+
return expectedDigits.length > 0 && normalizeDigits(actual).includes(expectedDigits);
|
|
49614
|
+
}
|
|
49615
|
+
if (kind === "currency" || kind === "number" || kind === "percentage") {
|
|
49616
|
+
const expectedNumber = normalizeSemanticNumber(candidate);
|
|
49617
|
+
return expectedNumber.length > 0 && normalizeSemanticNumber(actual).includes(expectedNumber);
|
|
49618
|
+
}
|
|
49619
|
+
const normalizedCandidate = normalizeText4(candidate);
|
|
49620
|
+
return normalizedCandidate.length > 0 && normalizeText4(actual).includes(normalizedCandidate);
|
|
49621
|
+
};
|
|
49622
|
+
var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
|
|
49623
|
+
const fields = expectedFields.map((field) => {
|
|
49624
|
+
const candidates = [field.value, ...field.aliases ?? []];
|
|
49625
|
+
const matchedAlias = candidates.find((candidate) => matchesCandidate(actualText, candidate, field.kind));
|
|
49626
|
+
return {
|
|
49627
|
+
...field,
|
|
49628
|
+
matched: matchedAlias !== undefined,
|
|
49629
|
+
matchedAlias
|
|
49630
|
+
};
|
|
49631
|
+
});
|
|
49632
|
+
const matchedCount = fields.filter((field) => field.matched).length;
|
|
49633
|
+
const totalCount = fields.length;
|
|
49634
|
+
return {
|
|
49635
|
+
accuracy: totalCount > 0 ? matchedCount / totalCount : 1,
|
|
49636
|
+
fields,
|
|
49637
|
+
matchedCount,
|
|
49638
|
+
missingFieldIds: fields.filter((field) => !field.matched).map((field) => field.id),
|
|
49639
|
+
passesRequired: fields.every((field) => field.required === false || field.matched),
|
|
49640
|
+
totalCount
|
|
49641
|
+
};
|
|
49642
|
+
};
|
|
49643
|
+
|
|
49561
49644
|
// src/testing/benchmark.ts
|
|
49562
49645
|
var resolveFixtureEnvironment = (fixture) => {
|
|
49563
49646
|
const tags = new Set(fixture.tags ?? []);
|
|
@@ -49788,10 +49871,21 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
49788
49871
|
const postSpeechTimeToFirstFinalMs = toPostSpeechLatency(result.finalEvents[0]?.receivedAt);
|
|
49789
49872
|
const postSpeechTimeToEndOfTurnMs = toPostSpeechLatency(result.endOfTurnEvents[0]?.receivedAt);
|
|
49790
49873
|
const expectedTerms = scoreExpectedTerms(result.finalText, fixture.expectedTerms);
|
|
49874
|
+
const criticalFields = scoreVoiceCriticalFields(result.finalText, fixture.expectedCriticalFields);
|
|
49791
49875
|
const speakerTurns = scoreSpeakerTurns(fixture, result);
|
|
49876
|
+
const transcriptConfidence = average2(result.finalEvents.map((event) => {
|
|
49877
|
+
if (typeof event.transcript.confidence === "number") {
|
|
49878
|
+
return event.transcript.confidence;
|
|
49879
|
+
}
|
|
49880
|
+
return average2([
|
|
49881
|
+
...(event.transcript.words ?? []).map((word) => word.confidence),
|
|
49882
|
+
...(event.transcript.tokens ?? []).map((token) => token.confidence)
|
|
49883
|
+
]);
|
|
49884
|
+
}));
|
|
49792
49885
|
return {
|
|
49793
49886
|
accuracy: result.accuracy,
|
|
49794
49887
|
closeCount: result.closeEvents.length,
|
|
49888
|
+
criticalFields,
|
|
49795
49889
|
difficulty: fixture.difficulty,
|
|
49796
49890
|
elapsedMs,
|
|
49797
49891
|
endOfTurnCount: result.endOfTurnEvents.length,
|
|
@@ -49802,7 +49896,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
49802
49896
|
fixtureId: fixture.id,
|
|
49803
49897
|
fragmentationCount: Math.max(0, result.finalEvents.length - 1),
|
|
49804
49898
|
group: resolveFixtureEnvironment(fixture),
|
|
49805
|
-
passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && (speakerTurns ? speakerTurns.passes : true),
|
|
49899
|
+
passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && criticalFields.passesRequired && (speakerTurns ? speakerTurns.passes : true),
|
|
49806
49900
|
partialCount: result.partialEvents.length,
|
|
49807
49901
|
speakerTurns,
|
|
49808
49902
|
postSpeechTimeToEndOfTurnMs,
|
|
@@ -49811,6 +49905,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
49811
49905
|
timeToEndOfTurnMs,
|
|
49812
49906
|
timeToFirstFinalMs,
|
|
49813
49907
|
timeToFirstPartialMs,
|
|
49908
|
+
transcriptConfidence: roundMetric4(transcriptConfidence),
|
|
49814
49909
|
title: fixture.title
|
|
49815
49910
|
};
|
|
49816
49911
|
};
|
|
@@ -49924,12 +50019,20 @@ var summarizeSTTBenchmark = (adapterId, fixtures) => {
|
|
|
49924
50019
|
const passCount = fixtures.filter((fixture) => fixture.passes).length;
|
|
49925
50020
|
return {
|
|
49926
50021
|
adapterId,
|
|
50022
|
+
confidenceCalibration: calibrateVoiceConfidence(fixtures.flatMap((fixture) => typeof fixture.transcriptConfidence === "number" ? [
|
|
50023
|
+
{
|
|
50024
|
+
confidence: fixture.transcriptConfidence,
|
|
50025
|
+
correct: fixture.passes
|
|
50026
|
+
}
|
|
50027
|
+
] : [])),
|
|
49927
50028
|
averageCharErrorRate: roundMetric4(average2(fixtures.map((fixture) => fixture.accuracy.charErrorRate))) ?? 0,
|
|
49928
50029
|
averageElapsedMs: roundMetric4(average2(fixtures.map((fixture) => fixture.elapsedMs)), 2) ?? 0,
|
|
49929
50030
|
averageEndOfTurnCount: roundMetric4(average2(fixtures.map((fixture) => fixture.endOfTurnCount)), 2) ?? 0,
|
|
49930
50031
|
averageFinalCount: roundMetric4(average2(fixtures.map((fixture) => fixture.finalCount)), 2) ?? 0,
|
|
49931
50032
|
averageSpeakerTurnMatchRate: roundMetric4(average2(fixtures.map((fixture) => fixture.speakerTurns?.patternMatchRate))),
|
|
49932
50033
|
averageTermRecall: roundMetric4(average2(fixtures.map((fixture) => fixture.expectedTerms.recall))) ?? 0,
|
|
50034
|
+
averageCriticalFieldAccuracy: roundMetric4(average2(fixtures.map((fixture) => fixture.criticalFields?.accuracy ?? 1))) ?? 0,
|
|
50035
|
+
requiredCriticalFieldPassRate: fixtureCount > 0 ? roundMetric4(fixtures.filter((fixture) => fixture.criticalFields?.passesRequired ?? true).length / fixtureCount) ?? 0 : 0,
|
|
49933
50036
|
averagePostSpeechTimeToEndOfTurnMs: roundMetric4(average2(fixtures.map((fixture) => fixture.postSpeechTimeToEndOfTurnMs)), 2),
|
|
49934
50037
|
averagePostSpeechTimeToFirstFinalMs: roundMetric4(average2(fixtures.map((fixture) => fixture.postSpeechTimeToFirstFinalMs)), 2),
|
|
49935
50038
|
averageTimeToEndOfTurnMs: roundMetric4(average2(fixtures.map((fixture) => fixture.timeToEndOfTurnMs)), 2),
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import type { STTAdapter, STTAdapterOpenOptions } from "../core/types";
|
|
2
2
|
import { type VoiceSTTAdapterHarnessOptions, type VoiceSTTAdapterHarnessResult } from "./stt";
|
|
3
3
|
import type { VoiceTestFixture } from "./fixtures";
|
|
4
|
+
import { type VoiceConfidenceCalibrationReport } from "./confidenceCalibration";
|
|
5
|
+
import { type VoiceCriticalFieldAccuracy } from "./criticalFields";
|
|
4
6
|
export type VoiceExpectedTermAccuracy = {
|
|
5
7
|
allMatched: boolean;
|
|
6
8
|
expectedTerms: string[];
|
|
@@ -25,6 +27,7 @@ export type VoiceSTTBenchmarkFixtureResult = {
|
|
|
25
27
|
endOfTurnCount: number;
|
|
26
28
|
errorCount: number;
|
|
27
29
|
expectedTerms: VoiceExpectedTermAccuracy;
|
|
30
|
+
criticalFields?: VoiceCriticalFieldAccuracy;
|
|
28
31
|
finalCount: number;
|
|
29
32
|
finalText: string;
|
|
30
33
|
fixtureId: string;
|
|
@@ -39,6 +42,7 @@ export type VoiceSTTBenchmarkFixtureResult = {
|
|
|
39
42
|
timeToEndOfTurnMs?: number;
|
|
40
43
|
timeToFirstFinalMs?: number;
|
|
41
44
|
timeToFirstPartialMs?: number;
|
|
45
|
+
transcriptConfidence?: number;
|
|
42
46
|
title: string;
|
|
43
47
|
};
|
|
44
48
|
export type VoiceSTTBenchmarkSummary = {
|
|
@@ -49,6 +53,8 @@ export type VoiceSTTBenchmarkSummary = {
|
|
|
49
53
|
averageFinalCount: number;
|
|
50
54
|
averageSpeakerTurnMatchRate?: number;
|
|
51
55
|
averageTermRecall: number;
|
|
56
|
+
averageCriticalFieldAccuracy: number;
|
|
57
|
+
requiredCriticalFieldPassRate: number;
|
|
52
58
|
averagePostSpeechTimeToEndOfTurnMs?: number;
|
|
53
59
|
averagePostSpeechTimeToFirstFinalMs?: number;
|
|
54
60
|
averageTimeToEndOfTurnMs?: number;
|
|
@@ -63,6 +69,7 @@ export type VoiceSTTBenchmarkSummary = {
|
|
|
63
69
|
totalErrorCount: number;
|
|
64
70
|
wordAccuracyRate: number;
|
|
65
71
|
groupSummaries: VoiceSTTBenchmarkFixtureSummary[];
|
|
72
|
+
confidenceCalibration: VoiceConfidenceCalibrationReport;
|
|
66
73
|
};
|
|
67
74
|
export type VoiceSTTBenchmarkFixtureSummary = {
|
|
68
75
|
group: VoiceSTTFixtureEnvironment;
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
export type VoiceConfidenceCalibrationSample = {
|
|
2
|
+
confidence: number;
|
|
3
|
+
correct: boolean;
|
|
4
|
+
metadata?: Record<string, unknown>;
|
|
5
|
+
};
|
|
6
|
+
export type VoiceConfidenceCalibrationBin = {
|
|
7
|
+
accuracy: number;
|
|
8
|
+
averageConfidence: number;
|
|
9
|
+
count: number;
|
|
10
|
+
lowerBound: number;
|
|
11
|
+
upperBound: number;
|
|
12
|
+
};
|
|
13
|
+
export type VoiceConfidenceCalibrationReport = {
|
|
14
|
+
bins: VoiceConfidenceCalibrationBin[];
|
|
15
|
+
brierScore: number;
|
|
16
|
+
expectedCalibrationError: number;
|
|
17
|
+
sampleCount: number;
|
|
18
|
+
};
|
|
19
|
+
export declare const calibrateVoiceConfidence: (samples: VoiceConfidenceCalibrationSample[], binCount?: number) => VoiceConfidenceCalibrationReport;
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
export type VoiceCriticalFieldKind = "acronym" | "brand" | "currency" | "custom" | "email" | "number" | "organization" | "percentage" | "person-name" | "phone";
|
|
2
|
+
export type VoiceExpectedCriticalField = {
|
|
3
|
+
aliases?: string[];
|
|
4
|
+
id: string;
|
|
5
|
+
kind: VoiceCriticalFieldKind;
|
|
6
|
+
metadata?: Record<string, unknown>;
|
|
7
|
+
required?: boolean;
|
|
8
|
+
value: string;
|
|
9
|
+
};
|
|
10
|
+
export type VoiceCriticalFieldResult = VoiceExpectedCriticalField & {
|
|
11
|
+
matched: boolean;
|
|
12
|
+
matchedAlias?: string;
|
|
13
|
+
};
|
|
14
|
+
export type VoiceCriticalFieldAccuracy = {
|
|
15
|
+
accuracy: number;
|
|
16
|
+
fields: VoiceCriticalFieldResult[];
|
|
17
|
+
matchedCount: number;
|
|
18
|
+
missingFieldIds: string[];
|
|
19
|
+
passesRequired: boolean;
|
|
20
|
+
totalCount: number;
|
|
21
|
+
};
|
|
22
|
+
export declare const scoreVoiceCriticalFields: (actualText: string, expectedFields?: VoiceExpectedCriticalField[]) => VoiceCriticalFieldAccuracy;
|
|
@@ -1,10 +1,12 @@
|
|
|
1
1
|
import type { AudioFormat, VoiceExpectedSpeakerTurn } from "../core/types";
|
|
2
|
+
import type { VoiceExpectedCriticalField } from "./criticalFields";
|
|
2
3
|
export type VoiceTestFixtureManifestEntry = {
|
|
3
4
|
id: string;
|
|
4
5
|
title: string;
|
|
5
6
|
audioPath: string;
|
|
6
7
|
expectedText: string;
|
|
7
8
|
expectedTerms?: string[];
|
|
9
|
+
expectedCriticalFields?: VoiceExpectedCriticalField[];
|
|
8
10
|
expectedSpeakerTurns?: VoiceExpectedSpeakerTurn[];
|
|
9
11
|
expectedTurnTexts?: string[];
|
|
10
12
|
chunkDurationMs?: number;
|
package/dist/testing/index.d.ts
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
export * from "./accuracy";
|
|
2
2
|
export * from "./benchmark";
|
|
3
|
+
export * from "./confidenceCalibration";
|
|
4
|
+
export * from "./criticalFields";
|
|
3
5
|
export * from "./corrected";
|
|
4
6
|
export * from "./duplex";
|
|
5
7
|
export * from "./fixtures";
|
|
@@ -7,6 +9,7 @@ export * from "./ioProviderSimulator";
|
|
|
7
9
|
export * from "./providerSimulator";
|
|
8
10
|
export * from "./resilience";
|
|
9
11
|
export * from "./review";
|
|
12
|
+
export * from "./routingBenchmark";
|
|
10
13
|
export * from "./sessionBenchmark";
|
|
11
14
|
export * from "./stt";
|
|
12
15
|
export * from "./telephony";
|
package/dist/testing/index.js
CHANGED
|
@@ -367,6 +367,269 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
|
|
|
367
367
|
};
|
|
368
368
|
};
|
|
369
369
|
|
|
370
|
+
// src/testing/confidenceCalibration.ts
|
|
371
|
+
var clampConfidence = (value) => Math.max(0, Math.min(1, value));
|
|
372
|
+
var calibrateVoiceConfidence = (samples, binCount = 10) => {
|
|
373
|
+
const safeBinCount = Math.max(1, Math.round(binCount));
|
|
374
|
+
const bins = Array.from({ length: safeBinCount }, (_, index) => {
|
|
375
|
+
const lowerBound = index / safeBinCount;
|
|
376
|
+
return {
|
|
377
|
+
accuracy: 0,
|
|
378
|
+
averageConfidence: 0,
|
|
379
|
+
count: 0,
|
|
380
|
+
lowerBound,
|
|
381
|
+
upperBound: (index + 1) / safeBinCount
|
|
382
|
+
};
|
|
383
|
+
});
|
|
384
|
+
let brierTotal = 0;
|
|
385
|
+
for (const sample of samples) {
|
|
386
|
+
const confidence = clampConfidence(sample.confidence);
|
|
387
|
+
const binIndex = Math.min(safeBinCount - 1, Math.floor(confidence * safeBinCount));
|
|
388
|
+
const bin = bins[binIndex];
|
|
389
|
+
bin.count += 1;
|
|
390
|
+
bin.averageConfidence += confidence;
|
|
391
|
+
bin.accuracy += sample.correct ? 1 : 0;
|
|
392
|
+
brierTotal += (confidence - (sample.correct ? 1 : 0)) ** 2;
|
|
393
|
+
}
|
|
394
|
+
let expectedCalibrationError = 0;
|
|
395
|
+
for (const bin of bins) {
|
|
396
|
+
if (bin.count === 0)
|
|
397
|
+
continue;
|
|
398
|
+
bin.averageConfidence /= bin.count;
|
|
399
|
+
bin.accuracy /= bin.count;
|
|
400
|
+
expectedCalibrationError += bin.count / Math.max(1, samples.length) * Math.abs(bin.accuracy - bin.averageConfidence);
|
|
401
|
+
}
|
|
402
|
+
return {
|
|
403
|
+
bins,
|
|
404
|
+
brierScore: samples.length > 0 ? brierTotal / samples.length : 0,
|
|
405
|
+
expectedCalibrationError,
|
|
406
|
+
sampleCount: samples.length
|
|
407
|
+
};
|
|
408
|
+
};
|
|
409
|
+
|
|
410
|
+
// src/core/numberNormalizer.ts
|
|
411
|
+
var ONES = {
|
|
412
|
+
eight: 8,
|
|
413
|
+
eighteen: 18,
|
|
414
|
+
eleven: 11,
|
|
415
|
+
fifteen: 15,
|
|
416
|
+
five: 5,
|
|
417
|
+
four: 4,
|
|
418
|
+
fourteen: 14,
|
|
419
|
+
nine: 9,
|
|
420
|
+
nineteen: 19,
|
|
421
|
+
one: 1,
|
|
422
|
+
seven: 7,
|
|
423
|
+
seventeen: 17,
|
|
424
|
+
six: 6,
|
|
425
|
+
sixteen: 16,
|
|
426
|
+
ten: 10,
|
|
427
|
+
thirteen: 13,
|
|
428
|
+
three: 3,
|
|
429
|
+
twelve: 12,
|
|
430
|
+
two: 2,
|
|
431
|
+
zero: 0
|
|
432
|
+
};
|
|
433
|
+
var TENS = {
|
|
434
|
+
eighty: 80,
|
|
435
|
+
fifty: 50,
|
|
436
|
+
forty: 40,
|
|
437
|
+
ninety: 90,
|
|
438
|
+
seventy: 70,
|
|
439
|
+
sixty: 60,
|
|
440
|
+
thirty: 30,
|
|
441
|
+
twenty: 20
|
|
442
|
+
};
|
|
443
|
+
var SCALES = {
|
|
444
|
+
billion: 1e9,
|
|
445
|
+
million: 1e6,
|
|
446
|
+
thousand: 1000,
|
|
447
|
+
trillion: 1000000000000
|
|
448
|
+
};
|
|
449
|
+
var MAGNITUDE_WORDS = [
|
|
450
|
+
[1000000000000, "trillion"],
|
|
451
|
+
[1e9, "billion"],
|
|
452
|
+
[1e6, "million"]
|
|
453
|
+
];
|
|
454
|
+
var FILLER = new Set(["and", "a", "an"]);
|
|
455
|
+
var DECIMAL_PLACES = 3;
|
|
456
|
+
var isValueWord = (word) => (word in ONES) || (word in TENS) || word === "hundred";
|
|
457
|
+
var isScaleWord = (word) => (word in SCALES);
|
|
458
|
+
var isNumberWord = (word) => isValueWord(word) || isScaleWord(word) || word === "point" || FILLER.has(word);
|
|
459
|
+
var trimNumber = (value) => {
|
|
460
|
+
if (Number.isInteger(value))
|
|
461
|
+
return value.toLocaleString("en-US");
|
|
462
|
+
return String(Number(value.toFixed(DECIMAL_PLACES)));
|
|
463
|
+
};
|
|
464
|
+
var renderValue = (value, usedMagnitude) => {
|
|
465
|
+
if (usedMagnitude) {
|
|
466
|
+
for (const [scale, word] of MAGNITUDE_WORDS) {
|
|
467
|
+
if (value >= scale) {
|
|
468
|
+
const scaled = value / scale;
|
|
469
|
+
if (Number(scaled.toFixed(DECIMAL_PLACES)) === scaled) {
|
|
470
|
+
return `${trimNumber(scaled)} ${word}`;
|
|
471
|
+
}
|
|
472
|
+
}
|
|
473
|
+
}
|
|
474
|
+
}
|
|
475
|
+
return trimNumber(value);
|
|
476
|
+
};
|
|
477
|
+
var parseNumberWords = (words) => {
|
|
478
|
+
let total = 0;
|
|
479
|
+
let current = 0;
|
|
480
|
+
let usedMagnitude = false;
|
|
481
|
+
let sawNumber = false;
|
|
482
|
+
let decimal = null;
|
|
483
|
+
const foldDecimal = () => {
|
|
484
|
+
if (decimal && decimal.length > 0)
|
|
485
|
+
current += Number(`0.${decimal}`);
|
|
486
|
+
decimal = null;
|
|
487
|
+
};
|
|
488
|
+
for (const word of words) {
|
|
489
|
+
if (word === "point") {
|
|
490
|
+
decimal = "";
|
|
491
|
+
continue;
|
|
492
|
+
}
|
|
493
|
+
const one = ONES[word];
|
|
494
|
+
const ten = TENS[word];
|
|
495
|
+
const scale = SCALES[word];
|
|
496
|
+
if (decimal !== null) {
|
|
497
|
+
if (one !== undefined && one <= 9) {
|
|
498
|
+
decimal += String(one);
|
|
499
|
+
sawNumber = true;
|
|
500
|
+
continue;
|
|
501
|
+
}
|
|
502
|
+
foldDecimal();
|
|
503
|
+
}
|
|
504
|
+
if (FILLER.has(word))
|
|
505
|
+
continue;
|
|
506
|
+
if (one !== undefined) {
|
|
507
|
+
current += one;
|
|
508
|
+
sawNumber = true;
|
|
509
|
+
} else if (ten !== undefined) {
|
|
510
|
+
current += ten;
|
|
511
|
+
sawNumber = true;
|
|
512
|
+
} else if (word === "hundred") {
|
|
513
|
+
current = (current === 0 ? 1 : current) * 100;
|
|
514
|
+
sawNumber = true;
|
|
515
|
+
} else if (scale !== undefined) {
|
|
516
|
+
total += (current === 0 ? 1 : current) * scale;
|
|
517
|
+
current = 0;
|
|
518
|
+
sawNumber = true;
|
|
519
|
+
usedMagnitude = true;
|
|
520
|
+
}
|
|
521
|
+
}
|
|
522
|
+
foldDecimal();
|
|
523
|
+
if (!sawNumber)
|
|
524
|
+
return null;
|
|
525
|
+
return { usedMagnitude, value: total + current };
|
|
526
|
+
};
|
|
527
|
+
var PERCENT_RE = /^(per ?cent|percent|percentage)$/;
|
|
528
|
+
var CURRENCY_RE = /^(dollars?|bucks?|usd)$/;
|
|
529
|
+
var WORD_RE = /^[A-Za-z]+(?:-[A-Za-z]+)*$/;
|
|
530
|
+
var wordsOf = (token) => token.toLowerCase().split("-");
|
|
531
|
+
var isNumberToken = (token) => WORD_RE.test(token) && wordsOf(token).every(isNumberWord);
|
|
532
|
+
var startsNumber = (token) => WORD_RE.test(token) && wordsOf(token).some(isValueWord);
|
|
533
|
+
var isSpace = (token) => /^\s+$/.test(token);
|
|
534
|
+
var normalizeSpokenNumbers = (input) => {
|
|
535
|
+
if (!input)
|
|
536
|
+
return input;
|
|
537
|
+
const parts = input.match(/[A-Za-z]+(?:-[A-Za-z]+)*|[^A-Za-z]+/g);
|
|
538
|
+
if (!parts)
|
|
539
|
+
return input;
|
|
540
|
+
const at = (idx) => parts[idx] ?? "";
|
|
541
|
+
const out = [];
|
|
542
|
+
let i = 0;
|
|
543
|
+
while (i < parts.length) {
|
|
544
|
+
const token = at(i);
|
|
545
|
+
const lower = token.toLowerCase();
|
|
546
|
+
if (lower === "a" || lower === "an") {
|
|
547
|
+
const nextWord = at(i + 2);
|
|
548
|
+
const nextHead = wordsOf(nextWord)[0] ?? "";
|
|
549
|
+
const nextIsScale = isSpace(at(i + 1)) && WORD_RE.test(nextWord) && (nextHead === "hundred" || isScaleWord(nextHead));
|
|
550
|
+
if (!nextIsScale) {
|
|
551
|
+
out.push(token);
|
|
552
|
+
i += 1;
|
|
553
|
+
continue;
|
|
554
|
+
}
|
|
555
|
+
} else if (!startsNumber(token)) {
|
|
556
|
+
out.push(token);
|
|
557
|
+
i += 1;
|
|
558
|
+
continue;
|
|
559
|
+
}
|
|
560
|
+
const spanIdx = [i];
|
|
561
|
+
let j = i + 1;
|
|
562
|
+
while (isSpace(at(j)) && isNumberToken(at(j + 1))) {
|
|
563
|
+
spanIdx.push(j + 1);
|
|
564
|
+
j += 2;
|
|
565
|
+
}
|
|
566
|
+
const words = spanIdx.flatMap((k) => wordsOf(at(k)));
|
|
567
|
+
while (words.length > 0 && FILLER.has(words[words.length - 1] ?? "")) {
|
|
568
|
+
words.pop();
|
|
569
|
+
}
|
|
570
|
+
const parsed = parseNumberWords(words);
|
|
571
|
+
if (!parsed) {
|
|
572
|
+
out.push(token);
|
|
573
|
+
i += 1;
|
|
574
|
+
continue;
|
|
575
|
+
}
|
|
576
|
+
let rendered = renderValue(parsed.value, parsed.usedMagnitude);
|
|
577
|
+
let lastIdx = spanIdx[spanIdx.length - 1] ?? i;
|
|
578
|
+
const unitWord = at(lastIdx + 2);
|
|
579
|
+
if (isSpace(at(lastIdx + 1)) && WORD_RE.test(unitWord)) {
|
|
580
|
+
const unit = unitWord.toLowerCase();
|
|
581
|
+
if (PERCENT_RE.test(unit)) {
|
|
582
|
+
rendered = `${rendered}%`;
|
|
583
|
+
lastIdx += 2;
|
|
584
|
+
} else if (CURRENCY_RE.test(unit)) {
|
|
585
|
+
rendered = rendered.startsWith("$") ? rendered : `$${rendered}`;
|
|
586
|
+
lastIdx += 2;
|
|
587
|
+
}
|
|
588
|
+
}
|
|
589
|
+
out.push(rendered);
|
|
590
|
+
i = lastIdx + 1;
|
|
591
|
+
}
|
|
592
|
+
return out.join("");
|
|
593
|
+
};
|
|
594
|
+
|
|
595
|
+
// src/testing/criticalFields.ts
|
|
596
|
+
var normalizeText2 = (value) => value.normalize("NFKC").toLowerCase().replace(/[^\p{L}\p{N}@.%+$'-]+/gu, " ").replace(/\s+/g, " ").trim();
|
|
597
|
+
var normalizeDigits = (value) => normalizeSpokenNumbers(value).replace(/\D/g, "");
|
|
598
|
+
var normalizeSemanticNumber = (value) => normalizeSpokenNumbers(value).toLowerCase().replace(/\bdollars?\b|\busd\b|\bpercent(age)?\b|[%,$]/g, "").replace(/\s+/g, "").trim();
|
|
599
|
+
var matchesCandidate = (actual, candidate, kind) => {
|
|
600
|
+
if (kind === "phone") {
|
|
601
|
+
const expectedDigits = normalizeDigits(candidate);
|
|
602
|
+
return expectedDigits.length > 0 && normalizeDigits(actual).includes(expectedDigits);
|
|
603
|
+
}
|
|
604
|
+
if (kind === "currency" || kind === "number" || kind === "percentage") {
|
|
605
|
+
const expectedNumber = normalizeSemanticNumber(candidate);
|
|
606
|
+
return expectedNumber.length > 0 && normalizeSemanticNumber(actual).includes(expectedNumber);
|
|
607
|
+
}
|
|
608
|
+
const normalizedCandidate = normalizeText2(candidate);
|
|
609
|
+
return normalizedCandidate.length > 0 && normalizeText2(actual).includes(normalizedCandidate);
|
|
610
|
+
};
|
|
611
|
+
var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
|
|
612
|
+
const fields = expectedFields.map((field) => {
|
|
613
|
+
const candidates = [field.value, ...field.aliases ?? []];
|
|
614
|
+
const matchedAlias = candidates.find((candidate) => matchesCandidate(actualText, candidate, field.kind));
|
|
615
|
+
return {
|
|
616
|
+
...field,
|
|
617
|
+
matched: matchedAlias !== undefined,
|
|
618
|
+
matchedAlias
|
|
619
|
+
};
|
|
620
|
+
});
|
|
621
|
+
const matchedCount = fields.filter((field) => field.matched).length;
|
|
622
|
+
const totalCount = fields.length;
|
|
623
|
+
return {
|
|
624
|
+
accuracy: totalCount > 0 ? matchedCount / totalCount : 1,
|
|
625
|
+
fields,
|
|
626
|
+
matchedCount,
|
|
627
|
+
missingFieldIds: fields.filter((field) => !field.matched).map((field) => field.id),
|
|
628
|
+
passesRequired: fields.every((field) => field.required === false || field.matched),
|
|
629
|
+
totalCount
|
|
630
|
+
};
|
|
631
|
+
};
|
|
632
|
+
|
|
370
633
|
// src/testing/benchmark.ts
|
|
371
634
|
var resolveFixtureEnvironment = (fixture) => {
|
|
372
635
|
const tags = new Set(fixture.tags ?? []);
|
|
@@ -597,10 +860,21 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
597
860
|
const postSpeechTimeToFirstFinalMs = toPostSpeechLatency(result.finalEvents[0]?.receivedAt);
|
|
598
861
|
const postSpeechTimeToEndOfTurnMs = toPostSpeechLatency(result.endOfTurnEvents[0]?.receivedAt);
|
|
599
862
|
const expectedTerms = scoreExpectedTerms(result.finalText, fixture.expectedTerms);
|
|
863
|
+
const criticalFields = scoreVoiceCriticalFields(result.finalText, fixture.expectedCriticalFields);
|
|
600
864
|
const speakerTurns = scoreSpeakerTurns(fixture, result);
|
|
865
|
+
const transcriptConfidence = average(result.finalEvents.map((event) => {
|
|
866
|
+
if (typeof event.transcript.confidence === "number") {
|
|
867
|
+
return event.transcript.confidence;
|
|
868
|
+
}
|
|
869
|
+
return average([
|
|
870
|
+
...(event.transcript.words ?? []).map((word) => word.confidence),
|
|
871
|
+
...(event.transcript.tokens ?? []).map((token) => token.confidence)
|
|
872
|
+
]);
|
|
873
|
+
}));
|
|
601
874
|
return {
|
|
602
875
|
accuracy: result.accuracy,
|
|
603
876
|
closeCount: result.closeEvents.length,
|
|
877
|
+
criticalFields,
|
|
604
878
|
difficulty: fixture.difficulty,
|
|
605
879
|
elapsedMs,
|
|
606
880
|
endOfTurnCount: result.endOfTurnEvents.length,
|
|
@@ -611,7 +885,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
611
885
|
fixtureId: fixture.id,
|
|
612
886
|
fragmentationCount: Math.max(0, result.finalEvents.length - 1),
|
|
613
887
|
group: resolveFixtureEnvironment(fixture),
|
|
614
|
-
passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && (speakerTurns ? speakerTurns.passes : true),
|
|
888
|
+
passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && criticalFields.passesRequired && (speakerTurns ? speakerTurns.passes : true),
|
|
615
889
|
partialCount: result.partialEvents.length,
|
|
616
890
|
speakerTurns,
|
|
617
891
|
postSpeechTimeToEndOfTurnMs,
|
|
@@ -620,6 +894,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
620
894
|
timeToEndOfTurnMs,
|
|
621
895
|
timeToFirstFinalMs,
|
|
622
896
|
timeToFirstPartialMs,
|
|
897
|
+
transcriptConfidence: roundMetric(transcriptConfidence),
|
|
623
898
|
title: fixture.title
|
|
624
899
|
};
|
|
625
900
|
};
|
|
@@ -733,12 +1008,20 @@ var summarizeSTTBenchmark = (adapterId, fixtures) => {
|
|
|
733
1008
|
const passCount = fixtures.filter((fixture) => fixture.passes).length;
|
|
734
1009
|
return {
|
|
735
1010
|
adapterId,
|
|
1011
|
+
confidenceCalibration: calibrateVoiceConfidence(fixtures.flatMap((fixture) => typeof fixture.transcriptConfidence === "number" ? [
|
|
1012
|
+
{
|
|
1013
|
+
confidence: fixture.transcriptConfidence,
|
|
1014
|
+
correct: fixture.passes
|
|
1015
|
+
}
|
|
1016
|
+
] : [])),
|
|
736
1017
|
averageCharErrorRate: roundMetric(average(fixtures.map((fixture) => fixture.accuracy.charErrorRate))) ?? 0,
|
|
737
1018
|
averageElapsedMs: roundMetric(average(fixtures.map((fixture) => fixture.elapsedMs)), 2) ?? 0,
|
|
738
1019
|
averageEndOfTurnCount: roundMetric(average(fixtures.map((fixture) => fixture.endOfTurnCount)), 2) ?? 0,
|
|
739
1020
|
averageFinalCount: roundMetric(average(fixtures.map((fixture) => fixture.finalCount)), 2) ?? 0,
|
|
740
1021
|
averageSpeakerTurnMatchRate: roundMetric(average(fixtures.map((fixture) => fixture.speakerTurns?.patternMatchRate))),
|
|
741
1022
|
averageTermRecall: roundMetric(average(fixtures.map((fixture) => fixture.expectedTerms.recall))) ?? 0,
|
|
1023
|
+
averageCriticalFieldAccuracy: roundMetric(average(fixtures.map((fixture) => fixture.criticalFields?.accuracy ?? 1))) ?? 0,
|
|
1024
|
+
requiredCriticalFieldPassRate: fixtureCount > 0 ? roundMetric(fixtures.filter((fixture) => fixture.criticalFields?.passesRequired ?? true).length / fixtureCount) ?? 0 : 0,
|
|
742
1025
|
averagePostSpeechTimeToEndOfTurnMs: roundMetric(average(fixtures.map((fixture) => fixture.postSpeechTimeToEndOfTurnMs)), 2),
|
|
743
1026
|
averagePostSpeechTimeToFirstFinalMs: roundMetric(average(fixtures.map((fixture) => fixture.postSpeechTimeToFirstFinalMs)), 2),
|
|
744
1027
|
averageTimeToEndOfTurnMs: roundMetric(average(fixtures.map((fixture) => fixture.timeToEndOfTurnMs)), 2),
|
|
@@ -5599,191 +5882,6 @@ var createVoiceMemoryStore = () => {
|
|
|
5599
5882
|
// src/core/session.ts
|
|
5600
5883
|
import { Buffer as Buffer2 } from "buffer";
|
|
5601
5884
|
|
|
5602
|
-
// src/core/numberNormalizer.ts
|
|
5603
|
-
var ONES = {
|
|
5604
|
-
eight: 8,
|
|
5605
|
-
eighteen: 18,
|
|
5606
|
-
eleven: 11,
|
|
5607
|
-
fifteen: 15,
|
|
5608
|
-
five: 5,
|
|
5609
|
-
four: 4,
|
|
5610
|
-
fourteen: 14,
|
|
5611
|
-
nine: 9,
|
|
5612
|
-
nineteen: 19,
|
|
5613
|
-
one: 1,
|
|
5614
|
-
seven: 7,
|
|
5615
|
-
seventeen: 17,
|
|
5616
|
-
six: 6,
|
|
5617
|
-
sixteen: 16,
|
|
5618
|
-
ten: 10,
|
|
5619
|
-
thirteen: 13,
|
|
5620
|
-
three: 3,
|
|
5621
|
-
twelve: 12,
|
|
5622
|
-
two: 2,
|
|
5623
|
-
zero: 0
|
|
5624
|
-
};
|
|
5625
|
-
var TENS = {
|
|
5626
|
-
eighty: 80,
|
|
5627
|
-
fifty: 50,
|
|
5628
|
-
forty: 40,
|
|
5629
|
-
ninety: 90,
|
|
5630
|
-
seventy: 70,
|
|
5631
|
-
sixty: 60,
|
|
5632
|
-
thirty: 30,
|
|
5633
|
-
twenty: 20
|
|
5634
|
-
};
|
|
5635
|
-
var SCALES = {
|
|
5636
|
-
billion: 1e9,
|
|
5637
|
-
million: 1e6,
|
|
5638
|
-
thousand: 1000,
|
|
5639
|
-
trillion: 1000000000000
|
|
5640
|
-
};
|
|
5641
|
-
var MAGNITUDE_WORDS = [
|
|
5642
|
-
[1000000000000, "trillion"],
|
|
5643
|
-
[1e9, "billion"],
|
|
5644
|
-
[1e6, "million"]
|
|
5645
|
-
];
|
|
5646
|
-
var FILLER = new Set(["and", "a", "an"]);
|
|
5647
|
-
var DECIMAL_PLACES = 3;
|
|
5648
|
-
var isValueWord = (word) => (word in ONES) || (word in TENS) || word === "hundred";
|
|
5649
|
-
var isScaleWord = (word) => (word in SCALES);
|
|
5650
|
-
var isNumberWord = (word) => isValueWord(word) || isScaleWord(word) || word === "point" || FILLER.has(word);
|
|
5651
|
-
var trimNumber = (value) => {
|
|
5652
|
-
if (Number.isInteger(value))
|
|
5653
|
-
return value.toLocaleString("en-US");
|
|
5654
|
-
return String(Number(value.toFixed(DECIMAL_PLACES)));
|
|
5655
|
-
};
|
|
5656
|
-
var renderValue = (value, usedMagnitude) => {
|
|
5657
|
-
if (usedMagnitude) {
|
|
5658
|
-
for (const [scale, word] of MAGNITUDE_WORDS) {
|
|
5659
|
-
if (value >= scale) {
|
|
5660
|
-
const scaled = value / scale;
|
|
5661
|
-
if (Number(scaled.toFixed(DECIMAL_PLACES)) === scaled) {
|
|
5662
|
-
return `${trimNumber(scaled)} ${word}`;
|
|
5663
|
-
}
|
|
5664
|
-
}
|
|
5665
|
-
}
|
|
5666
|
-
}
|
|
5667
|
-
return trimNumber(value);
|
|
5668
|
-
};
|
|
5669
|
-
var parseNumberWords = (words) => {
|
|
5670
|
-
let total = 0;
|
|
5671
|
-
let current = 0;
|
|
5672
|
-
let usedMagnitude = false;
|
|
5673
|
-
let sawNumber = false;
|
|
5674
|
-
let decimal = null;
|
|
5675
|
-
const foldDecimal = () => {
|
|
5676
|
-
if (decimal && decimal.length > 0)
|
|
5677
|
-
current += Number(`0.${decimal}`);
|
|
5678
|
-
decimal = null;
|
|
5679
|
-
};
|
|
5680
|
-
for (const word of words) {
|
|
5681
|
-
if (word === "point") {
|
|
5682
|
-
decimal = "";
|
|
5683
|
-
continue;
|
|
5684
|
-
}
|
|
5685
|
-
const one = ONES[word];
|
|
5686
|
-
const ten = TENS[word];
|
|
5687
|
-
const scale = SCALES[word];
|
|
5688
|
-
if (decimal !== null) {
|
|
5689
|
-
if (one !== undefined && one <= 9) {
|
|
5690
|
-
decimal += String(one);
|
|
5691
|
-
sawNumber = true;
|
|
5692
|
-
continue;
|
|
5693
|
-
}
|
|
5694
|
-
foldDecimal();
|
|
5695
|
-
}
|
|
5696
|
-
if (FILLER.has(word))
|
|
5697
|
-
continue;
|
|
5698
|
-
if (one !== undefined) {
|
|
5699
|
-
current += one;
|
|
5700
|
-
sawNumber = true;
|
|
5701
|
-
} else if (ten !== undefined) {
|
|
5702
|
-
current += ten;
|
|
5703
|
-
sawNumber = true;
|
|
5704
|
-
} else if (word === "hundred") {
|
|
5705
|
-
current = (current === 0 ? 1 : current) * 100;
|
|
5706
|
-
sawNumber = true;
|
|
5707
|
-
} else if (scale !== undefined) {
|
|
5708
|
-
total += (current === 0 ? 1 : current) * scale;
|
|
5709
|
-
current = 0;
|
|
5710
|
-
sawNumber = true;
|
|
5711
|
-
usedMagnitude = true;
|
|
5712
|
-
}
|
|
5713
|
-
}
|
|
5714
|
-
foldDecimal();
|
|
5715
|
-
if (!sawNumber)
|
|
5716
|
-
return null;
|
|
5717
|
-
return { usedMagnitude, value: total + current };
|
|
5718
|
-
};
|
|
5719
|
-
var PERCENT_RE = /^(per ?cent|percent|percentage)$/;
|
|
5720
|
-
var CURRENCY_RE = /^(dollars?|bucks?|usd)$/;
|
|
5721
|
-
var WORD_RE = /^[A-Za-z]+(?:-[A-Za-z]+)*$/;
|
|
5722
|
-
var wordsOf = (token) => token.toLowerCase().split("-");
|
|
5723
|
-
var isNumberToken = (token) => WORD_RE.test(token) && wordsOf(token).every(isNumberWord);
|
|
5724
|
-
var startsNumber = (token) => WORD_RE.test(token) && wordsOf(token).some(isValueWord);
|
|
5725
|
-
var isSpace = (token) => /^\s+$/.test(token);
|
|
5726
|
-
var normalizeSpokenNumbers = (input) => {
|
|
5727
|
-
if (!input)
|
|
5728
|
-
return input;
|
|
5729
|
-
const parts = input.match(/[A-Za-z]+(?:-[A-Za-z]+)*|[^A-Za-z]+/g);
|
|
5730
|
-
if (!parts)
|
|
5731
|
-
return input;
|
|
5732
|
-
const at = (idx) => parts[idx] ?? "";
|
|
5733
|
-
const out = [];
|
|
5734
|
-
let i = 0;
|
|
5735
|
-
while (i < parts.length) {
|
|
5736
|
-
const token = at(i);
|
|
5737
|
-
const lower = token.toLowerCase();
|
|
5738
|
-
if (lower === "a" || lower === "an") {
|
|
5739
|
-
const nextWord = at(i + 2);
|
|
5740
|
-
const nextHead = wordsOf(nextWord)[0] ?? "";
|
|
5741
|
-
const nextIsScale = isSpace(at(i + 1)) && WORD_RE.test(nextWord) && (nextHead === "hundred" || isScaleWord(nextHead));
|
|
5742
|
-
if (!nextIsScale) {
|
|
5743
|
-
out.push(token);
|
|
5744
|
-
i += 1;
|
|
5745
|
-
continue;
|
|
5746
|
-
}
|
|
5747
|
-
} else if (!startsNumber(token)) {
|
|
5748
|
-
out.push(token);
|
|
5749
|
-
i += 1;
|
|
5750
|
-
continue;
|
|
5751
|
-
}
|
|
5752
|
-
const spanIdx = [i];
|
|
5753
|
-
let j = i + 1;
|
|
5754
|
-
while (isSpace(at(j)) && isNumberToken(at(j + 1))) {
|
|
5755
|
-
spanIdx.push(j + 1);
|
|
5756
|
-
j += 2;
|
|
5757
|
-
}
|
|
5758
|
-
const words = spanIdx.flatMap((k) => wordsOf(at(k)));
|
|
5759
|
-
while (words.length > 0 && FILLER.has(words[words.length - 1] ?? "")) {
|
|
5760
|
-
words.pop();
|
|
5761
|
-
}
|
|
5762
|
-
const parsed = parseNumberWords(words);
|
|
5763
|
-
if (!parsed) {
|
|
5764
|
-
out.push(token);
|
|
5765
|
-
i += 1;
|
|
5766
|
-
continue;
|
|
5767
|
-
}
|
|
5768
|
-
let rendered = renderValue(parsed.value, parsed.usedMagnitude);
|
|
5769
|
-
let lastIdx = spanIdx[spanIdx.length - 1] ?? i;
|
|
5770
|
-
const unitWord = at(lastIdx + 2);
|
|
5771
|
-
if (isSpace(at(lastIdx + 1)) && WORD_RE.test(unitWord)) {
|
|
5772
|
-
const unit = unitWord.toLowerCase();
|
|
5773
|
-
if (PERCENT_RE.test(unit)) {
|
|
5774
|
-
rendered = `${rendered}%`;
|
|
5775
|
-
lastIdx += 2;
|
|
5776
|
-
} else if (CURRENCY_RE.test(unit)) {
|
|
5777
|
-
rendered = rendered.startsWith("$") ? rendered : `$${rendered}`;
|
|
5778
|
-
lastIdx += 2;
|
|
5779
|
-
}
|
|
5780
|
-
}
|
|
5781
|
-
out.push(rendered);
|
|
5782
|
-
i = lastIdx + 1;
|
|
5783
|
-
}
|
|
5784
|
-
return out.join("");
|
|
5785
|
-
};
|
|
5786
|
-
|
|
5787
5885
|
// src/core/backchannel.ts
|
|
5788
5886
|
var DEFAULT_CUES = [
|
|
5789
5887
|
{ text: "mm-hmm" },
|
|
@@ -6414,7 +6512,7 @@ var cloneTranscript = (transcript) => ({
|
|
|
6414
6512
|
});
|
|
6415
6513
|
var encodeBase64 = (chunk) => Buffer2.from(chunk).toString("base64");
|
|
6416
6514
|
var countWords2 = (text) => text.trim().split(/\s+/).filter(Boolean).length;
|
|
6417
|
-
var
|
|
6515
|
+
var normalizeText3 = (text) => text.trim().replace(/\s+/g, " ");
|
|
6418
6516
|
var getAudioChunkDurationMs = (chunk) => chunk.byteLength / (DEFAULT_FORMAT.sampleRateHz * DEFAULT_FORMAT.channels * 2) * 1000;
|
|
6419
6517
|
var getBufferedAudioDurationMs = (chunks) => chunks.reduce((total, chunk) => total + getAudioChunkDurationMs(chunk), 0);
|
|
6420
6518
|
var STREAM_SENTENCE_BOUNDARY = /[.!?\u2026]['")\]]*\s/;
|
|
@@ -6490,9 +6588,9 @@ var createTurnCostEstimate = (input) => {
|
|
|
6490
6588
|
totalBillableAudioMs: Math.max(0, input.primaryAudioMs) + Math.max(0, input.fallbackReplayAudioMs)
|
|
6491
6589
|
};
|
|
6492
6590
|
};
|
|
6493
|
-
var normalizeCorrectionText = (text) =>
|
|
6591
|
+
var normalizeCorrectionText = (text) => normalizeText3(text);
|
|
6494
6592
|
var evaluateFallbackNeed = (candidate, config) => {
|
|
6495
|
-
const trimmed =
|
|
6593
|
+
const trimmed = normalizeText3(candidate.text);
|
|
6496
6594
|
const wordCount = countWords2(trimmed);
|
|
6497
6595
|
const averageConfidence = calculateMeanConfidence(candidate.transcripts);
|
|
6498
6596
|
const words = collectTranscriptWords(candidate.transcripts);
|
|
@@ -7901,12 +7999,12 @@ var createVoiceSession = (options) => {
|
|
|
7901
7999
|
const fallbackCandidate = {
|
|
7902
8000
|
confidence: fallbackConfidence,
|
|
7903
8001
|
text: fallbackText,
|
|
7904
|
-
wordCount: countWords2(
|
|
8002
|
+
wordCount: countWords2(normalizeText3(fallbackText))
|
|
7905
8003
|
};
|
|
7906
8004
|
const primaryCandidate = {
|
|
7907
8005
|
confidence: calculateMeanConfidence(primaryTranscripts),
|
|
7908
8006
|
text: primaryText,
|
|
7909
|
-
wordCount: countWords2(
|
|
8007
|
+
wordCount: countWords2(normalizeText3(primaryText))
|
|
7910
8008
|
};
|
|
7911
8009
|
const policyPrefersFallback = fallbackCandidate.text.length > 0 && fallbackNeed.reason !== undefined && sttFallback.preferFallbackOn?.includes(fallbackNeed.reason);
|
|
7912
8010
|
const selection = policyPrefersFallback ? {
|
|
@@ -7999,7 +8097,7 @@ var createVoiceSession = (options) => {
|
|
|
7999
8097
|
};
|
|
8000
8098
|
const buildTurnSignature = (session, finalText, transcriptIdsOverride) => {
|
|
8001
8099
|
const finalTranscriptIds = transcriptIdsOverride ?? getFinalTranscriptIds(session.currentTurn.transcripts);
|
|
8002
|
-
return `${
|
|
8100
|
+
return `${normalizeText3(finalText)}|${finalTranscriptIds.join(",")}`;
|
|
8003
8101
|
};
|
|
8004
8102
|
const isDuplicateTurnCommit = (session, finalText) => {
|
|
8005
8103
|
const signature = buildTurnSignature(session, finalText);
|
|
@@ -8007,8 +8105,8 @@ var createVoiceSession = (options) => {
|
|
|
8007
8105
|
const isRecent = committedTurn && committedTurn.committedAt > 0 && Date.now() - committedTurn.committedAt < DEFAULT_DUPLICATE_TURN_WINDOW_MS;
|
|
8008
8106
|
const committedSignature = committedTurn?.signature ?? "";
|
|
8009
8107
|
const committedTranscriptIds = committedTurn?.transcriptIds ?? [];
|
|
8010
|
-
const committedText =
|
|
8011
|
-
const isSameText =
|
|
8108
|
+
const committedText = normalizeText3(committedTurn?.text ?? "");
|
|
8109
|
+
const isSameText = normalizeText3(finalText) === committedText;
|
|
8012
8110
|
const hasNoNewAudioSinceCommit = (session.currentTurn.lastAudioAt ?? 0) <= (committedTurn?.committedAt ?? 0);
|
|
8013
8111
|
if (!isRecent) {
|
|
8014
8112
|
return false;
|
|
@@ -8028,7 +8126,7 @@ var createVoiceSession = (options) => {
|
|
|
8028
8126
|
...session.lastCommittedTurn ?? {},
|
|
8029
8127
|
committedAt: Date.now(),
|
|
8030
8128
|
signature: buildTurnSignature(session, finalText, getFinalTranscriptIds(committedTranscripts)),
|
|
8031
|
-
text:
|
|
8129
|
+
text: normalizeText3(finalText),
|
|
8032
8130
|
transcriptIds: getFinalTranscriptIds(committedTranscripts)
|
|
8033
8131
|
};
|
|
8034
8132
|
};
|
|
@@ -9424,6 +9522,10 @@ var createVoiceSession = (options) => {
|
|
|
9424
9522
|
commitTurn: async (reason = "manual") => runSerial("api.commitTurn", async () => {
|
|
9425
9523
|
await commitTurnInternal(reason);
|
|
9426
9524
|
}),
|
|
9525
|
+
configureSTT: async (configuration) => runSerial("api.configureSTT", async () => {
|
|
9526
|
+
const adapter = await ensureAdapter();
|
|
9527
|
+
await adapter.configure?.(configuration);
|
|
9528
|
+
}),
|
|
9427
9529
|
complete: async (result) => runSerial("api.complete", async () => {
|
|
9428
9530
|
await completeInternal(result);
|
|
9429
9531
|
}),
|
|
@@ -10257,6 +10359,22 @@ var renderVoiceCallReviewMarkdown = (artifact) => {
|
|
|
10257
10359
|
].filter((value) => typeof value === "string").join(`
|
|
10258
10360
|
`);
|
|
10259
10361
|
};
|
|
10362
|
+
// src/testing/routingBenchmark.ts
|
|
10363
|
+
var evaluateVoiceSTTRouting = (fixtures) => {
|
|
10364
|
+
const attempted = fixtures.filter((fixture) => fixture.fallbackUsed);
|
|
10365
|
+
const improved = attempted.filter((fixture) => fixture.fallbackScore > fixture.primaryScore);
|
|
10366
|
+
const harmed = attempted.filter((fixture) => fixture.fallbackScore < fixture.primaryScore);
|
|
10367
|
+
const average3 = (values) => values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : 0;
|
|
10368
|
+
return {
|
|
10369
|
+
fallbackAttemptRate: fixtures.length > 0 ? attempted.length / fixtures.length : 0,
|
|
10370
|
+
fallbackHarmRate: attempted.length > 0 ? harmed.length / attempted.length : 0,
|
|
10371
|
+
fallbackImprovementRate: attempted.length > 0 ? improved.length / attempted.length : 0,
|
|
10372
|
+
fixtureCount: fixtures.length,
|
|
10373
|
+
oracleScore: average3(fixtures.map((fixture) => Math.max(fixture.primaryScore, fixture.fallbackScore))),
|
|
10374
|
+
primaryScore: average3(fixtures.map((fixture) => fixture.primaryScore)),
|
|
10375
|
+
selectedScore: average3(fixtures.map((fixture) => fixture.fallbackUsed ? fixture.fallbackScore : fixture.primaryScore))
|
|
10376
|
+
};
|
|
10377
|
+
};
|
|
10260
10378
|
// src/testing/sessionBenchmark.ts
|
|
10261
10379
|
var average3 = (values) => values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : 0;
|
|
10262
10380
|
var normalizeTurnText = (value) => value.toLowerCase().replace(/[^\p{L}\p{N}\s']/gu, " ").replace(/\s+/g, " ").trim();
|
|
@@ -15689,6 +15807,7 @@ export {
|
|
|
15689
15807
|
summarizeTTSBenchmark,
|
|
15690
15808
|
summarizeSTTBenchmarkSeries,
|
|
15691
15809
|
summarizeSTTBenchmark,
|
|
15810
|
+
scoreVoiceCriticalFields,
|
|
15692
15811
|
scoreTranscriptAccuracy,
|
|
15693
15812
|
scoreCorrectedExpectedTerms,
|
|
15694
15813
|
runVoiceTelephonyMediaOperationsSmoke,
|
|
@@ -15716,6 +15835,7 @@ export {
|
|
|
15716
15835
|
getDefaultVoiceTelephonyBenchmarkScenarios,
|
|
15717
15836
|
getDefaultVoiceDuplexBenchmarkScenarios,
|
|
15718
15837
|
getDefaultTTSBenchmarkFixtures,
|
|
15838
|
+
evaluateVoiceSTTRouting,
|
|
15719
15839
|
evaluateSTTBenchmarkAcceptance,
|
|
15720
15840
|
createVoiceProviderFailureSimulator,
|
|
15721
15841
|
createVoiceIOProviderFailureSimulator,
|
|
@@ -15727,6 +15847,7 @@ export {
|
|
|
15727
15847
|
createCodeSwitchBenchmarkCorrectionHandler,
|
|
15728
15848
|
createBenchmarkCorrectionHandler,
|
|
15729
15849
|
compareSTTBenchmarks,
|
|
15850
|
+
calibrateVoiceConfidence,
|
|
15730
15851
|
buildSessionCorrectionAudit,
|
|
15731
15852
|
buildFixturePhraseHints,
|
|
15732
15853
|
buildCorrectionBenchmarkAudit,
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
export type VoiceSTTRoutingFixture = {
|
|
2
|
+
fallbackScore: number;
|
|
3
|
+
fallbackUsed: boolean;
|
|
4
|
+
id: string;
|
|
5
|
+
primaryScore: number;
|
|
6
|
+
};
|
|
7
|
+
export type VoiceSTTRoutingBenchmarkReport = {
|
|
8
|
+
fallbackAttemptRate: number;
|
|
9
|
+
fallbackHarmRate: number;
|
|
10
|
+
fallbackImprovementRate: number;
|
|
11
|
+
fixtureCount: number;
|
|
12
|
+
oracleScore: number;
|
|
13
|
+
primaryScore: number;
|
|
14
|
+
selectedScore: number;
|
|
15
|
+
};
|
|
16
|
+
export declare const evaluateVoiceSTTRouting: (fixtures: VoiceSTTRoutingFixture[]) => VoiceSTTRoutingBenchmarkReport;
|