@absolutejs/voice 0.0.22-beta.634 → 0.0.22-beta.636
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -0
- package/dist/core/types.d.ts +28 -1
- package/dist/index.js +55 -3
- package/dist/testing/benchmark.d.ts +4 -0
- package/dist/testing/confidenceCalibration.d.ts +19 -0
- package/dist/testing/criticalFields.d.ts +22 -0
- package/dist/testing/fixtures.d.ts +2 -0
- package/dist/testing/index.d.ts +3 -0
- package/dist/testing/index.js +306 -197
- package/dist/testing/routingBenchmark.d.ts +16 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -5015,6 +5015,14 @@ That keeps HTMX pages declarative without inventing custom fragment endpoints fo
|
|
|
5015
5015
|
|
|
5016
5016
|
Performance & accuracy benchmarks (STT, TTS, duplex, telephony, sessions) and head-to-head comparisons against Vapi live in a dedicated repo: **[absolutejs/benchmarks](https://github.com/absolutejs/benchmarks)**. They consume the published `@absolutejs/voice` package and provider adapters.
|
|
5017
5017
|
|
|
5018
|
+
Reusable eval contracts stay in `@absolutejs/voice/testing`: fixture manifests
|
|
5019
|
+
can label critical names, organizations, currency, percentages, phone numbers,
|
|
5020
|
+
and other exact fields; benchmark reports score those fields independently from
|
|
5021
|
+
WER; confidence calibration reports ECE/Brier scores; and routing reports expose
|
|
5022
|
+
fallback improvement and harm rates. Audio corpora and executable provider runs
|
|
5023
|
+
remain separate so applications can consume the same contracts without shipping
|
|
5024
|
+
benchmark media in the runtime package.
|
|
5025
|
+
|
|
5018
5026
|
## Adapter Contract
|
|
5019
5027
|
|
|
5020
5028
|
Adapters normalize vendor behavior into a core event model so the plugin never branches on vendor names.
|
|
@@ -5035,6 +5043,7 @@ type STTAdapterSession = {
|
|
|
5035
5043
|
handler: (payload: STTSessionEventMap[K]) => void | Promise<void>,
|
|
5036
5044
|
) => () => void;
|
|
5037
5045
|
send: (audio: AudioChunk) => Promise<void>;
|
|
5046
|
+
configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
|
|
5038
5047
|
close: (reason?: string) => Promise<void>;
|
|
5039
5048
|
};
|
|
5040
5049
|
```
|
package/dist/core/types.d.ts
CHANGED
|
@@ -90,6 +90,15 @@ export type TranscriptWord = {
|
|
|
90
90
|
startedAtMs?: number;
|
|
91
91
|
text: string;
|
|
92
92
|
};
|
|
93
|
+
/** Provider token evidence. Tokens are intentionally kept separate from words:
|
|
94
|
+
* subword log probabilities are useful for calibration and routing, but are not
|
|
95
|
+
* word-level timestamps or confidence scores. */
|
|
96
|
+
export type TranscriptToken = {
|
|
97
|
+
bytes?: number[];
|
|
98
|
+
confidence?: number;
|
|
99
|
+
logProbability?: number;
|
|
100
|
+
text: string;
|
|
101
|
+
};
|
|
93
102
|
export type Transcript = {
|
|
94
103
|
id: string;
|
|
95
104
|
text: string;
|
|
@@ -101,8 +110,15 @@ export type Transcript = {
|
|
|
101
110
|
startedAtMs?: number;
|
|
102
111
|
endedAtMs?: number;
|
|
103
112
|
vendor?: string;
|
|
113
|
+
tokens?: TranscriptToken[];
|
|
104
114
|
words?: TranscriptWord[];
|
|
105
115
|
};
|
|
116
|
+
export type VoiceSTTSessionConfiguration = {
|
|
117
|
+
languageHints?: string[];
|
|
118
|
+
lexicon?: VoiceLexiconEntry[];
|
|
119
|
+
phraseHints?: VoicePhraseHint[];
|
|
120
|
+
turnDetection?: Partial<VoiceTurnDetectionConfig>;
|
|
121
|
+
};
|
|
106
122
|
export type VoiceTranscriptQuality = {
|
|
107
123
|
averageConfidence?: number;
|
|
108
124
|
confidenceSampleCount: number;
|
|
@@ -133,7 +149,7 @@ export type VoiceTurnCostEstimate = {
|
|
|
133
149
|
primaryAudioMs: number;
|
|
134
150
|
totalBillableAudioMs: number;
|
|
135
151
|
};
|
|
136
|
-
export type VoiceFallbackSelectionReason = "fallback-empty" | "primary-empty" | "word-count-margin" | "confidence-margin" | "word-count-tiebreak" | "kept-primary";
|
|
152
|
+
export type VoiceFallbackSelectionReason = "fallback-empty" | "primary-empty" | "word-count-margin" | "confidence-margin" | "word-count-tiebreak" | "policy-preference" | "kept-primary";
|
|
137
153
|
export type VoiceFallbackDiagnostics = {
|
|
138
154
|
attempted: boolean;
|
|
139
155
|
fallbackConfidence?: number;
|
|
@@ -184,6 +200,8 @@ export type STTSessionEventMap = {
|
|
|
184
200
|
export type STTAdapterSession = {
|
|
185
201
|
on: <K extends keyof STTSessionEventMap>(event: K, handler: (payload: STTSessionEventMap[K]) => void | Promise<void>) => () => void;
|
|
186
202
|
send: (audio: AudioChunk) => Promise<void>;
|
|
203
|
+
/** Update provider-supported STT context without restarting the stream. */
|
|
204
|
+
configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
|
|
187
205
|
close: (reason?: string) => Promise<void>;
|
|
188
206
|
};
|
|
189
207
|
export type STTAdapterOpenOptions = {
|
|
@@ -240,6 +258,8 @@ export type RealtimeSessionEventMap = STTSessionEventMap & {
|
|
|
240
258
|
export type RealtimeAdapterSession = {
|
|
241
259
|
on: <K extends keyof RealtimeSessionEventMap>(event: K, handler: (payload: RealtimeSessionEventMap[K]) => void | Promise<void>) => () => void;
|
|
242
260
|
send: (input: AudioChunk | string) => Promise<void>;
|
|
261
|
+
/** Update provider-supported input transcription context in place. */
|
|
262
|
+
configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
|
|
243
263
|
close: (reason?: string) => Promise<void>;
|
|
244
264
|
};
|
|
245
265
|
export type RealtimeAdapterOpenOptions = {
|
|
@@ -458,6 +478,9 @@ export type VoiceSTTFallbackConfig = {
|
|
|
458
478
|
* language that should receive a second transcription pass regardless of the
|
|
459
479
|
* aggregate confidence. */
|
|
460
480
|
riskPolicy?: VoiceSTTFallbackRiskPolicy;
|
|
481
|
+
/** Prefer a non-empty independent fallback for these audited trigger reasons,
|
|
482
|
+
* including providers that do not expose comparable confidence scores. */
|
|
483
|
+
preferFallbackOn?: VoiceFallbackTriggerReason[];
|
|
461
484
|
};
|
|
462
485
|
export type VoiceResolvedSTTFallbackConfig = {
|
|
463
486
|
adapter: STTAdapter;
|
|
@@ -470,6 +493,7 @@ export type VoiceResolvedSTTFallbackConfig = {
|
|
|
470
493
|
maxAttemptsPerTurn: number;
|
|
471
494
|
wordConfidenceThreshold?: number;
|
|
472
495
|
riskPolicy?: VoiceSTTFallbackRiskPolicy;
|
|
496
|
+
preferFallbackOn?: VoiceFallbackTriggerReason[];
|
|
473
497
|
};
|
|
474
498
|
export type VoiceTurnDetectionConfig = {
|
|
475
499
|
profile?: VoiceTurnProfile;
|
|
@@ -580,6 +604,9 @@ export type VoiceSessionHandle<TContext = unknown, TSession extends VoiceSession
|
|
|
580
604
|
speechThreshold: number;
|
|
581
605
|
transcriptStabilityMs: number;
|
|
582
606
|
}>;
|
|
607
|
+
/** Refresh vocabulary, language hints, or provider turn settings while a
|
|
608
|
+
* call is active. Unsupported fields are ignored by the active adapter. */
|
|
609
|
+
configureSTT: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
|
|
583
610
|
};
|
|
584
611
|
export type VoiceLLMUsage = {
|
|
585
612
|
provider?: string;
|
package/dist/index.js
CHANGED
|
@@ -4260,7 +4260,8 @@ var createVoiceSession = (options) => {
|
|
|
4260
4260
|
settleMs: options.sttFallback.settleMs ?? DEFAULT_FALLBACK_SETTLE_MS,
|
|
4261
4261
|
trigger: options.sttFallback.trigger ?? "empty-or-low-confidence",
|
|
4262
4262
|
wordConfidenceThreshold: options.sttFallback.wordConfidenceThreshold,
|
|
4263
|
-
riskPolicy: options.sttFallback.riskPolicy
|
|
4263
|
+
riskPolicy: options.sttFallback.riskPolicy,
|
|
4264
|
+
preferFallbackOn: options.sttFallback.preferFallbackOn
|
|
4264
4265
|
} : undefined;
|
|
4265
4266
|
const appendTrace = async (input) => {
|
|
4266
4267
|
await options.trace?.append({
|
|
@@ -5524,7 +5525,11 @@ var createVoiceSession = (options) => {
|
|
|
5524
5525
|
text: primaryText,
|
|
5525
5526
|
wordCount: countWords2(normalizeText2(primaryText))
|
|
5526
5527
|
};
|
|
5527
|
-
const
|
|
5528
|
+
const policyPrefersFallback = fallbackCandidate.text.length > 0 && fallbackNeed.reason !== undefined && sttFallback.preferFallbackOn?.includes(fallbackNeed.reason);
|
|
5529
|
+
const selection = policyPrefersFallback ? {
|
|
5530
|
+
reason: "policy-preference",
|
|
5531
|
+
winner: fallbackCandidate
|
|
5532
|
+
} : selectBetterTurnText(primaryCandidate, fallbackCandidate);
|
|
5528
5533
|
const diagnostics = {
|
|
5529
5534
|
attempted: true,
|
|
5530
5535
|
fallbackConfidence: fallbackCandidate.confidence,
|
|
@@ -7036,6 +7041,10 @@ var createVoiceSession = (options) => {
|
|
|
7036
7041
|
commitTurn: async (reason = "manual") => runSerial("api.commitTurn", async () => {
|
|
7037
7042
|
await commitTurnInternal(reason);
|
|
7038
7043
|
}),
|
|
7044
|
+
configureSTT: async (configuration) => runSerial("api.configureSTT", async () => {
|
|
7045
|
+
const adapter = await ensureAdapter();
|
|
7046
|
+
await adapter.configure?.(configuration);
|
|
7047
|
+
}),
|
|
7039
7048
|
complete: async (result) => runSerial("api.complete", async () => {
|
|
7040
7049
|
await completeInternal(result);
|
|
7041
7050
|
}),
|
|
@@ -44488,6 +44497,7 @@ var createContractApi = (session) => ({
|
|
|
44488
44497
|
id: session.id,
|
|
44489
44498
|
attachUserMedia: async () => {},
|
|
44490
44499
|
close: async () => {},
|
|
44500
|
+
configureSTT: async () => {},
|
|
44491
44501
|
commitTurn: async () => {},
|
|
44492
44502
|
complete: async () => {},
|
|
44493
44503
|
connect: async () => {},
|
|
@@ -49553,6 +49563,44 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
|
|
|
49553
49563
|
};
|
|
49554
49564
|
};
|
|
49555
49565
|
|
|
49566
|
+
// src/testing/criticalFields.ts
|
|
49567
|
+
var normalizeText4 = (value) => value.normalize("NFKC").toLowerCase().replace(/[^\p{L}\p{N}@.%+$'-]+/gu, " ").replace(/\s+/g, " ").trim();
|
|
49568
|
+
var normalizeDigits = (value) => normalizeSpokenNumbers(value).replace(/\D/g, "");
|
|
49569
|
+
var normalizeSemanticNumber = (value) => normalizeSpokenNumbers(value).toLowerCase().replace(/\bdollars?\b|\busd\b|\bpercent(age)?\b|[%,$]/g, "").replace(/\s+/g, "").trim();
|
|
49570
|
+
var matchesCandidate = (actual, candidate, kind) => {
|
|
49571
|
+
if (kind === "phone") {
|
|
49572
|
+
const expectedDigits = normalizeDigits(candidate);
|
|
49573
|
+
return expectedDigits.length > 0 && normalizeDigits(actual).includes(expectedDigits);
|
|
49574
|
+
}
|
|
49575
|
+
if (kind === "currency" || kind === "number" || kind === "percentage") {
|
|
49576
|
+
const expectedNumber = normalizeSemanticNumber(candidate);
|
|
49577
|
+
return expectedNumber.length > 0 && normalizeSemanticNumber(actual).includes(expectedNumber);
|
|
49578
|
+
}
|
|
49579
|
+
const normalizedCandidate = normalizeText4(candidate);
|
|
49580
|
+
return normalizedCandidate.length > 0 && normalizeText4(actual).includes(normalizedCandidate);
|
|
49581
|
+
};
|
|
49582
|
+
var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
|
|
49583
|
+
const fields = expectedFields.map((field) => {
|
|
49584
|
+
const candidates = [field.value, ...field.aliases ?? []];
|
|
49585
|
+
const matchedAlias = candidates.find((candidate) => matchesCandidate(actualText, candidate, field.kind));
|
|
49586
|
+
return {
|
|
49587
|
+
...field,
|
|
49588
|
+
matched: matchedAlias !== undefined,
|
|
49589
|
+
matchedAlias
|
|
49590
|
+
};
|
|
49591
|
+
});
|
|
49592
|
+
const matchedCount = fields.filter((field) => field.matched).length;
|
|
49593
|
+
const totalCount = fields.length;
|
|
49594
|
+
return {
|
|
49595
|
+
accuracy: totalCount > 0 ? matchedCount / totalCount : 1,
|
|
49596
|
+
fields,
|
|
49597
|
+
matchedCount,
|
|
49598
|
+
missingFieldIds: fields.filter((field) => !field.matched).map((field) => field.id),
|
|
49599
|
+
passesRequired: fields.every((field) => field.required === false || field.matched),
|
|
49600
|
+
totalCount
|
|
49601
|
+
};
|
|
49602
|
+
};
|
|
49603
|
+
|
|
49556
49604
|
// src/testing/benchmark.ts
|
|
49557
49605
|
var resolveFixtureEnvironment = (fixture) => {
|
|
49558
49606
|
const tags = new Set(fixture.tags ?? []);
|
|
@@ -49783,10 +49831,12 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
49783
49831
|
const postSpeechTimeToFirstFinalMs = toPostSpeechLatency(result.finalEvents[0]?.receivedAt);
|
|
49784
49832
|
const postSpeechTimeToEndOfTurnMs = toPostSpeechLatency(result.endOfTurnEvents[0]?.receivedAt);
|
|
49785
49833
|
const expectedTerms = scoreExpectedTerms(result.finalText, fixture.expectedTerms);
|
|
49834
|
+
const criticalFields = scoreVoiceCriticalFields(result.finalText, fixture.expectedCriticalFields);
|
|
49786
49835
|
const speakerTurns = scoreSpeakerTurns(fixture, result);
|
|
49787
49836
|
return {
|
|
49788
49837
|
accuracy: result.accuracy,
|
|
49789
49838
|
closeCount: result.closeEvents.length,
|
|
49839
|
+
criticalFields,
|
|
49790
49840
|
difficulty: fixture.difficulty,
|
|
49791
49841
|
elapsedMs,
|
|
49792
49842
|
endOfTurnCount: result.endOfTurnEvents.length,
|
|
@@ -49797,7 +49847,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
49797
49847
|
fixtureId: fixture.id,
|
|
49798
49848
|
fragmentationCount: Math.max(0, result.finalEvents.length - 1),
|
|
49799
49849
|
group: resolveFixtureEnvironment(fixture),
|
|
49800
|
-
passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && (speakerTurns ? speakerTurns.passes : true),
|
|
49850
|
+
passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && criticalFields.passesRequired && (speakerTurns ? speakerTurns.passes : true),
|
|
49801
49851
|
partialCount: result.partialEvents.length,
|
|
49802
49852
|
speakerTurns,
|
|
49803
49853
|
postSpeechTimeToEndOfTurnMs,
|
|
@@ -49925,6 +49975,8 @@ var summarizeSTTBenchmark = (adapterId, fixtures) => {
|
|
|
49925
49975
|
averageFinalCount: roundMetric4(average2(fixtures.map((fixture) => fixture.finalCount)), 2) ?? 0,
|
|
49926
49976
|
averageSpeakerTurnMatchRate: roundMetric4(average2(fixtures.map((fixture) => fixture.speakerTurns?.patternMatchRate))),
|
|
49927
49977
|
averageTermRecall: roundMetric4(average2(fixtures.map((fixture) => fixture.expectedTerms.recall))) ?? 0,
|
|
49978
|
+
averageCriticalFieldAccuracy: roundMetric4(average2(fixtures.map((fixture) => fixture.criticalFields?.accuracy ?? 1))) ?? 0,
|
|
49979
|
+
requiredCriticalFieldPassRate: fixtureCount > 0 ? roundMetric4(fixtures.filter((fixture) => fixture.criticalFields?.passesRequired ?? true).length / fixtureCount) ?? 0 : 0,
|
|
49928
49980
|
averagePostSpeechTimeToEndOfTurnMs: roundMetric4(average2(fixtures.map((fixture) => fixture.postSpeechTimeToEndOfTurnMs)), 2),
|
|
49929
49981
|
averagePostSpeechTimeToFirstFinalMs: roundMetric4(average2(fixtures.map((fixture) => fixture.postSpeechTimeToFirstFinalMs)), 2),
|
|
49930
49982
|
averageTimeToEndOfTurnMs: roundMetric4(average2(fixtures.map((fixture) => fixture.timeToEndOfTurnMs)), 2),
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import type { STTAdapter, STTAdapterOpenOptions } from "../core/types";
|
|
2
2
|
import { type VoiceSTTAdapterHarnessOptions, type VoiceSTTAdapterHarnessResult } from "./stt";
|
|
3
3
|
import type { VoiceTestFixture } from "./fixtures";
|
|
4
|
+
import { type VoiceCriticalFieldAccuracy } from "./criticalFields";
|
|
4
5
|
export type VoiceExpectedTermAccuracy = {
|
|
5
6
|
allMatched: boolean;
|
|
6
7
|
expectedTerms: string[];
|
|
@@ -25,6 +26,7 @@ export type VoiceSTTBenchmarkFixtureResult = {
|
|
|
25
26
|
endOfTurnCount: number;
|
|
26
27
|
errorCount: number;
|
|
27
28
|
expectedTerms: VoiceExpectedTermAccuracy;
|
|
29
|
+
criticalFields?: VoiceCriticalFieldAccuracy;
|
|
28
30
|
finalCount: number;
|
|
29
31
|
finalText: string;
|
|
30
32
|
fixtureId: string;
|
|
@@ -49,6 +51,8 @@ export type VoiceSTTBenchmarkSummary = {
|
|
|
49
51
|
averageFinalCount: number;
|
|
50
52
|
averageSpeakerTurnMatchRate?: number;
|
|
51
53
|
averageTermRecall: number;
|
|
54
|
+
averageCriticalFieldAccuracy: number;
|
|
55
|
+
requiredCriticalFieldPassRate: number;
|
|
52
56
|
averagePostSpeechTimeToEndOfTurnMs?: number;
|
|
53
57
|
averagePostSpeechTimeToFirstFinalMs?: number;
|
|
54
58
|
averageTimeToEndOfTurnMs?: number;
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
export type VoiceConfidenceCalibrationSample = {
|
|
2
|
+
confidence: number;
|
|
3
|
+
correct: boolean;
|
|
4
|
+
metadata?: Record<string, unknown>;
|
|
5
|
+
};
|
|
6
|
+
export type VoiceConfidenceCalibrationBin = {
|
|
7
|
+
accuracy: number;
|
|
8
|
+
averageConfidence: number;
|
|
9
|
+
count: number;
|
|
10
|
+
lowerBound: number;
|
|
11
|
+
upperBound: number;
|
|
12
|
+
};
|
|
13
|
+
export type VoiceConfidenceCalibrationReport = {
|
|
14
|
+
bins: VoiceConfidenceCalibrationBin[];
|
|
15
|
+
brierScore: number;
|
|
16
|
+
expectedCalibrationError: number;
|
|
17
|
+
sampleCount: number;
|
|
18
|
+
};
|
|
19
|
+
export declare const calibrateVoiceConfidence: (samples: VoiceConfidenceCalibrationSample[], binCount?: number) => VoiceConfidenceCalibrationReport;
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
export type VoiceCriticalFieldKind = "acronym" | "brand" | "currency" | "custom" | "email" | "number" | "organization" | "percentage" | "person-name" | "phone";
|
|
2
|
+
export type VoiceExpectedCriticalField = {
|
|
3
|
+
aliases?: string[];
|
|
4
|
+
id: string;
|
|
5
|
+
kind: VoiceCriticalFieldKind;
|
|
6
|
+
metadata?: Record<string, unknown>;
|
|
7
|
+
required?: boolean;
|
|
8
|
+
value: string;
|
|
9
|
+
};
|
|
10
|
+
export type VoiceCriticalFieldResult = VoiceExpectedCriticalField & {
|
|
11
|
+
matched: boolean;
|
|
12
|
+
matchedAlias?: string;
|
|
13
|
+
};
|
|
14
|
+
export type VoiceCriticalFieldAccuracy = {
|
|
15
|
+
accuracy: number;
|
|
16
|
+
fields: VoiceCriticalFieldResult[];
|
|
17
|
+
matchedCount: number;
|
|
18
|
+
missingFieldIds: string[];
|
|
19
|
+
passesRequired: boolean;
|
|
20
|
+
totalCount: number;
|
|
21
|
+
};
|
|
22
|
+
export declare const scoreVoiceCriticalFields: (actualText: string, expectedFields?: VoiceExpectedCriticalField[]) => VoiceCriticalFieldAccuracy;
|
|
@@ -1,10 +1,12 @@
|
|
|
1
1
|
import type { AudioFormat, VoiceExpectedSpeakerTurn } from "../core/types";
|
|
2
|
+
import type { VoiceExpectedCriticalField } from "./criticalFields";
|
|
2
3
|
export type VoiceTestFixtureManifestEntry = {
|
|
3
4
|
id: string;
|
|
4
5
|
title: string;
|
|
5
6
|
audioPath: string;
|
|
6
7
|
expectedText: string;
|
|
7
8
|
expectedTerms?: string[];
|
|
9
|
+
expectedCriticalFields?: VoiceExpectedCriticalField[];
|
|
8
10
|
expectedSpeakerTurns?: VoiceExpectedSpeakerTurn[];
|
|
9
11
|
expectedTurnTexts?: string[];
|
|
10
12
|
chunkDurationMs?: number;
|
package/dist/testing/index.d.ts
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
export * from "./accuracy";
|
|
2
2
|
export * from "./benchmark";
|
|
3
|
+
export * from "./confidenceCalibration";
|
|
4
|
+
export * from "./criticalFields";
|
|
3
5
|
export * from "./corrected";
|
|
4
6
|
export * from "./duplex";
|
|
5
7
|
export * from "./fixtures";
|
|
@@ -7,6 +9,7 @@ export * from "./ioProviderSimulator";
|
|
|
7
9
|
export * from "./providerSimulator";
|
|
8
10
|
export * from "./resilience";
|
|
9
11
|
export * from "./review";
|
|
12
|
+
export * from "./routingBenchmark";
|
|
10
13
|
export * from "./sessionBenchmark";
|
|
11
14
|
export * from "./stt";
|
|
12
15
|
export * from "./telephony";
|
package/dist/testing/index.js
CHANGED
|
@@ -367,6 +367,229 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
|
|
|
367
367
|
};
|
|
368
368
|
};
|
|
369
369
|
|
|
370
|
+
// src/core/numberNormalizer.ts
|
|
371
|
+
var ONES = {
|
|
372
|
+
eight: 8,
|
|
373
|
+
eighteen: 18,
|
|
374
|
+
eleven: 11,
|
|
375
|
+
fifteen: 15,
|
|
376
|
+
five: 5,
|
|
377
|
+
four: 4,
|
|
378
|
+
fourteen: 14,
|
|
379
|
+
nine: 9,
|
|
380
|
+
nineteen: 19,
|
|
381
|
+
one: 1,
|
|
382
|
+
seven: 7,
|
|
383
|
+
seventeen: 17,
|
|
384
|
+
six: 6,
|
|
385
|
+
sixteen: 16,
|
|
386
|
+
ten: 10,
|
|
387
|
+
thirteen: 13,
|
|
388
|
+
three: 3,
|
|
389
|
+
twelve: 12,
|
|
390
|
+
two: 2,
|
|
391
|
+
zero: 0
|
|
392
|
+
};
|
|
393
|
+
var TENS = {
|
|
394
|
+
eighty: 80,
|
|
395
|
+
fifty: 50,
|
|
396
|
+
forty: 40,
|
|
397
|
+
ninety: 90,
|
|
398
|
+
seventy: 70,
|
|
399
|
+
sixty: 60,
|
|
400
|
+
thirty: 30,
|
|
401
|
+
twenty: 20
|
|
402
|
+
};
|
|
403
|
+
var SCALES = {
|
|
404
|
+
billion: 1e9,
|
|
405
|
+
million: 1e6,
|
|
406
|
+
thousand: 1000,
|
|
407
|
+
trillion: 1000000000000
|
|
408
|
+
};
|
|
409
|
+
var MAGNITUDE_WORDS = [
|
|
410
|
+
[1000000000000, "trillion"],
|
|
411
|
+
[1e9, "billion"],
|
|
412
|
+
[1e6, "million"]
|
|
413
|
+
];
|
|
414
|
+
var FILLER = new Set(["and", "a", "an"]);
|
|
415
|
+
var DECIMAL_PLACES = 3;
|
|
416
|
+
var isValueWord = (word) => (word in ONES) || (word in TENS) || word === "hundred";
|
|
417
|
+
var isScaleWord = (word) => (word in SCALES);
|
|
418
|
+
var isNumberWord = (word) => isValueWord(word) || isScaleWord(word) || word === "point" || FILLER.has(word);
|
|
419
|
+
var trimNumber = (value) => {
|
|
420
|
+
if (Number.isInteger(value))
|
|
421
|
+
return value.toLocaleString("en-US");
|
|
422
|
+
return String(Number(value.toFixed(DECIMAL_PLACES)));
|
|
423
|
+
};
|
|
424
|
+
var renderValue = (value, usedMagnitude) => {
|
|
425
|
+
if (usedMagnitude) {
|
|
426
|
+
for (const [scale, word] of MAGNITUDE_WORDS) {
|
|
427
|
+
if (value >= scale) {
|
|
428
|
+
const scaled = value / scale;
|
|
429
|
+
if (Number(scaled.toFixed(DECIMAL_PLACES)) === scaled) {
|
|
430
|
+
return `${trimNumber(scaled)} ${word}`;
|
|
431
|
+
}
|
|
432
|
+
}
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
return trimNumber(value);
|
|
436
|
+
};
|
|
437
|
+
var parseNumberWords = (words) => {
|
|
438
|
+
let total = 0;
|
|
439
|
+
let current = 0;
|
|
440
|
+
let usedMagnitude = false;
|
|
441
|
+
let sawNumber = false;
|
|
442
|
+
let decimal = null;
|
|
443
|
+
const foldDecimal = () => {
|
|
444
|
+
if (decimal && decimal.length > 0)
|
|
445
|
+
current += Number(`0.${decimal}`);
|
|
446
|
+
decimal = null;
|
|
447
|
+
};
|
|
448
|
+
for (const word of words) {
|
|
449
|
+
if (word === "point") {
|
|
450
|
+
decimal = "";
|
|
451
|
+
continue;
|
|
452
|
+
}
|
|
453
|
+
const one = ONES[word];
|
|
454
|
+
const ten = TENS[word];
|
|
455
|
+
const scale = SCALES[word];
|
|
456
|
+
if (decimal !== null) {
|
|
457
|
+
if (one !== undefined && one <= 9) {
|
|
458
|
+
decimal += String(one);
|
|
459
|
+
sawNumber = true;
|
|
460
|
+
continue;
|
|
461
|
+
}
|
|
462
|
+
foldDecimal();
|
|
463
|
+
}
|
|
464
|
+
if (FILLER.has(word))
|
|
465
|
+
continue;
|
|
466
|
+
if (one !== undefined) {
|
|
467
|
+
current += one;
|
|
468
|
+
sawNumber = true;
|
|
469
|
+
} else if (ten !== undefined) {
|
|
470
|
+
current += ten;
|
|
471
|
+
sawNumber = true;
|
|
472
|
+
} else if (word === "hundred") {
|
|
473
|
+
current = (current === 0 ? 1 : current) * 100;
|
|
474
|
+
sawNumber = true;
|
|
475
|
+
} else if (scale !== undefined) {
|
|
476
|
+
total += (current === 0 ? 1 : current) * scale;
|
|
477
|
+
current = 0;
|
|
478
|
+
sawNumber = true;
|
|
479
|
+
usedMagnitude = true;
|
|
480
|
+
}
|
|
481
|
+
}
|
|
482
|
+
foldDecimal();
|
|
483
|
+
if (!sawNumber)
|
|
484
|
+
return null;
|
|
485
|
+
return { usedMagnitude, value: total + current };
|
|
486
|
+
};
|
|
487
|
+
var PERCENT_RE = /^(per ?cent|percent|percentage)$/;
|
|
488
|
+
var CURRENCY_RE = /^(dollars?|bucks?|usd)$/;
|
|
489
|
+
var WORD_RE = /^[A-Za-z]+(?:-[A-Za-z]+)*$/;
|
|
490
|
+
var wordsOf = (token) => token.toLowerCase().split("-");
|
|
491
|
+
var isNumberToken = (token) => WORD_RE.test(token) && wordsOf(token).every(isNumberWord);
|
|
492
|
+
var startsNumber = (token) => WORD_RE.test(token) && wordsOf(token).some(isValueWord);
|
|
493
|
+
var isSpace = (token) => /^\s+$/.test(token);
|
|
494
|
+
var normalizeSpokenNumbers = (input) => {
|
|
495
|
+
if (!input)
|
|
496
|
+
return input;
|
|
497
|
+
const parts = input.match(/[A-Za-z]+(?:-[A-Za-z]+)*|[^A-Za-z]+/g);
|
|
498
|
+
if (!parts)
|
|
499
|
+
return input;
|
|
500
|
+
const at = (idx) => parts[idx] ?? "";
|
|
501
|
+
const out = [];
|
|
502
|
+
let i = 0;
|
|
503
|
+
while (i < parts.length) {
|
|
504
|
+
const token = at(i);
|
|
505
|
+
const lower = token.toLowerCase();
|
|
506
|
+
if (lower === "a" || lower === "an") {
|
|
507
|
+
const nextWord = at(i + 2);
|
|
508
|
+
const nextHead = wordsOf(nextWord)[0] ?? "";
|
|
509
|
+
const nextIsScale = isSpace(at(i + 1)) && WORD_RE.test(nextWord) && (nextHead === "hundred" || isScaleWord(nextHead));
|
|
510
|
+
if (!nextIsScale) {
|
|
511
|
+
out.push(token);
|
|
512
|
+
i += 1;
|
|
513
|
+
continue;
|
|
514
|
+
}
|
|
515
|
+
} else if (!startsNumber(token)) {
|
|
516
|
+
out.push(token);
|
|
517
|
+
i += 1;
|
|
518
|
+
continue;
|
|
519
|
+
}
|
|
520
|
+
const spanIdx = [i];
|
|
521
|
+
let j = i + 1;
|
|
522
|
+
while (isSpace(at(j)) && isNumberToken(at(j + 1))) {
|
|
523
|
+
spanIdx.push(j + 1);
|
|
524
|
+
j += 2;
|
|
525
|
+
}
|
|
526
|
+
const words = spanIdx.flatMap((k) => wordsOf(at(k)));
|
|
527
|
+
while (words.length > 0 && FILLER.has(words[words.length - 1] ?? "")) {
|
|
528
|
+
words.pop();
|
|
529
|
+
}
|
|
530
|
+
const parsed = parseNumberWords(words);
|
|
531
|
+
if (!parsed) {
|
|
532
|
+
out.push(token);
|
|
533
|
+
i += 1;
|
|
534
|
+
continue;
|
|
535
|
+
}
|
|
536
|
+
let rendered = renderValue(parsed.value, parsed.usedMagnitude);
|
|
537
|
+
let lastIdx = spanIdx[spanIdx.length - 1] ?? i;
|
|
538
|
+
const unitWord = at(lastIdx + 2);
|
|
539
|
+
if (isSpace(at(lastIdx + 1)) && WORD_RE.test(unitWord)) {
|
|
540
|
+
const unit = unitWord.toLowerCase();
|
|
541
|
+
if (PERCENT_RE.test(unit)) {
|
|
542
|
+
rendered = `${rendered}%`;
|
|
543
|
+
lastIdx += 2;
|
|
544
|
+
} else if (CURRENCY_RE.test(unit)) {
|
|
545
|
+
rendered = rendered.startsWith("$") ? rendered : `$${rendered}`;
|
|
546
|
+
lastIdx += 2;
|
|
547
|
+
}
|
|
548
|
+
}
|
|
549
|
+
out.push(rendered);
|
|
550
|
+
i = lastIdx + 1;
|
|
551
|
+
}
|
|
552
|
+
return out.join("");
|
|
553
|
+
};
|
|
554
|
+
|
|
555
|
+
// src/testing/criticalFields.ts
|
|
556
|
+
var normalizeText2 = (value) => value.normalize("NFKC").toLowerCase().replace(/[^\p{L}\p{N}@.%+$'-]+/gu, " ").replace(/\s+/g, " ").trim();
|
|
557
|
+
var normalizeDigits = (value) => normalizeSpokenNumbers(value).replace(/\D/g, "");
|
|
558
|
+
var normalizeSemanticNumber = (value) => normalizeSpokenNumbers(value).toLowerCase().replace(/\bdollars?\b|\busd\b|\bpercent(age)?\b|[%,$]/g, "").replace(/\s+/g, "").trim();
|
|
559
|
+
var matchesCandidate = (actual, candidate, kind) => {
|
|
560
|
+
if (kind === "phone") {
|
|
561
|
+
const expectedDigits = normalizeDigits(candidate);
|
|
562
|
+
return expectedDigits.length > 0 && normalizeDigits(actual).includes(expectedDigits);
|
|
563
|
+
}
|
|
564
|
+
if (kind === "currency" || kind === "number" || kind === "percentage") {
|
|
565
|
+
const expectedNumber = normalizeSemanticNumber(candidate);
|
|
566
|
+
return expectedNumber.length > 0 && normalizeSemanticNumber(actual).includes(expectedNumber);
|
|
567
|
+
}
|
|
568
|
+
const normalizedCandidate = normalizeText2(candidate);
|
|
569
|
+
return normalizedCandidate.length > 0 && normalizeText2(actual).includes(normalizedCandidate);
|
|
570
|
+
};
|
|
571
|
+
var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
|
|
572
|
+
const fields = expectedFields.map((field) => {
|
|
573
|
+
const candidates = [field.value, ...field.aliases ?? []];
|
|
574
|
+
const matchedAlias = candidates.find((candidate) => matchesCandidate(actualText, candidate, field.kind));
|
|
575
|
+
return {
|
|
576
|
+
...field,
|
|
577
|
+
matched: matchedAlias !== undefined,
|
|
578
|
+
matchedAlias
|
|
579
|
+
};
|
|
580
|
+
});
|
|
581
|
+
const matchedCount = fields.filter((field) => field.matched).length;
|
|
582
|
+
const totalCount = fields.length;
|
|
583
|
+
return {
|
|
584
|
+
accuracy: totalCount > 0 ? matchedCount / totalCount : 1,
|
|
585
|
+
fields,
|
|
586
|
+
matchedCount,
|
|
587
|
+
missingFieldIds: fields.filter((field) => !field.matched).map((field) => field.id),
|
|
588
|
+
passesRequired: fields.every((field) => field.required === false || field.matched),
|
|
589
|
+
totalCount
|
|
590
|
+
};
|
|
591
|
+
};
|
|
592
|
+
|
|
370
593
|
// src/testing/benchmark.ts
|
|
371
594
|
var resolveFixtureEnvironment = (fixture) => {
|
|
372
595
|
const tags = new Set(fixture.tags ?? []);
|
|
@@ -597,10 +820,12 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
597
820
|
const postSpeechTimeToFirstFinalMs = toPostSpeechLatency(result.finalEvents[0]?.receivedAt);
|
|
598
821
|
const postSpeechTimeToEndOfTurnMs = toPostSpeechLatency(result.endOfTurnEvents[0]?.receivedAt);
|
|
599
822
|
const expectedTerms = scoreExpectedTerms(result.finalText, fixture.expectedTerms);
|
|
823
|
+
const criticalFields = scoreVoiceCriticalFields(result.finalText, fixture.expectedCriticalFields);
|
|
600
824
|
const speakerTurns = scoreSpeakerTurns(fixture, result);
|
|
601
825
|
return {
|
|
602
826
|
accuracy: result.accuracy,
|
|
603
827
|
closeCount: result.closeEvents.length,
|
|
828
|
+
criticalFields,
|
|
604
829
|
difficulty: fixture.difficulty,
|
|
605
830
|
elapsedMs,
|
|
606
831
|
endOfTurnCount: result.endOfTurnEvents.length,
|
|
@@ -611,7 +836,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
611
836
|
fixtureId: fixture.id,
|
|
612
837
|
fragmentationCount: Math.max(0, result.finalEvents.length - 1),
|
|
613
838
|
group: resolveFixtureEnvironment(fixture),
|
|
614
|
-
passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && (speakerTurns ? speakerTurns.passes : true),
|
|
839
|
+
passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && criticalFields.passesRequired && (speakerTurns ? speakerTurns.passes : true),
|
|
615
840
|
partialCount: result.partialEvents.length,
|
|
616
841
|
speakerTurns,
|
|
617
842
|
postSpeechTimeToEndOfTurnMs,
|
|
@@ -739,6 +964,8 @@ var summarizeSTTBenchmark = (adapterId, fixtures) => {
|
|
|
739
964
|
averageFinalCount: roundMetric(average(fixtures.map((fixture) => fixture.finalCount)), 2) ?? 0,
|
|
740
965
|
averageSpeakerTurnMatchRate: roundMetric(average(fixtures.map((fixture) => fixture.speakerTurns?.patternMatchRate))),
|
|
741
966
|
averageTermRecall: roundMetric(average(fixtures.map((fixture) => fixture.expectedTerms.recall))) ?? 0,
|
|
967
|
+
averageCriticalFieldAccuracy: roundMetric(average(fixtures.map((fixture) => fixture.criticalFields?.accuracy ?? 1))) ?? 0,
|
|
968
|
+
requiredCriticalFieldPassRate: fixtureCount > 0 ? roundMetric(fixtures.filter((fixture) => fixture.criticalFields?.passesRequired ?? true).length / fixtureCount) ?? 0 : 0,
|
|
742
969
|
averagePostSpeechTimeToEndOfTurnMs: roundMetric(average(fixtures.map((fixture) => fixture.postSpeechTimeToEndOfTurnMs)), 2),
|
|
743
970
|
averagePostSpeechTimeToFirstFinalMs: roundMetric(average(fixtures.map((fixture) => fixture.postSpeechTimeToFirstFinalMs)), 2),
|
|
744
971
|
averageTimeToEndOfTurnMs: roundMetric(average(fixtures.map((fixture) => fixture.timeToEndOfTurnMs)), 2),
|
|
@@ -803,6 +1030,45 @@ var summarizeSTTBenchmarkSeries = (input) => {
|
|
|
803
1030
|
}
|
|
804
1031
|
};
|
|
805
1032
|
};
|
|
1033
|
+
// src/testing/confidenceCalibration.ts
|
|
1034
|
+
var clampConfidence = (value) => Math.max(0, Math.min(1, value));
|
|
1035
|
+
var calibrateVoiceConfidence = (samples, binCount = 10) => {
|
|
1036
|
+
const safeBinCount = Math.max(1, Math.round(binCount));
|
|
1037
|
+
const bins = Array.from({ length: safeBinCount }, (_, index) => {
|
|
1038
|
+
const lowerBound = index / safeBinCount;
|
|
1039
|
+
return {
|
|
1040
|
+
accuracy: 0,
|
|
1041
|
+
averageConfidence: 0,
|
|
1042
|
+
count: 0,
|
|
1043
|
+
lowerBound,
|
|
1044
|
+
upperBound: (index + 1) / safeBinCount
|
|
1045
|
+
};
|
|
1046
|
+
});
|
|
1047
|
+
let brierTotal = 0;
|
|
1048
|
+
for (const sample of samples) {
|
|
1049
|
+
const confidence = clampConfidence(sample.confidence);
|
|
1050
|
+
const binIndex = Math.min(safeBinCount - 1, Math.floor(confidence * safeBinCount));
|
|
1051
|
+
const bin = bins[binIndex];
|
|
1052
|
+
bin.count += 1;
|
|
1053
|
+
bin.averageConfidence += confidence;
|
|
1054
|
+
bin.accuracy += sample.correct ? 1 : 0;
|
|
1055
|
+
brierTotal += (confidence - (sample.correct ? 1 : 0)) ** 2;
|
|
1056
|
+
}
|
|
1057
|
+
let expectedCalibrationError = 0;
|
|
1058
|
+
for (const bin of bins) {
|
|
1059
|
+
if (bin.count === 0)
|
|
1060
|
+
continue;
|
|
1061
|
+
bin.averageConfidence /= bin.count;
|
|
1062
|
+
bin.accuracy /= bin.count;
|
|
1063
|
+
expectedCalibrationError += bin.count / Math.max(1, samples.length) * Math.abs(bin.accuracy - bin.averageConfidence);
|
|
1064
|
+
}
|
|
1065
|
+
return {
|
|
1066
|
+
bins,
|
|
1067
|
+
brierScore: samples.length > 0 ? brierTotal / samples.length : 0,
|
|
1068
|
+
expectedCalibrationError,
|
|
1069
|
+
sampleCount: samples.length
|
|
1070
|
+
};
|
|
1071
|
+
};
|
|
806
1072
|
// src/core/correction.ts
|
|
807
1073
|
var escapeRegExp = (value) => value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
808
1074
|
var buildAliasMatcher = (alias) => new RegExp(`(?<![\\p{L}\\p{N}'])${escapeRegExp(alias)}(?![\\p{L}\\p{N}'])`, "giu");
|
|
@@ -5599,191 +5865,6 @@ var createVoiceMemoryStore = () => {
|
|
|
5599
5865
|
// src/core/session.ts
|
|
5600
5866
|
import { Buffer as Buffer2 } from "buffer";
|
|
5601
5867
|
|
|
5602
|
-
// src/core/numberNormalizer.ts
|
|
5603
|
-
var ONES = {
|
|
5604
|
-
eight: 8,
|
|
5605
|
-
eighteen: 18,
|
|
5606
|
-
eleven: 11,
|
|
5607
|
-
fifteen: 15,
|
|
5608
|
-
five: 5,
|
|
5609
|
-
four: 4,
|
|
5610
|
-
fourteen: 14,
|
|
5611
|
-
nine: 9,
|
|
5612
|
-
nineteen: 19,
|
|
5613
|
-
one: 1,
|
|
5614
|
-
seven: 7,
|
|
5615
|
-
seventeen: 17,
|
|
5616
|
-
six: 6,
|
|
5617
|
-
sixteen: 16,
|
|
5618
|
-
ten: 10,
|
|
5619
|
-
thirteen: 13,
|
|
5620
|
-
three: 3,
|
|
5621
|
-
twelve: 12,
|
|
5622
|
-
two: 2,
|
|
5623
|
-
zero: 0
|
|
5624
|
-
};
|
|
5625
|
-
var TENS = {
|
|
5626
|
-
eighty: 80,
|
|
5627
|
-
fifty: 50,
|
|
5628
|
-
forty: 40,
|
|
5629
|
-
ninety: 90,
|
|
5630
|
-
seventy: 70,
|
|
5631
|
-
sixty: 60,
|
|
5632
|
-
thirty: 30,
|
|
5633
|
-
twenty: 20
|
|
5634
|
-
};
|
|
5635
|
-
var SCALES = {
|
|
5636
|
-
billion: 1e9,
|
|
5637
|
-
million: 1e6,
|
|
5638
|
-
thousand: 1000,
|
|
5639
|
-
trillion: 1000000000000
|
|
5640
|
-
};
|
|
5641
|
-
var MAGNITUDE_WORDS = [
|
|
5642
|
-
[1000000000000, "trillion"],
|
|
5643
|
-
[1e9, "billion"],
|
|
5644
|
-
[1e6, "million"]
|
|
5645
|
-
];
|
|
5646
|
-
var FILLER = new Set(["and", "a", "an"]);
|
|
5647
|
-
var DECIMAL_PLACES = 3;
|
|
5648
|
-
var isValueWord = (word) => (word in ONES) || (word in TENS) || word === "hundred";
|
|
5649
|
-
var isScaleWord = (word) => (word in SCALES);
|
|
5650
|
-
var isNumberWord = (word) => isValueWord(word) || isScaleWord(word) || word === "point" || FILLER.has(word);
|
|
5651
|
-
var trimNumber = (value) => {
|
|
5652
|
-
if (Number.isInteger(value))
|
|
5653
|
-
return value.toLocaleString("en-US");
|
|
5654
|
-
return String(Number(value.toFixed(DECIMAL_PLACES)));
|
|
5655
|
-
};
|
|
5656
|
-
var renderValue = (value, usedMagnitude) => {
|
|
5657
|
-
if (usedMagnitude) {
|
|
5658
|
-
for (const [scale, word] of MAGNITUDE_WORDS) {
|
|
5659
|
-
if (value >= scale) {
|
|
5660
|
-
const scaled = value / scale;
|
|
5661
|
-
if (Number(scaled.toFixed(DECIMAL_PLACES)) === scaled) {
|
|
5662
|
-
return `${trimNumber(scaled)} ${word}`;
|
|
5663
|
-
}
|
|
5664
|
-
}
|
|
5665
|
-
}
|
|
5666
|
-
}
|
|
5667
|
-
return trimNumber(value);
|
|
5668
|
-
};
|
|
5669
|
-
var parseNumberWords = (words) => {
|
|
5670
|
-
let total = 0;
|
|
5671
|
-
let current = 0;
|
|
5672
|
-
let usedMagnitude = false;
|
|
5673
|
-
let sawNumber = false;
|
|
5674
|
-
let decimal = null;
|
|
5675
|
-
const foldDecimal = () => {
|
|
5676
|
-
if (decimal && decimal.length > 0)
|
|
5677
|
-
current += Number(`0.${decimal}`);
|
|
5678
|
-
decimal = null;
|
|
5679
|
-
};
|
|
5680
|
-
for (const word of words) {
|
|
5681
|
-
if (word === "point") {
|
|
5682
|
-
decimal = "";
|
|
5683
|
-
continue;
|
|
5684
|
-
}
|
|
5685
|
-
const one = ONES[word];
|
|
5686
|
-
const ten = TENS[word];
|
|
5687
|
-
const scale = SCALES[word];
|
|
5688
|
-
if (decimal !== null) {
|
|
5689
|
-
if (one !== undefined && one <= 9) {
|
|
5690
|
-
decimal += String(one);
|
|
5691
|
-
sawNumber = true;
|
|
5692
|
-
continue;
|
|
5693
|
-
}
|
|
5694
|
-
foldDecimal();
|
|
5695
|
-
}
|
|
5696
|
-
if (FILLER.has(word))
|
|
5697
|
-
continue;
|
|
5698
|
-
if (one !== undefined) {
|
|
5699
|
-
current += one;
|
|
5700
|
-
sawNumber = true;
|
|
5701
|
-
} else if (ten !== undefined) {
|
|
5702
|
-
current += ten;
|
|
5703
|
-
sawNumber = true;
|
|
5704
|
-
} else if (word === "hundred") {
|
|
5705
|
-
current = (current === 0 ? 1 : current) * 100;
|
|
5706
|
-
sawNumber = true;
|
|
5707
|
-
} else if (scale !== undefined) {
|
|
5708
|
-
total += (current === 0 ? 1 : current) * scale;
|
|
5709
|
-
current = 0;
|
|
5710
|
-
sawNumber = true;
|
|
5711
|
-
usedMagnitude = true;
|
|
5712
|
-
}
|
|
5713
|
-
}
|
|
5714
|
-
foldDecimal();
|
|
5715
|
-
if (!sawNumber)
|
|
5716
|
-
return null;
|
|
5717
|
-
return { usedMagnitude, value: total + current };
|
|
5718
|
-
};
|
|
5719
|
-
var PERCENT_RE = /^(per ?cent|percent|percentage)$/;
|
|
5720
|
-
var CURRENCY_RE = /^(dollars?|bucks?|usd)$/;
|
|
5721
|
-
var WORD_RE = /^[A-Za-z]+(?:-[A-Za-z]+)*$/;
|
|
5722
|
-
var wordsOf = (token) => token.toLowerCase().split("-");
|
|
5723
|
-
var isNumberToken = (token) => WORD_RE.test(token) && wordsOf(token).every(isNumberWord);
|
|
5724
|
-
var startsNumber = (token) => WORD_RE.test(token) && wordsOf(token).some(isValueWord);
|
|
5725
|
-
var isSpace = (token) => /^\s+$/.test(token);
|
|
5726
|
-
var normalizeSpokenNumbers = (input) => {
|
|
5727
|
-
if (!input)
|
|
5728
|
-
return input;
|
|
5729
|
-
const parts = input.match(/[A-Za-z]+(?:-[A-Za-z]+)*|[^A-Za-z]+/g);
|
|
5730
|
-
if (!parts)
|
|
5731
|
-
return input;
|
|
5732
|
-
const at = (idx) => parts[idx] ?? "";
|
|
5733
|
-
const out = [];
|
|
5734
|
-
let i = 0;
|
|
5735
|
-
while (i < parts.length) {
|
|
5736
|
-
const token = at(i);
|
|
5737
|
-
const lower = token.toLowerCase();
|
|
5738
|
-
if (lower === "a" || lower === "an") {
|
|
5739
|
-
const nextWord = at(i + 2);
|
|
5740
|
-
const nextHead = wordsOf(nextWord)[0] ?? "";
|
|
5741
|
-
const nextIsScale = isSpace(at(i + 1)) && WORD_RE.test(nextWord) && (nextHead === "hundred" || isScaleWord(nextHead));
|
|
5742
|
-
if (!nextIsScale) {
|
|
5743
|
-
out.push(token);
|
|
5744
|
-
i += 1;
|
|
5745
|
-
continue;
|
|
5746
|
-
}
|
|
5747
|
-
} else if (!startsNumber(token)) {
|
|
5748
|
-
out.push(token);
|
|
5749
|
-
i += 1;
|
|
5750
|
-
continue;
|
|
5751
|
-
}
|
|
5752
|
-
const spanIdx = [i];
|
|
5753
|
-
let j = i + 1;
|
|
5754
|
-
while (isSpace(at(j)) && isNumberToken(at(j + 1))) {
|
|
5755
|
-
spanIdx.push(j + 1);
|
|
5756
|
-
j += 2;
|
|
5757
|
-
}
|
|
5758
|
-
const words = spanIdx.flatMap((k) => wordsOf(at(k)));
|
|
5759
|
-
while (words.length > 0 && FILLER.has(words[words.length - 1] ?? "")) {
|
|
5760
|
-
words.pop();
|
|
5761
|
-
}
|
|
5762
|
-
const parsed = parseNumberWords(words);
|
|
5763
|
-
if (!parsed) {
|
|
5764
|
-
out.push(token);
|
|
5765
|
-
i += 1;
|
|
5766
|
-
continue;
|
|
5767
|
-
}
|
|
5768
|
-
let rendered = renderValue(parsed.value, parsed.usedMagnitude);
|
|
5769
|
-
let lastIdx = spanIdx[spanIdx.length - 1] ?? i;
|
|
5770
|
-
const unitWord = at(lastIdx + 2);
|
|
5771
|
-
if (isSpace(at(lastIdx + 1)) && WORD_RE.test(unitWord)) {
|
|
5772
|
-
const unit = unitWord.toLowerCase();
|
|
5773
|
-
if (PERCENT_RE.test(unit)) {
|
|
5774
|
-
rendered = `${rendered}%`;
|
|
5775
|
-
lastIdx += 2;
|
|
5776
|
-
} else if (CURRENCY_RE.test(unit)) {
|
|
5777
|
-
rendered = rendered.startsWith("$") ? rendered : `$${rendered}`;
|
|
5778
|
-
lastIdx += 2;
|
|
5779
|
-
}
|
|
5780
|
-
}
|
|
5781
|
-
out.push(rendered);
|
|
5782
|
-
i = lastIdx + 1;
|
|
5783
|
-
}
|
|
5784
|
-
return out.join("");
|
|
5785
|
-
};
|
|
5786
|
-
|
|
5787
5868
|
// src/core/backchannel.ts
|
|
5788
5869
|
var DEFAULT_CUES = [
|
|
5789
5870
|
{ text: "mm-hmm" },
|
|
@@ -6414,7 +6495,7 @@ var cloneTranscript = (transcript) => ({
|
|
|
6414
6495
|
});
|
|
6415
6496
|
var encodeBase64 = (chunk) => Buffer2.from(chunk).toString("base64");
|
|
6416
6497
|
var countWords2 = (text) => text.trim().split(/\s+/).filter(Boolean).length;
|
|
6417
|
-
var
|
|
6498
|
+
var normalizeText3 = (text) => text.trim().replace(/\s+/g, " ");
|
|
6418
6499
|
var getAudioChunkDurationMs = (chunk) => chunk.byteLength / (DEFAULT_FORMAT.sampleRateHz * DEFAULT_FORMAT.channels * 2) * 1000;
|
|
6419
6500
|
var getBufferedAudioDurationMs = (chunks) => chunks.reduce((total, chunk) => total + getAudioChunkDurationMs(chunk), 0);
|
|
6420
6501
|
var STREAM_SENTENCE_BOUNDARY = /[.!?\u2026]['")\]]*\s/;
|
|
@@ -6490,9 +6571,9 @@ var createTurnCostEstimate = (input) => {
|
|
|
6490
6571
|
totalBillableAudioMs: Math.max(0, input.primaryAudioMs) + Math.max(0, input.fallbackReplayAudioMs)
|
|
6491
6572
|
};
|
|
6492
6573
|
};
|
|
6493
|
-
var normalizeCorrectionText = (text) =>
|
|
6574
|
+
var normalizeCorrectionText = (text) => normalizeText3(text);
|
|
6494
6575
|
var evaluateFallbackNeed = (candidate, config) => {
|
|
6495
|
-
const trimmed =
|
|
6576
|
+
const trimmed = normalizeText3(candidate.text);
|
|
6496
6577
|
const wordCount = countWords2(trimmed);
|
|
6497
6578
|
const averageConfidence = calculateMeanConfidence(candidate.transcripts);
|
|
6498
6579
|
const words = collectTranscriptWords(candidate.transcripts);
|
|
@@ -6643,7 +6724,8 @@ var createVoiceSession = (options) => {
|
|
|
6643
6724
|
settleMs: options.sttFallback.settleMs ?? DEFAULT_FALLBACK_SETTLE_MS,
|
|
6644
6725
|
trigger: options.sttFallback.trigger ?? "empty-or-low-confidence",
|
|
6645
6726
|
wordConfidenceThreshold: options.sttFallback.wordConfidenceThreshold,
|
|
6646
|
-
riskPolicy: options.sttFallback.riskPolicy
|
|
6727
|
+
riskPolicy: options.sttFallback.riskPolicy,
|
|
6728
|
+
preferFallbackOn: options.sttFallback.preferFallbackOn
|
|
6647
6729
|
} : undefined;
|
|
6648
6730
|
const appendTrace = async (input) => {
|
|
6649
6731
|
await options.trace?.append({
|
|
@@ -7900,14 +7982,18 @@ var createVoiceSession = (options) => {
|
|
|
7900
7982
|
const fallbackCandidate = {
|
|
7901
7983
|
confidence: fallbackConfidence,
|
|
7902
7984
|
text: fallbackText,
|
|
7903
|
-
wordCount: countWords2(
|
|
7985
|
+
wordCount: countWords2(normalizeText3(fallbackText))
|
|
7904
7986
|
};
|
|
7905
7987
|
const primaryCandidate = {
|
|
7906
7988
|
confidence: calculateMeanConfidence(primaryTranscripts),
|
|
7907
7989
|
text: primaryText,
|
|
7908
|
-
wordCount: countWords2(
|
|
7990
|
+
wordCount: countWords2(normalizeText3(primaryText))
|
|
7909
7991
|
};
|
|
7910
|
-
const
|
|
7992
|
+
const policyPrefersFallback = fallbackCandidate.text.length > 0 && fallbackNeed.reason !== undefined && sttFallback.preferFallbackOn?.includes(fallbackNeed.reason);
|
|
7993
|
+
const selection = policyPrefersFallback ? {
|
|
7994
|
+
reason: "policy-preference",
|
|
7995
|
+
winner: fallbackCandidate
|
|
7996
|
+
} : selectBetterTurnText(primaryCandidate, fallbackCandidate);
|
|
7911
7997
|
const diagnostics = {
|
|
7912
7998
|
attempted: true,
|
|
7913
7999
|
fallbackConfidence: fallbackCandidate.confidence,
|
|
@@ -7994,7 +8080,7 @@ var createVoiceSession = (options) => {
|
|
|
7994
8080
|
};
|
|
7995
8081
|
const buildTurnSignature = (session, finalText, transcriptIdsOverride) => {
|
|
7996
8082
|
const finalTranscriptIds = transcriptIdsOverride ?? getFinalTranscriptIds(session.currentTurn.transcripts);
|
|
7997
|
-
return `${
|
|
8083
|
+
return `${normalizeText3(finalText)}|${finalTranscriptIds.join(",")}`;
|
|
7998
8084
|
};
|
|
7999
8085
|
const isDuplicateTurnCommit = (session, finalText) => {
|
|
8000
8086
|
const signature = buildTurnSignature(session, finalText);
|
|
@@ -8002,8 +8088,8 @@ var createVoiceSession = (options) => {
|
|
|
8002
8088
|
const isRecent = committedTurn && committedTurn.committedAt > 0 && Date.now() - committedTurn.committedAt < DEFAULT_DUPLICATE_TURN_WINDOW_MS;
|
|
8003
8089
|
const committedSignature = committedTurn?.signature ?? "";
|
|
8004
8090
|
const committedTranscriptIds = committedTurn?.transcriptIds ?? [];
|
|
8005
|
-
const committedText =
|
|
8006
|
-
const isSameText =
|
|
8091
|
+
const committedText = normalizeText3(committedTurn?.text ?? "");
|
|
8092
|
+
const isSameText = normalizeText3(finalText) === committedText;
|
|
8007
8093
|
const hasNoNewAudioSinceCommit = (session.currentTurn.lastAudioAt ?? 0) <= (committedTurn?.committedAt ?? 0);
|
|
8008
8094
|
if (!isRecent) {
|
|
8009
8095
|
return false;
|
|
@@ -8023,7 +8109,7 @@ var createVoiceSession = (options) => {
|
|
|
8023
8109
|
...session.lastCommittedTurn ?? {},
|
|
8024
8110
|
committedAt: Date.now(),
|
|
8025
8111
|
signature: buildTurnSignature(session, finalText, getFinalTranscriptIds(committedTranscripts)),
|
|
8026
|
-
text:
|
|
8112
|
+
text: normalizeText3(finalText),
|
|
8027
8113
|
transcriptIds: getFinalTranscriptIds(committedTranscripts)
|
|
8028
8114
|
};
|
|
8029
8115
|
};
|
|
@@ -9419,6 +9505,10 @@ var createVoiceSession = (options) => {
|
|
|
9419
9505
|
commitTurn: async (reason = "manual") => runSerial("api.commitTurn", async () => {
|
|
9420
9506
|
await commitTurnInternal(reason);
|
|
9421
9507
|
}),
|
|
9508
|
+
configureSTT: async (configuration) => runSerial("api.configureSTT", async () => {
|
|
9509
|
+
const adapter = await ensureAdapter();
|
|
9510
|
+
await adapter.configure?.(configuration);
|
|
9511
|
+
}),
|
|
9422
9512
|
complete: async (result) => runSerial("api.complete", async () => {
|
|
9423
9513
|
await completeInternal(result);
|
|
9424
9514
|
}),
|
|
@@ -10252,6 +10342,22 @@ var renderVoiceCallReviewMarkdown = (artifact) => {
|
|
|
10252
10342
|
].filter((value) => typeof value === "string").join(`
|
|
10253
10343
|
`);
|
|
10254
10344
|
};
|
|
10345
|
+
// src/testing/routingBenchmark.ts
|
|
10346
|
+
var evaluateVoiceSTTRouting = (fixtures) => {
|
|
10347
|
+
const attempted = fixtures.filter((fixture) => fixture.fallbackUsed);
|
|
10348
|
+
const improved = attempted.filter((fixture) => fixture.fallbackScore > fixture.primaryScore);
|
|
10349
|
+
const harmed = attempted.filter((fixture) => fixture.fallbackScore < fixture.primaryScore);
|
|
10350
|
+
const average3 = (values) => values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : 0;
|
|
10351
|
+
return {
|
|
10352
|
+
fallbackAttemptRate: fixtures.length > 0 ? attempted.length / fixtures.length : 0,
|
|
10353
|
+
fallbackHarmRate: attempted.length > 0 ? harmed.length / attempted.length : 0,
|
|
10354
|
+
fallbackImprovementRate: attempted.length > 0 ? improved.length / attempted.length : 0,
|
|
10355
|
+
fixtureCount: fixtures.length,
|
|
10356
|
+
oracleScore: average3(fixtures.map((fixture) => Math.max(fixture.primaryScore, fixture.fallbackScore))),
|
|
10357
|
+
primaryScore: average3(fixtures.map((fixture) => fixture.primaryScore)),
|
|
10358
|
+
selectedScore: average3(fixtures.map((fixture) => fixture.fallbackUsed ? fixture.fallbackScore : fixture.primaryScore))
|
|
10359
|
+
};
|
|
10360
|
+
};
|
|
10255
10361
|
// src/testing/sessionBenchmark.ts
|
|
10256
10362
|
var average3 = (values) => values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : 0;
|
|
10257
10363
|
var normalizeTurnText = (value) => value.toLowerCase().replace(/[^\p{L}\p{N}\s']/gu, " ").replace(/\s+/g, " ").trim();
|
|
@@ -15684,6 +15790,7 @@ export {
|
|
|
15684
15790
|
summarizeTTSBenchmark,
|
|
15685
15791
|
summarizeSTTBenchmarkSeries,
|
|
15686
15792
|
summarizeSTTBenchmark,
|
|
15793
|
+
scoreVoiceCriticalFields,
|
|
15687
15794
|
scoreTranscriptAccuracy,
|
|
15688
15795
|
scoreCorrectedExpectedTerms,
|
|
15689
15796
|
runVoiceTelephonyMediaOperationsSmoke,
|
|
@@ -15711,6 +15818,7 @@ export {
|
|
|
15711
15818
|
getDefaultVoiceTelephonyBenchmarkScenarios,
|
|
15712
15819
|
getDefaultVoiceDuplexBenchmarkScenarios,
|
|
15713
15820
|
getDefaultTTSBenchmarkFixtures,
|
|
15821
|
+
evaluateVoiceSTTRouting,
|
|
15714
15822
|
evaluateSTTBenchmarkAcceptance,
|
|
15715
15823
|
createVoiceProviderFailureSimulator,
|
|
15716
15824
|
createVoiceIOProviderFailureSimulator,
|
|
@@ -15722,6 +15830,7 @@ export {
|
|
|
15722
15830
|
createCodeSwitchBenchmarkCorrectionHandler,
|
|
15723
15831
|
createBenchmarkCorrectionHandler,
|
|
15724
15832
|
compareSTTBenchmarks,
|
|
15833
|
+
calibrateVoiceConfidence,
|
|
15725
15834
|
buildSessionCorrectionAudit,
|
|
15726
15835
|
buildFixturePhraseHints,
|
|
15727
15836
|
buildCorrectionBenchmarkAudit,
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
export type VoiceSTTRoutingFixture = {
|
|
2
|
+
fallbackScore: number;
|
|
3
|
+
fallbackUsed: boolean;
|
|
4
|
+
id: string;
|
|
5
|
+
primaryScore: number;
|
|
6
|
+
};
|
|
7
|
+
export type VoiceSTTRoutingBenchmarkReport = {
|
|
8
|
+
fallbackAttemptRate: number;
|
|
9
|
+
fallbackHarmRate: number;
|
|
10
|
+
fallbackImprovementRate: number;
|
|
11
|
+
fixtureCount: number;
|
|
12
|
+
oracleScore: number;
|
|
13
|
+
primaryScore: number;
|
|
14
|
+
selectedScore: number;
|
|
15
|
+
};
|
|
16
|
+
export declare const evaluateVoiceSTTRouting: (fixtures: VoiceSTTRoutingFixture[]) => VoiceSTTRoutingBenchmarkReport;
|