@absolutejs/voice 0.0.22-beta.635 → 0.0.22-beta.636
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -0
- package/dist/core/types.d.ts +23 -0
- package/dist/index.js +48 -1
- package/dist/testing/benchmark.d.ts +4 -0
- package/dist/testing/confidenceCalibration.d.ts +19 -0
- package/dist/testing/criticalFields.d.ts +22 -0
- package/dist/testing/fixtures.d.ts +2 -0
- package/dist/testing/index.d.ts +3 -0
- package/dist/testing/index.js +299 -195
- package/dist/testing/routingBenchmark.d.ts +16 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -5015,6 +5015,14 @@ That keeps HTMX pages declarative without inventing custom fragment endpoints fo
|
|
|
5015
5015
|
|
|
5016
5016
|
Performance & accuracy benchmarks (STT, TTS, duplex, telephony, sessions) and head-to-head comparisons against Vapi live in a dedicated repo: **[absolutejs/benchmarks](https://github.com/absolutejs/benchmarks)**. They consume the published `@absolutejs/voice` package and provider adapters.
|
|
5017
5017
|
|
|
5018
|
+
Reusable eval contracts stay in `@absolutejs/voice/testing`: fixture manifests
|
|
5019
|
+
can label critical names, organizations, currency, percentages, phone numbers,
|
|
5020
|
+
and other exact fields; benchmark reports score those fields independently from
|
|
5021
|
+
WER; confidence calibration reports ECE/Brier scores; and routing reports expose
|
|
5022
|
+
fallback improvement and harm rates. Audio corpora and executable provider runs
|
|
5023
|
+
remain separate so applications can consume the same contracts without shipping
|
|
5024
|
+
benchmark media in the runtime package.
|
|
5025
|
+
|
|
5018
5026
|
## Adapter Contract
|
|
5019
5027
|
|
|
5020
5028
|
Adapters normalize vendor behavior into a core event model so the plugin never branches on vendor names.
|
|
@@ -5035,6 +5043,7 @@ type STTAdapterSession = {
|
|
|
5035
5043
|
handler: (payload: STTSessionEventMap[K]) => void | Promise<void>,
|
|
5036
5044
|
) => () => void;
|
|
5037
5045
|
send: (audio: AudioChunk) => Promise<void>;
|
|
5046
|
+
configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
|
|
5038
5047
|
close: (reason?: string) => Promise<void>;
|
|
5039
5048
|
};
|
|
5040
5049
|
```
|
package/dist/core/types.d.ts
CHANGED
|
@@ -90,6 +90,15 @@ export type TranscriptWord = {
|
|
|
90
90
|
startedAtMs?: number;
|
|
91
91
|
text: string;
|
|
92
92
|
};
|
|
93
|
+
/** Provider token evidence. Tokens are intentionally kept separate from words:
|
|
94
|
+
* subword log probabilities are useful for calibration and routing, but are not
|
|
95
|
+
* word-level timestamps or confidence scores. */
|
|
96
|
+
export type TranscriptToken = {
|
|
97
|
+
bytes?: number[];
|
|
98
|
+
confidence?: number;
|
|
99
|
+
logProbability?: number;
|
|
100
|
+
text: string;
|
|
101
|
+
};
|
|
93
102
|
export type Transcript = {
|
|
94
103
|
id: string;
|
|
95
104
|
text: string;
|
|
@@ -101,8 +110,15 @@ export type Transcript = {
|
|
|
101
110
|
startedAtMs?: number;
|
|
102
111
|
endedAtMs?: number;
|
|
103
112
|
vendor?: string;
|
|
113
|
+
tokens?: TranscriptToken[];
|
|
104
114
|
words?: TranscriptWord[];
|
|
105
115
|
};
|
|
116
|
+
export type VoiceSTTSessionConfiguration = {
|
|
117
|
+
languageHints?: string[];
|
|
118
|
+
lexicon?: VoiceLexiconEntry[];
|
|
119
|
+
phraseHints?: VoicePhraseHint[];
|
|
120
|
+
turnDetection?: Partial<VoiceTurnDetectionConfig>;
|
|
121
|
+
};
|
|
106
122
|
export type VoiceTranscriptQuality = {
|
|
107
123
|
averageConfidence?: number;
|
|
108
124
|
confidenceSampleCount: number;
|
|
@@ -184,6 +200,8 @@ export type STTSessionEventMap = {
|
|
|
184
200
|
export type STTAdapterSession = {
|
|
185
201
|
on: <K extends keyof STTSessionEventMap>(event: K, handler: (payload: STTSessionEventMap[K]) => void | Promise<void>) => () => void;
|
|
186
202
|
send: (audio: AudioChunk) => Promise<void>;
|
|
203
|
+
/** Update provider-supported STT context without restarting the stream. */
|
|
204
|
+
configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
|
|
187
205
|
close: (reason?: string) => Promise<void>;
|
|
188
206
|
};
|
|
189
207
|
export type STTAdapterOpenOptions = {
|
|
@@ -240,6 +258,8 @@ export type RealtimeSessionEventMap = STTSessionEventMap & {
|
|
|
240
258
|
export type RealtimeAdapterSession = {
|
|
241
259
|
on: <K extends keyof RealtimeSessionEventMap>(event: K, handler: (payload: RealtimeSessionEventMap[K]) => void | Promise<void>) => () => void;
|
|
242
260
|
send: (input: AudioChunk | string) => Promise<void>;
|
|
261
|
+
/** Update provider-supported input transcription context in place. */
|
|
262
|
+
configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
|
|
243
263
|
close: (reason?: string) => Promise<void>;
|
|
244
264
|
};
|
|
245
265
|
export type RealtimeAdapterOpenOptions = {
|
|
@@ -584,6 +604,9 @@ export type VoiceSessionHandle<TContext = unknown, TSession extends VoiceSession
|
|
|
584
604
|
speechThreshold: number;
|
|
585
605
|
transcriptStabilityMs: number;
|
|
586
606
|
}>;
|
|
607
|
+
/** Refresh vocabulary, language hints, or provider turn settings while a
|
|
608
|
+
* call is active. Unsupported fields are ignored by the active adapter. */
|
|
609
|
+
configureSTT: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
|
|
587
610
|
};
|
|
588
611
|
export type VoiceLLMUsage = {
|
|
589
612
|
provider?: string;
|
package/dist/index.js
CHANGED
|
@@ -7041,6 +7041,10 @@ var createVoiceSession = (options) => {
|
|
|
7041
7041
|
commitTurn: async (reason = "manual") => runSerial("api.commitTurn", async () => {
|
|
7042
7042
|
await commitTurnInternal(reason);
|
|
7043
7043
|
}),
|
|
7044
|
+
configureSTT: async (configuration) => runSerial("api.configureSTT", async () => {
|
|
7045
|
+
const adapter = await ensureAdapter();
|
|
7046
|
+
await adapter.configure?.(configuration);
|
|
7047
|
+
}),
|
|
7044
7048
|
complete: async (result) => runSerial("api.complete", async () => {
|
|
7045
7049
|
await completeInternal(result);
|
|
7046
7050
|
}),
|
|
@@ -44493,6 +44497,7 @@ var createContractApi = (session) => ({
|
|
|
44493
44497
|
id: session.id,
|
|
44494
44498
|
attachUserMedia: async () => {},
|
|
44495
44499
|
close: async () => {},
|
|
44500
|
+
configureSTT: async () => {},
|
|
44496
44501
|
commitTurn: async () => {},
|
|
44497
44502
|
complete: async () => {},
|
|
44498
44503
|
connect: async () => {},
|
|
@@ -49558,6 +49563,44 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
|
|
|
49558
49563
|
};
|
|
49559
49564
|
};
|
|
49560
49565
|
|
|
49566
|
+
// src/testing/criticalFields.ts
|
|
49567
|
+
var normalizeText4 = (value) => value.normalize("NFKC").toLowerCase().replace(/[^\p{L}\p{N}@.%+$'-]+/gu, " ").replace(/\s+/g, " ").trim();
|
|
49568
|
+
var normalizeDigits = (value) => normalizeSpokenNumbers(value).replace(/\D/g, "");
|
|
49569
|
+
var normalizeSemanticNumber = (value) => normalizeSpokenNumbers(value).toLowerCase().replace(/\bdollars?\b|\busd\b|\bpercent(age)?\b|[%,$]/g, "").replace(/\s+/g, "").trim();
|
|
49570
|
+
var matchesCandidate = (actual, candidate, kind) => {
|
|
49571
|
+
if (kind === "phone") {
|
|
49572
|
+
const expectedDigits = normalizeDigits(candidate);
|
|
49573
|
+
return expectedDigits.length > 0 && normalizeDigits(actual).includes(expectedDigits);
|
|
49574
|
+
}
|
|
49575
|
+
if (kind === "currency" || kind === "number" || kind === "percentage") {
|
|
49576
|
+
const expectedNumber = normalizeSemanticNumber(candidate);
|
|
49577
|
+
return expectedNumber.length > 0 && normalizeSemanticNumber(actual).includes(expectedNumber);
|
|
49578
|
+
}
|
|
49579
|
+
const normalizedCandidate = normalizeText4(candidate);
|
|
49580
|
+
return normalizedCandidate.length > 0 && normalizeText4(actual).includes(normalizedCandidate);
|
|
49581
|
+
};
|
|
49582
|
+
var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
|
|
49583
|
+
const fields = expectedFields.map((field) => {
|
|
49584
|
+
const candidates = [field.value, ...field.aliases ?? []];
|
|
49585
|
+
const matchedAlias = candidates.find((candidate) => matchesCandidate(actualText, candidate, field.kind));
|
|
49586
|
+
return {
|
|
49587
|
+
...field,
|
|
49588
|
+
matched: matchedAlias !== undefined,
|
|
49589
|
+
matchedAlias
|
|
49590
|
+
};
|
|
49591
|
+
});
|
|
49592
|
+
const matchedCount = fields.filter((field) => field.matched).length;
|
|
49593
|
+
const totalCount = fields.length;
|
|
49594
|
+
return {
|
|
49595
|
+
accuracy: totalCount > 0 ? matchedCount / totalCount : 1,
|
|
49596
|
+
fields,
|
|
49597
|
+
matchedCount,
|
|
49598
|
+
missingFieldIds: fields.filter((field) => !field.matched).map((field) => field.id),
|
|
49599
|
+
passesRequired: fields.every((field) => field.required === false || field.matched),
|
|
49600
|
+
totalCount
|
|
49601
|
+
};
|
|
49602
|
+
};
|
|
49603
|
+
|
|
49561
49604
|
// src/testing/benchmark.ts
|
|
49562
49605
|
var resolveFixtureEnvironment = (fixture) => {
|
|
49563
49606
|
const tags = new Set(fixture.tags ?? []);
|
|
@@ -49788,10 +49831,12 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
49788
49831
|
const postSpeechTimeToFirstFinalMs = toPostSpeechLatency(result.finalEvents[0]?.receivedAt);
|
|
49789
49832
|
const postSpeechTimeToEndOfTurnMs = toPostSpeechLatency(result.endOfTurnEvents[0]?.receivedAt);
|
|
49790
49833
|
const expectedTerms = scoreExpectedTerms(result.finalText, fixture.expectedTerms);
|
|
49834
|
+
const criticalFields = scoreVoiceCriticalFields(result.finalText, fixture.expectedCriticalFields);
|
|
49791
49835
|
const speakerTurns = scoreSpeakerTurns(fixture, result);
|
|
49792
49836
|
return {
|
|
49793
49837
|
accuracy: result.accuracy,
|
|
49794
49838
|
closeCount: result.closeEvents.length,
|
|
49839
|
+
criticalFields,
|
|
49795
49840
|
difficulty: fixture.difficulty,
|
|
49796
49841
|
elapsedMs,
|
|
49797
49842
|
endOfTurnCount: result.endOfTurnEvents.length,
|
|
@@ -49802,7 +49847,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
49802
49847
|
fixtureId: fixture.id,
|
|
49803
49848
|
fragmentationCount: Math.max(0, result.finalEvents.length - 1),
|
|
49804
49849
|
group: resolveFixtureEnvironment(fixture),
|
|
49805
|
-
passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && (speakerTurns ? speakerTurns.passes : true),
|
|
49850
|
+
passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && criticalFields.passesRequired && (speakerTurns ? speakerTurns.passes : true),
|
|
49806
49851
|
partialCount: result.partialEvents.length,
|
|
49807
49852
|
speakerTurns,
|
|
49808
49853
|
postSpeechTimeToEndOfTurnMs,
|
|
@@ -49930,6 +49975,8 @@ var summarizeSTTBenchmark = (adapterId, fixtures) => {
|
|
|
49930
49975
|
averageFinalCount: roundMetric4(average2(fixtures.map((fixture) => fixture.finalCount)), 2) ?? 0,
|
|
49931
49976
|
averageSpeakerTurnMatchRate: roundMetric4(average2(fixtures.map((fixture) => fixture.speakerTurns?.patternMatchRate))),
|
|
49932
49977
|
averageTermRecall: roundMetric4(average2(fixtures.map((fixture) => fixture.expectedTerms.recall))) ?? 0,
|
|
49978
|
+
averageCriticalFieldAccuracy: roundMetric4(average2(fixtures.map((fixture) => fixture.criticalFields?.accuracy ?? 1))) ?? 0,
|
|
49979
|
+
requiredCriticalFieldPassRate: fixtureCount > 0 ? roundMetric4(fixtures.filter((fixture) => fixture.criticalFields?.passesRequired ?? true).length / fixtureCount) ?? 0 : 0,
|
|
49933
49980
|
averagePostSpeechTimeToEndOfTurnMs: roundMetric4(average2(fixtures.map((fixture) => fixture.postSpeechTimeToEndOfTurnMs)), 2),
|
|
49934
49981
|
averagePostSpeechTimeToFirstFinalMs: roundMetric4(average2(fixtures.map((fixture) => fixture.postSpeechTimeToFirstFinalMs)), 2),
|
|
49935
49982
|
averageTimeToEndOfTurnMs: roundMetric4(average2(fixtures.map((fixture) => fixture.timeToEndOfTurnMs)), 2),
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import type { STTAdapter, STTAdapterOpenOptions } from "../core/types";
|
|
2
2
|
import { type VoiceSTTAdapterHarnessOptions, type VoiceSTTAdapterHarnessResult } from "./stt";
|
|
3
3
|
import type { VoiceTestFixture } from "./fixtures";
|
|
4
|
+
import { type VoiceCriticalFieldAccuracy } from "./criticalFields";
|
|
4
5
|
export type VoiceExpectedTermAccuracy = {
|
|
5
6
|
allMatched: boolean;
|
|
6
7
|
expectedTerms: string[];
|
|
@@ -25,6 +26,7 @@ export type VoiceSTTBenchmarkFixtureResult = {
|
|
|
25
26
|
endOfTurnCount: number;
|
|
26
27
|
errorCount: number;
|
|
27
28
|
expectedTerms: VoiceExpectedTermAccuracy;
|
|
29
|
+
criticalFields?: VoiceCriticalFieldAccuracy;
|
|
28
30
|
finalCount: number;
|
|
29
31
|
finalText: string;
|
|
30
32
|
fixtureId: string;
|
|
@@ -49,6 +51,8 @@ export type VoiceSTTBenchmarkSummary = {
|
|
|
49
51
|
averageFinalCount: number;
|
|
50
52
|
averageSpeakerTurnMatchRate?: number;
|
|
51
53
|
averageTermRecall: number;
|
|
54
|
+
averageCriticalFieldAccuracy: number;
|
|
55
|
+
requiredCriticalFieldPassRate: number;
|
|
52
56
|
averagePostSpeechTimeToEndOfTurnMs?: number;
|
|
53
57
|
averagePostSpeechTimeToFirstFinalMs?: number;
|
|
54
58
|
averageTimeToEndOfTurnMs?: number;
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
export type VoiceConfidenceCalibrationSample = {
|
|
2
|
+
confidence: number;
|
|
3
|
+
correct: boolean;
|
|
4
|
+
metadata?: Record<string, unknown>;
|
|
5
|
+
};
|
|
6
|
+
export type VoiceConfidenceCalibrationBin = {
|
|
7
|
+
accuracy: number;
|
|
8
|
+
averageConfidence: number;
|
|
9
|
+
count: number;
|
|
10
|
+
lowerBound: number;
|
|
11
|
+
upperBound: number;
|
|
12
|
+
};
|
|
13
|
+
export type VoiceConfidenceCalibrationReport = {
|
|
14
|
+
bins: VoiceConfidenceCalibrationBin[];
|
|
15
|
+
brierScore: number;
|
|
16
|
+
expectedCalibrationError: number;
|
|
17
|
+
sampleCount: number;
|
|
18
|
+
};
|
|
19
|
+
export declare const calibrateVoiceConfidence: (samples: VoiceConfidenceCalibrationSample[], binCount?: number) => VoiceConfidenceCalibrationReport;
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
export type VoiceCriticalFieldKind = "acronym" | "brand" | "currency" | "custom" | "email" | "number" | "organization" | "percentage" | "person-name" | "phone";
|
|
2
|
+
export type VoiceExpectedCriticalField = {
|
|
3
|
+
aliases?: string[];
|
|
4
|
+
id: string;
|
|
5
|
+
kind: VoiceCriticalFieldKind;
|
|
6
|
+
metadata?: Record<string, unknown>;
|
|
7
|
+
required?: boolean;
|
|
8
|
+
value: string;
|
|
9
|
+
};
|
|
10
|
+
export type VoiceCriticalFieldResult = VoiceExpectedCriticalField & {
|
|
11
|
+
matched: boolean;
|
|
12
|
+
matchedAlias?: string;
|
|
13
|
+
};
|
|
14
|
+
export type VoiceCriticalFieldAccuracy = {
|
|
15
|
+
accuracy: number;
|
|
16
|
+
fields: VoiceCriticalFieldResult[];
|
|
17
|
+
matchedCount: number;
|
|
18
|
+
missingFieldIds: string[];
|
|
19
|
+
passesRequired: boolean;
|
|
20
|
+
totalCount: number;
|
|
21
|
+
};
|
|
22
|
+
export declare const scoreVoiceCriticalFields: (actualText: string, expectedFields?: VoiceExpectedCriticalField[]) => VoiceCriticalFieldAccuracy;
|
|
@@ -1,10 +1,12 @@
|
|
|
1
1
|
import type { AudioFormat, VoiceExpectedSpeakerTurn } from "../core/types";
|
|
2
|
+
import type { VoiceExpectedCriticalField } from "./criticalFields";
|
|
2
3
|
export type VoiceTestFixtureManifestEntry = {
|
|
3
4
|
id: string;
|
|
4
5
|
title: string;
|
|
5
6
|
audioPath: string;
|
|
6
7
|
expectedText: string;
|
|
7
8
|
expectedTerms?: string[];
|
|
9
|
+
expectedCriticalFields?: VoiceExpectedCriticalField[];
|
|
8
10
|
expectedSpeakerTurns?: VoiceExpectedSpeakerTurn[];
|
|
9
11
|
expectedTurnTexts?: string[];
|
|
10
12
|
chunkDurationMs?: number;
|
package/dist/testing/index.d.ts
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
export * from "./accuracy";
|
|
2
2
|
export * from "./benchmark";
|
|
3
|
+
export * from "./confidenceCalibration";
|
|
4
|
+
export * from "./criticalFields";
|
|
3
5
|
export * from "./corrected";
|
|
4
6
|
export * from "./duplex";
|
|
5
7
|
export * from "./fixtures";
|
|
@@ -7,6 +9,7 @@ export * from "./ioProviderSimulator";
|
|
|
7
9
|
export * from "./providerSimulator";
|
|
8
10
|
export * from "./resilience";
|
|
9
11
|
export * from "./review";
|
|
12
|
+
export * from "./routingBenchmark";
|
|
10
13
|
export * from "./sessionBenchmark";
|
|
11
14
|
export * from "./stt";
|
|
12
15
|
export * from "./telephony";
|
package/dist/testing/index.js
CHANGED
|
@@ -367,6 +367,229 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
|
|
|
367
367
|
};
|
|
368
368
|
};
|
|
369
369
|
|
|
370
|
+
// src/core/numberNormalizer.ts
|
|
371
|
+
var ONES = {
|
|
372
|
+
eight: 8,
|
|
373
|
+
eighteen: 18,
|
|
374
|
+
eleven: 11,
|
|
375
|
+
fifteen: 15,
|
|
376
|
+
five: 5,
|
|
377
|
+
four: 4,
|
|
378
|
+
fourteen: 14,
|
|
379
|
+
nine: 9,
|
|
380
|
+
nineteen: 19,
|
|
381
|
+
one: 1,
|
|
382
|
+
seven: 7,
|
|
383
|
+
seventeen: 17,
|
|
384
|
+
six: 6,
|
|
385
|
+
sixteen: 16,
|
|
386
|
+
ten: 10,
|
|
387
|
+
thirteen: 13,
|
|
388
|
+
three: 3,
|
|
389
|
+
twelve: 12,
|
|
390
|
+
two: 2,
|
|
391
|
+
zero: 0
|
|
392
|
+
};
|
|
393
|
+
var TENS = {
|
|
394
|
+
eighty: 80,
|
|
395
|
+
fifty: 50,
|
|
396
|
+
forty: 40,
|
|
397
|
+
ninety: 90,
|
|
398
|
+
seventy: 70,
|
|
399
|
+
sixty: 60,
|
|
400
|
+
thirty: 30,
|
|
401
|
+
twenty: 20
|
|
402
|
+
};
|
|
403
|
+
var SCALES = {
|
|
404
|
+
billion: 1e9,
|
|
405
|
+
million: 1e6,
|
|
406
|
+
thousand: 1000,
|
|
407
|
+
trillion: 1000000000000
|
|
408
|
+
};
|
|
409
|
+
var MAGNITUDE_WORDS = [
|
|
410
|
+
[1000000000000, "trillion"],
|
|
411
|
+
[1e9, "billion"],
|
|
412
|
+
[1e6, "million"]
|
|
413
|
+
];
|
|
414
|
+
var FILLER = new Set(["and", "a", "an"]);
|
|
415
|
+
var DECIMAL_PLACES = 3;
|
|
416
|
+
var isValueWord = (word) => (word in ONES) || (word in TENS) || word === "hundred";
|
|
417
|
+
var isScaleWord = (word) => (word in SCALES);
|
|
418
|
+
var isNumberWord = (word) => isValueWord(word) || isScaleWord(word) || word === "point" || FILLER.has(word);
|
|
419
|
+
var trimNumber = (value) => {
|
|
420
|
+
if (Number.isInteger(value))
|
|
421
|
+
return value.toLocaleString("en-US");
|
|
422
|
+
return String(Number(value.toFixed(DECIMAL_PLACES)));
|
|
423
|
+
};
|
|
424
|
+
var renderValue = (value, usedMagnitude) => {
|
|
425
|
+
if (usedMagnitude) {
|
|
426
|
+
for (const [scale, word] of MAGNITUDE_WORDS) {
|
|
427
|
+
if (value >= scale) {
|
|
428
|
+
const scaled = value / scale;
|
|
429
|
+
if (Number(scaled.toFixed(DECIMAL_PLACES)) === scaled) {
|
|
430
|
+
return `${trimNumber(scaled)} ${word}`;
|
|
431
|
+
}
|
|
432
|
+
}
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
return trimNumber(value);
|
|
436
|
+
};
|
|
437
|
+
var parseNumberWords = (words) => {
|
|
438
|
+
let total = 0;
|
|
439
|
+
let current = 0;
|
|
440
|
+
let usedMagnitude = false;
|
|
441
|
+
let sawNumber = false;
|
|
442
|
+
let decimal = null;
|
|
443
|
+
const foldDecimal = () => {
|
|
444
|
+
if (decimal && decimal.length > 0)
|
|
445
|
+
current += Number(`0.${decimal}`);
|
|
446
|
+
decimal = null;
|
|
447
|
+
};
|
|
448
|
+
for (const word of words) {
|
|
449
|
+
if (word === "point") {
|
|
450
|
+
decimal = "";
|
|
451
|
+
continue;
|
|
452
|
+
}
|
|
453
|
+
const one = ONES[word];
|
|
454
|
+
const ten = TENS[word];
|
|
455
|
+
const scale = SCALES[word];
|
|
456
|
+
if (decimal !== null) {
|
|
457
|
+
if (one !== undefined && one <= 9) {
|
|
458
|
+
decimal += String(one);
|
|
459
|
+
sawNumber = true;
|
|
460
|
+
continue;
|
|
461
|
+
}
|
|
462
|
+
foldDecimal();
|
|
463
|
+
}
|
|
464
|
+
if (FILLER.has(word))
|
|
465
|
+
continue;
|
|
466
|
+
if (one !== undefined) {
|
|
467
|
+
current += one;
|
|
468
|
+
sawNumber = true;
|
|
469
|
+
} else if (ten !== undefined) {
|
|
470
|
+
current += ten;
|
|
471
|
+
sawNumber = true;
|
|
472
|
+
} else if (word === "hundred") {
|
|
473
|
+
current = (current === 0 ? 1 : current) * 100;
|
|
474
|
+
sawNumber = true;
|
|
475
|
+
} else if (scale !== undefined) {
|
|
476
|
+
total += (current === 0 ? 1 : current) * scale;
|
|
477
|
+
current = 0;
|
|
478
|
+
sawNumber = true;
|
|
479
|
+
usedMagnitude = true;
|
|
480
|
+
}
|
|
481
|
+
}
|
|
482
|
+
foldDecimal();
|
|
483
|
+
if (!sawNumber)
|
|
484
|
+
return null;
|
|
485
|
+
return { usedMagnitude, value: total + current };
|
|
486
|
+
};
|
|
487
|
+
var PERCENT_RE = /^(per ?cent|percent|percentage)$/;
|
|
488
|
+
var CURRENCY_RE = /^(dollars?|bucks?|usd)$/;
|
|
489
|
+
var WORD_RE = /^[A-Za-z]+(?:-[A-Za-z]+)*$/;
|
|
490
|
+
var wordsOf = (token) => token.toLowerCase().split("-");
|
|
491
|
+
var isNumberToken = (token) => WORD_RE.test(token) && wordsOf(token).every(isNumberWord);
|
|
492
|
+
var startsNumber = (token) => WORD_RE.test(token) && wordsOf(token).some(isValueWord);
|
|
493
|
+
var isSpace = (token) => /^\s+$/.test(token);
|
|
494
|
+
var normalizeSpokenNumbers = (input) => {
|
|
495
|
+
if (!input)
|
|
496
|
+
return input;
|
|
497
|
+
const parts = input.match(/[A-Za-z]+(?:-[A-Za-z]+)*|[^A-Za-z]+/g);
|
|
498
|
+
if (!parts)
|
|
499
|
+
return input;
|
|
500
|
+
const at = (idx) => parts[idx] ?? "";
|
|
501
|
+
const out = [];
|
|
502
|
+
let i = 0;
|
|
503
|
+
while (i < parts.length) {
|
|
504
|
+
const token = at(i);
|
|
505
|
+
const lower = token.toLowerCase();
|
|
506
|
+
if (lower === "a" || lower === "an") {
|
|
507
|
+
const nextWord = at(i + 2);
|
|
508
|
+
const nextHead = wordsOf(nextWord)[0] ?? "";
|
|
509
|
+
const nextIsScale = isSpace(at(i + 1)) && WORD_RE.test(nextWord) && (nextHead === "hundred" || isScaleWord(nextHead));
|
|
510
|
+
if (!nextIsScale) {
|
|
511
|
+
out.push(token);
|
|
512
|
+
i += 1;
|
|
513
|
+
continue;
|
|
514
|
+
}
|
|
515
|
+
} else if (!startsNumber(token)) {
|
|
516
|
+
out.push(token);
|
|
517
|
+
i += 1;
|
|
518
|
+
continue;
|
|
519
|
+
}
|
|
520
|
+
const spanIdx = [i];
|
|
521
|
+
let j = i + 1;
|
|
522
|
+
while (isSpace(at(j)) && isNumberToken(at(j + 1))) {
|
|
523
|
+
spanIdx.push(j + 1);
|
|
524
|
+
j += 2;
|
|
525
|
+
}
|
|
526
|
+
const words = spanIdx.flatMap((k) => wordsOf(at(k)));
|
|
527
|
+
while (words.length > 0 && FILLER.has(words[words.length - 1] ?? "")) {
|
|
528
|
+
words.pop();
|
|
529
|
+
}
|
|
530
|
+
const parsed = parseNumberWords(words);
|
|
531
|
+
if (!parsed) {
|
|
532
|
+
out.push(token);
|
|
533
|
+
i += 1;
|
|
534
|
+
continue;
|
|
535
|
+
}
|
|
536
|
+
let rendered = renderValue(parsed.value, parsed.usedMagnitude);
|
|
537
|
+
let lastIdx = spanIdx[spanIdx.length - 1] ?? i;
|
|
538
|
+
const unitWord = at(lastIdx + 2);
|
|
539
|
+
if (isSpace(at(lastIdx + 1)) && WORD_RE.test(unitWord)) {
|
|
540
|
+
const unit = unitWord.toLowerCase();
|
|
541
|
+
if (PERCENT_RE.test(unit)) {
|
|
542
|
+
rendered = `${rendered}%`;
|
|
543
|
+
lastIdx += 2;
|
|
544
|
+
} else if (CURRENCY_RE.test(unit)) {
|
|
545
|
+
rendered = rendered.startsWith("$") ? rendered : `$${rendered}`;
|
|
546
|
+
lastIdx += 2;
|
|
547
|
+
}
|
|
548
|
+
}
|
|
549
|
+
out.push(rendered);
|
|
550
|
+
i = lastIdx + 1;
|
|
551
|
+
}
|
|
552
|
+
return out.join("");
|
|
553
|
+
};
|
|
554
|
+
|
|
555
|
+
// src/testing/criticalFields.ts
|
|
556
|
+
var normalizeText2 = (value) => value.normalize("NFKC").toLowerCase().replace(/[^\p{L}\p{N}@.%+$'-]+/gu, " ").replace(/\s+/g, " ").trim();
|
|
557
|
+
var normalizeDigits = (value) => normalizeSpokenNumbers(value).replace(/\D/g, "");
|
|
558
|
+
var normalizeSemanticNumber = (value) => normalizeSpokenNumbers(value).toLowerCase().replace(/\bdollars?\b|\busd\b|\bpercent(age)?\b|[%,$]/g, "").replace(/\s+/g, "").trim();
|
|
559
|
+
var matchesCandidate = (actual, candidate, kind) => {
|
|
560
|
+
if (kind === "phone") {
|
|
561
|
+
const expectedDigits = normalizeDigits(candidate);
|
|
562
|
+
return expectedDigits.length > 0 && normalizeDigits(actual).includes(expectedDigits);
|
|
563
|
+
}
|
|
564
|
+
if (kind === "currency" || kind === "number" || kind === "percentage") {
|
|
565
|
+
const expectedNumber = normalizeSemanticNumber(candidate);
|
|
566
|
+
return expectedNumber.length > 0 && normalizeSemanticNumber(actual).includes(expectedNumber);
|
|
567
|
+
}
|
|
568
|
+
const normalizedCandidate = normalizeText2(candidate);
|
|
569
|
+
return normalizedCandidate.length > 0 && normalizeText2(actual).includes(normalizedCandidate);
|
|
570
|
+
};
|
|
571
|
+
var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
|
|
572
|
+
const fields = expectedFields.map((field) => {
|
|
573
|
+
const candidates = [field.value, ...field.aliases ?? []];
|
|
574
|
+
const matchedAlias = candidates.find((candidate) => matchesCandidate(actualText, candidate, field.kind));
|
|
575
|
+
return {
|
|
576
|
+
...field,
|
|
577
|
+
matched: matchedAlias !== undefined,
|
|
578
|
+
matchedAlias
|
|
579
|
+
};
|
|
580
|
+
});
|
|
581
|
+
const matchedCount = fields.filter((field) => field.matched).length;
|
|
582
|
+
const totalCount = fields.length;
|
|
583
|
+
return {
|
|
584
|
+
accuracy: totalCount > 0 ? matchedCount / totalCount : 1,
|
|
585
|
+
fields,
|
|
586
|
+
matchedCount,
|
|
587
|
+
missingFieldIds: fields.filter((field) => !field.matched).map((field) => field.id),
|
|
588
|
+
passesRequired: fields.every((field) => field.required === false || field.matched),
|
|
589
|
+
totalCount
|
|
590
|
+
};
|
|
591
|
+
};
|
|
592
|
+
|
|
370
593
|
// src/testing/benchmark.ts
|
|
371
594
|
var resolveFixtureEnvironment = (fixture) => {
|
|
372
595
|
const tags = new Set(fixture.tags ?? []);
|
|
@@ -597,10 +820,12 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
597
820
|
const postSpeechTimeToFirstFinalMs = toPostSpeechLatency(result.finalEvents[0]?.receivedAt);
|
|
598
821
|
const postSpeechTimeToEndOfTurnMs = toPostSpeechLatency(result.endOfTurnEvents[0]?.receivedAt);
|
|
599
822
|
const expectedTerms = scoreExpectedTerms(result.finalText, fixture.expectedTerms);
|
|
823
|
+
const criticalFields = scoreVoiceCriticalFields(result.finalText, fixture.expectedCriticalFields);
|
|
600
824
|
const speakerTurns = scoreSpeakerTurns(fixture, result);
|
|
601
825
|
return {
|
|
602
826
|
accuracy: result.accuracy,
|
|
603
827
|
closeCount: result.closeEvents.length,
|
|
828
|
+
criticalFields,
|
|
604
829
|
difficulty: fixture.difficulty,
|
|
605
830
|
elapsedMs,
|
|
606
831
|
endOfTurnCount: result.endOfTurnEvents.length,
|
|
@@ -611,7 +836,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
611
836
|
fixtureId: fixture.id,
|
|
612
837
|
fragmentationCount: Math.max(0, result.finalEvents.length - 1),
|
|
613
838
|
group: resolveFixtureEnvironment(fixture),
|
|
614
|
-
passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && (speakerTurns ? speakerTurns.passes : true),
|
|
839
|
+
passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && criticalFields.passesRequired && (speakerTurns ? speakerTurns.passes : true),
|
|
615
840
|
partialCount: result.partialEvents.length,
|
|
616
841
|
speakerTurns,
|
|
617
842
|
postSpeechTimeToEndOfTurnMs,
|
|
@@ -739,6 +964,8 @@ var summarizeSTTBenchmark = (adapterId, fixtures) => {
|
|
|
739
964
|
averageFinalCount: roundMetric(average(fixtures.map((fixture) => fixture.finalCount)), 2) ?? 0,
|
|
740
965
|
averageSpeakerTurnMatchRate: roundMetric(average(fixtures.map((fixture) => fixture.speakerTurns?.patternMatchRate))),
|
|
741
966
|
averageTermRecall: roundMetric(average(fixtures.map((fixture) => fixture.expectedTerms.recall))) ?? 0,
|
|
967
|
+
averageCriticalFieldAccuracy: roundMetric(average(fixtures.map((fixture) => fixture.criticalFields?.accuracy ?? 1))) ?? 0,
|
|
968
|
+
requiredCriticalFieldPassRate: fixtureCount > 0 ? roundMetric(fixtures.filter((fixture) => fixture.criticalFields?.passesRequired ?? true).length / fixtureCount) ?? 0 : 0,
|
|
742
969
|
averagePostSpeechTimeToEndOfTurnMs: roundMetric(average(fixtures.map((fixture) => fixture.postSpeechTimeToEndOfTurnMs)), 2),
|
|
743
970
|
averagePostSpeechTimeToFirstFinalMs: roundMetric(average(fixtures.map((fixture) => fixture.postSpeechTimeToFirstFinalMs)), 2),
|
|
744
971
|
averageTimeToEndOfTurnMs: roundMetric(average(fixtures.map((fixture) => fixture.timeToEndOfTurnMs)), 2),
|
|
@@ -803,6 +1030,45 @@ var summarizeSTTBenchmarkSeries = (input) => {
|
|
|
803
1030
|
}
|
|
804
1031
|
};
|
|
805
1032
|
};
|
|
1033
|
+
// src/testing/confidenceCalibration.ts
|
|
1034
|
+
var clampConfidence = (value) => Math.max(0, Math.min(1, value));
|
|
1035
|
+
var calibrateVoiceConfidence = (samples, binCount = 10) => {
|
|
1036
|
+
const safeBinCount = Math.max(1, Math.round(binCount));
|
|
1037
|
+
const bins = Array.from({ length: safeBinCount }, (_, index) => {
|
|
1038
|
+
const lowerBound = index / safeBinCount;
|
|
1039
|
+
return {
|
|
1040
|
+
accuracy: 0,
|
|
1041
|
+
averageConfidence: 0,
|
|
1042
|
+
count: 0,
|
|
1043
|
+
lowerBound,
|
|
1044
|
+
upperBound: (index + 1) / safeBinCount
|
|
1045
|
+
};
|
|
1046
|
+
});
|
|
1047
|
+
let brierTotal = 0;
|
|
1048
|
+
for (const sample of samples) {
|
|
1049
|
+
const confidence = clampConfidence(sample.confidence);
|
|
1050
|
+
const binIndex = Math.min(safeBinCount - 1, Math.floor(confidence * safeBinCount));
|
|
1051
|
+
const bin = bins[binIndex];
|
|
1052
|
+
bin.count += 1;
|
|
1053
|
+
bin.averageConfidence += confidence;
|
|
1054
|
+
bin.accuracy += sample.correct ? 1 : 0;
|
|
1055
|
+
brierTotal += (confidence - (sample.correct ? 1 : 0)) ** 2;
|
|
1056
|
+
}
|
|
1057
|
+
let expectedCalibrationError = 0;
|
|
1058
|
+
for (const bin of bins) {
|
|
1059
|
+
if (bin.count === 0)
|
|
1060
|
+
continue;
|
|
1061
|
+
bin.averageConfidence /= bin.count;
|
|
1062
|
+
bin.accuracy /= bin.count;
|
|
1063
|
+
expectedCalibrationError += bin.count / Math.max(1, samples.length) * Math.abs(bin.accuracy - bin.averageConfidence);
|
|
1064
|
+
}
|
|
1065
|
+
return {
|
|
1066
|
+
bins,
|
|
1067
|
+
brierScore: samples.length > 0 ? brierTotal / samples.length : 0,
|
|
1068
|
+
expectedCalibrationError,
|
|
1069
|
+
sampleCount: samples.length
|
|
1070
|
+
};
|
|
1071
|
+
};
|
|
806
1072
|
// src/core/correction.ts
|
|
807
1073
|
var escapeRegExp = (value) => value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
808
1074
|
var buildAliasMatcher = (alias) => new RegExp(`(?<![\\p{L}\\p{N}'])${escapeRegExp(alias)}(?![\\p{L}\\p{N}'])`, "giu");
|
|
@@ -5599,191 +5865,6 @@ var createVoiceMemoryStore = () => {
|
|
|
5599
5865
|
// src/core/session.ts
|
|
5600
5866
|
import { Buffer as Buffer2 } from "buffer";
|
|
5601
5867
|
|
|
5602
|
-
// src/core/numberNormalizer.ts
|
|
5603
|
-
var ONES = {
|
|
5604
|
-
eight: 8,
|
|
5605
|
-
eighteen: 18,
|
|
5606
|
-
eleven: 11,
|
|
5607
|
-
fifteen: 15,
|
|
5608
|
-
five: 5,
|
|
5609
|
-
four: 4,
|
|
5610
|
-
fourteen: 14,
|
|
5611
|
-
nine: 9,
|
|
5612
|
-
nineteen: 19,
|
|
5613
|
-
one: 1,
|
|
5614
|
-
seven: 7,
|
|
5615
|
-
seventeen: 17,
|
|
5616
|
-
six: 6,
|
|
5617
|
-
sixteen: 16,
|
|
5618
|
-
ten: 10,
|
|
5619
|
-
thirteen: 13,
|
|
5620
|
-
three: 3,
|
|
5621
|
-
twelve: 12,
|
|
5622
|
-
two: 2,
|
|
5623
|
-
zero: 0
|
|
5624
|
-
};
|
|
5625
|
-
var TENS = {
|
|
5626
|
-
eighty: 80,
|
|
5627
|
-
fifty: 50,
|
|
5628
|
-
forty: 40,
|
|
5629
|
-
ninety: 90,
|
|
5630
|
-
seventy: 70,
|
|
5631
|
-
sixty: 60,
|
|
5632
|
-
thirty: 30,
|
|
5633
|
-
twenty: 20
|
|
5634
|
-
};
|
|
5635
|
-
var SCALES = {
|
|
5636
|
-
billion: 1e9,
|
|
5637
|
-
million: 1e6,
|
|
5638
|
-
thousand: 1000,
|
|
5639
|
-
trillion: 1000000000000
|
|
5640
|
-
};
|
|
5641
|
-
var MAGNITUDE_WORDS = [
|
|
5642
|
-
[1000000000000, "trillion"],
|
|
5643
|
-
[1e9, "billion"],
|
|
5644
|
-
[1e6, "million"]
|
|
5645
|
-
];
|
|
5646
|
-
var FILLER = new Set(["and", "a", "an"]);
|
|
5647
|
-
var DECIMAL_PLACES = 3;
|
|
5648
|
-
var isValueWord = (word) => (word in ONES) || (word in TENS) || word === "hundred";
|
|
5649
|
-
var isScaleWord = (word) => (word in SCALES);
|
|
5650
|
-
var isNumberWord = (word) => isValueWord(word) || isScaleWord(word) || word === "point" || FILLER.has(word);
|
|
5651
|
-
var trimNumber = (value) => {
|
|
5652
|
-
if (Number.isInteger(value))
|
|
5653
|
-
return value.toLocaleString("en-US");
|
|
5654
|
-
return String(Number(value.toFixed(DECIMAL_PLACES)));
|
|
5655
|
-
};
|
|
5656
|
-
var renderValue = (value, usedMagnitude) => {
|
|
5657
|
-
if (usedMagnitude) {
|
|
5658
|
-
for (const [scale, word] of MAGNITUDE_WORDS) {
|
|
5659
|
-
if (value >= scale) {
|
|
5660
|
-
const scaled = value / scale;
|
|
5661
|
-
if (Number(scaled.toFixed(DECIMAL_PLACES)) === scaled) {
|
|
5662
|
-
return `${trimNumber(scaled)} ${word}`;
|
|
5663
|
-
}
|
|
5664
|
-
}
|
|
5665
|
-
}
|
|
5666
|
-
}
|
|
5667
|
-
return trimNumber(value);
|
|
5668
|
-
};
|
|
5669
|
-
var parseNumberWords = (words) => {
|
|
5670
|
-
let total = 0;
|
|
5671
|
-
let current = 0;
|
|
5672
|
-
let usedMagnitude = false;
|
|
5673
|
-
let sawNumber = false;
|
|
5674
|
-
let decimal = null;
|
|
5675
|
-
const foldDecimal = () => {
|
|
5676
|
-
if (decimal && decimal.length > 0)
|
|
5677
|
-
current += Number(`0.${decimal}`);
|
|
5678
|
-
decimal = null;
|
|
5679
|
-
};
|
|
5680
|
-
for (const word of words) {
|
|
5681
|
-
if (word === "point") {
|
|
5682
|
-
decimal = "";
|
|
5683
|
-
continue;
|
|
5684
|
-
}
|
|
5685
|
-
const one = ONES[word];
|
|
5686
|
-
const ten = TENS[word];
|
|
5687
|
-
const scale = SCALES[word];
|
|
5688
|
-
if (decimal !== null) {
|
|
5689
|
-
if (one !== undefined && one <= 9) {
|
|
5690
|
-
decimal += String(one);
|
|
5691
|
-
sawNumber = true;
|
|
5692
|
-
continue;
|
|
5693
|
-
}
|
|
5694
|
-
foldDecimal();
|
|
5695
|
-
}
|
|
5696
|
-
if (FILLER.has(word))
|
|
5697
|
-
continue;
|
|
5698
|
-
if (one !== undefined) {
|
|
5699
|
-
current += one;
|
|
5700
|
-
sawNumber = true;
|
|
5701
|
-
} else if (ten !== undefined) {
|
|
5702
|
-
current += ten;
|
|
5703
|
-
sawNumber = true;
|
|
5704
|
-
} else if (word === "hundred") {
|
|
5705
|
-
current = (current === 0 ? 1 : current) * 100;
|
|
5706
|
-
sawNumber = true;
|
|
5707
|
-
} else if (scale !== undefined) {
|
|
5708
|
-
total += (current === 0 ? 1 : current) * scale;
|
|
5709
|
-
current = 0;
|
|
5710
|
-
sawNumber = true;
|
|
5711
|
-
usedMagnitude = true;
|
|
5712
|
-
}
|
|
5713
|
-
}
|
|
5714
|
-
foldDecimal();
|
|
5715
|
-
if (!sawNumber)
|
|
5716
|
-
return null;
|
|
5717
|
-
return { usedMagnitude, value: total + current };
|
|
5718
|
-
};
|
|
5719
|
-
var PERCENT_RE = /^(per ?cent|percent|percentage)$/;
|
|
5720
|
-
var CURRENCY_RE = /^(dollars?|bucks?|usd)$/;
|
|
5721
|
-
var WORD_RE = /^[A-Za-z]+(?:-[A-Za-z]+)*$/;
|
|
5722
|
-
var wordsOf = (token) => token.toLowerCase().split("-");
|
|
5723
|
-
var isNumberToken = (token) => WORD_RE.test(token) && wordsOf(token).every(isNumberWord);
|
|
5724
|
-
var startsNumber = (token) => WORD_RE.test(token) && wordsOf(token).some(isValueWord);
|
|
5725
|
-
var isSpace = (token) => /^\s+$/.test(token);
|
|
5726
|
-
var normalizeSpokenNumbers = (input) => {
|
|
5727
|
-
if (!input)
|
|
5728
|
-
return input;
|
|
5729
|
-
const parts = input.match(/[A-Za-z]+(?:-[A-Za-z]+)*|[^A-Za-z]+/g);
|
|
5730
|
-
if (!parts)
|
|
5731
|
-
return input;
|
|
5732
|
-
const at = (idx) => parts[idx] ?? "";
|
|
5733
|
-
const out = [];
|
|
5734
|
-
let i = 0;
|
|
5735
|
-
while (i < parts.length) {
|
|
5736
|
-
const token = at(i);
|
|
5737
|
-
const lower = token.toLowerCase();
|
|
5738
|
-
if (lower === "a" || lower === "an") {
|
|
5739
|
-
const nextWord = at(i + 2);
|
|
5740
|
-
const nextHead = wordsOf(nextWord)[0] ?? "";
|
|
5741
|
-
const nextIsScale = isSpace(at(i + 1)) && WORD_RE.test(nextWord) && (nextHead === "hundred" || isScaleWord(nextHead));
|
|
5742
|
-
if (!nextIsScale) {
|
|
5743
|
-
out.push(token);
|
|
5744
|
-
i += 1;
|
|
5745
|
-
continue;
|
|
5746
|
-
}
|
|
5747
|
-
} else if (!startsNumber(token)) {
|
|
5748
|
-
out.push(token);
|
|
5749
|
-
i += 1;
|
|
5750
|
-
continue;
|
|
5751
|
-
}
|
|
5752
|
-
const spanIdx = [i];
|
|
5753
|
-
let j = i + 1;
|
|
5754
|
-
while (isSpace(at(j)) && isNumberToken(at(j + 1))) {
|
|
5755
|
-
spanIdx.push(j + 1);
|
|
5756
|
-
j += 2;
|
|
5757
|
-
}
|
|
5758
|
-
const words = spanIdx.flatMap((k) => wordsOf(at(k)));
|
|
5759
|
-
while (words.length > 0 && FILLER.has(words[words.length - 1] ?? "")) {
|
|
5760
|
-
words.pop();
|
|
5761
|
-
}
|
|
5762
|
-
const parsed = parseNumberWords(words);
|
|
5763
|
-
if (!parsed) {
|
|
5764
|
-
out.push(token);
|
|
5765
|
-
i += 1;
|
|
5766
|
-
continue;
|
|
5767
|
-
}
|
|
5768
|
-
let rendered = renderValue(parsed.value, parsed.usedMagnitude);
|
|
5769
|
-
let lastIdx = spanIdx[spanIdx.length - 1] ?? i;
|
|
5770
|
-
const unitWord = at(lastIdx + 2);
|
|
5771
|
-
if (isSpace(at(lastIdx + 1)) && WORD_RE.test(unitWord)) {
|
|
5772
|
-
const unit = unitWord.toLowerCase();
|
|
5773
|
-
if (PERCENT_RE.test(unit)) {
|
|
5774
|
-
rendered = `${rendered}%`;
|
|
5775
|
-
lastIdx += 2;
|
|
5776
|
-
} else if (CURRENCY_RE.test(unit)) {
|
|
5777
|
-
rendered = rendered.startsWith("$") ? rendered : `$${rendered}`;
|
|
5778
|
-
lastIdx += 2;
|
|
5779
|
-
}
|
|
5780
|
-
}
|
|
5781
|
-
out.push(rendered);
|
|
5782
|
-
i = lastIdx + 1;
|
|
5783
|
-
}
|
|
5784
|
-
return out.join("");
|
|
5785
|
-
};
|
|
5786
|
-
|
|
5787
5868
|
// src/core/backchannel.ts
|
|
5788
5869
|
var DEFAULT_CUES = [
|
|
5789
5870
|
{ text: "mm-hmm" },
|
|
@@ -6414,7 +6495,7 @@ var cloneTranscript = (transcript) => ({
|
|
|
6414
6495
|
});
|
|
6415
6496
|
var encodeBase64 = (chunk) => Buffer2.from(chunk).toString("base64");
|
|
6416
6497
|
var countWords2 = (text) => text.trim().split(/\s+/).filter(Boolean).length;
|
|
6417
|
-
var
|
|
6498
|
+
var normalizeText3 = (text) => text.trim().replace(/\s+/g, " ");
|
|
6418
6499
|
var getAudioChunkDurationMs = (chunk) => chunk.byteLength / (DEFAULT_FORMAT.sampleRateHz * DEFAULT_FORMAT.channels * 2) * 1000;
|
|
6419
6500
|
var getBufferedAudioDurationMs = (chunks) => chunks.reduce((total, chunk) => total + getAudioChunkDurationMs(chunk), 0);
|
|
6420
6501
|
var STREAM_SENTENCE_BOUNDARY = /[.!?\u2026]['")\]]*\s/;
|
|
@@ -6490,9 +6571,9 @@ var createTurnCostEstimate = (input) => {
|
|
|
6490
6571
|
totalBillableAudioMs: Math.max(0, input.primaryAudioMs) + Math.max(0, input.fallbackReplayAudioMs)
|
|
6491
6572
|
};
|
|
6492
6573
|
};
|
|
6493
|
-
var normalizeCorrectionText = (text) =>
|
|
6574
|
+
var normalizeCorrectionText = (text) => normalizeText3(text);
|
|
6494
6575
|
var evaluateFallbackNeed = (candidate, config) => {
|
|
6495
|
-
const trimmed =
|
|
6576
|
+
const trimmed = normalizeText3(candidate.text);
|
|
6496
6577
|
const wordCount = countWords2(trimmed);
|
|
6497
6578
|
const averageConfidence = calculateMeanConfidence(candidate.transcripts);
|
|
6498
6579
|
const words = collectTranscriptWords(candidate.transcripts);
|
|
@@ -7901,12 +7982,12 @@ var createVoiceSession = (options) => {
|
|
|
7901
7982
|
const fallbackCandidate = {
|
|
7902
7983
|
confidence: fallbackConfidence,
|
|
7903
7984
|
text: fallbackText,
|
|
7904
|
-
wordCount: countWords2(
|
|
7985
|
+
wordCount: countWords2(normalizeText3(fallbackText))
|
|
7905
7986
|
};
|
|
7906
7987
|
const primaryCandidate = {
|
|
7907
7988
|
confidence: calculateMeanConfidence(primaryTranscripts),
|
|
7908
7989
|
text: primaryText,
|
|
7909
|
-
wordCount: countWords2(
|
|
7990
|
+
wordCount: countWords2(normalizeText3(primaryText))
|
|
7910
7991
|
};
|
|
7911
7992
|
const policyPrefersFallback = fallbackCandidate.text.length > 0 && fallbackNeed.reason !== undefined && sttFallback.preferFallbackOn?.includes(fallbackNeed.reason);
|
|
7912
7993
|
const selection = policyPrefersFallback ? {
|
|
@@ -7999,7 +8080,7 @@ var createVoiceSession = (options) => {
|
|
|
7999
8080
|
};
|
|
8000
8081
|
const buildTurnSignature = (session, finalText, transcriptIdsOverride) => {
|
|
8001
8082
|
const finalTranscriptIds = transcriptIdsOverride ?? getFinalTranscriptIds(session.currentTurn.transcripts);
|
|
8002
|
-
return `${
|
|
8083
|
+
return `${normalizeText3(finalText)}|${finalTranscriptIds.join(",")}`;
|
|
8003
8084
|
};
|
|
8004
8085
|
const isDuplicateTurnCommit = (session, finalText) => {
|
|
8005
8086
|
const signature = buildTurnSignature(session, finalText);
|
|
@@ -8007,8 +8088,8 @@ var createVoiceSession = (options) => {
|
|
|
8007
8088
|
const isRecent = committedTurn && committedTurn.committedAt > 0 && Date.now() - committedTurn.committedAt < DEFAULT_DUPLICATE_TURN_WINDOW_MS;
|
|
8008
8089
|
const committedSignature = committedTurn?.signature ?? "";
|
|
8009
8090
|
const committedTranscriptIds = committedTurn?.transcriptIds ?? [];
|
|
8010
|
-
const committedText =
|
|
8011
|
-
const isSameText =
|
|
8091
|
+
const committedText = normalizeText3(committedTurn?.text ?? "");
|
|
8092
|
+
const isSameText = normalizeText3(finalText) === committedText;
|
|
8012
8093
|
const hasNoNewAudioSinceCommit = (session.currentTurn.lastAudioAt ?? 0) <= (committedTurn?.committedAt ?? 0);
|
|
8013
8094
|
if (!isRecent) {
|
|
8014
8095
|
return false;
|
|
@@ -8028,7 +8109,7 @@ var createVoiceSession = (options) => {
|
|
|
8028
8109
|
...session.lastCommittedTurn ?? {},
|
|
8029
8110
|
committedAt: Date.now(),
|
|
8030
8111
|
signature: buildTurnSignature(session, finalText, getFinalTranscriptIds(committedTranscripts)),
|
|
8031
|
-
text:
|
|
8112
|
+
text: normalizeText3(finalText),
|
|
8032
8113
|
transcriptIds: getFinalTranscriptIds(committedTranscripts)
|
|
8033
8114
|
};
|
|
8034
8115
|
};
|
|
@@ -9424,6 +9505,10 @@ var createVoiceSession = (options) => {
|
|
|
9424
9505
|
commitTurn: async (reason = "manual") => runSerial("api.commitTurn", async () => {
|
|
9425
9506
|
await commitTurnInternal(reason);
|
|
9426
9507
|
}),
|
|
9508
|
+
configureSTT: async (configuration) => runSerial("api.configureSTT", async () => {
|
|
9509
|
+
const adapter = await ensureAdapter();
|
|
9510
|
+
await adapter.configure?.(configuration);
|
|
9511
|
+
}),
|
|
9427
9512
|
complete: async (result) => runSerial("api.complete", async () => {
|
|
9428
9513
|
await completeInternal(result);
|
|
9429
9514
|
}),
|
|
@@ -10257,6 +10342,22 @@ var renderVoiceCallReviewMarkdown = (artifact) => {
|
|
|
10257
10342
|
].filter((value) => typeof value === "string").join(`
|
|
10258
10343
|
`);
|
|
10259
10344
|
};
|
|
10345
|
+
// src/testing/routingBenchmark.ts
|
|
10346
|
+
var evaluateVoiceSTTRouting = (fixtures) => {
|
|
10347
|
+
const attempted = fixtures.filter((fixture) => fixture.fallbackUsed);
|
|
10348
|
+
const improved = attempted.filter((fixture) => fixture.fallbackScore > fixture.primaryScore);
|
|
10349
|
+
const harmed = attempted.filter((fixture) => fixture.fallbackScore < fixture.primaryScore);
|
|
10350
|
+
const average3 = (values) => values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : 0;
|
|
10351
|
+
return {
|
|
10352
|
+
fallbackAttemptRate: fixtures.length > 0 ? attempted.length / fixtures.length : 0,
|
|
10353
|
+
fallbackHarmRate: attempted.length > 0 ? harmed.length / attempted.length : 0,
|
|
10354
|
+
fallbackImprovementRate: attempted.length > 0 ? improved.length / attempted.length : 0,
|
|
10355
|
+
fixtureCount: fixtures.length,
|
|
10356
|
+
oracleScore: average3(fixtures.map((fixture) => Math.max(fixture.primaryScore, fixture.fallbackScore))),
|
|
10357
|
+
primaryScore: average3(fixtures.map((fixture) => fixture.primaryScore)),
|
|
10358
|
+
selectedScore: average3(fixtures.map((fixture) => fixture.fallbackUsed ? fixture.fallbackScore : fixture.primaryScore))
|
|
10359
|
+
};
|
|
10360
|
+
};
|
|
10260
10361
|
// src/testing/sessionBenchmark.ts
|
|
10261
10362
|
var average3 = (values) => values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : 0;
|
|
10262
10363
|
var normalizeTurnText = (value) => value.toLowerCase().replace(/[^\p{L}\p{N}\s']/gu, " ").replace(/\s+/g, " ").trim();
|
|
@@ -15689,6 +15790,7 @@ export {
|
|
|
15689
15790
|
summarizeTTSBenchmark,
|
|
15690
15791
|
summarizeSTTBenchmarkSeries,
|
|
15691
15792
|
summarizeSTTBenchmark,
|
|
15793
|
+
scoreVoiceCriticalFields,
|
|
15692
15794
|
scoreTranscriptAccuracy,
|
|
15693
15795
|
scoreCorrectedExpectedTerms,
|
|
15694
15796
|
runVoiceTelephonyMediaOperationsSmoke,
|
|
@@ -15716,6 +15818,7 @@ export {
|
|
|
15716
15818
|
getDefaultVoiceTelephonyBenchmarkScenarios,
|
|
15717
15819
|
getDefaultVoiceDuplexBenchmarkScenarios,
|
|
15718
15820
|
getDefaultTTSBenchmarkFixtures,
|
|
15821
|
+
evaluateVoiceSTTRouting,
|
|
15719
15822
|
evaluateSTTBenchmarkAcceptance,
|
|
15720
15823
|
createVoiceProviderFailureSimulator,
|
|
15721
15824
|
createVoiceIOProviderFailureSimulator,
|
|
@@ -15727,6 +15830,7 @@ export {
|
|
|
15727
15830
|
createCodeSwitchBenchmarkCorrectionHandler,
|
|
15728
15831
|
createBenchmarkCorrectionHandler,
|
|
15729
15832
|
compareSTTBenchmarks,
|
|
15833
|
+
calibrateVoiceConfidence,
|
|
15730
15834
|
buildSessionCorrectionAudit,
|
|
15731
15835
|
buildFixturePhraseHints,
|
|
15732
15836
|
buildCorrectionBenchmarkAudit,
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
export type VoiceSTTRoutingFixture = {
|
|
2
|
+
fallbackScore: number;
|
|
3
|
+
fallbackUsed: boolean;
|
|
4
|
+
id: string;
|
|
5
|
+
primaryScore: number;
|
|
6
|
+
};
|
|
7
|
+
export type VoiceSTTRoutingBenchmarkReport = {
|
|
8
|
+
fallbackAttemptRate: number;
|
|
9
|
+
fallbackHarmRate: number;
|
|
10
|
+
fallbackImprovementRate: number;
|
|
11
|
+
fixtureCount: number;
|
|
12
|
+
oracleScore: number;
|
|
13
|
+
primaryScore: number;
|
|
14
|
+
selectedScore: number;
|
|
15
|
+
};
|
|
16
|
+
export declare const evaluateVoiceSTTRouting: (fixtures: VoiceSTTRoutingFixture[]) => VoiceSTTRoutingBenchmarkReport;
|