@absolutejs/voice 0.0.22-beta.635 → 0.0.22-beta.637

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -5015,6 +5015,14 @@ That keeps HTMX pages declarative without inventing custom fragment endpoints fo
5015
5015
 
5016
5016
  Performance & accuracy benchmarks (STT, TTS, duplex, telephony, sessions) and head-to-head comparisons against Vapi live in a dedicated repo: **[absolutejs/benchmarks](https://github.com/absolutejs/benchmarks)**. They consume the published `@absolutejs/voice` package and provider adapters.
5017
5017
 
5018
+ Reusable eval contracts stay in `@absolutejs/voice/testing`: fixture manifests
5019
+ can label critical names, organizations, currency, percentages, phone numbers,
5020
+ and other exact fields; benchmark reports score those fields independently from
5021
+ WER; confidence calibration reports ECE/Brier scores; and routing reports expose
5022
+ fallback improvement and harm rates. Audio corpora and executable provider runs
5023
+ remain separate so applications can consume the same contracts without shipping
5024
+ benchmark media in the runtime package.
5025
+
5018
5026
  ## Adapter Contract
5019
5027
 
5020
5028
  Adapters normalize vendor behavior into a core event model so the plugin never branches on vendor names.
@@ -5035,6 +5043,7 @@ type STTAdapterSession = {
5035
5043
  handler: (payload: STTSessionEventMap[K]) => void | Promise<void>,
5036
5044
  ) => () => void;
5037
5045
  send: (audio: AudioChunk) => Promise<void>;
5046
+ configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
5038
5047
  close: (reason?: string) => Promise<void>;
5039
5048
  };
5040
5049
  ```
@@ -90,6 +90,15 @@ export type TranscriptWord = {
90
90
  startedAtMs?: number;
91
91
  text: string;
92
92
  };
93
+ /** Provider token evidence. Tokens are intentionally kept separate from words:
94
+ * subword log probabilities are useful for calibration and routing, but are not
95
+ * word-level timestamps or confidence scores. */
96
+ export type TranscriptToken = {
97
+ bytes?: number[];
98
+ confidence?: number;
99
+ logProbability?: number;
100
+ text: string;
101
+ };
93
102
  export type Transcript = {
94
103
  id: string;
95
104
  text: string;
@@ -101,8 +110,15 @@ export type Transcript = {
101
110
  startedAtMs?: number;
102
111
  endedAtMs?: number;
103
112
  vendor?: string;
113
+ tokens?: TranscriptToken[];
104
114
  words?: TranscriptWord[];
105
115
  };
116
+ export type VoiceSTTSessionConfiguration = {
117
+ languageHints?: string[];
118
+ lexicon?: VoiceLexiconEntry[];
119
+ phraseHints?: VoicePhraseHint[];
120
+ turnDetection?: Partial<VoiceTurnDetectionConfig>;
121
+ };
106
122
  export type VoiceTranscriptQuality = {
107
123
  averageConfidence?: number;
108
124
  confidenceSampleCount: number;
@@ -184,6 +200,8 @@ export type STTSessionEventMap = {
184
200
  export type STTAdapterSession = {
185
201
  on: <K extends keyof STTSessionEventMap>(event: K, handler: (payload: STTSessionEventMap[K]) => void | Promise<void>) => () => void;
186
202
  send: (audio: AudioChunk) => Promise<void>;
203
+ /** Update provider-supported STT context without restarting the stream. */
204
+ configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
187
205
  close: (reason?: string) => Promise<void>;
188
206
  };
189
207
  export type STTAdapterOpenOptions = {
@@ -240,6 +258,8 @@ export type RealtimeSessionEventMap = STTSessionEventMap & {
240
258
  export type RealtimeAdapterSession = {
241
259
  on: <K extends keyof RealtimeSessionEventMap>(event: K, handler: (payload: RealtimeSessionEventMap[K]) => void | Promise<void>) => () => void;
242
260
  send: (input: AudioChunk | string) => Promise<void>;
261
+ /** Update provider-supported input transcription context in place. */
262
+ configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
243
263
  close: (reason?: string) => Promise<void>;
244
264
  };
245
265
  export type RealtimeAdapterOpenOptions = {
@@ -584,6 +604,9 @@ export type VoiceSessionHandle<TContext = unknown, TSession extends VoiceSession
584
604
  speechThreshold: number;
585
605
  transcriptStabilityMs: number;
586
606
  }>;
607
+ /** Refresh vocabulary, language hints, or provider turn settings while a
608
+ * call is active. Unsupported fields are ignored by the active adapter. */
609
+ configureSTT: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
587
610
  };
588
611
  export type VoiceLLMUsage = {
589
612
  provider?: string;
package/dist/index.js CHANGED
@@ -7041,6 +7041,10 @@ var createVoiceSession = (options) => {
7041
7041
  commitTurn: async (reason = "manual") => runSerial("api.commitTurn", async () => {
7042
7042
  await commitTurnInternal(reason);
7043
7043
  }),
7044
+ configureSTT: async (configuration) => runSerial("api.configureSTT", async () => {
7045
+ const adapter = await ensureAdapter();
7046
+ await adapter.configure?.(configuration);
7047
+ }),
7044
7048
  complete: async (result) => runSerial("api.complete", async () => {
7045
7049
  await completeInternal(result);
7046
7050
  }),
@@ -44493,6 +44497,7 @@ var createContractApi = (session) => ({
44493
44497
  id: session.id,
44494
44498
  attachUserMedia: async () => {},
44495
44499
  close: async () => {},
44500
+ configureSTT: async () => {},
44496
44501
  commitTurn: async () => {},
44497
44502
  complete: async () => {},
44498
44503
  connect: async () => {},
@@ -49558,6 +49563,84 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
49558
49563
  };
49559
49564
  };
49560
49565
 
49566
+ // src/testing/confidenceCalibration.ts
49567
+ var clampConfidence2 = (value) => Math.max(0, Math.min(1, value));
49568
+ var calibrateVoiceConfidence = (samples, binCount = 10) => {
49569
+ const safeBinCount = Math.max(1, Math.round(binCount));
49570
+ const bins = Array.from({ length: safeBinCount }, (_, index) => {
49571
+ const lowerBound = index / safeBinCount;
49572
+ return {
49573
+ accuracy: 0,
49574
+ averageConfidence: 0,
49575
+ count: 0,
49576
+ lowerBound,
49577
+ upperBound: (index + 1) / safeBinCount
49578
+ };
49579
+ });
49580
+ let brierTotal = 0;
49581
+ for (const sample of samples) {
49582
+ const confidence = clampConfidence2(sample.confidence);
49583
+ const binIndex = Math.min(safeBinCount - 1, Math.floor(confidence * safeBinCount));
49584
+ const bin = bins[binIndex];
49585
+ bin.count += 1;
49586
+ bin.averageConfidence += confidence;
49587
+ bin.accuracy += sample.correct ? 1 : 0;
49588
+ brierTotal += (confidence - (sample.correct ? 1 : 0)) ** 2;
49589
+ }
49590
+ let expectedCalibrationError = 0;
49591
+ for (const bin of bins) {
49592
+ if (bin.count === 0)
49593
+ continue;
49594
+ bin.averageConfidence /= bin.count;
49595
+ bin.accuracy /= bin.count;
49596
+ expectedCalibrationError += bin.count / Math.max(1, samples.length) * Math.abs(bin.accuracy - bin.averageConfidence);
49597
+ }
49598
+ return {
49599
+ bins,
49600
+ brierScore: samples.length > 0 ? brierTotal / samples.length : 0,
49601
+ expectedCalibrationError,
49602
+ sampleCount: samples.length
49603
+ };
49604
+ };
49605
+
49606
+ // src/testing/criticalFields.ts
49607
+ var normalizeText4 = (value) => value.normalize("NFKC").toLowerCase().replace(/[^\p{L}\p{N}@.%+$'-]+/gu, " ").replace(/\s+/g, " ").trim();
49608
+ var normalizeDigits = (value) => normalizeSpokenNumbers(value).replace(/\D/g, "");
49609
+ var normalizeSemanticNumber = (value) => normalizeSpokenNumbers(value).toLowerCase().replace(/\bdollars?\b|\busd\b|\bpercent(age)?\b|[%,$]/g, "").replace(/\s+/g, "").trim();
49610
+ var matchesCandidate = (actual, candidate, kind) => {
49611
+ if (kind === "phone") {
49612
+ const expectedDigits = normalizeDigits(candidate);
49613
+ return expectedDigits.length > 0 && normalizeDigits(actual).includes(expectedDigits);
49614
+ }
49615
+ if (kind === "currency" || kind === "number" || kind === "percentage") {
49616
+ const expectedNumber = normalizeSemanticNumber(candidate);
49617
+ return expectedNumber.length > 0 && normalizeSemanticNumber(actual).includes(expectedNumber);
49618
+ }
49619
+ const normalizedCandidate = normalizeText4(candidate);
49620
+ return normalizedCandidate.length > 0 && normalizeText4(actual).includes(normalizedCandidate);
49621
+ };
49622
+ var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
49623
+ const fields = expectedFields.map((field) => {
49624
+ const candidates = [field.value, ...field.aliases ?? []];
49625
+ const matchedAlias = candidates.find((candidate) => matchesCandidate(actualText, candidate, field.kind));
49626
+ return {
49627
+ ...field,
49628
+ matched: matchedAlias !== undefined,
49629
+ matchedAlias
49630
+ };
49631
+ });
49632
+ const matchedCount = fields.filter((field) => field.matched).length;
49633
+ const totalCount = fields.length;
49634
+ return {
49635
+ accuracy: totalCount > 0 ? matchedCount / totalCount : 1,
49636
+ fields,
49637
+ matchedCount,
49638
+ missingFieldIds: fields.filter((field) => !field.matched).map((field) => field.id),
49639
+ passesRequired: fields.every((field) => field.required === false || field.matched),
49640
+ totalCount
49641
+ };
49642
+ };
49643
+
49561
49644
  // src/testing/benchmark.ts
49562
49645
  var resolveFixtureEnvironment = (fixture) => {
49563
49646
  const tags = new Set(fixture.tags ?? []);
@@ -49788,10 +49871,21 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
49788
49871
  const postSpeechTimeToFirstFinalMs = toPostSpeechLatency(result.finalEvents[0]?.receivedAt);
49789
49872
  const postSpeechTimeToEndOfTurnMs = toPostSpeechLatency(result.endOfTurnEvents[0]?.receivedAt);
49790
49873
  const expectedTerms = scoreExpectedTerms(result.finalText, fixture.expectedTerms);
49874
+ const criticalFields = scoreVoiceCriticalFields(result.finalText, fixture.expectedCriticalFields);
49791
49875
  const speakerTurns = scoreSpeakerTurns(fixture, result);
49876
+ const transcriptConfidence = average2(result.finalEvents.map((event) => {
49877
+ if (typeof event.transcript.confidence === "number") {
49878
+ return event.transcript.confidence;
49879
+ }
49880
+ return average2([
49881
+ ...(event.transcript.words ?? []).map((word) => word.confidence),
49882
+ ...(event.transcript.tokens ?? []).map((token) => token.confidence)
49883
+ ]);
49884
+ }));
49792
49885
  return {
49793
49886
  accuracy: result.accuracy,
49794
49887
  closeCount: result.closeEvents.length,
49888
+ criticalFields,
49795
49889
  difficulty: fixture.difficulty,
49796
49890
  elapsedMs,
49797
49891
  endOfTurnCount: result.endOfTurnEvents.length,
@@ -49802,7 +49896,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
49802
49896
  fixtureId: fixture.id,
49803
49897
  fragmentationCount: Math.max(0, result.finalEvents.length - 1),
49804
49898
  group: resolveFixtureEnvironment(fixture),
49805
- passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && (speakerTurns ? speakerTurns.passes : true),
49899
+ passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && criticalFields.passesRequired && (speakerTurns ? speakerTurns.passes : true),
49806
49900
  partialCount: result.partialEvents.length,
49807
49901
  speakerTurns,
49808
49902
  postSpeechTimeToEndOfTurnMs,
@@ -49811,6 +49905,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
49811
49905
  timeToEndOfTurnMs,
49812
49906
  timeToFirstFinalMs,
49813
49907
  timeToFirstPartialMs,
49908
+ transcriptConfidence: roundMetric4(transcriptConfidence),
49814
49909
  title: fixture.title
49815
49910
  };
49816
49911
  };
@@ -49924,12 +50019,20 @@ var summarizeSTTBenchmark = (adapterId, fixtures) => {
49924
50019
  const passCount = fixtures.filter((fixture) => fixture.passes).length;
49925
50020
  return {
49926
50021
  adapterId,
50022
+ confidenceCalibration: calibrateVoiceConfidence(fixtures.flatMap((fixture) => typeof fixture.transcriptConfidence === "number" ? [
50023
+ {
50024
+ confidence: fixture.transcriptConfidence,
50025
+ correct: fixture.passes
50026
+ }
50027
+ ] : [])),
49927
50028
  averageCharErrorRate: roundMetric4(average2(fixtures.map((fixture) => fixture.accuracy.charErrorRate))) ?? 0,
49928
50029
  averageElapsedMs: roundMetric4(average2(fixtures.map((fixture) => fixture.elapsedMs)), 2) ?? 0,
49929
50030
  averageEndOfTurnCount: roundMetric4(average2(fixtures.map((fixture) => fixture.endOfTurnCount)), 2) ?? 0,
49930
50031
  averageFinalCount: roundMetric4(average2(fixtures.map((fixture) => fixture.finalCount)), 2) ?? 0,
49931
50032
  averageSpeakerTurnMatchRate: roundMetric4(average2(fixtures.map((fixture) => fixture.speakerTurns?.patternMatchRate))),
49932
50033
  averageTermRecall: roundMetric4(average2(fixtures.map((fixture) => fixture.expectedTerms.recall))) ?? 0,
50034
+ averageCriticalFieldAccuracy: roundMetric4(average2(fixtures.map((fixture) => fixture.criticalFields?.accuracy ?? 1))) ?? 0,
50035
+ requiredCriticalFieldPassRate: fixtureCount > 0 ? roundMetric4(fixtures.filter((fixture) => fixture.criticalFields?.passesRequired ?? true).length / fixtureCount) ?? 0 : 0,
49933
50036
  averagePostSpeechTimeToEndOfTurnMs: roundMetric4(average2(fixtures.map((fixture) => fixture.postSpeechTimeToEndOfTurnMs)), 2),
49934
50037
  averagePostSpeechTimeToFirstFinalMs: roundMetric4(average2(fixtures.map((fixture) => fixture.postSpeechTimeToFirstFinalMs)), 2),
49935
50038
  averageTimeToEndOfTurnMs: roundMetric4(average2(fixtures.map((fixture) => fixture.timeToEndOfTurnMs)), 2),
@@ -1,6 +1,8 @@
1
1
  import type { STTAdapter, STTAdapterOpenOptions } from "../core/types";
2
2
  import { type VoiceSTTAdapterHarnessOptions, type VoiceSTTAdapterHarnessResult } from "./stt";
3
3
  import type { VoiceTestFixture } from "./fixtures";
4
+ import { type VoiceConfidenceCalibrationReport } from "./confidenceCalibration";
5
+ import { type VoiceCriticalFieldAccuracy } from "./criticalFields";
4
6
  export type VoiceExpectedTermAccuracy = {
5
7
  allMatched: boolean;
6
8
  expectedTerms: string[];
@@ -25,6 +27,7 @@ export type VoiceSTTBenchmarkFixtureResult = {
25
27
  endOfTurnCount: number;
26
28
  errorCount: number;
27
29
  expectedTerms: VoiceExpectedTermAccuracy;
30
+ criticalFields?: VoiceCriticalFieldAccuracy;
28
31
  finalCount: number;
29
32
  finalText: string;
30
33
  fixtureId: string;
@@ -39,6 +42,7 @@ export type VoiceSTTBenchmarkFixtureResult = {
39
42
  timeToEndOfTurnMs?: number;
40
43
  timeToFirstFinalMs?: number;
41
44
  timeToFirstPartialMs?: number;
45
+ transcriptConfidence?: number;
42
46
  title: string;
43
47
  };
44
48
  export type VoiceSTTBenchmarkSummary = {
@@ -49,6 +53,8 @@ export type VoiceSTTBenchmarkSummary = {
49
53
  averageFinalCount: number;
50
54
  averageSpeakerTurnMatchRate?: number;
51
55
  averageTermRecall: number;
56
+ averageCriticalFieldAccuracy: number;
57
+ requiredCriticalFieldPassRate: number;
52
58
  averagePostSpeechTimeToEndOfTurnMs?: number;
53
59
  averagePostSpeechTimeToFirstFinalMs?: number;
54
60
  averageTimeToEndOfTurnMs?: number;
@@ -63,6 +69,7 @@ export type VoiceSTTBenchmarkSummary = {
63
69
  totalErrorCount: number;
64
70
  wordAccuracyRate: number;
65
71
  groupSummaries: VoiceSTTBenchmarkFixtureSummary[];
72
+ confidenceCalibration: VoiceConfidenceCalibrationReport;
66
73
  };
67
74
  export type VoiceSTTBenchmarkFixtureSummary = {
68
75
  group: VoiceSTTFixtureEnvironment;
@@ -0,0 +1,19 @@
1
+ export type VoiceConfidenceCalibrationSample = {
2
+ confidence: number;
3
+ correct: boolean;
4
+ metadata?: Record<string, unknown>;
5
+ };
6
+ export type VoiceConfidenceCalibrationBin = {
7
+ accuracy: number;
8
+ averageConfidence: number;
9
+ count: number;
10
+ lowerBound: number;
11
+ upperBound: number;
12
+ };
13
+ export type VoiceConfidenceCalibrationReport = {
14
+ bins: VoiceConfidenceCalibrationBin[];
15
+ brierScore: number;
16
+ expectedCalibrationError: number;
17
+ sampleCount: number;
18
+ };
19
+ export declare const calibrateVoiceConfidence: (samples: VoiceConfidenceCalibrationSample[], binCount?: number) => VoiceConfidenceCalibrationReport;
@@ -0,0 +1,22 @@
1
+ export type VoiceCriticalFieldKind = "acronym" | "brand" | "currency" | "custom" | "email" | "number" | "organization" | "percentage" | "person-name" | "phone";
2
+ export type VoiceExpectedCriticalField = {
3
+ aliases?: string[];
4
+ id: string;
5
+ kind: VoiceCriticalFieldKind;
6
+ metadata?: Record<string, unknown>;
7
+ required?: boolean;
8
+ value: string;
9
+ };
10
+ export type VoiceCriticalFieldResult = VoiceExpectedCriticalField & {
11
+ matched: boolean;
12
+ matchedAlias?: string;
13
+ };
14
+ export type VoiceCriticalFieldAccuracy = {
15
+ accuracy: number;
16
+ fields: VoiceCriticalFieldResult[];
17
+ matchedCount: number;
18
+ missingFieldIds: string[];
19
+ passesRequired: boolean;
20
+ totalCount: number;
21
+ };
22
+ export declare const scoreVoiceCriticalFields: (actualText: string, expectedFields?: VoiceExpectedCriticalField[]) => VoiceCriticalFieldAccuracy;
@@ -1,10 +1,12 @@
1
1
  import type { AudioFormat, VoiceExpectedSpeakerTurn } from "../core/types";
2
+ import type { VoiceExpectedCriticalField } from "./criticalFields";
2
3
  export type VoiceTestFixtureManifestEntry = {
3
4
  id: string;
4
5
  title: string;
5
6
  audioPath: string;
6
7
  expectedText: string;
7
8
  expectedTerms?: string[];
9
+ expectedCriticalFields?: VoiceExpectedCriticalField[];
8
10
  expectedSpeakerTurns?: VoiceExpectedSpeakerTurn[];
9
11
  expectedTurnTexts?: string[];
10
12
  chunkDurationMs?: number;
@@ -1,5 +1,7 @@
1
1
  export * from "./accuracy";
2
2
  export * from "./benchmark";
3
+ export * from "./confidenceCalibration";
4
+ export * from "./criticalFields";
3
5
  export * from "./corrected";
4
6
  export * from "./duplex";
5
7
  export * from "./fixtures";
@@ -7,6 +9,7 @@ export * from "./ioProviderSimulator";
7
9
  export * from "./providerSimulator";
8
10
  export * from "./resilience";
9
11
  export * from "./review";
12
+ export * from "./routingBenchmark";
10
13
  export * from "./sessionBenchmark";
11
14
  export * from "./stt";
12
15
  export * from "./telephony";
@@ -367,6 +367,269 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
367
367
  };
368
368
  };
369
369
 
370
+ // src/testing/confidenceCalibration.ts
371
+ var clampConfidence = (value) => Math.max(0, Math.min(1, value));
372
+ var calibrateVoiceConfidence = (samples, binCount = 10) => {
373
+ const safeBinCount = Math.max(1, Math.round(binCount));
374
+ const bins = Array.from({ length: safeBinCount }, (_, index) => {
375
+ const lowerBound = index / safeBinCount;
376
+ return {
377
+ accuracy: 0,
378
+ averageConfidence: 0,
379
+ count: 0,
380
+ lowerBound,
381
+ upperBound: (index + 1) / safeBinCount
382
+ };
383
+ });
384
+ let brierTotal = 0;
385
+ for (const sample of samples) {
386
+ const confidence = clampConfidence(sample.confidence);
387
+ const binIndex = Math.min(safeBinCount - 1, Math.floor(confidence * safeBinCount));
388
+ const bin = bins[binIndex];
389
+ bin.count += 1;
390
+ bin.averageConfidence += confidence;
391
+ bin.accuracy += sample.correct ? 1 : 0;
392
+ brierTotal += (confidence - (sample.correct ? 1 : 0)) ** 2;
393
+ }
394
+ let expectedCalibrationError = 0;
395
+ for (const bin of bins) {
396
+ if (bin.count === 0)
397
+ continue;
398
+ bin.averageConfidence /= bin.count;
399
+ bin.accuracy /= bin.count;
400
+ expectedCalibrationError += bin.count / Math.max(1, samples.length) * Math.abs(bin.accuracy - bin.averageConfidence);
401
+ }
402
+ return {
403
+ bins,
404
+ brierScore: samples.length > 0 ? brierTotal / samples.length : 0,
405
+ expectedCalibrationError,
406
+ sampleCount: samples.length
407
+ };
408
+ };
409
+
410
+ // src/core/numberNormalizer.ts
411
+ var ONES = {
412
+ eight: 8,
413
+ eighteen: 18,
414
+ eleven: 11,
415
+ fifteen: 15,
416
+ five: 5,
417
+ four: 4,
418
+ fourteen: 14,
419
+ nine: 9,
420
+ nineteen: 19,
421
+ one: 1,
422
+ seven: 7,
423
+ seventeen: 17,
424
+ six: 6,
425
+ sixteen: 16,
426
+ ten: 10,
427
+ thirteen: 13,
428
+ three: 3,
429
+ twelve: 12,
430
+ two: 2,
431
+ zero: 0
432
+ };
433
+ var TENS = {
434
+ eighty: 80,
435
+ fifty: 50,
436
+ forty: 40,
437
+ ninety: 90,
438
+ seventy: 70,
439
+ sixty: 60,
440
+ thirty: 30,
441
+ twenty: 20
442
+ };
443
+ var SCALES = {
444
+ billion: 1e9,
445
+ million: 1e6,
446
+ thousand: 1000,
447
+ trillion: 1000000000000
448
+ };
449
+ var MAGNITUDE_WORDS = [
450
+ [1000000000000, "trillion"],
451
+ [1e9, "billion"],
452
+ [1e6, "million"]
453
+ ];
454
+ var FILLER = new Set(["and", "a", "an"]);
455
+ var DECIMAL_PLACES = 3;
456
+ var isValueWord = (word) => (word in ONES) || (word in TENS) || word === "hundred";
457
+ var isScaleWord = (word) => (word in SCALES);
458
+ var isNumberWord = (word) => isValueWord(word) || isScaleWord(word) || word === "point" || FILLER.has(word);
459
+ var trimNumber = (value) => {
460
+ if (Number.isInteger(value))
461
+ return value.toLocaleString("en-US");
462
+ return String(Number(value.toFixed(DECIMAL_PLACES)));
463
+ };
464
+ var renderValue = (value, usedMagnitude) => {
465
+ if (usedMagnitude) {
466
+ for (const [scale, word] of MAGNITUDE_WORDS) {
467
+ if (value >= scale) {
468
+ const scaled = value / scale;
469
+ if (Number(scaled.toFixed(DECIMAL_PLACES)) === scaled) {
470
+ return `${trimNumber(scaled)} ${word}`;
471
+ }
472
+ }
473
+ }
474
+ }
475
+ return trimNumber(value);
476
+ };
477
+ var parseNumberWords = (words) => {
478
+ let total = 0;
479
+ let current = 0;
480
+ let usedMagnitude = false;
481
+ let sawNumber = false;
482
+ let decimal = null;
483
+ const foldDecimal = () => {
484
+ if (decimal && decimal.length > 0)
485
+ current += Number(`0.${decimal}`);
486
+ decimal = null;
487
+ };
488
+ for (const word of words) {
489
+ if (word === "point") {
490
+ decimal = "";
491
+ continue;
492
+ }
493
+ const one = ONES[word];
494
+ const ten = TENS[word];
495
+ const scale = SCALES[word];
496
+ if (decimal !== null) {
497
+ if (one !== undefined && one <= 9) {
498
+ decimal += String(one);
499
+ sawNumber = true;
500
+ continue;
501
+ }
502
+ foldDecimal();
503
+ }
504
+ if (FILLER.has(word))
505
+ continue;
506
+ if (one !== undefined) {
507
+ current += one;
508
+ sawNumber = true;
509
+ } else if (ten !== undefined) {
510
+ current += ten;
511
+ sawNumber = true;
512
+ } else if (word === "hundred") {
513
+ current = (current === 0 ? 1 : current) * 100;
514
+ sawNumber = true;
515
+ } else if (scale !== undefined) {
516
+ total += (current === 0 ? 1 : current) * scale;
517
+ current = 0;
518
+ sawNumber = true;
519
+ usedMagnitude = true;
520
+ }
521
+ }
522
+ foldDecimal();
523
+ if (!sawNumber)
524
+ return null;
525
+ return { usedMagnitude, value: total + current };
526
+ };
527
+ var PERCENT_RE = /^(per ?cent|percent|percentage)$/;
528
+ var CURRENCY_RE = /^(dollars?|bucks?|usd)$/;
529
+ var WORD_RE = /^[A-Za-z]+(?:-[A-Za-z]+)*$/;
530
+ var wordsOf = (token) => token.toLowerCase().split("-");
531
+ var isNumberToken = (token) => WORD_RE.test(token) && wordsOf(token).every(isNumberWord);
532
+ var startsNumber = (token) => WORD_RE.test(token) && wordsOf(token).some(isValueWord);
533
+ var isSpace = (token) => /^\s+$/.test(token);
534
+ var normalizeSpokenNumbers = (input) => {
535
+ if (!input)
536
+ return input;
537
+ const parts = input.match(/[A-Za-z]+(?:-[A-Za-z]+)*|[^A-Za-z]+/g);
538
+ if (!parts)
539
+ return input;
540
+ const at = (idx) => parts[idx] ?? "";
541
+ const out = [];
542
+ let i = 0;
543
+ while (i < parts.length) {
544
+ const token = at(i);
545
+ const lower = token.toLowerCase();
546
+ if (lower === "a" || lower === "an") {
547
+ const nextWord = at(i + 2);
548
+ const nextHead = wordsOf(nextWord)[0] ?? "";
549
+ const nextIsScale = isSpace(at(i + 1)) && WORD_RE.test(nextWord) && (nextHead === "hundred" || isScaleWord(nextHead));
550
+ if (!nextIsScale) {
551
+ out.push(token);
552
+ i += 1;
553
+ continue;
554
+ }
555
+ } else if (!startsNumber(token)) {
556
+ out.push(token);
557
+ i += 1;
558
+ continue;
559
+ }
560
+ const spanIdx = [i];
561
+ let j = i + 1;
562
+ while (isSpace(at(j)) && isNumberToken(at(j + 1))) {
563
+ spanIdx.push(j + 1);
564
+ j += 2;
565
+ }
566
+ const words = spanIdx.flatMap((k) => wordsOf(at(k)));
567
+ while (words.length > 0 && FILLER.has(words[words.length - 1] ?? "")) {
568
+ words.pop();
569
+ }
570
+ const parsed = parseNumberWords(words);
571
+ if (!parsed) {
572
+ out.push(token);
573
+ i += 1;
574
+ continue;
575
+ }
576
+ let rendered = renderValue(parsed.value, parsed.usedMagnitude);
577
+ let lastIdx = spanIdx[spanIdx.length - 1] ?? i;
578
+ const unitWord = at(lastIdx + 2);
579
+ if (isSpace(at(lastIdx + 1)) && WORD_RE.test(unitWord)) {
580
+ const unit = unitWord.toLowerCase();
581
+ if (PERCENT_RE.test(unit)) {
582
+ rendered = `${rendered}%`;
583
+ lastIdx += 2;
584
+ } else if (CURRENCY_RE.test(unit)) {
585
+ rendered = rendered.startsWith("$") ? rendered : `$${rendered}`;
586
+ lastIdx += 2;
587
+ }
588
+ }
589
+ out.push(rendered);
590
+ i = lastIdx + 1;
591
+ }
592
+ return out.join("");
593
+ };
594
+
595
+ // src/testing/criticalFields.ts
596
+ var normalizeText2 = (value) => value.normalize("NFKC").toLowerCase().replace(/[^\p{L}\p{N}@.%+$'-]+/gu, " ").replace(/\s+/g, " ").trim();
597
+ var normalizeDigits = (value) => normalizeSpokenNumbers(value).replace(/\D/g, "");
598
+ var normalizeSemanticNumber = (value) => normalizeSpokenNumbers(value).toLowerCase().replace(/\bdollars?\b|\busd\b|\bpercent(age)?\b|[%,$]/g, "").replace(/\s+/g, "").trim();
599
+ var matchesCandidate = (actual, candidate, kind) => {
600
+ if (kind === "phone") {
601
+ const expectedDigits = normalizeDigits(candidate);
602
+ return expectedDigits.length > 0 && normalizeDigits(actual).includes(expectedDigits);
603
+ }
604
+ if (kind === "currency" || kind === "number" || kind === "percentage") {
605
+ const expectedNumber = normalizeSemanticNumber(candidate);
606
+ return expectedNumber.length > 0 && normalizeSemanticNumber(actual).includes(expectedNumber);
607
+ }
608
+ const normalizedCandidate = normalizeText2(candidate);
609
+ return normalizedCandidate.length > 0 && normalizeText2(actual).includes(normalizedCandidate);
610
+ };
611
+ var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
612
+ const fields = expectedFields.map((field) => {
613
+ const candidates = [field.value, ...field.aliases ?? []];
614
+ const matchedAlias = candidates.find((candidate) => matchesCandidate(actualText, candidate, field.kind));
615
+ return {
616
+ ...field,
617
+ matched: matchedAlias !== undefined,
618
+ matchedAlias
619
+ };
620
+ });
621
+ const matchedCount = fields.filter((field) => field.matched).length;
622
+ const totalCount = fields.length;
623
+ return {
624
+ accuracy: totalCount > 0 ? matchedCount / totalCount : 1,
625
+ fields,
626
+ matchedCount,
627
+ missingFieldIds: fields.filter((field) => !field.matched).map((field) => field.id),
628
+ passesRequired: fields.every((field) => field.required === false || field.matched),
629
+ totalCount
630
+ };
631
+ };
632
+
370
633
  // src/testing/benchmark.ts
371
634
  var resolveFixtureEnvironment = (fixture) => {
372
635
  const tags = new Set(fixture.tags ?? []);
@@ -597,10 +860,21 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
597
860
  const postSpeechTimeToFirstFinalMs = toPostSpeechLatency(result.finalEvents[0]?.receivedAt);
598
861
  const postSpeechTimeToEndOfTurnMs = toPostSpeechLatency(result.endOfTurnEvents[0]?.receivedAt);
599
862
  const expectedTerms = scoreExpectedTerms(result.finalText, fixture.expectedTerms);
863
+ const criticalFields = scoreVoiceCriticalFields(result.finalText, fixture.expectedCriticalFields);
600
864
  const speakerTurns = scoreSpeakerTurns(fixture, result);
865
+ const transcriptConfidence = average(result.finalEvents.map((event) => {
866
+ if (typeof event.transcript.confidence === "number") {
867
+ return event.transcript.confidence;
868
+ }
869
+ return average([
870
+ ...(event.transcript.words ?? []).map((word) => word.confidence),
871
+ ...(event.transcript.tokens ?? []).map((token) => token.confidence)
872
+ ]);
873
+ }));
601
874
  return {
602
875
  accuracy: result.accuracy,
603
876
  closeCount: result.closeEvents.length,
877
+ criticalFields,
604
878
  difficulty: fixture.difficulty,
605
879
  elapsedMs,
606
880
  endOfTurnCount: result.endOfTurnEvents.length,
@@ -611,7 +885,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
611
885
  fixtureId: fixture.id,
612
886
  fragmentationCount: Math.max(0, result.finalEvents.length - 1),
613
887
  group: resolveFixtureEnvironment(fixture),
614
- passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && (speakerTurns ? speakerTurns.passes : true),
888
+ passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && criticalFields.passesRequired && (speakerTurns ? speakerTurns.passes : true),
615
889
  partialCount: result.partialEvents.length,
616
890
  speakerTurns,
617
891
  postSpeechTimeToEndOfTurnMs,
@@ -620,6 +894,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
620
894
  timeToEndOfTurnMs,
621
895
  timeToFirstFinalMs,
622
896
  timeToFirstPartialMs,
897
+ transcriptConfidence: roundMetric(transcriptConfidence),
623
898
  title: fixture.title
624
899
  };
625
900
  };
@@ -733,12 +1008,20 @@ var summarizeSTTBenchmark = (adapterId, fixtures) => {
733
1008
  const passCount = fixtures.filter((fixture) => fixture.passes).length;
734
1009
  return {
735
1010
  adapterId,
1011
+ confidenceCalibration: calibrateVoiceConfidence(fixtures.flatMap((fixture) => typeof fixture.transcriptConfidence === "number" ? [
1012
+ {
1013
+ confidence: fixture.transcriptConfidence,
1014
+ correct: fixture.passes
1015
+ }
1016
+ ] : [])),
736
1017
  averageCharErrorRate: roundMetric(average(fixtures.map((fixture) => fixture.accuracy.charErrorRate))) ?? 0,
737
1018
  averageElapsedMs: roundMetric(average(fixtures.map((fixture) => fixture.elapsedMs)), 2) ?? 0,
738
1019
  averageEndOfTurnCount: roundMetric(average(fixtures.map((fixture) => fixture.endOfTurnCount)), 2) ?? 0,
739
1020
  averageFinalCount: roundMetric(average(fixtures.map((fixture) => fixture.finalCount)), 2) ?? 0,
740
1021
  averageSpeakerTurnMatchRate: roundMetric(average(fixtures.map((fixture) => fixture.speakerTurns?.patternMatchRate))),
741
1022
  averageTermRecall: roundMetric(average(fixtures.map((fixture) => fixture.expectedTerms.recall))) ?? 0,
1023
+ averageCriticalFieldAccuracy: roundMetric(average(fixtures.map((fixture) => fixture.criticalFields?.accuracy ?? 1))) ?? 0,
1024
+ requiredCriticalFieldPassRate: fixtureCount > 0 ? roundMetric(fixtures.filter((fixture) => fixture.criticalFields?.passesRequired ?? true).length / fixtureCount) ?? 0 : 0,
742
1025
  averagePostSpeechTimeToEndOfTurnMs: roundMetric(average(fixtures.map((fixture) => fixture.postSpeechTimeToEndOfTurnMs)), 2),
743
1026
  averagePostSpeechTimeToFirstFinalMs: roundMetric(average(fixtures.map((fixture) => fixture.postSpeechTimeToFirstFinalMs)), 2),
744
1027
  averageTimeToEndOfTurnMs: roundMetric(average(fixtures.map((fixture) => fixture.timeToEndOfTurnMs)), 2),
@@ -5599,191 +5882,6 @@ var createVoiceMemoryStore = () => {
5599
5882
  // src/core/session.ts
5600
5883
  import { Buffer as Buffer2 } from "buffer";
5601
5884
 
5602
- // src/core/numberNormalizer.ts
5603
- var ONES = {
5604
- eight: 8,
5605
- eighteen: 18,
5606
- eleven: 11,
5607
- fifteen: 15,
5608
- five: 5,
5609
- four: 4,
5610
- fourteen: 14,
5611
- nine: 9,
5612
- nineteen: 19,
5613
- one: 1,
5614
- seven: 7,
5615
- seventeen: 17,
5616
- six: 6,
5617
- sixteen: 16,
5618
- ten: 10,
5619
- thirteen: 13,
5620
- three: 3,
5621
- twelve: 12,
5622
- two: 2,
5623
- zero: 0
5624
- };
5625
- var TENS = {
5626
- eighty: 80,
5627
- fifty: 50,
5628
- forty: 40,
5629
- ninety: 90,
5630
- seventy: 70,
5631
- sixty: 60,
5632
- thirty: 30,
5633
- twenty: 20
5634
- };
5635
- var SCALES = {
5636
- billion: 1e9,
5637
- million: 1e6,
5638
- thousand: 1000,
5639
- trillion: 1000000000000
5640
- };
5641
- var MAGNITUDE_WORDS = [
5642
- [1000000000000, "trillion"],
5643
- [1e9, "billion"],
5644
- [1e6, "million"]
5645
- ];
5646
- var FILLER = new Set(["and", "a", "an"]);
5647
- var DECIMAL_PLACES = 3;
5648
- var isValueWord = (word) => (word in ONES) || (word in TENS) || word === "hundred";
5649
- var isScaleWord = (word) => (word in SCALES);
5650
- var isNumberWord = (word) => isValueWord(word) || isScaleWord(word) || word === "point" || FILLER.has(word);
5651
- var trimNumber = (value) => {
5652
- if (Number.isInteger(value))
5653
- return value.toLocaleString("en-US");
5654
- return String(Number(value.toFixed(DECIMAL_PLACES)));
5655
- };
5656
- var renderValue = (value, usedMagnitude) => {
5657
- if (usedMagnitude) {
5658
- for (const [scale, word] of MAGNITUDE_WORDS) {
5659
- if (value >= scale) {
5660
- const scaled = value / scale;
5661
- if (Number(scaled.toFixed(DECIMAL_PLACES)) === scaled) {
5662
- return `${trimNumber(scaled)} ${word}`;
5663
- }
5664
- }
5665
- }
5666
- }
5667
- return trimNumber(value);
5668
- };
5669
- var parseNumberWords = (words) => {
5670
- let total = 0;
5671
- let current = 0;
5672
- let usedMagnitude = false;
5673
- let sawNumber = false;
5674
- let decimal = null;
5675
- const foldDecimal = () => {
5676
- if (decimal && decimal.length > 0)
5677
- current += Number(`0.${decimal}`);
5678
- decimal = null;
5679
- };
5680
- for (const word of words) {
5681
- if (word === "point") {
5682
- decimal = "";
5683
- continue;
5684
- }
5685
- const one = ONES[word];
5686
- const ten = TENS[word];
5687
- const scale = SCALES[word];
5688
- if (decimal !== null) {
5689
- if (one !== undefined && one <= 9) {
5690
- decimal += String(one);
5691
- sawNumber = true;
5692
- continue;
5693
- }
5694
- foldDecimal();
5695
- }
5696
- if (FILLER.has(word))
5697
- continue;
5698
- if (one !== undefined) {
5699
- current += one;
5700
- sawNumber = true;
5701
- } else if (ten !== undefined) {
5702
- current += ten;
5703
- sawNumber = true;
5704
- } else if (word === "hundred") {
5705
- current = (current === 0 ? 1 : current) * 100;
5706
- sawNumber = true;
5707
- } else if (scale !== undefined) {
5708
- total += (current === 0 ? 1 : current) * scale;
5709
- current = 0;
5710
- sawNumber = true;
5711
- usedMagnitude = true;
5712
- }
5713
- }
5714
- foldDecimal();
5715
- if (!sawNumber)
5716
- return null;
5717
- return { usedMagnitude, value: total + current };
5718
- };
5719
- var PERCENT_RE = /^(per ?cent|percent|percentage)$/;
5720
- var CURRENCY_RE = /^(dollars?|bucks?|usd)$/;
5721
- var WORD_RE = /^[A-Za-z]+(?:-[A-Za-z]+)*$/;
5722
- var wordsOf = (token) => token.toLowerCase().split("-");
5723
- var isNumberToken = (token) => WORD_RE.test(token) && wordsOf(token).every(isNumberWord);
5724
- var startsNumber = (token) => WORD_RE.test(token) && wordsOf(token).some(isValueWord);
5725
- var isSpace = (token) => /^\s+$/.test(token);
5726
- var normalizeSpokenNumbers = (input) => {
5727
- if (!input)
5728
- return input;
5729
- const parts = input.match(/[A-Za-z]+(?:-[A-Za-z]+)*|[^A-Za-z]+/g);
5730
- if (!parts)
5731
- return input;
5732
- const at = (idx) => parts[idx] ?? "";
5733
- const out = [];
5734
- let i = 0;
5735
- while (i < parts.length) {
5736
- const token = at(i);
5737
- const lower = token.toLowerCase();
5738
- if (lower === "a" || lower === "an") {
5739
- const nextWord = at(i + 2);
5740
- const nextHead = wordsOf(nextWord)[0] ?? "";
5741
- const nextIsScale = isSpace(at(i + 1)) && WORD_RE.test(nextWord) && (nextHead === "hundred" || isScaleWord(nextHead));
5742
- if (!nextIsScale) {
5743
- out.push(token);
5744
- i += 1;
5745
- continue;
5746
- }
5747
- } else if (!startsNumber(token)) {
5748
- out.push(token);
5749
- i += 1;
5750
- continue;
5751
- }
5752
- const spanIdx = [i];
5753
- let j = i + 1;
5754
- while (isSpace(at(j)) && isNumberToken(at(j + 1))) {
5755
- spanIdx.push(j + 1);
5756
- j += 2;
5757
- }
5758
- const words = spanIdx.flatMap((k) => wordsOf(at(k)));
5759
- while (words.length > 0 && FILLER.has(words[words.length - 1] ?? "")) {
5760
- words.pop();
5761
- }
5762
- const parsed = parseNumberWords(words);
5763
- if (!parsed) {
5764
- out.push(token);
5765
- i += 1;
5766
- continue;
5767
- }
5768
- let rendered = renderValue(parsed.value, parsed.usedMagnitude);
5769
- let lastIdx = spanIdx[spanIdx.length - 1] ?? i;
5770
- const unitWord = at(lastIdx + 2);
5771
- if (isSpace(at(lastIdx + 1)) && WORD_RE.test(unitWord)) {
5772
- const unit = unitWord.toLowerCase();
5773
- if (PERCENT_RE.test(unit)) {
5774
- rendered = `${rendered}%`;
5775
- lastIdx += 2;
5776
- } else if (CURRENCY_RE.test(unit)) {
5777
- rendered = rendered.startsWith("$") ? rendered : `$${rendered}`;
5778
- lastIdx += 2;
5779
- }
5780
- }
5781
- out.push(rendered);
5782
- i = lastIdx + 1;
5783
- }
5784
- return out.join("");
5785
- };
5786
-
5787
5885
  // src/core/backchannel.ts
5788
5886
  var DEFAULT_CUES = [
5789
5887
  { text: "mm-hmm" },
@@ -6414,7 +6512,7 @@ var cloneTranscript = (transcript) => ({
6414
6512
  });
6415
6513
  var encodeBase64 = (chunk) => Buffer2.from(chunk).toString("base64");
6416
6514
  var countWords2 = (text) => text.trim().split(/\s+/).filter(Boolean).length;
6417
- var normalizeText2 = (text) => text.trim().replace(/\s+/g, " ");
6515
+ var normalizeText3 = (text) => text.trim().replace(/\s+/g, " ");
6418
6516
  var getAudioChunkDurationMs = (chunk) => chunk.byteLength / (DEFAULT_FORMAT.sampleRateHz * DEFAULT_FORMAT.channels * 2) * 1000;
6419
6517
  var getBufferedAudioDurationMs = (chunks) => chunks.reduce((total, chunk) => total + getAudioChunkDurationMs(chunk), 0);
6420
6518
  var STREAM_SENTENCE_BOUNDARY = /[.!?\u2026]['")\]]*\s/;
@@ -6490,9 +6588,9 @@ var createTurnCostEstimate = (input) => {
6490
6588
  totalBillableAudioMs: Math.max(0, input.primaryAudioMs) + Math.max(0, input.fallbackReplayAudioMs)
6491
6589
  };
6492
6590
  };
6493
- var normalizeCorrectionText = (text) => normalizeText2(text);
6591
+ var normalizeCorrectionText = (text) => normalizeText3(text);
6494
6592
  var evaluateFallbackNeed = (candidate, config) => {
6495
- const trimmed = normalizeText2(candidate.text);
6593
+ const trimmed = normalizeText3(candidate.text);
6496
6594
  const wordCount = countWords2(trimmed);
6497
6595
  const averageConfidence = calculateMeanConfidence(candidate.transcripts);
6498
6596
  const words = collectTranscriptWords(candidate.transcripts);
@@ -7901,12 +7999,12 @@ var createVoiceSession = (options) => {
7901
7999
  const fallbackCandidate = {
7902
8000
  confidence: fallbackConfidence,
7903
8001
  text: fallbackText,
7904
- wordCount: countWords2(normalizeText2(fallbackText))
8002
+ wordCount: countWords2(normalizeText3(fallbackText))
7905
8003
  };
7906
8004
  const primaryCandidate = {
7907
8005
  confidence: calculateMeanConfidence(primaryTranscripts),
7908
8006
  text: primaryText,
7909
- wordCount: countWords2(normalizeText2(primaryText))
8007
+ wordCount: countWords2(normalizeText3(primaryText))
7910
8008
  };
7911
8009
  const policyPrefersFallback = fallbackCandidate.text.length > 0 && fallbackNeed.reason !== undefined && sttFallback.preferFallbackOn?.includes(fallbackNeed.reason);
7912
8010
  const selection = policyPrefersFallback ? {
@@ -7999,7 +8097,7 @@ var createVoiceSession = (options) => {
7999
8097
  };
8000
8098
  const buildTurnSignature = (session, finalText, transcriptIdsOverride) => {
8001
8099
  const finalTranscriptIds = transcriptIdsOverride ?? getFinalTranscriptIds(session.currentTurn.transcripts);
8002
- return `${normalizeText2(finalText)}|${finalTranscriptIds.join(",")}`;
8100
+ return `${normalizeText3(finalText)}|${finalTranscriptIds.join(",")}`;
8003
8101
  };
8004
8102
  const isDuplicateTurnCommit = (session, finalText) => {
8005
8103
  const signature = buildTurnSignature(session, finalText);
@@ -8007,8 +8105,8 @@ var createVoiceSession = (options) => {
8007
8105
  const isRecent = committedTurn && committedTurn.committedAt > 0 && Date.now() - committedTurn.committedAt < DEFAULT_DUPLICATE_TURN_WINDOW_MS;
8008
8106
  const committedSignature = committedTurn?.signature ?? "";
8009
8107
  const committedTranscriptIds = committedTurn?.transcriptIds ?? [];
8010
- const committedText = normalizeText2(committedTurn?.text ?? "");
8011
- const isSameText = normalizeText2(finalText) === committedText;
8108
+ const committedText = normalizeText3(committedTurn?.text ?? "");
8109
+ const isSameText = normalizeText3(finalText) === committedText;
8012
8110
  const hasNoNewAudioSinceCommit = (session.currentTurn.lastAudioAt ?? 0) <= (committedTurn?.committedAt ?? 0);
8013
8111
  if (!isRecent) {
8014
8112
  return false;
@@ -8028,7 +8126,7 @@ var createVoiceSession = (options) => {
8028
8126
  ...session.lastCommittedTurn ?? {},
8029
8127
  committedAt: Date.now(),
8030
8128
  signature: buildTurnSignature(session, finalText, getFinalTranscriptIds(committedTranscripts)),
8031
- text: normalizeText2(finalText),
8129
+ text: normalizeText3(finalText),
8032
8130
  transcriptIds: getFinalTranscriptIds(committedTranscripts)
8033
8131
  };
8034
8132
  };
@@ -9424,6 +9522,10 @@ var createVoiceSession = (options) => {
9424
9522
  commitTurn: async (reason = "manual") => runSerial("api.commitTurn", async () => {
9425
9523
  await commitTurnInternal(reason);
9426
9524
  }),
9525
+ configureSTT: async (configuration) => runSerial("api.configureSTT", async () => {
9526
+ const adapter = await ensureAdapter();
9527
+ await adapter.configure?.(configuration);
9528
+ }),
9427
9529
  complete: async (result) => runSerial("api.complete", async () => {
9428
9530
  await completeInternal(result);
9429
9531
  }),
@@ -10257,6 +10359,22 @@ var renderVoiceCallReviewMarkdown = (artifact) => {
10257
10359
  ].filter((value) => typeof value === "string").join(`
10258
10360
  `);
10259
10361
  };
10362
+ // src/testing/routingBenchmark.ts
10363
+ var evaluateVoiceSTTRouting = (fixtures) => {
10364
+ const attempted = fixtures.filter((fixture) => fixture.fallbackUsed);
10365
+ const improved = attempted.filter((fixture) => fixture.fallbackScore > fixture.primaryScore);
10366
+ const harmed = attempted.filter((fixture) => fixture.fallbackScore < fixture.primaryScore);
10367
+ const average3 = (values) => values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : 0;
10368
+ return {
10369
+ fallbackAttemptRate: fixtures.length > 0 ? attempted.length / fixtures.length : 0,
10370
+ fallbackHarmRate: attempted.length > 0 ? harmed.length / attempted.length : 0,
10371
+ fallbackImprovementRate: attempted.length > 0 ? improved.length / attempted.length : 0,
10372
+ fixtureCount: fixtures.length,
10373
+ oracleScore: average3(fixtures.map((fixture) => Math.max(fixture.primaryScore, fixture.fallbackScore))),
10374
+ primaryScore: average3(fixtures.map((fixture) => fixture.primaryScore)),
10375
+ selectedScore: average3(fixtures.map((fixture) => fixture.fallbackUsed ? fixture.fallbackScore : fixture.primaryScore))
10376
+ };
10377
+ };
10260
10378
  // src/testing/sessionBenchmark.ts
10261
10379
  var average3 = (values) => values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : 0;
10262
10380
  var normalizeTurnText = (value) => value.toLowerCase().replace(/[^\p{L}\p{N}\s']/gu, " ").replace(/\s+/g, " ").trim();
@@ -15689,6 +15807,7 @@ export {
15689
15807
  summarizeTTSBenchmark,
15690
15808
  summarizeSTTBenchmarkSeries,
15691
15809
  summarizeSTTBenchmark,
15810
+ scoreVoiceCriticalFields,
15692
15811
  scoreTranscriptAccuracy,
15693
15812
  scoreCorrectedExpectedTerms,
15694
15813
  runVoiceTelephonyMediaOperationsSmoke,
@@ -15716,6 +15835,7 @@ export {
15716
15835
  getDefaultVoiceTelephonyBenchmarkScenarios,
15717
15836
  getDefaultVoiceDuplexBenchmarkScenarios,
15718
15837
  getDefaultTTSBenchmarkFixtures,
15838
+ evaluateVoiceSTTRouting,
15719
15839
  evaluateSTTBenchmarkAcceptance,
15720
15840
  createVoiceProviderFailureSimulator,
15721
15841
  createVoiceIOProviderFailureSimulator,
@@ -15727,6 +15847,7 @@ export {
15727
15847
  createCodeSwitchBenchmarkCorrectionHandler,
15728
15848
  createBenchmarkCorrectionHandler,
15729
15849
  compareSTTBenchmarks,
15850
+ calibrateVoiceConfidence,
15730
15851
  buildSessionCorrectionAudit,
15731
15852
  buildFixturePhraseHints,
15732
15853
  buildCorrectionBenchmarkAudit,
@@ -0,0 +1,16 @@
1
+ export type VoiceSTTRoutingFixture = {
2
+ fallbackScore: number;
3
+ fallbackUsed: boolean;
4
+ id: string;
5
+ primaryScore: number;
6
+ };
7
+ export type VoiceSTTRoutingBenchmarkReport = {
8
+ fallbackAttemptRate: number;
9
+ fallbackHarmRate: number;
10
+ fallbackImprovementRate: number;
11
+ fixtureCount: number;
12
+ oracleScore: number;
13
+ primaryScore: number;
14
+ selectedScore: number;
15
+ };
16
+ export declare const evaluateVoiceSTTRouting: (fixtures: VoiceSTTRoutingFixture[]) => VoiceSTTRoutingBenchmarkReport;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@absolutejs/voice",
3
- "version": "0.0.22-beta.635",
3
+ "version": "0.0.22-beta.637",
4
4
  "description": "Voice primitives and Elysia plugin for AbsoluteJS",
5
5
  "repository": {
6
6
  "type": "git",