@absolutejs/voice 0.0.22-beta.635 → 0.0.22-beta.636

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -5015,6 +5015,14 @@ That keeps HTMX pages declarative without inventing custom fragment endpoints fo
5015
5015
 
5016
5016
  Performance & accuracy benchmarks (STT, TTS, duplex, telephony, sessions) and head-to-head comparisons against Vapi live in a dedicated repo: **[absolutejs/benchmarks](https://github.com/absolutejs/benchmarks)**. They consume the published `@absolutejs/voice` package and provider adapters.
5017
5017
 
5018
+ Reusable eval contracts stay in `@absolutejs/voice/testing`: fixture manifests
5019
+ can label critical names, organizations, currency, percentages, phone numbers,
5020
+ and other exact fields; benchmark reports score those fields independently from
5021
+ WER; confidence calibration reports ECE/Brier scores; and routing reports expose
5022
+ fallback improvement and harm rates. Audio corpora and executable provider runs
5023
+ remain separate so applications can consume the same contracts without shipping
5024
+ benchmark media in the runtime package.
5025
+
5018
5026
  ## Adapter Contract
5019
5027
 
5020
5028
  Adapters normalize vendor behavior into a core event model so the plugin never branches on vendor names.
@@ -5035,6 +5043,7 @@ type STTAdapterSession = {
5035
5043
  handler: (payload: STTSessionEventMap[K]) => void | Promise<void>,
5036
5044
  ) => () => void;
5037
5045
  send: (audio: AudioChunk) => Promise<void>;
5046
+ configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
5038
5047
  close: (reason?: string) => Promise<void>;
5039
5048
  };
5040
5049
  ```
@@ -90,6 +90,15 @@ export type TranscriptWord = {
90
90
  startedAtMs?: number;
91
91
  text: string;
92
92
  };
93
+ /** Provider token evidence. Tokens are intentionally kept separate from words:
94
+ * subword log probabilities are useful for calibration and routing, but are not
95
+ * word-level timestamps or confidence scores. */
96
+ export type TranscriptToken = {
97
+ bytes?: number[];
98
+ confidence?: number;
99
+ logProbability?: number;
100
+ text: string;
101
+ };
93
102
  export type Transcript = {
94
103
  id: string;
95
104
  text: string;
@@ -101,8 +110,15 @@ export type Transcript = {
101
110
  startedAtMs?: number;
102
111
  endedAtMs?: number;
103
112
  vendor?: string;
113
+ tokens?: TranscriptToken[];
104
114
  words?: TranscriptWord[];
105
115
  };
116
+ export type VoiceSTTSessionConfiguration = {
117
+ languageHints?: string[];
118
+ lexicon?: VoiceLexiconEntry[];
119
+ phraseHints?: VoicePhraseHint[];
120
+ turnDetection?: Partial<VoiceTurnDetectionConfig>;
121
+ };
106
122
  export type VoiceTranscriptQuality = {
107
123
  averageConfidence?: number;
108
124
  confidenceSampleCount: number;
@@ -184,6 +200,8 @@ export type STTSessionEventMap = {
184
200
  export type STTAdapterSession = {
185
201
  on: <K extends keyof STTSessionEventMap>(event: K, handler: (payload: STTSessionEventMap[K]) => void | Promise<void>) => () => void;
186
202
  send: (audio: AudioChunk) => Promise<void>;
203
+ /** Update provider-supported STT context without restarting the stream. */
204
+ configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
187
205
  close: (reason?: string) => Promise<void>;
188
206
  };
189
207
  export type STTAdapterOpenOptions = {
@@ -240,6 +258,8 @@ export type RealtimeSessionEventMap = STTSessionEventMap & {
240
258
  export type RealtimeAdapterSession = {
241
259
  on: <K extends keyof RealtimeSessionEventMap>(event: K, handler: (payload: RealtimeSessionEventMap[K]) => void | Promise<void>) => () => void;
242
260
  send: (input: AudioChunk | string) => Promise<void>;
261
+ /** Update provider-supported input transcription context in place. */
262
+ configure?: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
243
263
  close: (reason?: string) => Promise<void>;
244
264
  };
245
265
  export type RealtimeAdapterOpenOptions = {
@@ -584,6 +604,9 @@ export type VoiceSessionHandle<TContext = unknown, TSession extends VoiceSession
584
604
  speechThreshold: number;
585
605
  transcriptStabilityMs: number;
586
606
  }>;
607
+ /** Refresh vocabulary, language hints, or provider turn settings while a
608
+ * call is active. Unsupported fields are ignored by the active adapter. */
609
+ configureSTT: (configuration: VoiceSTTSessionConfiguration) => Promise<void>;
587
610
  };
588
611
  export type VoiceLLMUsage = {
589
612
  provider?: string;
package/dist/index.js CHANGED
@@ -7041,6 +7041,10 @@ var createVoiceSession = (options) => {
7041
7041
  commitTurn: async (reason = "manual") => runSerial("api.commitTurn", async () => {
7042
7042
  await commitTurnInternal(reason);
7043
7043
  }),
7044
+ configureSTT: async (configuration) => runSerial("api.configureSTT", async () => {
7045
+ const adapter = await ensureAdapter();
7046
+ await adapter.configure?.(configuration);
7047
+ }),
7044
7048
  complete: async (result) => runSerial("api.complete", async () => {
7045
7049
  await completeInternal(result);
7046
7050
  }),
@@ -44493,6 +44497,7 @@ var createContractApi = (session) => ({
44493
44497
  id: session.id,
44494
44498
  attachUserMedia: async () => {},
44495
44499
  close: async () => {},
44500
+ configureSTT: async () => {},
44496
44501
  commitTurn: async () => {},
44497
44502
  complete: async () => {},
44498
44503
  connect: async () => {},
@@ -49558,6 +49563,44 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
49558
49563
  };
49559
49564
  };
49560
49565
 
49566
+ // src/testing/criticalFields.ts
49567
+ var normalizeText4 = (value) => value.normalize("NFKC").toLowerCase().replace(/[^\p{L}\p{N}@.%+$'-]+/gu, " ").replace(/\s+/g, " ").trim();
49568
+ var normalizeDigits = (value) => normalizeSpokenNumbers(value).replace(/\D/g, "");
49569
+ var normalizeSemanticNumber = (value) => normalizeSpokenNumbers(value).toLowerCase().replace(/\bdollars?\b|\busd\b|\bpercent(age)?\b|[%,$]/g, "").replace(/\s+/g, "").trim();
49570
+ var matchesCandidate = (actual, candidate, kind) => {
49571
+ if (kind === "phone") {
49572
+ const expectedDigits = normalizeDigits(candidate);
49573
+ return expectedDigits.length > 0 && normalizeDigits(actual).includes(expectedDigits);
49574
+ }
49575
+ if (kind === "currency" || kind === "number" || kind === "percentage") {
49576
+ const expectedNumber = normalizeSemanticNumber(candidate);
49577
+ return expectedNumber.length > 0 && normalizeSemanticNumber(actual).includes(expectedNumber);
49578
+ }
49579
+ const normalizedCandidate = normalizeText4(candidate);
49580
+ return normalizedCandidate.length > 0 && normalizeText4(actual).includes(normalizedCandidate);
49581
+ };
49582
+ var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
49583
+ const fields = expectedFields.map((field) => {
49584
+ const candidates = [field.value, ...field.aliases ?? []];
49585
+ const matchedAlias = candidates.find((candidate) => matchesCandidate(actualText, candidate, field.kind));
49586
+ return {
49587
+ ...field,
49588
+ matched: matchedAlias !== undefined,
49589
+ matchedAlias
49590
+ };
49591
+ });
49592
+ const matchedCount = fields.filter((field) => field.matched).length;
49593
+ const totalCount = fields.length;
49594
+ return {
49595
+ accuracy: totalCount > 0 ? matchedCount / totalCount : 1,
49596
+ fields,
49597
+ matchedCount,
49598
+ missingFieldIds: fields.filter((field) => !field.matched).map((field) => field.id),
49599
+ passesRequired: fields.every((field) => field.required === false || field.matched),
49600
+ totalCount
49601
+ };
49602
+ };
49603
+
49561
49604
  // src/testing/benchmark.ts
49562
49605
  var resolveFixtureEnvironment = (fixture) => {
49563
49606
  const tags = new Set(fixture.tags ?? []);
@@ -49788,10 +49831,12 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
49788
49831
  const postSpeechTimeToFirstFinalMs = toPostSpeechLatency(result.finalEvents[0]?.receivedAt);
49789
49832
  const postSpeechTimeToEndOfTurnMs = toPostSpeechLatency(result.endOfTurnEvents[0]?.receivedAt);
49790
49833
  const expectedTerms = scoreExpectedTerms(result.finalText, fixture.expectedTerms);
49834
+ const criticalFields = scoreVoiceCriticalFields(result.finalText, fixture.expectedCriticalFields);
49791
49835
  const speakerTurns = scoreSpeakerTurns(fixture, result);
49792
49836
  return {
49793
49837
  accuracy: result.accuracy,
49794
49838
  closeCount: result.closeEvents.length,
49839
+ criticalFields,
49795
49840
  difficulty: fixture.difficulty,
49796
49841
  elapsedMs,
49797
49842
  endOfTurnCount: result.endOfTurnEvents.length,
@@ -49802,7 +49847,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
49802
49847
  fixtureId: fixture.id,
49803
49848
  fragmentationCount: Math.max(0, result.finalEvents.length - 1),
49804
49849
  group: resolveFixtureEnvironment(fixture),
49805
- passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && (speakerTurns ? speakerTurns.passes : true),
49850
+ passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && criticalFields.passesRequired && (speakerTurns ? speakerTurns.passes : true),
49806
49851
  partialCount: result.partialEvents.length,
49807
49852
  speakerTurns,
49808
49853
  postSpeechTimeToEndOfTurnMs,
@@ -49930,6 +49975,8 @@ var summarizeSTTBenchmark = (adapterId, fixtures) => {
49930
49975
  averageFinalCount: roundMetric4(average2(fixtures.map((fixture) => fixture.finalCount)), 2) ?? 0,
49931
49976
  averageSpeakerTurnMatchRate: roundMetric4(average2(fixtures.map((fixture) => fixture.speakerTurns?.patternMatchRate))),
49932
49977
  averageTermRecall: roundMetric4(average2(fixtures.map((fixture) => fixture.expectedTerms.recall))) ?? 0,
49978
+ averageCriticalFieldAccuracy: roundMetric4(average2(fixtures.map((fixture) => fixture.criticalFields?.accuracy ?? 1))) ?? 0,
49979
+ requiredCriticalFieldPassRate: fixtureCount > 0 ? roundMetric4(fixtures.filter((fixture) => fixture.criticalFields?.passesRequired ?? true).length / fixtureCount) ?? 0 : 0,
49933
49980
  averagePostSpeechTimeToEndOfTurnMs: roundMetric4(average2(fixtures.map((fixture) => fixture.postSpeechTimeToEndOfTurnMs)), 2),
49934
49981
  averagePostSpeechTimeToFirstFinalMs: roundMetric4(average2(fixtures.map((fixture) => fixture.postSpeechTimeToFirstFinalMs)), 2),
49935
49982
  averageTimeToEndOfTurnMs: roundMetric4(average2(fixtures.map((fixture) => fixture.timeToEndOfTurnMs)), 2),
@@ -1,6 +1,7 @@
1
1
  import type { STTAdapter, STTAdapterOpenOptions } from "../core/types";
2
2
  import { type VoiceSTTAdapterHarnessOptions, type VoiceSTTAdapterHarnessResult } from "./stt";
3
3
  import type { VoiceTestFixture } from "./fixtures";
4
+ import { type VoiceCriticalFieldAccuracy } from "./criticalFields";
4
5
  export type VoiceExpectedTermAccuracy = {
5
6
  allMatched: boolean;
6
7
  expectedTerms: string[];
@@ -25,6 +26,7 @@ export type VoiceSTTBenchmarkFixtureResult = {
25
26
  endOfTurnCount: number;
26
27
  errorCount: number;
27
28
  expectedTerms: VoiceExpectedTermAccuracy;
29
+ criticalFields?: VoiceCriticalFieldAccuracy;
28
30
  finalCount: number;
29
31
  finalText: string;
30
32
  fixtureId: string;
@@ -49,6 +51,8 @@ export type VoiceSTTBenchmarkSummary = {
49
51
  averageFinalCount: number;
50
52
  averageSpeakerTurnMatchRate?: number;
51
53
  averageTermRecall: number;
54
+ averageCriticalFieldAccuracy: number;
55
+ requiredCriticalFieldPassRate: number;
52
56
  averagePostSpeechTimeToEndOfTurnMs?: number;
53
57
  averagePostSpeechTimeToFirstFinalMs?: number;
54
58
  averageTimeToEndOfTurnMs?: number;
@@ -0,0 +1,19 @@
1
+ export type VoiceConfidenceCalibrationSample = {
2
+ confidence: number;
3
+ correct: boolean;
4
+ metadata?: Record<string, unknown>;
5
+ };
6
+ export type VoiceConfidenceCalibrationBin = {
7
+ accuracy: number;
8
+ averageConfidence: number;
9
+ count: number;
10
+ lowerBound: number;
11
+ upperBound: number;
12
+ };
13
+ export type VoiceConfidenceCalibrationReport = {
14
+ bins: VoiceConfidenceCalibrationBin[];
15
+ brierScore: number;
16
+ expectedCalibrationError: number;
17
+ sampleCount: number;
18
+ };
19
+ export declare const calibrateVoiceConfidence: (samples: VoiceConfidenceCalibrationSample[], binCount?: number) => VoiceConfidenceCalibrationReport;
@@ -0,0 +1,22 @@
1
+ export type VoiceCriticalFieldKind = "acronym" | "brand" | "currency" | "custom" | "email" | "number" | "organization" | "percentage" | "person-name" | "phone";
2
+ export type VoiceExpectedCriticalField = {
3
+ aliases?: string[];
4
+ id: string;
5
+ kind: VoiceCriticalFieldKind;
6
+ metadata?: Record<string, unknown>;
7
+ required?: boolean;
8
+ value: string;
9
+ };
10
+ export type VoiceCriticalFieldResult = VoiceExpectedCriticalField & {
11
+ matched: boolean;
12
+ matchedAlias?: string;
13
+ };
14
+ export type VoiceCriticalFieldAccuracy = {
15
+ accuracy: number;
16
+ fields: VoiceCriticalFieldResult[];
17
+ matchedCount: number;
18
+ missingFieldIds: string[];
19
+ passesRequired: boolean;
20
+ totalCount: number;
21
+ };
22
+ export declare const scoreVoiceCriticalFields: (actualText: string, expectedFields?: VoiceExpectedCriticalField[]) => VoiceCriticalFieldAccuracy;
@@ -1,10 +1,12 @@
1
1
  import type { AudioFormat, VoiceExpectedSpeakerTurn } from "../core/types";
2
+ import type { VoiceExpectedCriticalField } from "./criticalFields";
2
3
  export type VoiceTestFixtureManifestEntry = {
3
4
  id: string;
4
5
  title: string;
5
6
  audioPath: string;
6
7
  expectedText: string;
7
8
  expectedTerms?: string[];
9
+ expectedCriticalFields?: VoiceExpectedCriticalField[];
8
10
  expectedSpeakerTurns?: VoiceExpectedSpeakerTurn[];
9
11
  expectedTurnTexts?: string[];
10
12
  chunkDurationMs?: number;
@@ -1,5 +1,7 @@
1
1
  export * from "./accuracy";
2
2
  export * from "./benchmark";
3
+ export * from "./confidenceCalibration";
4
+ export * from "./criticalFields";
3
5
  export * from "./corrected";
4
6
  export * from "./duplex";
5
7
  export * from "./fixtures";
@@ -7,6 +9,7 @@ export * from "./ioProviderSimulator";
7
9
  export * from "./providerSimulator";
8
10
  export * from "./resilience";
9
11
  export * from "./review";
12
+ export * from "./routingBenchmark";
10
13
  export * from "./sessionBenchmark";
11
14
  export * from "./stt";
12
15
  export * from "./telephony";
@@ -367,6 +367,229 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
367
367
  };
368
368
  };
369
369
 
370
+ // src/core/numberNormalizer.ts
371
+ var ONES = {
372
+ eight: 8,
373
+ eighteen: 18,
374
+ eleven: 11,
375
+ fifteen: 15,
376
+ five: 5,
377
+ four: 4,
378
+ fourteen: 14,
379
+ nine: 9,
380
+ nineteen: 19,
381
+ one: 1,
382
+ seven: 7,
383
+ seventeen: 17,
384
+ six: 6,
385
+ sixteen: 16,
386
+ ten: 10,
387
+ thirteen: 13,
388
+ three: 3,
389
+ twelve: 12,
390
+ two: 2,
391
+ zero: 0
392
+ };
393
+ var TENS = {
394
+ eighty: 80,
395
+ fifty: 50,
396
+ forty: 40,
397
+ ninety: 90,
398
+ seventy: 70,
399
+ sixty: 60,
400
+ thirty: 30,
401
+ twenty: 20
402
+ };
403
+ var SCALES = {
404
+ billion: 1e9,
405
+ million: 1e6,
406
+ thousand: 1000,
407
+ trillion: 1000000000000
408
+ };
409
+ var MAGNITUDE_WORDS = [
410
+ [1000000000000, "trillion"],
411
+ [1e9, "billion"],
412
+ [1e6, "million"]
413
+ ];
414
+ var FILLER = new Set(["and", "a", "an"]);
415
+ var DECIMAL_PLACES = 3;
416
+ var isValueWord = (word) => (word in ONES) || (word in TENS) || word === "hundred";
417
+ var isScaleWord = (word) => (word in SCALES);
418
+ var isNumberWord = (word) => isValueWord(word) || isScaleWord(word) || word === "point" || FILLER.has(word);
419
+ var trimNumber = (value) => {
420
+ if (Number.isInteger(value))
421
+ return value.toLocaleString("en-US");
422
+ return String(Number(value.toFixed(DECIMAL_PLACES)));
423
+ };
424
+ var renderValue = (value, usedMagnitude) => {
425
+ if (usedMagnitude) {
426
+ for (const [scale, word] of MAGNITUDE_WORDS) {
427
+ if (value >= scale) {
428
+ const scaled = value / scale;
429
+ if (Number(scaled.toFixed(DECIMAL_PLACES)) === scaled) {
430
+ return `${trimNumber(scaled)} ${word}`;
431
+ }
432
+ }
433
+ }
434
+ }
435
+ return trimNumber(value);
436
+ };
437
+ var parseNumberWords = (words) => {
438
+ let total = 0;
439
+ let current = 0;
440
+ let usedMagnitude = false;
441
+ let sawNumber = false;
442
+ let decimal = null;
443
+ const foldDecimal = () => {
444
+ if (decimal && decimal.length > 0)
445
+ current += Number(`0.${decimal}`);
446
+ decimal = null;
447
+ };
448
+ for (const word of words) {
449
+ if (word === "point") {
450
+ decimal = "";
451
+ continue;
452
+ }
453
+ const one = ONES[word];
454
+ const ten = TENS[word];
455
+ const scale = SCALES[word];
456
+ if (decimal !== null) {
457
+ if (one !== undefined && one <= 9) {
458
+ decimal += String(one);
459
+ sawNumber = true;
460
+ continue;
461
+ }
462
+ foldDecimal();
463
+ }
464
+ if (FILLER.has(word))
465
+ continue;
466
+ if (one !== undefined) {
467
+ current += one;
468
+ sawNumber = true;
469
+ } else if (ten !== undefined) {
470
+ current += ten;
471
+ sawNumber = true;
472
+ } else if (word === "hundred") {
473
+ current = (current === 0 ? 1 : current) * 100;
474
+ sawNumber = true;
475
+ } else if (scale !== undefined) {
476
+ total += (current === 0 ? 1 : current) * scale;
477
+ current = 0;
478
+ sawNumber = true;
479
+ usedMagnitude = true;
480
+ }
481
+ }
482
+ foldDecimal();
483
+ if (!sawNumber)
484
+ return null;
485
+ return { usedMagnitude, value: total + current };
486
+ };
487
+ var PERCENT_RE = /^(per ?cent|percent|percentage)$/;
488
+ var CURRENCY_RE = /^(dollars?|bucks?|usd)$/;
489
+ var WORD_RE = /^[A-Za-z]+(?:-[A-Za-z]+)*$/;
490
+ var wordsOf = (token) => token.toLowerCase().split("-");
491
+ var isNumberToken = (token) => WORD_RE.test(token) && wordsOf(token).every(isNumberWord);
492
+ var startsNumber = (token) => WORD_RE.test(token) && wordsOf(token).some(isValueWord);
493
+ var isSpace = (token) => /^\s+$/.test(token);
494
+ var normalizeSpokenNumbers = (input) => {
495
+ if (!input)
496
+ return input;
497
+ const parts = input.match(/[A-Za-z]+(?:-[A-Za-z]+)*|[^A-Za-z]+/g);
498
+ if (!parts)
499
+ return input;
500
+ const at = (idx) => parts[idx] ?? "";
501
+ const out = [];
502
+ let i = 0;
503
+ while (i < parts.length) {
504
+ const token = at(i);
505
+ const lower = token.toLowerCase();
506
+ if (lower === "a" || lower === "an") {
507
+ const nextWord = at(i + 2);
508
+ const nextHead = wordsOf(nextWord)[0] ?? "";
509
+ const nextIsScale = isSpace(at(i + 1)) && WORD_RE.test(nextWord) && (nextHead === "hundred" || isScaleWord(nextHead));
510
+ if (!nextIsScale) {
511
+ out.push(token);
512
+ i += 1;
513
+ continue;
514
+ }
515
+ } else if (!startsNumber(token)) {
516
+ out.push(token);
517
+ i += 1;
518
+ continue;
519
+ }
520
+ const spanIdx = [i];
521
+ let j = i + 1;
522
+ while (isSpace(at(j)) && isNumberToken(at(j + 1))) {
523
+ spanIdx.push(j + 1);
524
+ j += 2;
525
+ }
526
+ const words = spanIdx.flatMap((k) => wordsOf(at(k)));
527
+ while (words.length > 0 && FILLER.has(words[words.length - 1] ?? "")) {
528
+ words.pop();
529
+ }
530
+ const parsed = parseNumberWords(words);
531
+ if (!parsed) {
532
+ out.push(token);
533
+ i += 1;
534
+ continue;
535
+ }
536
+ let rendered = renderValue(parsed.value, parsed.usedMagnitude);
537
+ let lastIdx = spanIdx[spanIdx.length - 1] ?? i;
538
+ const unitWord = at(lastIdx + 2);
539
+ if (isSpace(at(lastIdx + 1)) && WORD_RE.test(unitWord)) {
540
+ const unit = unitWord.toLowerCase();
541
+ if (PERCENT_RE.test(unit)) {
542
+ rendered = `${rendered}%`;
543
+ lastIdx += 2;
544
+ } else if (CURRENCY_RE.test(unit)) {
545
+ rendered = rendered.startsWith("$") ? rendered : `$${rendered}`;
546
+ lastIdx += 2;
547
+ }
548
+ }
549
+ out.push(rendered);
550
+ i = lastIdx + 1;
551
+ }
552
+ return out.join("");
553
+ };
554
+
555
+ // src/testing/criticalFields.ts
556
+ var normalizeText2 = (value) => value.normalize("NFKC").toLowerCase().replace(/[^\p{L}\p{N}@.%+$'-]+/gu, " ").replace(/\s+/g, " ").trim();
557
+ var normalizeDigits = (value) => normalizeSpokenNumbers(value).replace(/\D/g, "");
558
+ var normalizeSemanticNumber = (value) => normalizeSpokenNumbers(value).toLowerCase().replace(/\bdollars?\b|\busd\b|\bpercent(age)?\b|[%,$]/g, "").replace(/\s+/g, "").trim();
559
+ var matchesCandidate = (actual, candidate, kind) => {
560
+ if (kind === "phone") {
561
+ const expectedDigits = normalizeDigits(candidate);
562
+ return expectedDigits.length > 0 && normalizeDigits(actual).includes(expectedDigits);
563
+ }
564
+ if (kind === "currency" || kind === "number" || kind === "percentage") {
565
+ const expectedNumber = normalizeSemanticNumber(candidate);
566
+ return expectedNumber.length > 0 && normalizeSemanticNumber(actual).includes(expectedNumber);
567
+ }
568
+ const normalizedCandidate = normalizeText2(candidate);
569
+ return normalizedCandidate.length > 0 && normalizeText2(actual).includes(normalizedCandidate);
570
+ };
571
+ var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
572
+ const fields = expectedFields.map((field) => {
573
+ const candidates = [field.value, ...field.aliases ?? []];
574
+ const matchedAlias = candidates.find((candidate) => matchesCandidate(actualText, candidate, field.kind));
575
+ return {
576
+ ...field,
577
+ matched: matchedAlias !== undefined,
578
+ matchedAlias
579
+ };
580
+ });
581
+ const matchedCount = fields.filter((field) => field.matched).length;
582
+ const totalCount = fields.length;
583
+ return {
584
+ accuracy: totalCount > 0 ? matchedCount / totalCount : 1,
585
+ fields,
586
+ matchedCount,
587
+ missingFieldIds: fields.filter((field) => !field.matched).map((field) => field.id),
588
+ passesRequired: fields.every((field) => field.required === false || field.matched),
589
+ totalCount
590
+ };
591
+ };
592
+
370
593
  // src/testing/benchmark.ts
371
594
  var resolveFixtureEnvironment = (fixture) => {
372
595
  const tags = new Set(fixture.tags ?? []);
@@ -597,10 +820,12 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
597
820
  const postSpeechTimeToFirstFinalMs = toPostSpeechLatency(result.finalEvents[0]?.receivedAt);
598
821
  const postSpeechTimeToEndOfTurnMs = toPostSpeechLatency(result.endOfTurnEvents[0]?.receivedAt);
599
822
  const expectedTerms = scoreExpectedTerms(result.finalText, fixture.expectedTerms);
823
+ const criticalFields = scoreVoiceCriticalFields(result.finalText, fixture.expectedCriticalFields);
600
824
  const speakerTurns = scoreSpeakerTurns(fixture, result);
601
825
  return {
602
826
  accuracy: result.accuracy,
603
827
  closeCount: result.closeEvents.length,
828
+ criticalFields,
604
829
  difficulty: fixture.difficulty,
605
830
  elapsedMs,
606
831
  endOfTurnCount: result.endOfTurnEvents.length,
@@ -611,7 +836,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
611
836
  fixtureId: fixture.id,
612
837
  fragmentationCount: Math.max(0, result.finalEvents.length - 1),
613
838
  group: resolveFixtureEnvironment(fixture),
614
- passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && (speakerTurns ? speakerTurns.passes : true),
839
+ passes: result.errorEvents.length === 0 && result.finalText.trim().length > 0 && result.accuracy.passesThreshold && criticalFields.passesRequired && (speakerTurns ? speakerTurns.passes : true),
615
840
  partialCount: result.partialEvents.length,
616
841
  speakerTurns,
617
842
  postSpeechTimeToEndOfTurnMs,
@@ -739,6 +964,8 @@ var summarizeSTTBenchmark = (adapterId, fixtures) => {
739
964
  averageFinalCount: roundMetric(average(fixtures.map((fixture) => fixture.finalCount)), 2) ?? 0,
740
965
  averageSpeakerTurnMatchRate: roundMetric(average(fixtures.map((fixture) => fixture.speakerTurns?.patternMatchRate))),
741
966
  averageTermRecall: roundMetric(average(fixtures.map((fixture) => fixture.expectedTerms.recall))) ?? 0,
967
+ averageCriticalFieldAccuracy: roundMetric(average(fixtures.map((fixture) => fixture.criticalFields?.accuracy ?? 1))) ?? 0,
968
+ requiredCriticalFieldPassRate: fixtureCount > 0 ? roundMetric(fixtures.filter((fixture) => fixture.criticalFields?.passesRequired ?? true).length / fixtureCount) ?? 0 : 0,
742
969
  averagePostSpeechTimeToEndOfTurnMs: roundMetric(average(fixtures.map((fixture) => fixture.postSpeechTimeToEndOfTurnMs)), 2),
743
970
  averagePostSpeechTimeToFirstFinalMs: roundMetric(average(fixtures.map((fixture) => fixture.postSpeechTimeToFirstFinalMs)), 2),
744
971
  averageTimeToEndOfTurnMs: roundMetric(average(fixtures.map((fixture) => fixture.timeToEndOfTurnMs)), 2),
@@ -803,6 +1030,45 @@ var summarizeSTTBenchmarkSeries = (input) => {
803
1030
  }
804
1031
  };
805
1032
  };
1033
+ // src/testing/confidenceCalibration.ts
1034
+ var clampConfidence = (value) => Math.max(0, Math.min(1, value));
1035
+ var calibrateVoiceConfidence = (samples, binCount = 10) => {
1036
+ const safeBinCount = Math.max(1, Math.round(binCount));
1037
+ const bins = Array.from({ length: safeBinCount }, (_, index) => {
1038
+ const lowerBound = index / safeBinCount;
1039
+ return {
1040
+ accuracy: 0,
1041
+ averageConfidence: 0,
1042
+ count: 0,
1043
+ lowerBound,
1044
+ upperBound: (index + 1) / safeBinCount
1045
+ };
1046
+ });
1047
+ let brierTotal = 0;
1048
+ for (const sample of samples) {
1049
+ const confidence = clampConfidence(sample.confidence);
1050
+ const binIndex = Math.min(safeBinCount - 1, Math.floor(confidence * safeBinCount));
1051
+ const bin = bins[binIndex];
1052
+ bin.count += 1;
1053
+ bin.averageConfidence += confidence;
1054
+ bin.accuracy += sample.correct ? 1 : 0;
1055
+ brierTotal += (confidence - (sample.correct ? 1 : 0)) ** 2;
1056
+ }
1057
+ let expectedCalibrationError = 0;
1058
+ for (const bin of bins) {
1059
+ if (bin.count === 0)
1060
+ continue;
1061
+ bin.averageConfidence /= bin.count;
1062
+ bin.accuracy /= bin.count;
1063
+ expectedCalibrationError += bin.count / Math.max(1, samples.length) * Math.abs(bin.accuracy - bin.averageConfidence);
1064
+ }
1065
+ return {
1066
+ bins,
1067
+ brierScore: samples.length > 0 ? brierTotal / samples.length : 0,
1068
+ expectedCalibrationError,
1069
+ sampleCount: samples.length
1070
+ };
1071
+ };
806
1072
  // src/core/correction.ts
807
1073
  var escapeRegExp = (value) => value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
808
1074
  var buildAliasMatcher = (alias) => new RegExp(`(?<![\\p{L}\\p{N}'])${escapeRegExp(alias)}(?![\\p{L}\\p{N}'])`, "giu");
@@ -5599,191 +5865,6 @@ var createVoiceMemoryStore = () => {
5599
5865
  // src/core/session.ts
5600
5866
  import { Buffer as Buffer2 } from "buffer";
5601
5867
 
5602
- // src/core/numberNormalizer.ts
5603
- var ONES = {
5604
- eight: 8,
5605
- eighteen: 18,
5606
- eleven: 11,
5607
- fifteen: 15,
5608
- five: 5,
5609
- four: 4,
5610
- fourteen: 14,
5611
- nine: 9,
5612
- nineteen: 19,
5613
- one: 1,
5614
- seven: 7,
5615
- seventeen: 17,
5616
- six: 6,
5617
- sixteen: 16,
5618
- ten: 10,
5619
- thirteen: 13,
5620
- three: 3,
5621
- twelve: 12,
5622
- two: 2,
5623
- zero: 0
5624
- };
5625
- var TENS = {
5626
- eighty: 80,
5627
- fifty: 50,
5628
- forty: 40,
5629
- ninety: 90,
5630
- seventy: 70,
5631
- sixty: 60,
5632
- thirty: 30,
5633
- twenty: 20
5634
- };
5635
- var SCALES = {
5636
- billion: 1e9,
5637
- million: 1e6,
5638
- thousand: 1000,
5639
- trillion: 1000000000000
5640
- };
5641
- var MAGNITUDE_WORDS = [
5642
- [1000000000000, "trillion"],
5643
- [1e9, "billion"],
5644
- [1e6, "million"]
5645
- ];
5646
- var FILLER = new Set(["and", "a", "an"]);
5647
- var DECIMAL_PLACES = 3;
5648
- var isValueWord = (word) => (word in ONES) || (word in TENS) || word === "hundred";
5649
- var isScaleWord = (word) => (word in SCALES);
5650
- var isNumberWord = (word) => isValueWord(word) || isScaleWord(word) || word === "point" || FILLER.has(word);
5651
- var trimNumber = (value) => {
5652
- if (Number.isInteger(value))
5653
- return value.toLocaleString("en-US");
5654
- return String(Number(value.toFixed(DECIMAL_PLACES)));
5655
- };
5656
- var renderValue = (value, usedMagnitude) => {
5657
- if (usedMagnitude) {
5658
- for (const [scale, word] of MAGNITUDE_WORDS) {
5659
- if (value >= scale) {
5660
- const scaled = value / scale;
5661
- if (Number(scaled.toFixed(DECIMAL_PLACES)) === scaled) {
5662
- return `${trimNumber(scaled)} ${word}`;
5663
- }
5664
- }
5665
- }
5666
- }
5667
- return trimNumber(value);
5668
- };
5669
- var parseNumberWords = (words) => {
5670
- let total = 0;
5671
- let current = 0;
5672
- let usedMagnitude = false;
5673
- let sawNumber = false;
5674
- let decimal = null;
5675
- const foldDecimal = () => {
5676
- if (decimal && decimal.length > 0)
5677
- current += Number(`0.${decimal}`);
5678
- decimal = null;
5679
- };
5680
- for (const word of words) {
5681
- if (word === "point") {
5682
- decimal = "";
5683
- continue;
5684
- }
5685
- const one = ONES[word];
5686
- const ten = TENS[word];
5687
- const scale = SCALES[word];
5688
- if (decimal !== null) {
5689
- if (one !== undefined && one <= 9) {
5690
- decimal += String(one);
5691
- sawNumber = true;
5692
- continue;
5693
- }
5694
- foldDecimal();
5695
- }
5696
- if (FILLER.has(word))
5697
- continue;
5698
- if (one !== undefined) {
5699
- current += one;
5700
- sawNumber = true;
5701
- } else if (ten !== undefined) {
5702
- current += ten;
5703
- sawNumber = true;
5704
- } else if (word === "hundred") {
5705
- current = (current === 0 ? 1 : current) * 100;
5706
- sawNumber = true;
5707
- } else if (scale !== undefined) {
5708
- total += (current === 0 ? 1 : current) * scale;
5709
- current = 0;
5710
- sawNumber = true;
5711
- usedMagnitude = true;
5712
- }
5713
- }
5714
- foldDecimal();
5715
- if (!sawNumber)
5716
- return null;
5717
- return { usedMagnitude, value: total + current };
5718
- };
5719
- var PERCENT_RE = /^(per ?cent|percent|percentage)$/;
5720
- var CURRENCY_RE = /^(dollars?|bucks?|usd)$/;
5721
- var WORD_RE = /^[A-Za-z]+(?:-[A-Za-z]+)*$/;
5722
- var wordsOf = (token) => token.toLowerCase().split("-");
5723
- var isNumberToken = (token) => WORD_RE.test(token) && wordsOf(token).every(isNumberWord);
5724
- var startsNumber = (token) => WORD_RE.test(token) && wordsOf(token).some(isValueWord);
5725
- var isSpace = (token) => /^\s+$/.test(token);
5726
- var normalizeSpokenNumbers = (input) => {
5727
- if (!input)
5728
- return input;
5729
- const parts = input.match(/[A-Za-z]+(?:-[A-Za-z]+)*|[^A-Za-z]+/g);
5730
- if (!parts)
5731
- return input;
5732
- const at = (idx) => parts[idx] ?? "";
5733
- const out = [];
5734
- let i = 0;
5735
- while (i < parts.length) {
5736
- const token = at(i);
5737
- const lower = token.toLowerCase();
5738
- if (lower === "a" || lower === "an") {
5739
- const nextWord = at(i + 2);
5740
- const nextHead = wordsOf(nextWord)[0] ?? "";
5741
- const nextIsScale = isSpace(at(i + 1)) && WORD_RE.test(nextWord) && (nextHead === "hundred" || isScaleWord(nextHead));
5742
- if (!nextIsScale) {
5743
- out.push(token);
5744
- i += 1;
5745
- continue;
5746
- }
5747
- } else if (!startsNumber(token)) {
5748
- out.push(token);
5749
- i += 1;
5750
- continue;
5751
- }
5752
- const spanIdx = [i];
5753
- let j = i + 1;
5754
- while (isSpace(at(j)) && isNumberToken(at(j + 1))) {
5755
- spanIdx.push(j + 1);
5756
- j += 2;
5757
- }
5758
- const words = spanIdx.flatMap((k) => wordsOf(at(k)));
5759
- while (words.length > 0 && FILLER.has(words[words.length - 1] ?? "")) {
5760
- words.pop();
5761
- }
5762
- const parsed = parseNumberWords(words);
5763
- if (!parsed) {
5764
- out.push(token);
5765
- i += 1;
5766
- continue;
5767
- }
5768
- let rendered = renderValue(parsed.value, parsed.usedMagnitude);
5769
- let lastIdx = spanIdx[spanIdx.length - 1] ?? i;
5770
- const unitWord = at(lastIdx + 2);
5771
- if (isSpace(at(lastIdx + 1)) && WORD_RE.test(unitWord)) {
5772
- const unit = unitWord.toLowerCase();
5773
- if (PERCENT_RE.test(unit)) {
5774
- rendered = `${rendered}%`;
5775
- lastIdx += 2;
5776
- } else if (CURRENCY_RE.test(unit)) {
5777
- rendered = rendered.startsWith("$") ? rendered : `$${rendered}`;
5778
- lastIdx += 2;
5779
- }
5780
- }
5781
- out.push(rendered);
5782
- i = lastIdx + 1;
5783
- }
5784
- return out.join("");
5785
- };
5786
-
5787
5868
  // src/core/backchannel.ts
5788
5869
  var DEFAULT_CUES = [
5789
5870
  { text: "mm-hmm" },
@@ -6414,7 +6495,7 @@ var cloneTranscript = (transcript) => ({
6414
6495
  });
6415
6496
  var encodeBase64 = (chunk) => Buffer2.from(chunk).toString("base64");
6416
6497
  var countWords2 = (text) => text.trim().split(/\s+/).filter(Boolean).length;
6417
- var normalizeText2 = (text) => text.trim().replace(/\s+/g, " ");
6498
+ var normalizeText3 = (text) => text.trim().replace(/\s+/g, " ");
6418
6499
  var getAudioChunkDurationMs = (chunk) => chunk.byteLength / (DEFAULT_FORMAT.sampleRateHz * DEFAULT_FORMAT.channels * 2) * 1000;
6419
6500
  var getBufferedAudioDurationMs = (chunks) => chunks.reduce((total, chunk) => total + getAudioChunkDurationMs(chunk), 0);
6420
6501
  var STREAM_SENTENCE_BOUNDARY = /[.!?\u2026]['")\]]*\s/;
@@ -6490,9 +6571,9 @@ var createTurnCostEstimate = (input) => {
6490
6571
  totalBillableAudioMs: Math.max(0, input.primaryAudioMs) + Math.max(0, input.fallbackReplayAudioMs)
6491
6572
  };
6492
6573
  };
6493
- var normalizeCorrectionText = (text) => normalizeText2(text);
6574
+ var normalizeCorrectionText = (text) => normalizeText3(text);
6494
6575
  var evaluateFallbackNeed = (candidate, config) => {
6495
- const trimmed = normalizeText2(candidate.text);
6576
+ const trimmed = normalizeText3(candidate.text);
6496
6577
  const wordCount = countWords2(trimmed);
6497
6578
  const averageConfidence = calculateMeanConfidence(candidate.transcripts);
6498
6579
  const words = collectTranscriptWords(candidate.transcripts);
@@ -7901,12 +7982,12 @@ var createVoiceSession = (options) => {
7901
7982
  const fallbackCandidate = {
7902
7983
  confidence: fallbackConfidence,
7903
7984
  text: fallbackText,
7904
- wordCount: countWords2(normalizeText2(fallbackText))
7985
+ wordCount: countWords2(normalizeText3(fallbackText))
7905
7986
  };
7906
7987
  const primaryCandidate = {
7907
7988
  confidence: calculateMeanConfidence(primaryTranscripts),
7908
7989
  text: primaryText,
7909
- wordCount: countWords2(normalizeText2(primaryText))
7990
+ wordCount: countWords2(normalizeText3(primaryText))
7910
7991
  };
7911
7992
  const policyPrefersFallback = fallbackCandidate.text.length > 0 && fallbackNeed.reason !== undefined && sttFallback.preferFallbackOn?.includes(fallbackNeed.reason);
7912
7993
  const selection = policyPrefersFallback ? {
@@ -7999,7 +8080,7 @@ var createVoiceSession = (options) => {
7999
8080
  };
8000
8081
  const buildTurnSignature = (session, finalText, transcriptIdsOverride) => {
8001
8082
  const finalTranscriptIds = transcriptIdsOverride ?? getFinalTranscriptIds(session.currentTurn.transcripts);
8002
- return `${normalizeText2(finalText)}|${finalTranscriptIds.join(",")}`;
8083
+ return `${normalizeText3(finalText)}|${finalTranscriptIds.join(",")}`;
8003
8084
  };
8004
8085
  const isDuplicateTurnCommit = (session, finalText) => {
8005
8086
  const signature = buildTurnSignature(session, finalText);
@@ -8007,8 +8088,8 @@ var createVoiceSession = (options) => {
8007
8088
  const isRecent = committedTurn && committedTurn.committedAt > 0 && Date.now() - committedTurn.committedAt < DEFAULT_DUPLICATE_TURN_WINDOW_MS;
8008
8089
  const committedSignature = committedTurn?.signature ?? "";
8009
8090
  const committedTranscriptIds = committedTurn?.transcriptIds ?? [];
8010
- const committedText = normalizeText2(committedTurn?.text ?? "");
8011
- const isSameText = normalizeText2(finalText) === committedText;
8091
+ const committedText = normalizeText3(committedTurn?.text ?? "");
8092
+ const isSameText = normalizeText3(finalText) === committedText;
8012
8093
  const hasNoNewAudioSinceCommit = (session.currentTurn.lastAudioAt ?? 0) <= (committedTurn?.committedAt ?? 0);
8013
8094
  if (!isRecent) {
8014
8095
  return false;
@@ -8028,7 +8109,7 @@ var createVoiceSession = (options) => {
8028
8109
  ...session.lastCommittedTurn ?? {},
8029
8110
  committedAt: Date.now(),
8030
8111
  signature: buildTurnSignature(session, finalText, getFinalTranscriptIds(committedTranscripts)),
8031
- text: normalizeText2(finalText),
8112
+ text: normalizeText3(finalText),
8032
8113
  transcriptIds: getFinalTranscriptIds(committedTranscripts)
8033
8114
  };
8034
8115
  };
@@ -9424,6 +9505,10 @@ var createVoiceSession = (options) => {
9424
9505
  commitTurn: async (reason = "manual") => runSerial("api.commitTurn", async () => {
9425
9506
  await commitTurnInternal(reason);
9426
9507
  }),
9508
+ configureSTT: async (configuration) => runSerial("api.configureSTT", async () => {
9509
+ const adapter = await ensureAdapter();
9510
+ await adapter.configure?.(configuration);
9511
+ }),
9427
9512
  complete: async (result) => runSerial("api.complete", async () => {
9428
9513
  await completeInternal(result);
9429
9514
  }),
@@ -10257,6 +10342,22 @@ var renderVoiceCallReviewMarkdown = (artifact) => {
10257
10342
  ].filter((value) => typeof value === "string").join(`
10258
10343
  `);
10259
10344
  };
10345
+ // src/testing/routingBenchmark.ts
10346
+ var evaluateVoiceSTTRouting = (fixtures) => {
10347
+ const attempted = fixtures.filter((fixture) => fixture.fallbackUsed);
10348
+ const improved = attempted.filter((fixture) => fixture.fallbackScore > fixture.primaryScore);
10349
+ const harmed = attempted.filter((fixture) => fixture.fallbackScore < fixture.primaryScore);
10350
+ const average3 = (values) => values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : 0;
10351
+ return {
10352
+ fallbackAttemptRate: fixtures.length > 0 ? attempted.length / fixtures.length : 0,
10353
+ fallbackHarmRate: attempted.length > 0 ? harmed.length / attempted.length : 0,
10354
+ fallbackImprovementRate: attempted.length > 0 ? improved.length / attempted.length : 0,
10355
+ fixtureCount: fixtures.length,
10356
+ oracleScore: average3(fixtures.map((fixture) => Math.max(fixture.primaryScore, fixture.fallbackScore))),
10357
+ primaryScore: average3(fixtures.map((fixture) => fixture.primaryScore)),
10358
+ selectedScore: average3(fixtures.map((fixture) => fixture.fallbackUsed ? fixture.fallbackScore : fixture.primaryScore))
10359
+ };
10360
+ };
10260
10361
  // src/testing/sessionBenchmark.ts
10261
10362
  var average3 = (values) => values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : 0;
10262
10363
  var normalizeTurnText = (value) => value.toLowerCase().replace(/[^\p{L}\p{N}\s']/gu, " ").replace(/\s+/g, " ").trim();
@@ -15689,6 +15790,7 @@ export {
15689
15790
  summarizeTTSBenchmark,
15690
15791
  summarizeSTTBenchmarkSeries,
15691
15792
  summarizeSTTBenchmark,
15793
+ scoreVoiceCriticalFields,
15692
15794
  scoreTranscriptAccuracy,
15693
15795
  scoreCorrectedExpectedTerms,
15694
15796
  runVoiceTelephonyMediaOperationsSmoke,
@@ -15716,6 +15818,7 @@ export {
15716
15818
  getDefaultVoiceTelephonyBenchmarkScenarios,
15717
15819
  getDefaultVoiceDuplexBenchmarkScenarios,
15718
15820
  getDefaultTTSBenchmarkFixtures,
15821
+ evaluateVoiceSTTRouting,
15719
15822
  evaluateSTTBenchmarkAcceptance,
15720
15823
  createVoiceProviderFailureSimulator,
15721
15824
  createVoiceIOProviderFailureSimulator,
@@ -15727,6 +15830,7 @@ export {
15727
15830
  createCodeSwitchBenchmarkCorrectionHandler,
15728
15831
  createBenchmarkCorrectionHandler,
15729
15832
  compareSTTBenchmarks,
15833
+ calibrateVoiceConfidence,
15730
15834
  buildSessionCorrectionAudit,
15731
15835
  buildFixturePhraseHints,
15732
15836
  buildCorrectionBenchmarkAudit,
@@ -0,0 +1,16 @@
1
+ export type VoiceSTTRoutingFixture = {
2
+ fallbackScore: number;
3
+ fallbackUsed: boolean;
4
+ id: string;
5
+ primaryScore: number;
6
+ };
7
+ export type VoiceSTTRoutingBenchmarkReport = {
8
+ fallbackAttemptRate: number;
9
+ fallbackHarmRate: number;
10
+ fallbackImprovementRate: number;
11
+ fixtureCount: number;
12
+ oracleScore: number;
13
+ primaryScore: number;
14
+ selectedScore: number;
15
+ };
16
+ export declare const evaluateVoiceSTTRouting: (fixtures: VoiceSTTRoutingFixture[]) => VoiceSTTRoutingBenchmarkReport;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@absolutejs/voice",
3
- "version": "0.0.22-beta.635",
3
+ "version": "0.0.22-beta.636",
4
4
  "description": "Voice primitives and Elysia plugin for AbsoluteJS",
5
5
  "repository": {
6
6
  "type": "git",