@absolutejs/voice 0.0.22-beta.633 → 0.0.22-beta.635

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -78,6 +78,18 @@ export type VoiceTranscriptSentiment = {
78
78
  metadata?: Record<string, unknown>;
79
79
  score?: number;
80
80
  };
81
+ /** A provider-normalized word hypothesis. Keeping this evidence attached to the
82
+ * transcript lets applications detect a single risky name or number that would
83
+ * otherwise disappear inside a high turn-level average confidence. */
84
+ export type TranscriptWord = {
85
+ confidence?: number;
86
+ endedAtMs?: number;
87
+ language?: string;
88
+ punctuatedText?: string;
89
+ speaker?: string | number;
90
+ startedAtMs?: number;
91
+ text: string;
92
+ };
81
93
  export type Transcript = {
82
94
  id: string;
83
95
  text: string;
@@ -89,6 +101,7 @@ export type Transcript = {
89
101
  startedAtMs?: number;
90
102
  endedAtMs?: number;
91
103
  vendor?: string;
104
+ words?: TranscriptWord[];
92
105
  };
93
106
  export type VoiceTranscriptQuality = {
94
107
  averageConfidence?: number;
@@ -101,6 +114,8 @@ export type VoiceTranscriptQuality = {
101
114
  partialTranscriptCount: number;
102
115
  selectedTranscriptCount: number;
103
116
  source: "fallback" | "primary";
117
+ lowestWordConfidence?: number;
118
+ wordConfidenceSampleCount?: number;
104
119
  };
105
120
  export type VoiceTurnCorrectionDiagnostics = {
106
121
  attempted: boolean;
@@ -118,7 +133,7 @@ export type VoiceTurnCostEstimate = {
118
133
  primaryAudioMs: number;
119
134
  totalBillableAudioMs: number;
120
135
  };
121
- export type VoiceFallbackSelectionReason = "fallback-empty" | "primary-empty" | "word-count-margin" | "confidence-margin" | "word-count-tiebreak" | "kept-primary";
136
+ export type VoiceFallbackSelectionReason = "fallback-empty" | "primary-empty" | "word-count-margin" | "confidence-margin" | "word-count-tiebreak" | "policy-preference" | "kept-primary";
122
137
  export type VoiceFallbackDiagnostics = {
123
138
  attempted: boolean;
124
139
  fallbackConfidence?: number;
@@ -130,6 +145,7 @@ export type VoiceFallbackDiagnostics = {
130
145
  selected: boolean;
131
146
  selectionReason: VoiceFallbackSelectionReason;
132
147
  trigger: "empty-turn" | "low-confidence" | "empty-or-low-confidence" | "always";
148
+ triggerReason?: VoiceFallbackTriggerReason;
133
149
  };
134
150
  export type VoicePartialEvent = {
135
151
  type: "partial";
@@ -417,6 +433,16 @@ export type VoiceSTTLifecycle = "continuous" | "turn-scoped";
417
433
  export type VoiceTurnProfile = "fast" | "balanced" | "long-form";
418
434
  export type VoiceTurnQualityProfile = "general" | "accent-heavy" | "noisy-room" | "short-command";
419
435
  export type VoiceTurnFallbackTrigger = "empty-turn" | "low-confidence" | "empty-or-low-confidence" | "always";
436
+ export type VoiceFallbackTriggerReason = "always" | "empty-turn" | "risk-policy" | "transcript-confidence" | "word-confidence";
437
+ export type VoiceSTTFallbackCandidate = {
438
+ averageConfidence: number;
439
+ lowestWordConfidence?: number;
440
+ text: string;
441
+ transcripts: Transcript[];
442
+ wordCount: number;
443
+ words: TranscriptWord[];
444
+ };
445
+ export type VoiceSTTFallbackRiskPolicy = (candidate: VoiceSTTFallbackCandidate) => boolean;
420
446
  export type VoiceSTTFallbackConfig = {
421
447
  adapter: STTAdapter;
422
448
  trigger?: VoiceTurnFallbackTrigger;
@@ -426,6 +452,15 @@ export type VoiceSTTFallbackConfig = {
426
452
  settleMs?: number;
427
453
  completionTimeoutMs?: number;
428
454
  maxAttemptsPerTurn?: number;
455
+ /** Run the fallback when any provider word falls below this value. */
456
+ wordConfidenceThreshold?: number;
457
+ /** Application-specific routing for high-value turns, entities, or correction
458
+ * language that should receive a second transcription pass regardless of the
459
+ * aggregate confidence. */
460
+ riskPolicy?: VoiceSTTFallbackRiskPolicy;
461
+ /** Prefer a non-empty independent fallback for these audited trigger reasons,
462
+ * including providers that do not expose comparable confidence scores. */
463
+ preferFallbackOn?: VoiceFallbackTriggerReason[];
429
464
  };
430
465
  export type VoiceResolvedSTTFallbackConfig = {
431
466
  adapter: STTAdapter;
@@ -436,6 +471,9 @@ export type VoiceResolvedSTTFallbackConfig = {
436
471
  settleMs: number;
437
472
  completionTimeoutMs: number;
438
473
  maxAttemptsPerTurn: number;
474
+ wordConfidenceThreshold?: number;
475
+ riskPolicy?: VoiceSTTFallbackRiskPolicy;
476
+ preferFallbackOn?: VoiceFallbackTriggerReason[];
439
477
  };
440
478
  export type VoiceTurnDetectionConfig = {
441
479
  profile?: VoiceTurnProfile;
package/dist/index.js CHANGED
@@ -4025,7 +4025,10 @@ var createEmptyCurrentTurn = () => ({
4025
4025
  silenceStartedAt: undefined,
4026
4026
  transcripts: []
4027
4027
  });
4028
- var cloneTranscript = (transcript) => ({ ...transcript });
4028
+ var cloneTranscript = (transcript) => ({
4029
+ ...transcript,
4030
+ words: transcript.words?.map((word) => ({ ...word }))
4031
+ });
4029
4032
  var encodeBase64 = (chunk) => Buffer2.from(chunk).toString("base64");
4030
4033
  var countWords2 = (text) => text.trim().split(/\s+/).filter(Boolean).length;
4031
4034
  var normalizeText2 = (text) => text.trim().replace(/\s+/g, " ");
@@ -4070,9 +4073,12 @@ var calculateMeanConfidence = (transcripts) => {
4070
4073
  }
4071
4074
  return sum / total;
4072
4075
  };
4076
+ var collectTranscriptWords = (transcripts) => transcripts.flatMap((transcript) => transcript.words ?? []);
4077
+ var collectWordConfidences = (transcripts) => collectTranscriptWords(transcripts).map((word) => word.confidence).filter((confidence) => typeof confidence === "number" && Number.isFinite(confidence));
4073
4078
  var createTurnQuality = (transcripts, source, fallbackUsed, fallbackDiagnostics, correctionDiagnostics, costEstimate) => {
4074
4079
  const sampledTranscripts = transcripts.filter((transcript) => typeof transcript.confidence === "number");
4075
4080
  const confidenceSampleCount = sampledTranscripts.length;
4081
+ const wordConfidences = collectWordConfidences(transcripts);
4076
4082
  return {
4077
4083
  averageConfidence: confidenceSampleCount > 0 ? sampledTranscripts.reduce((sum, transcript) => sum + transcript.confidence, 0) / confidenceSampleCount : undefined,
4078
4084
  confidenceSampleCount,
@@ -4083,7 +4089,9 @@ var createTurnQuality = (transcripts, source, fallbackUsed, fallbackDiagnostics,
4083
4089
  finalTranscriptCount: transcripts.filter((transcript) => transcript.isFinal).length,
4084
4090
  partialTranscriptCount: transcripts.filter((transcript) => !transcript.isFinal).length,
4085
4091
  selectedTranscriptCount: transcripts.length,
4086
- source
4092
+ source,
4093
+ lowestWordConfidence: wordConfidences.length > 0 ? Math.min(...wordConfidences) : undefined,
4094
+ wordConfidenceSampleCount: wordConfidences.length
4087
4095
  };
4088
4096
  };
4089
4097
  var createTurnCostEstimate = (input) => {
@@ -4100,20 +4108,38 @@ var createTurnCostEstimate = (input) => {
4100
4108
  };
4101
4109
  };
4102
4110
  var normalizeCorrectionText = (text) => normalizeText2(text);
4103
- var isFallbackNeeded = (candidate, config) => {
4111
+ var evaluateFallbackNeed = (candidate, config) => {
4104
4112
  const trimmed = normalizeText2(candidate.text);
4105
4113
  const wordCount = countWords2(trimmed);
4114
+ const averageConfidence = calculateMeanConfidence(candidate.transcripts);
4115
+ const words = collectTranscriptWords(candidate.transcripts);
4116
+ const wordConfidences = collectWordConfidences(candidate.transcripts);
4117
+ const lowestWordConfidence = wordConfidences.length > 0 ? Math.min(...wordConfidences) : undefined;
4118
+ const resolvedCandidate = {
4119
+ averageConfidence,
4120
+ lowestWordConfidence,
4121
+ text: candidate.text,
4122
+ transcripts: candidate.transcripts,
4123
+ wordCount,
4124
+ words
4125
+ };
4106
4126
  if (config.trigger === "always") {
4107
- return true;
4127
+ return { needed: true, reason: "always" };
4108
4128
  }
4109
- if (config.trigger === "empty-turn") {
4110
- return wordCount < config.minTextLength;
4129
+ if (wordCount < config.minTextLength && (config.trigger === "empty-turn" || config.trigger === "empty-or-low-confidence")) {
4130
+ return { needed: true, reason: "empty-turn" };
4111
4131
  }
4112
- const averageConfidence = calculateMeanConfidence(candidate.transcripts);
4113
- if (config.trigger === "low-confidence") {
4114
- return averageConfidence > 0 && averageConfidence < config.confidenceThreshold;
4132
+ if (config.riskPolicy?.(resolvedCandidate)) {
4133
+ return { needed: true, reason: "risk-policy" };
4134
+ }
4135
+ if (config.wordConfidenceThreshold !== undefined && lowestWordConfidence !== undefined && lowestWordConfidence < config.wordConfidenceThreshold) {
4136
+ return { needed: true, reason: "word-confidence" };
4115
4137
  }
4116
- return averageConfidence > 0 && averageConfidence < config.confidenceThreshold || wordCount < config.minTextLength;
4138
+ const checksConfidence = config.trigger === "low-confidence" || config.trigger === "empty-or-low-confidence";
4139
+ if (checksConfidence && averageConfidence > 0 && averageConfidence < config.confidenceThreshold) {
4140
+ return { needed: true, reason: "transcript-confidence" };
4141
+ }
4142
+ return { needed: false, reason: undefined };
4117
4143
  };
4118
4144
  var selectBetterTurnText = (candidate, fallback) => {
4119
4145
  if (!fallback.text) {
@@ -4232,7 +4258,10 @@ var createVoiceSession = (options) => {
4232
4258
  minTextLength: options.sttFallback.minTextLength ?? DEFAULT_FALLBACK_MIN_TEXT_LENGTH,
4233
4259
  replayWindowMs: options.sttFallback.replayWindowMs ?? DEFAULT_FALLBACK_REPLAY_MS,
4234
4260
  settleMs: options.sttFallback.settleMs ?? DEFAULT_FALLBACK_SETTLE_MS,
4235
- trigger: options.sttFallback.trigger ?? "empty-or-low-confidence"
4261
+ trigger: options.sttFallback.trigger ?? "empty-or-low-confidence",
4262
+ wordConfidenceThreshold: options.sttFallback.wordConfidenceThreshold,
4263
+ riskPolicy: options.sttFallback.riskPolicy,
4264
+ preferFallbackOn: options.sttFallback.preferFallbackOn
4236
4265
  } : undefined;
4237
4266
  const appendTrace = async (input) => {
4238
4267
  await options.trace?.append({
@@ -5379,7 +5408,8 @@ var createVoiceSession = (options) => {
5379
5408
  text: primaryText,
5380
5409
  transcripts: primaryTranscripts
5381
5410
  };
5382
- if (!isFallbackNeeded(candidate, sttFallback)) {
5411
+ const fallbackNeed = evaluateFallbackNeed(candidate, sttFallback);
5412
+ if (!fallbackNeed.needed) {
5383
5413
  return null;
5384
5414
  }
5385
5415
  fallbackAttemptsForCurrentTurn += 1;
@@ -5495,7 +5525,11 @@ var createVoiceSession = (options) => {
5495
5525
  text: primaryText,
5496
5526
  wordCount: countWords2(normalizeText2(primaryText))
5497
5527
  };
5498
- const selection = selectBetterTurnText(primaryCandidate, fallbackCandidate);
5528
+ const policyPrefersFallback = fallbackCandidate.text.length > 0 && fallbackNeed.reason !== undefined && sttFallback.preferFallbackOn?.includes(fallbackNeed.reason);
5529
+ const selection = policyPrefersFallback ? {
5530
+ reason: "policy-preference",
5531
+ winner: fallbackCandidate
5532
+ } : selectBetterTurnText(primaryCandidate, fallbackCandidate);
5499
5533
  const diagnostics = {
5500
5534
  attempted: true,
5501
5535
  fallbackConfidence: fallbackCandidate.confidence,
@@ -5506,7 +5540,8 @@ var createVoiceSession = (options) => {
5506
5540
  primaryWordCount: primaryCandidate.wordCount,
5507
5541
  selected: selection.winner.text === fallbackCandidate.text,
5508
5542
  selectionReason: selection.reason,
5509
- trigger: sttFallback.trigger
5543
+ trigger: sttFallback.trigger,
5544
+ triggerReason: fallbackNeed.reason
5510
5545
  };
5511
5546
  if (selection.winner.text === primaryCandidate.text) {
5512
5547
  return {
@@ -25397,7 +25432,9 @@ var resolveSTTFallbackConfig = (config) => {
25397
25432
  minTextLength: config.minTextLength ?? 2,
25398
25433
  replayWindowMs: config.replayWindowMs ?? 8000,
25399
25434
  settleMs: config.settleMs ?? 220,
25400
- trigger: config.trigger ?? "empty-or-low-confidence"
25435
+ trigger: config.trigger ?? "empty-or-low-confidence",
25436
+ wordConfidenceThreshold: config.wordConfidenceThreshold,
25437
+ riskPolicy: config.riskPolicy
25401
25438
  };
25402
25439
  };
25403
25440
  var normalizePhraseHints = (hints) => (hints ?? []).map((hint) => ({
@@ -6408,7 +6408,10 @@ var createEmptyCurrentTurn = () => ({
6408
6408
  silenceStartedAt: undefined,
6409
6409
  transcripts: []
6410
6410
  });
6411
- var cloneTranscript = (transcript) => ({ ...transcript });
6411
+ var cloneTranscript = (transcript) => ({
6412
+ ...transcript,
6413
+ words: transcript.words?.map((word) => ({ ...word }))
6414
+ });
6412
6415
  var encodeBase64 = (chunk) => Buffer2.from(chunk).toString("base64");
6413
6416
  var countWords2 = (text) => text.trim().split(/\s+/).filter(Boolean).length;
6414
6417
  var normalizeText2 = (text) => text.trim().replace(/\s+/g, " ");
@@ -6453,9 +6456,12 @@ var calculateMeanConfidence = (transcripts) => {
6453
6456
  }
6454
6457
  return sum / total;
6455
6458
  };
6459
+ var collectTranscriptWords = (transcripts) => transcripts.flatMap((transcript) => transcript.words ?? []);
6460
+ var collectWordConfidences = (transcripts) => collectTranscriptWords(transcripts).map((word) => word.confidence).filter((confidence) => typeof confidence === "number" && Number.isFinite(confidence));
6456
6461
  var createTurnQuality = (transcripts, source, fallbackUsed, fallbackDiagnostics, correctionDiagnostics, costEstimate) => {
6457
6462
  const sampledTranscripts = transcripts.filter((transcript) => typeof transcript.confidence === "number");
6458
6463
  const confidenceSampleCount = sampledTranscripts.length;
6464
+ const wordConfidences = collectWordConfidences(transcripts);
6459
6465
  return {
6460
6466
  averageConfidence: confidenceSampleCount > 0 ? sampledTranscripts.reduce((sum, transcript) => sum + transcript.confidence, 0) / confidenceSampleCount : undefined,
6461
6467
  confidenceSampleCount,
@@ -6466,7 +6472,9 @@ var createTurnQuality = (transcripts, source, fallbackUsed, fallbackDiagnostics,
6466
6472
  finalTranscriptCount: transcripts.filter((transcript) => transcript.isFinal).length,
6467
6473
  partialTranscriptCount: transcripts.filter((transcript) => !transcript.isFinal).length,
6468
6474
  selectedTranscriptCount: transcripts.length,
6469
- source
6475
+ source,
6476
+ lowestWordConfidence: wordConfidences.length > 0 ? Math.min(...wordConfidences) : undefined,
6477
+ wordConfidenceSampleCount: wordConfidences.length
6470
6478
  };
6471
6479
  };
6472
6480
  var createTurnCostEstimate = (input) => {
@@ -6483,20 +6491,38 @@ var createTurnCostEstimate = (input) => {
6483
6491
  };
6484
6492
  };
6485
6493
  var normalizeCorrectionText = (text) => normalizeText2(text);
6486
- var isFallbackNeeded = (candidate, config) => {
6494
+ var evaluateFallbackNeed = (candidate, config) => {
6487
6495
  const trimmed = normalizeText2(candidate.text);
6488
6496
  const wordCount = countWords2(trimmed);
6497
+ const averageConfidence = calculateMeanConfidence(candidate.transcripts);
6498
+ const words = collectTranscriptWords(candidate.transcripts);
6499
+ const wordConfidences = collectWordConfidences(candidate.transcripts);
6500
+ const lowestWordConfidence = wordConfidences.length > 0 ? Math.min(...wordConfidences) : undefined;
6501
+ const resolvedCandidate = {
6502
+ averageConfidence,
6503
+ lowestWordConfidence,
6504
+ text: candidate.text,
6505
+ transcripts: candidate.transcripts,
6506
+ wordCount,
6507
+ words
6508
+ };
6489
6509
  if (config.trigger === "always") {
6490
- return true;
6510
+ return { needed: true, reason: "always" };
6491
6511
  }
6492
- if (config.trigger === "empty-turn") {
6493
- return wordCount < config.minTextLength;
6512
+ if (wordCount < config.minTextLength && (config.trigger === "empty-turn" || config.trigger === "empty-or-low-confidence")) {
6513
+ return { needed: true, reason: "empty-turn" };
6494
6514
  }
6495
- const averageConfidence = calculateMeanConfidence(candidate.transcripts);
6496
- if (config.trigger === "low-confidence") {
6497
- return averageConfidence > 0 && averageConfidence < config.confidenceThreshold;
6515
+ if (config.riskPolicy?.(resolvedCandidate)) {
6516
+ return { needed: true, reason: "risk-policy" };
6517
+ }
6518
+ if (config.wordConfidenceThreshold !== undefined && lowestWordConfidence !== undefined && lowestWordConfidence < config.wordConfidenceThreshold) {
6519
+ return { needed: true, reason: "word-confidence" };
6520
+ }
6521
+ const checksConfidence = config.trigger === "low-confidence" || config.trigger === "empty-or-low-confidence";
6522
+ if (checksConfidence && averageConfidence > 0 && averageConfidence < config.confidenceThreshold) {
6523
+ return { needed: true, reason: "transcript-confidence" };
6498
6524
  }
6499
- return averageConfidence > 0 && averageConfidence < config.confidenceThreshold || wordCount < config.minTextLength;
6525
+ return { needed: false, reason: undefined };
6500
6526
  };
6501
6527
  var selectBetterTurnText = (candidate, fallback) => {
6502
6528
  if (!fallback.text) {
@@ -6615,7 +6641,10 @@ var createVoiceSession = (options) => {
6615
6641
  minTextLength: options.sttFallback.minTextLength ?? DEFAULT_FALLBACK_MIN_TEXT_LENGTH,
6616
6642
  replayWindowMs: options.sttFallback.replayWindowMs ?? DEFAULT_FALLBACK_REPLAY_MS,
6617
6643
  settleMs: options.sttFallback.settleMs ?? DEFAULT_FALLBACK_SETTLE_MS,
6618
- trigger: options.sttFallback.trigger ?? "empty-or-low-confidence"
6644
+ trigger: options.sttFallback.trigger ?? "empty-or-low-confidence",
6645
+ wordConfidenceThreshold: options.sttFallback.wordConfidenceThreshold,
6646
+ riskPolicy: options.sttFallback.riskPolicy,
6647
+ preferFallbackOn: options.sttFallback.preferFallbackOn
6619
6648
  } : undefined;
6620
6649
  const appendTrace = async (input) => {
6621
6650
  await options.trace?.append({
@@ -7762,7 +7791,8 @@ var createVoiceSession = (options) => {
7762
7791
  text: primaryText,
7763
7792
  transcripts: primaryTranscripts
7764
7793
  };
7765
- if (!isFallbackNeeded(candidate, sttFallback)) {
7794
+ const fallbackNeed = evaluateFallbackNeed(candidate, sttFallback);
7795
+ if (!fallbackNeed.needed) {
7766
7796
  return null;
7767
7797
  }
7768
7798
  fallbackAttemptsForCurrentTurn += 1;
@@ -7878,7 +7908,11 @@ var createVoiceSession = (options) => {
7878
7908
  text: primaryText,
7879
7909
  wordCount: countWords2(normalizeText2(primaryText))
7880
7910
  };
7881
- const selection = selectBetterTurnText(primaryCandidate, fallbackCandidate);
7911
+ const policyPrefersFallback = fallbackCandidate.text.length > 0 && fallbackNeed.reason !== undefined && sttFallback.preferFallbackOn?.includes(fallbackNeed.reason);
7912
+ const selection = policyPrefersFallback ? {
7913
+ reason: "policy-preference",
7914
+ winner: fallbackCandidate
7915
+ } : selectBetterTurnText(primaryCandidate, fallbackCandidate);
7882
7916
  const diagnostics = {
7883
7917
  attempted: true,
7884
7918
  fallbackConfidence: fallbackCandidate.confidence,
@@ -7889,7 +7923,8 @@ var createVoiceSession = (options) => {
7889
7923
  primaryWordCount: primaryCandidate.wordCount,
7890
7924
  selected: selection.winner.text === fallbackCandidate.text,
7891
7925
  selectionReason: selection.reason,
7892
- trigger: sttFallback.trigger
7926
+ trigger: sttFallback.trigger,
7927
+ triggerReason: fallbackNeed.reason
7893
7928
  };
7894
7929
  if (selection.winner.text === primaryCandidate.text) {
7895
7930
  return {
@@ -10251,7 +10286,9 @@ var resolveBenchmarkFallbackConfig = (config) => {
10251
10286
  minTextLength: config.minTextLength ?? 2,
10252
10287
  replayWindowMs: config.replayWindowMs ?? 8000,
10253
10288
  settleMs: config.settleMs ?? 220,
10254
- trigger: config.trigger ?? "empty-or-low-confidence"
10289
+ trigger: config.trigger ?? "empty-or-low-confidence",
10290
+ wordConfidenceThreshold: config.wordConfidenceThreshold,
10291
+ riskPolicy: config.riskPolicy
10255
10292
  };
10256
10293
  };
10257
10294
  var chunkAudio2 = (audio, bytesPerChunk) => {
@@ -14426,7 +14463,9 @@ var resolveSTTFallbackConfig = (config) => {
14426
14463
  minTextLength: config.minTextLength ?? 2,
14427
14464
  replayWindowMs: config.replayWindowMs ?? 8000,
14428
14465
  settleMs: config.settleMs ?? 220,
14429
- trigger: config.trigger ?? "empty-or-low-confidence"
14466
+ trigger: config.trigger ?? "empty-or-low-confidence",
14467
+ wordConfidenceThreshold: config.wordConfidenceThreshold,
14468
+ riskPolicy: config.riskPolicy
14430
14469
  };
14431
14470
  };
14432
14471
  var normalizePhraseHints = (hints) => (hints ?? []).map((hint) => ({
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@absolutejs/voice",
3
- "version": "0.0.22-beta.633",
3
+ "version": "0.0.22-beta.635",
4
4
  "description": "Voice primitives and Elysia plugin for AbsoluteJS",
5
5
  "repository": {
6
6
  "type": "git",