@absolutejs/voice 0.0.22-beta.636 → 0.0.22-beta.638
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +148 -1
- package/dist/testing/accuracy.d.ts +17 -0
- package/dist/testing/audioMatrix.d.ts +22 -0
- package/dist/testing/benchmark.d.ts +5 -0
- package/dist/testing/conformance.d.ts +11 -0
- package/dist/testing/fixtures.d.ts +1 -0
- package/dist/testing/index.d.ts +5 -0
- package/dist/testing/index.js +321 -85
- package/dist/testing/outcomes.d.ts +12 -0
- package/dist/testing/provenance.d.ts +44 -0
- package/dist/testing/statistics.d.ts +30 -0
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -49422,18 +49422,72 @@ var levenshteinDistance2 = (left, right) => {
|
|
|
49422
49422
|
}
|
|
49423
49423
|
return previous[right.length];
|
|
49424
49424
|
};
|
|
49425
|
+
var alignTranscriptWords = (actualWords, expectedWords) => {
|
|
49426
|
+
const rows = expectedWords.length + 1;
|
|
49427
|
+
const columns = actualWords.length + 1;
|
|
49428
|
+
const costs = Array.from({ length: rows }, () => new Array(columns).fill(0));
|
|
49429
|
+
for (let row2 = 0;row2 < rows; row2 += 1)
|
|
49430
|
+
costs[row2][0] = row2;
|
|
49431
|
+
for (let column2 = 0;column2 < columns; column2 += 1)
|
|
49432
|
+
costs[0][column2] = column2;
|
|
49433
|
+
for (let row2 = 1;row2 < rows; row2 += 1) {
|
|
49434
|
+
for (let column2 = 1;column2 < columns; column2 += 1) {
|
|
49435
|
+
const substitution = costs[row2 - 1][column2 - 1] + (expectedWords[row2 - 1] === actualWords[column2 - 1] ? 0 : 1);
|
|
49436
|
+
costs[row2][column2] = Math.min(substitution, costs[row2 - 1][column2] + 1, costs[row2][column2 - 1] + 1);
|
|
49437
|
+
}
|
|
49438
|
+
}
|
|
49439
|
+
const operations = [];
|
|
49440
|
+
let row = expectedWords.length;
|
|
49441
|
+
let column = actualWords.length;
|
|
49442
|
+
while (row > 0 || column > 0) {
|
|
49443
|
+
const expected = expectedWords[row - 1];
|
|
49444
|
+
const actual = actualWords[column - 1];
|
|
49445
|
+
if (row > 0 && column > 0 && expected === actual && costs[row][column] === costs[row - 1][column - 1]) {
|
|
49446
|
+
operations.push({ actual, expected, type: "correct" });
|
|
49447
|
+
row -= 1;
|
|
49448
|
+
column -= 1;
|
|
49449
|
+
} else if (row > 0 && column > 0 && costs[row][column] === costs[row - 1][column - 1] + 1) {
|
|
49450
|
+
operations.push({ actual, expected, type: "substitution" });
|
|
49451
|
+
row -= 1;
|
|
49452
|
+
column -= 1;
|
|
49453
|
+
} else if (row > 0 && costs[row][column] === costs[row - 1][column] + 1) {
|
|
49454
|
+
operations.push({ expected, type: "deletion" });
|
|
49455
|
+
row -= 1;
|
|
49456
|
+
} else {
|
|
49457
|
+
operations.push({ actual, type: "insertion" });
|
|
49458
|
+
column -= 1;
|
|
49459
|
+
}
|
|
49460
|
+
}
|
|
49461
|
+
operations.reverse();
|
|
49462
|
+
const count = (type) => operations.filter((operation) => operation.type === type).length;
|
|
49463
|
+
const substitutions = count("substitution");
|
|
49464
|
+
const deletions = count("deletion");
|
|
49465
|
+
const insertions = count("insertion");
|
|
49466
|
+
return {
|
|
49467
|
+
correct: count("correct"),
|
|
49468
|
+
deletions,
|
|
49469
|
+
hypothesisWordCount: actualWords.length,
|
|
49470
|
+
insertions,
|
|
49471
|
+
operations,
|
|
49472
|
+
referenceWordCount: expectedWords.length,
|
|
49473
|
+
sentenceError: substitutions + deletions + insertions > 0,
|
|
49474
|
+
substitutions
|
|
49475
|
+
};
|
|
49476
|
+
};
|
|
49425
49477
|
var mergeFinalTranscriptText = (transcripts) => buildTurnText(transcripts.filter((transcript) => transcript.isFinal), "");
|
|
49426
49478
|
var scoreTranscriptAccuracy = (actualText, expectedText, threshold = 0.35) => {
|
|
49427
49479
|
const normalizedActual = normalizeAccuracyText(actualText);
|
|
49428
49480
|
const normalizedExpected = normalizeAccuracyText(expectedText);
|
|
49429
49481
|
const actualWords = normalizedActual ? normalizedActual.split(" ") : [];
|
|
49430
49482
|
const expectedWords = normalizedExpected ? normalizedExpected.split(" ") : [];
|
|
49431
|
-
const
|
|
49483
|
+
const alignment = alignTranscriptWords(actualWords, expectedWords);
|
|
49484
|
+
const wordDistance = alignment.substitutions + alignment.deletions + alignment.insertions;
|
|
49432
49485
|
const charDistance = levenshteinDistance2(Array.from(normalizedActual), Array.from(normalizedExpected));
|
|
49433
49486
|
const wordErrorRate = expectedWords.length > 0 ? wordDistance / expectedWords.length : 0;
|
|
49434
49487
|
const charErrorRate = normalizedExpected.length > 0 ? charDistance / normalizedExpected.length : 0;
|
|
49435
49488
|
return {
|
|
49436
49489
|
actualText: normalizedActual,
|
|
49490
|
+
alignment,
|
|
49437
49491
|
charDistance,
|
|
49438
49492
|
charErrorRate,
|
|
49439
49493
|
expectedText: normalizedExpected,
|
|
@@ -49563,6 +49617,46 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
|
|
|
49563
49617
|
};
|
|
49564
49618
|
};
|
|
49565
49619
|
|
|
49620
|
+
// src/testing/confidenceCalibration.ts
|
|
49621
|
+
var clampConfidence2 = (value) => Math.max(0, Math.min(1, value));
|
|
49622
|
+
var calibrateVoiceConfidence = (samples, binCount = 10) => {
|
|
49623
|
+
const safeBinCount = Math.max(1, Math.round(binCount));
|
|
49624
|
+
const bins = Array.from({ length: safeBinCount }, (_, index) => {
|
|
49625
|
+
const lowerBound = index / safeBinCount;
|
|
49626
|
+
return {
|
|
49627
|
+
accuracy: 0,
|
|
49628
|
+
averageConfidence: 0,
|
|
49629
|
+
count: 0,
|
|
49630
|
+
lowerBound,
|
|
49631
|
+
upperBound: (index + 1) / safeBinCount
|
|
49632
|
+
};
|
|
49633
|
+
});
|
|
49634
|
+
let brierTotal = 0;
|
|
49635
|
+
for (const sample of samples) {
|
|
49636
|
+
const confidence = clampConfidence2(sample.confidence);
|
|
49637
|
+
const binIndex = Math.min(safeBinCount - 1, Math.floor(confidence * safeBinCount));
|
|
49638
|
+
const bin = bins[binIndex];
|
|
49639
|
+
bin.count += 1;
|
|
49640
|
+
bin.averageConfidence += confidence;
|
|
49641
|
+
bin.accuracy += sample.correct ? 1 : 0;
|
|
49642
|
+
brierTotal += (confidence - (sample.correct ? 1 : 0)) ** 2;
|
|
49643
|
+
}
|
|
49644
|
+
let expectedCalibrationError = 0;
|
|
49645
|
+
for (const bin of bins) {
|
|
49646
|
+
if (bin.count === 0)
|
|
49647
|
+
continue;
|
|
49648
|
+
bin.averageConfidence /= bin.count;
|
|
49649
|
+
bin.accuracy /= bin.count;
|
|
49650
|
+
expectedCalibrationError += bin.count / Math.max(1, samples.length) * Math.abs(bin.accuracy - bin.averageConfidence);
|
|
49651
|
+
}
|
|
49652
|
+
return {
|
|
49653
|
+
bins,
|
|
49654
|
+
brierScore: samples.length > 0 ? brierTotal / samples.length : 0,
|
|
49655
|
+
expectedCalibrationError,
|
|
49656
|
+
sampleCount: samples.length
|
|
49657
|
+
};
|
|
49658
|
+
};
|
|
49659
|
+
|
|
49566
49660
|
// src/testing/criticalFields.ts
|
|
49567
49661
|
var normalizeText4 = (value) => value.normalize("NFKC").toLowerCase().replace(/[^\p{L}\p{N}@.%+$'-]+/gu, " ").replace(/\s+/g, " ").trim();
|
|
49568
49662
|
var normalizeDigits = (value) => normalizeSpokenNumbers(value).replace(/\D/g, "");
|
|
@@ -49601,6 +49695,42 @@ var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
|
|
|
49601
49695
|
};
|
|
49602
49696
|
};
|
|
49603
49697
|
|
|
49698
|
+
// src/testing/conformance.ts
|
|
49699
|
+
var evaluateSTTAdapterConformance = (result) => {
|
|
49700
|
+
const transcripts = [
|
|
49701
|
+
...result.partialEvents.map((event) => event.transcript),
|
|
49702
|
+
...result.finalEvents.map((event) => event.transcript)
|
|
49703
|
+
];
|
|
49704
|
+
const checks = [
|
|
49705
|
+
{
|
|
49706
|
+
detail: "The adapter emitted no error events for a valid fixture.",
|
|
49707
|
+
id: "no-errors",
|
|
49708
|
+
passed: result.errorEvents.length === 0
|
|
49709
|
+
},
|
|
49710
|
+
{
|
|
49711
|
+
detail: "Every transcript has a stable id and finite start time.",
|
|
49712
|
+
id: "transcript-identity",
|
|
49713
|
+
passed: transcripts.every((transcript) => transcript.id.trim().length > 0 && Number.isFinite(transcript.startedAtMs))
|
|
49714
|
+
},
|
|
49715
|
+
{
|
|
49716
|
+
detail: "Events marked final contain final transcripts.",
|
|
49717
|
+
id: "final-semantics",
|
|
49718
|
+
passed: result.finalEvents.every((event) => event.transcript.isFinal)
|
|
49719
|
+
},
|
|
49720
|
+
{
|
|
49721
|
+
detail: "Events marked partial contain non-final transcripts.",
|
|
49722
|
+
id: "partial-semantics",
|
|
49723
|
+
passed: result.partialEvents.every((event) => !event.transcript.isFinal)
|
|
49724
|
+
},
|
|
49725
|
+
{
|
|
49726
|
+
detail: "The assembled transcript is non-empty when finals were emitted.",
|
|
49727
|
+
id: "assembly",
|
|
49728
|
+
passed: result.finalEvents.length === 0 || result.finalText.trim().length > 0
|
|
49729
|
+
}
|
|
49730
|
+
];
|
|
49731
|
+
return { checks, passed: checks.every((check) => check.passed) };
|
|
49732
|
+
};
|
|
49733
|
+
|
|
49604
49734
|
// src/testing/benchmark.ts
|
|
49605
49735
|
var resolveFixtureEnvironment = (fixture) => {
|
|
49606
49736
|
const tags = new Set(fixture.tags ?? []);
|
|
@@ -49833,9 +49963,19 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
49833
49963
|
const expectedTerms = scoreExpectedTerms(result.finalText, fixture.expectedTerms);
|
|
49834
49964
|
const criticalFields = scoreVoiceCriticalFields(result.finalText, fixture.expectedCriticalFields);
|
|
49835
49965
|
const speakerTurns = scoreSpeakerTurns(fixture, result);
|
|
49966
|
+
const transcriptConfidence = average2(result.finalEvents.map((event) => {
|
|
49967
|
+
if (typeof event.transcript.confidence === "number") {
|
|
49968
|
+
return event.transcript.confidence;
|
|
49969
|
+
}
|
|
49970
|
+
return average2([
|
|
49971
|
+
...(event.transcript.words ?? []).map((word) => word.confidence),
|
|
49972
|
+
...(event.transcript.tokens ?? []).map((token) => token.confidence)
|
|
49973
|
+
]);
|
|
49974
|
+
}));
|
|
49836
49975
|
return {
|
|
49837
49976
|
accuracy: result.accuracy,
|
|
49838
49977
|
closeCount: result.closeEvents.length,
|
|
49978
|
+
conformance: evaluateSTTAdapterConformance(result),
|
|
49839
49979
|
criticalFields,
|
|
49840
49980
|
difficulty: fixture.difficulty,
|
|
49841
49981
|
elapsedMs,
|
|
@@ -49856,6 +49996,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
49856
49996
|
timeToEndOfTurnMs,
|
|
49857
49997
|
timeToFirstFinalMs,
|
|
49858
49998
|
timeToFirstPartialMs,
|
|
49999
|
+
transcriptConfidence: roundMetric4(transcriptConfidence),
|
|
49859
50000
|
title: fixture.title
|
|
49860
50001
|
};
|
|
49861
50002
|
};
|
|
@@ -49969,6 +50110,12 @@ var summarizeSTTBenchmark = (adapterId, fixtures) => {
|
|
|
49969
50110
|
const passCount = fixtures.filter((fixture) => fixture.passes).length;
|
|
49970
50111
|
return {
|
|
49971
50112
|
adapterId,
|
|
50113
|
+
confidenceCalibration: calibrateVoiceConfidence(fixtures.flatMap((fixture) => typeof fixture.transcriptConfidence === "number" ? [
|
|
50114
|
+
{
|
|
50115
|
+
confidence: fixture.transcriptConfidence,
|
|
50116
|
+
correct: fixture.passes
|
|
50117
|
+
}
|
|
50118
|
+
] : [])),
|
|
49972
50119
|
averageCharErrorRate: roundMetric4(average2(fixtures.map((fixture) => fixture.accuracy.charErrorRate))) ?? 0,
|
|
49973
50120
|
averageElapsedMs: roundMetric4(average2(fixtures.map((fixture) => fixture.elapsedMs)), 2) ?? 0,
|
|
49974
50121
|
averageEndOfTurnCount: roundMetric4(average2(fixtures.map((fixture) => fixture.endOfTurnCount)), 2) ?? 0,
|
|
@@ -8,6 +8,23 @@ export type VoiceTranscriptAccuracy = {
|
|
|
8
8
|
threshold: number;
|
|
9
9
|
wordDistance: number;
|
|
10
10
|
wordErrorRate: number;
|
|
11
|
+
alignment?: VoiceWordAlignment;
|
|
11
12
|
};
|
|
13
|
+
export type VoiceWordAlignmentOperation = {
|
|
14
|
+
actual?: string;
|
|
15
|
+
expected?: string;
|
|
16
|
+
type: "correct" | "substitution" | "deletion" | "insertion";
|
|
17
|
+
};
|
|
18
|
+
export type VoiceWordAlignment = {
|
|
19
|
+
correct: number;
|
|
20
|
+
deletions: number;
|
|
21
|
+
hypothesisWordCount: number;
|
|
22
|
+
insertions: number;
|
|
23
|
+
operations: VoiceWordAlignmentOperation[];
|
|
24
|
+
referenceWordCount: number;
|
|
25
|
+
sentenceError: boolean;
|
|
26
|
+
substitutions: number;
|
|
27
|
+
};
|
|
28
|
+
export declare const alignTranscriptWords: (actualWords: string[], expectedWords: string[]) => VoiceWordAlignment;
|
|
12
29
|
export declare const mergeFinalTranscriptText: (transcripts: Transcript[]) => string;
|
|
13
30
|
export declare const scoreTranscriptAccuracy: (actualText: string, expectedText: string, threshold?: number) => VoiceTranscriptAccuracy;
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import type { VoiceTestFixture } from "./fixtures";
|
|
2
|
+
export type VoiceAudioCondition = {
|
|
3
|
+
id: string;
|
|
4
|
+
type: "gain";
|
|
5
|
+
gain: number;
|
|
6
|
+
} | {
|
|
7
|
+
id: string;
|
|
8
|
+
type: "clip";
|
|
9
|
+
ceiling: number;
|
|
10
|
+
} | {
|
|
11
|
+
id: string;
|
|
12
|
+
type: "drop-chunks";
|
|
13
|
+
every: number;
|
|
14
|
+
chunkDurationMs: number;
|
|
15
|
+
} | {
|
|
16
|
+
id: string;
|
|
17
|
+
type: "noise";
|
|
18
|
+
seed: number;
|
|
19
|
+
snrDb: number;
|
|
20
|
+
};
|
|
21
|
+
export declare const applyVoiceAudioCondition: (fixture: VoiceTestFixture, condition: VoiceAudioCondition) => VoiceTestFixture;
|
|
22
|
+
export declare const buildVoiceAudioMatrix: (fixtures: VoiceTestFixture[], conditions: VoiceAudioCondition[]) => VoiceTestFixture[];
|
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
import type { STTAdapter, STTAdapterOpenOptions } from "../core/types";
|
|
2
2
|
import { type VoiceSTTAdapterHarnessOptions, type VoiceSTTAdapterHarnessResult } from "./stt";
|
|
3
3
|
import type { VoiceTestFixture } from "./fixtures";
|
|
4
|
+
import { type VoiceConfidenceCalibrationReport } from "./confidenceCalibration";
|
|
4
5
|
import { type VoiceCriticalFieldAccuracy } from "./criticalFields";
|
|
6
|
+
import { type VoiceSTTConformanceReport } from "./conformance";
|
|
5
7
|
export type VoiceExpectedTermAccuracy = {
|
|
6
8
|
allMatched: boolean;
|
|
7
9
|
expectedTerms: string[];
|
|
@@ -21,6 +23,7 @@ export type VoiceSpeakerTurnAccuracy = {
|
|
|
21
23
|
export type VoiceSTTBenchmarkFixtureResult = {
|
|
22
24
|
accuracy: VoiceSTTAdapterHarnessResult["accuracy"];
|
|
23
25
|
closeCount: number;
|
|
26
|
+
conformance?: VoiceSTTConformanceReport;
|
|
24
27
|
difficulty?: VoiceTestFixture["difficulty"];
|
|
25
28
|
elapsedMs: number;
|
|
26
29
|
endOfTurnCount: number;
|
|
@@ -41,6 +44,7 @@ export type VoiceSTTBenchmarkFixtureResult = {
|
|
|
41
44
|
timeToEndOfTurnMs?: number;
|
|
42
45
|
timeToFirstFinalMs?: number;
|
|
43
46
|
timeToFirstPartialMs?: number;
|
|
47
|
+
transcriptConfidence?: number;
|
|
44
48
|
title: string;
|
|
45
49
|
};
|
|
46
50
|
export type VoiceSTTBenchmarkSummary = {
|
|
@@ -67,6 +71,7 @@ export type VoiceSTTBenchmarkSummary = {
|
|
|
67
71
|
totalErrorCount: number;
|
|
68
72
|
wordAccuracyRate: number;
|
|
69
73
|
groupSummaries: VoiceSTTBenchmarkFixtureSummary[];
|
|
74
|
+
confidenceCalibration: VoiceConfidenceCalibrationReport;
|
|
70
75
|
};
|
|
71
76
|
export type VoiceSTTBenchmarkFixtureSummary = {
|
|
72
77
|
group: VoiceSTTFixtureEnvironment;
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import type { VoiceSTTAdapterHarnessResult } from "./stt";
|
|
2
|
+
export type VoiceSTTConformanceCheck = {
|
|
3
|
+
detail: string;
|
|
4
|
+
id: string;
|
|
5
|
+
passed: boolean;
|
|
6
|
+
};
|
|
7
|
+
export type VoiceSTTConformanceReport = {
|
|
8
|
+
checks: VoiceSTTConformanceCheck[];
|
|
9
|
+
passed: boolean;
|
|
10
|
+
};
|
|
11
|
+
export declare const evaluateSTTAdapterConformance: (result: VoiceSTTAdapterHarnessResult) => VoiceSTTConformanceReport;
|
|
@@ -15,6 +15,7 @@ export type VoiceTestFixtureManifestEntry = {
|
|
|
15
15
|
tags?: string[];
|
|
16
16
|
tailPaddingMs?: number;
|
|
17
17
|
format?: Partial<AudioFormat>;
|
|
18
|
+
provenance?: Omit<import("./provenance").VoiceCorpusFixtureProvenance, "audioSha256" | "fixtureId">;
|
|
18
19
|
};
|
|
19
20
|
export type VoiceTestFixture = Omit<VoiceTestFixtureManifestEntry, "audioPath"> & {
|
|
20
21
|
audio: Uint8Array;
|
package/dist/testing/index.d.ts
CHANGED
|
@@ -1,16 +1,21 @@
|
|
|
1
1
|
export * from "./accuracy";
|
|
2
|
+
export * from "./audioMatrix";
|
|
2
3
|
export * from "./benchmark";
|
|
3
4
|
export * from "./confidenceCalibration";
|
|
5
|
+
export * from "./conformance";
|
|
4
6
|
export * from "./criticalFields";
|
|
5
7
|
export * from "./corrected";
|
|
6
8
|
export * from "./duplex";
|
|
7
9
|
export * from "./fixtures";
|
|
8
10
|
export * from "./ioProviderSimulator";
|
|
11
|
+
export * from "./outcomes";
|
|
9
12
|
export * from "./providerSimulator";
|
|
13
|
+
export * from "./provenance";
|
|
10
14
|
export * from "./resilience";
|
|
11
15
|
export * from "./review";
|
|
12
16
|
export * from "./routingBenchmark";
|
|
13
17
|
export * from "./sessionBenchmark";
|
|
14
18
|
export * from "./stt";
|
|
19
|
+
export * from "./statistics";
|
|
15
20
|
export * from "./telephony";
|
|
16
21
|
export * from "./tts";
|
package/dist/testing/index.js
CHANGED
|
@@ -227,18 +227,72 @@ var levenshteinDistance = (left, right) => {
|
|
|
227
227
|
}
|
|
228
228
|
return previous[right.length];
|
|
229
229
|
};
|
|
230
|
+
var alignTranscriptWords = (actualWords, expectedWords) => {
|
|
231
|
+
const rows = expectedWords.length + 1;
|
|
232
|
+
const columns = actualWords.length + 1;
|
|
233
|
+
const costs = Array.from({ length: rows }, () => new Array(columns).fill(0));
|
|
234
|
+
for (let row2 = 0;row2 < rows; row2 += 1)
|
|
235
|
+
costs[row2][0] = row2;
|
|
236
|
+
for (let column2 = 0;column2 < columns; column2 += 1)
|
|
237
|
+
costs[0][column2] = column2;
|
|
238
|
+
for (let row2 = 1;row2 < rows; row2 += 1) {
|
|
239
|
+
for (let column2 = 1;column2 < columns; column2 += 1) {
|
|
240
|
+
const substitution = costs[row2 - 1][column2 - 1] + (expectedWords[row2 - 1] === actualWords[column2 - 1] ? 0 : 1);
|
|
241
|
+
costs[row2][column2] = Math.min(substitution, costs[row2 - 1][column2] + 1, costs[row2][column2 - 1] + 1);
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
const operations = [];
|
|
245
|
+
let row = expectedWords.length;
|
|
246
|
+
let column = actualWords.length;
|
|
247
|
+
while (row > 0 || column > 0) {
|
|
248
|
+
const expected = expectedWords[row - 1];
|
|
249
|
+
const actual = actualWords[column - 1];
|
|
250
|
+
if (row > 0 && column > 0 && expected === actual && costs[row][column] === costs[row - 1][column - 1]) {
|
|
251
|
+
operations.push({ actual, expected, type: "correct" });
|
|
252
|
+
row -= 1;
|
|
253
|
+
column -= 1;
|
|
254
|
+
} else if (row > 0 && column > 0 && costs[row][column] === costs[row - 1][column - 1] + 1) {
|
|
255
|
+
operations.push({ actual, expected, type: "substitution" });
|
|
256
|
+
row -= 1;
|
|
257
|
+
column -= 1;
|
|
258
|
+
} else if (row > 0 && costs[row][column] === costs[row - 1][column] + 1) {
|
|
259
|
+
operations.push({ expected, type: "deletion" });
|
|
260
|
+
row -= 1;
|
|
261
|
+
} else {
|
|
262
|
+
operations.push({ actual, type: "insertion" });
|
|
263
|
+
column -= 1;
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
operations.reverse();
|
|
267
|
+
const count = (type) => operations.filter((operation) => operation.type === type).length;
|
|
268
|
+
const substitutions = count("substitution");
|
|
269
|
+
const deletions = count("deletion");
|
|
270
|
+
const insertions = count("insertion");
|
|
271
|
+
return {
|
|
272
|
+
correct: count("correct"),
|
|
273
|
+
deletions,
|
|
274
|
+
hypothesisWordCount: actualWords.length,
|
|
275
|
+
insertions,
|
|
276
|
+
operations,
|
|
277
|
+
referenceWordCount: expectedWords.length,
|
|
278
|
+
sentenceError: substitutions + deletions + insertions > 0,
|
|
279
|
+
substitutions
|
|
280
|
+
};
|
|
281
|
+
};
|
|
230
282
|
var mergeFinalTranscriptText = (transcripts) => buildTurnText(transcripts.filter((transcript) => transcript.isFinal), "");
|
|
231
283
|
var scoreTranscriptAccuracy = (actualText, expectedText, threshold = 0.35) => {
|
|
232
284
|
const normalizedActual = normalizeAccuracyText(actualText);
|
|
233
285
|
const normalizedExpected = normalizeAccuracyText(expectedText);
|
|
234
286
|
const actualWords = normalizedActual ? normalizedActual.split(" ") : [];
|
|
235
287
|
const expectedWords = normalizedExpected ? normalizedExpected.split(" ") : [];
|
|
236
|
-
const
|
|
288
|
+
const alignment = alignTranscriptWords(actualWords, expectedWords);
|
|
289
|
+
const wordDistance = alignment.substitutions + alignment.deletions + alignment.insertions;
|
|
237
290
|
const charDistance = levenshteinDistance(Array.from(normalizedActual), Array.from(normalizedExpected));
|
|
238
291
|
const wordErrorRate = expectedWords.length > 0 ? wordDistance / expectedWords.length : 0;
|
|
239
292
|
const charErrorRate = normalizedExpected.length > 0 ? charDistance / normalizedExpected.length : 0;
|
|
240
293
|
return {
|
|
241
294
|
actualText: normalizedActual,
|
|
295
|
+
alignment,
|
|
242
296
|
charDistance,
|
|
243
297
|
charErrorRate,
|
|
244
298
|
expectedText: normalizedExpected,
|
|
@@ -248,6 +302,35 @@ var scoreTranscriptAccuracy = (actualText, expectedText, threshold = 0.35) => {
|
|
|
248
302
|
wordErrorRate
|
|
249
303
|
};
|
|
250
304
|
};
|
|
305
|
+
// src/testing/audioMatrix.ts
|
|
306
|
+
var clamp = (value) => Math.max(-32768, Math.min(32767, Math.round(value)));
|
|
307
|
+
var samples = (audio) => new Int16Array(audio.buffer.slice(audio.byteOffset, audio.byteOffset + audio.byteLength));
|
|
308
|
+
var bytes = (audio) => new Uint8Array(audio.buffer);
|
|
309
|
+
var applyVoiceAudioCondition = (fixture, condition) => {
|
|
310
|
+
const input = samples(fixture.audio);
|
|
311
|
+
const output = new Int16Array(input);
|
|
312
|
+
if (condition.type === "gain") {
|
|
313
|
+
for (let index = 0;index < output.length; index += 1)
|
|
314
|
+
output[index] = clamp(output[index] * condition.gain);
|
|
315
|
+
} else if (condition.type === "clip") {
|
|
316
|
+
const ceiling = Math.round(32767 * condition.ceiling);
|
|
317
|
+
for (let index = 0;index < output.length; index += 1)
|
|
318
|
+
output[index] = Math.max(-ceiling, Math.min(ceiling, output[index]));
|
|
319
|
+
} else if (condition.type === "drop-chunks") {
|
|
320
|
+
const chunkSamples = Math.max(1, Math.round(fixture.format.sampleRateHz * condition.chunkDurationMs / 1000));
|
|
321
|
+
for (let start = chunkSamples * (condition.every - 1);start < output.length; start += chunkSamples * condition.every)
|
|
322
|
+
output.fill(0, start, Math.min(output.length, start + chunkSamples));
|
|
323
|
+
} else {
|
|
324
|
+
let state = condition.seed >>> 0;
|
|
325
|
+
const random = () => (state = state * 1664525 + 1013904223 >>> 0) / 4294967296 * 2 - 1;
|
|
326
|
+
const signalPower = output.reduce((sum, value) => sum + value * value, 0) / Math.max(1, output.length);
|
|
327
|
+
const noiseRms = Math.sqrt(signalPower / 10 ** (condition.snrDb / 10));
|
|
328
|
+
for (let index = 0;index < output.length; index += 1)
|
|
329
|
+
output[index] = clamp(output[index] + random() * Math.sqrt(3) * noiseRms);
|
|
330
|
+
}
|
|
331
|
+
return { ...fixture, audio: bytes(output), id: `${fixture.id}--${condition.id}`, tags: [...fixture.tags ?? [], "conditioned", condition.id] };
|
|
332
|
+
};
|
|
333
|
+
var buildVoiceAudioMatrix = (fixtures, conditions) => fixtures.flatMap((fixture) => conditions.map((condition) => applyVoiceAudioCondition(fixture, condition)));
|
|
251
334
|
// src/testing/stt.ts
|
|
252
335
|
var chunkAudio = (audio, bytesPerChunk) => {
|
|
253
336
|
const chunks = [];
|
|
@@ -367,6 +450,46 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
|
|
|
367
450
|
};
|
|
368
451
|
};
|
|
369
452
|
|
|
453
|
+
// src/testing/confidenceCalibration.ts
|
|
454
|
+
var clampConfidence = (value) => Math.max(0, Math.min(1, value));
|
|
455
|
+
var calibrateVoiceConfidence = (samples2, binCount = 10) => {
|
|
456
|
+
const safeBinCount = Math.max(1, Math.round(binCount));
|
|
457
|
+
const bins = Array.from({ length: safeBinCount }, (_, index) => {
|
|
458
|
+
const lowerBound = index / safeBinCount;
|
|
459
|
+
return {
|
|
460
|
+
accuracy: 0,
|
|
461
|
+
averageConfidence: 0,
|
|
462
|
+
count: 0,
|
|
463
|
+
lowerBound,
|
|
464
|
+
upperBound: (index + 1) / safeBinCount
|
|
465
|
+
};
|
|
466
|
+
});
|
|
467
|
+
let brierTotal = 0;
|
|
468
|
+
for (const sample of samples2) {
|
|
469
|
+
const confidence = clampConfidence(sample.confidence);
|
|
470
|
+
const binIndex = Math.min(safeBinCount - 1, Math.floor(confidence * safeBinCount));
|
|
471
|
+
const bin = bins[binIndex];
|
|
472
|
+
bin.count += 1;
|
|
473
|
+
bin.averageConfidence += confidence;
|
|
474
|
+
bin.accuracy += sample.correct ? 1 : 0;
|
|
475
|
+
brierTotal += (confidence - (sample.correct ? 1 : 0)) ** 2;
|
|
476
|
+
}
|
|
477
|
+
let expectedCalibrationError = 0;
|
|
478
|
+
for (const bin of bins) {
|
|
479
|
+
if (bin.count === 0)
|
|
480
|
+
continue;
|
|
481
|
+
bin.averageConfidence /= bin.count;
|
|
482
|
+
bin.accuracy /= bin.count;
|
|
483
|
+
expectedCalibrationError += bin.count / Math.max(1, samples2.length) * Math.abs(bin.accuracy - bin.averageConfidence);
|
|
484
|
+
}
|
|
485
|
+
return {
|
|
486
|
+
bins,
|
|
487
|
+
brierScore: samples2.length > 0 ? brierTotal / samples2.length : 0,
|
|
488
|
+
expectedCalibrationError,
|
|
489
|
+
sampleCount: samples2.length
|
|
490
|
+
};
|
|
491
|
+
};
|
|
492
|
+
|
|
370
493
|
// src/core/numberNormalizer.ts
|
|
371
494
|
var ONES = {
|
|
372
495
|
eight: 8,
|
|
@@ -590,6 +713,42 @@ var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
|
|
|
590
713
|
};
|
|
591
714
|
};
|
|
592
715
|
|
|
716
|
+
// src/testing/conformance.ts
|
|
717
|
+
var evaluateSTTAdapterConformance = (result) => {
|
|
718
|
+
const transcripts = [
|
|
719
|
+
...result.partialEvents.map((event) => event.transcript),
|
|
720
|
+
...result.finalEvents.map((event) => event.transcript)
|
|
721
|
+
];
|
|
722
|
+
const checks = [
|
|
723
|
+
{
|
|
724
|
+
detail: "The adapter emitted no error events for a valid fixture.",
|
|
725
|
+
id: "no-errors",
|
|
726
|
+
passed: result.errorEvents.length === 0
|
|
727
|
+
},
|
|
728
|
+
{
|
|
729
|
+
detail: "Every transcript has a stable id and finite start time.",
|
|
730
|
+
id: "transcript-identity",
|
|
731
|
+
passed: transcripts.every((transcript) => transcript.id.trim().length > 0 && Number.isFinite(transcript.startedAtMs))
|
|
732
|
+
},
|
|
733
|
+
{
|
|
734
|
+
detail: "Events marked final contain final transcripts.",
|
|
735
|
+
id: "final-semantics",
|
|
736
|
+
passed: result.finalEvents.every((event) => event.transcript.isFinal)
|
|
737
|
+
},
|
|
738
|
+
{
|
|
739
|
+
detail: "Events marked partial contain non-final transcripts.",
|
|
740
|
+
id: "partial-semantics",
|
|
741
|
+
passed: result.partialEvents.every((event) => !event.transcript.isFinal)
|
|
742
|
+
},
|
|
743
|
+
{
|
|
744
|
+
detail: "The assembled transcript is non-empty when finals were emitted.",
|
|
745
|
+
id: "assembly",
|
|
746
|
+
passed: result.finalEvents.length === 0 || result.finalText.trim().length > 0
|
|
747
|
+
}
|
|
748
|
+
];
|
|
749
|
+
return { checks, passed: checks.every((check) => check.passed) };
|
|
750
|
+
};
|
|
751
|
+
|
|
593
752
|
// src/testing/benchmark.ts
|
|
594
753
|
var resolveFixtureEnvironment = (fixture) => {
|
|
595
754
|
const tags = new Set(fixture.tags ?? []);
|
|
@@ -822,9 +981,19 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
822
981
|
const expectedTerms = scoreExpectedTerms(result.finalText, fixture.expectedTerms);
|
|
823
982
|
const criticalFields = scoreVoiceCriticalFields(result.finalText, fixture.expectedCriticalFields);
|
|
824
983
|
const speakerTurns = scoreSpeakerTurns(fixture, result);
|
|
984
|
+
const transcriptConfidence = average(result.finalEvents.map((event) => {
|
|
985
|
+
if (typeof event.transcript.confidence === "number") {
|
|
986
|
+
return event.transcript.confidence;
|
|
987
|
+
}
|
|
988
|
+
return average([
|
|
989
|
+
...(event.transcript.words ?? []).map((word) => word.confidence),
|
|
990
|
+
...(event.transcript.tokens ?? []).map((token) => token.confidence)
|
|
991
|
+
]);
|
|
992
|
+
}));
|
|
825
993
|
return {
|
|
826
994
|
accuracy: result.accuracy,
|
|
827
995
|
closeCount: result.closeEvents.length,
|
|
996
|
+
conformance: evaluateSTTAdapterConformance(result),
|
|
828
997
|
criticalFields,
|
|
829
998
|
difficulty: fixture.difficulty,
|
|
830
999
|
elapsedMs,
|
|
@@ -845,6 +1014,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
845
1014
|
timeToEndOfTurnMs,
|
|
846
1015
|
timeToFirstFinalMs,
|
|
847
1016
|
timeToFirstPartialMs,
|
|
1017
|
+
transcriptConfidence: roundMetric(transcriptConfidence),
|
|
848
1018
|
title: fixture.title
|
|
849
1019
|
};
|
|
850
1020
|
};
|
|
@@ -958,6 +1128,12 @@ var summarizeSTTBenchmark = (adapterId, fixtures) => {
|
|
|
958
1128
|
const passCount = fixtures.filter((fixture) => fixture.passes).length;
|
|
959
1129
|
return {
|
|
960
1130
|
adapterId,
|
|
1131
|
+
confidenceCalibration: calibrateVoiceConfidence(fixtures.flatMap((fixture) => typeof fixture.transcriptConfidence === "number" ? [
|
|
1132
|
+
{
|
|
1133
|
+
confidence: fixture.transcriptConfidence,
|
|
1134
|
+
correct: fixture.passes
|
|
1135
|
+
}
|
|
1136
|
+
] : [])),
|
|
961
1137
|
averageCharErrorRate: roundMetric(average(fixtures.map((fixture) => fixture.accuracy.charErrorRate))) ?? 0,
|
|
962
1138
|
averageElapsedMs: roundMetric(average(fixtures.map((fixture) => fixture.elapsedMs)), 2) ?? 0,
|
|
963
1139
|
averageEndOfTurnCount: roundMetric(average(fixtures.map((fixture) => fixture.endOfTurnCount)), 2) ?? 0,
|
|
@@ -1030,45 +1206,6 @@ var summarizeSTTBenchmarkSeries = (input) => {
|
|
|
1030
1206
|
}
|
|
1031
1207
|
};
|
|
1032
1208
|
};
|
|
1033
|
-
// src/testing/confidenceCalibration.ts
|
|
1034
|
-
var clampConfidence = (value) => Math.max(0, Math.min(1, value));
|
|
1035
|
-
var calibrateVoiceConfidence = (samples, binCount = 10) => {
|
|
1036
|
-
const safeBinCount = Math.max(1, Math.round(binCount));
|
|
1037
|
-
const bins = Array.from({ length: safeBinCount }, (_, index) => {
|
|
1038
|
-
const lowerBound = index / safeBinCount;
|
|
1039
|
-
return {
|
|
1040
|
-
accuracy: 0,
|
|
1041
|
-
averageConfidence: 0,
|
|
1042
|
-
count: 0,
|
|
1043
|
-
lowerBound,
|
|
1044
|
-
upperBound: (index + 1) / safeBinCount
|
|
1045
|
-
};
|
|
1046
|
-
});
|
|
1047
|
-
let brierTotal = 0;
|
|
1048
|
-
for (const sample of samples) {
|
|
1049
|
-
const confidence = clampConfidence(sample.confidence);
|
|
1050
|
-
const binIndex = Math.min(safeBinCount - 1, Math.floor(confidence * safeBinCount));
|
|
1051
|
-
const bin = bins[binIndex];
|
|
1052
|
-
bin.count += 1;
|
|
1053
|
-
bin.averageConfidence += confidence;
|
|
1054
|
-
bin.accuracy += sample.correct ? 1 : 0;
|
|
1055
|
-
brierTotal += (confidence - (sample.correct ? 1 : 0)) ** 2;
|
|
1056
|
-
}
|
|
1057
|
-
let expectedCalibrationError = 0;
|
|
1058
|
-
for (const bin of bins) {
|
|
1059
|
-
if (bin.count === 0)
|
|
1060
|
-
continue;
|
|
1061
|
-
bin.averageConfidence /= bin.count;
|
|
1062
|
-
bin.accuracy /= bin.count;
|
|
1063
|
-
expectedCalibrationError += bin.count / Math.max(1, samples.length) * Math.abs(bin.accuracy - bin.averageConfidence);
|
|
1064
|
-
}
|
|
1065
|
-
return {
|
|
1066
|
-
bins,
|
|
1067
|
-
brierScore: samples.length > 0 ? brierTotal / samples.length : 0,
|
|
1068
|
-
expectedCalibrationError,
|
|
1069
|
-
sampleCount: samples.length
|
|
1070
|
-
};
|
|
1071
|
-
};
|
|
1072
1209
|
// src/core/correction.ts
|
|
1073
1210
|
var escapeRegExp = (value) => value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
1074
1211
|
var buildAliasMatcher = (alias) => new RegExp(`(?<![\\p{L}\\p{N}'])${escapeRegExp(alias)}(?![\\p{L}\\p{N}'])`, "giu");
|
|
@@ -2040,18 +2177,18 @@ var decodePCM16LEChunk = (audioContext, chunk) => {
|
|
|
2040
2177
|
if (format.container !== "raw" || format.encoding !== "pcm_s16le") {
|
|
2041
2178
|
throw new Error(`Unsupported assistant audio format: ${format.container}/${format.encoding}`);
|
|
2042
2179
|
}
|
|
2043
|
-
const
|
|
2180
|
+
const bytes2 = chunk.chunk;
|
|
2044
2181
|
const channels = Math.max(1, format.channels);
|
|
2045
|
-
const sampleCount = Math.floor(
|
|
2182
|
+
const sampleCount = Math.floor(bytes2.byteLength / 2);
|
|
2046
2183
|
const frameCount = Math.max(1, Math.floor(sampleCount / channels));
|
|
2047
2184
|
const audioBuffer = audioContext.createBuffer(channels, frameCount, format.sampleRateHz);
|
|
2048
|
-
const view = new DataView(
|
|
2185
|
+
const view = new DataView(bytes2.buffer, bytes2.byteOffset, bytes2.byteLength);
|
|
2049
2186
|
for (let channelIndex = 0;channelIndex < channels; channelIndex += 1) {
|
|
2050
2187
|
const channelData = audioBuffer.getChannelData(channelIndex);
|
|
2051
2188
|
for (let frameIndex = 0;frameIndex < frameCount; frameIndex += 1) {
|
|
2052
2189
|
const sampleIndex = frameIndex * channels + channelIndex;
|
|
2053
2190
|
const sampleOffset = sampleIndex * 2;
|
|
2054
|
-
if (sampleOffset + 1 >=
|
|
2191
|
+
if (sampleOffset + 1 >= bytes2.byteLength) {
|
|
2055
2192
|
channelData[frameIndex] = 0;
|
|
2056
2193
|
continue;
|
|
2057
2194
|
}
|
|
@@ -2553,20 +2690,20 @@ var floatTo16BitPCM = (input) => {
|
|
|
2553
2690
|
return new Uint8Array(output.buffer);
|
|
2554
2691
|
};
|
|
2555
2692
|
var getPcmLevel = (audio) => {
|
|
2556
|
-
const
|
|
2557
|
-
if (
|
|
2693
|
+
const bytes2 = audio instanceof Uint8Array ? audio : new Uint8Array(audio);
|
|
2694
|
+
if (bytes2.byteLength < 2) {
|
|
2558
2695
|
return 0;
|
|
2559
2696
|
}
|
|
2560
|
-
const
|
|
2561
|
-
if (
|
|
2697
|
+
const samples2 = new Int16Array(bytes2.buffer, bytes2.byteOffset, Math.floor(bytes2.byteLength / 2));
|
|
2698
|
+
if (samples2.length === 0) {
|
|
2562
2699
|
return 0;
|
|
2563
2700
|
}
|
|
2564
2701
|
let sumSquares = 0;
|
|
2565
|
-
for (const sample of
|
|
2702
|
+
for (const sample of samples2) {
|
|
2566
2703
|
const normalized = sample / 32768;
|
|
2567
2704
|
sumSquares += normalized * normalized;
|
|
2568
2705
|
}
|
|
2569
|
-
return Math.min(1, Math.max(0, Math.sqrt(sumSquares /
|
|
2706
|
+
return Math.min(1, Math.max(0, Math.sqrt(sumSquares / samples2.length) * 5.5));
|
|
2570
2707
|
};
|
|
2571
2708
|
var downsampleBuffer = (input, sourceRate, targetRate) => {
|
|
2572
2709
|
if (sourceRate === targetRate) {
|
|
@@ -3444,16 +3581,16 @@ var toInt16Array = (audio) => {
|
|
|
3444
3581
|
}
|
|
3445
3582
|
return new Int16Array(audio.buffer, audio.byteOffset, Math.floor(audio.byteLength / 2));
|
|
3446
3583
|
};
|
|
3447
|
-
var computeRms = (
|
|
3448
|
-
if (
|
|
3584
|
+
var computeRms = (samples2) => {
|
|
3585
|
+
if (samples2.length === 0) {
|
|
3449
3586
|
return 0;
|
|
3450
3587
|
}
|
|
3451
3588
|
let sumSquares = 0;
|
|
3452
|
-
for (const sample of
|
|
3589
|
+
for (const sample of samples2) {
|
|
3453
3590
|
const normalized = sample / 32768;
|
|
3454
3591
|
sumSquares += normalized * normalized;
|
|
3455
3592
|
}
|
|
3456
|
-
return Math.sqrt(sumSquares /
|
|
3593
|
+
return Math.sqrt(sumSquares / samples2.length);
|
|
3457
3594
|
};
|
|
3458
3595
|
var conditionAudioChunk = (audio, config) => {
|
|
3459
3596
|
if (!config) {
|
|
@@ -4256,7 +4393,7 @@ var resolveVoiceFixtureDirectories = async (input) => {
|
|
|
4256
4393
|
};
|
|
4257
4394
|
var clampSample2 = (value) => Math.max(-32768, Math.min(32767, Math.round(value)));
|
|
4258
4395
|
var toPcm16Samples = (audio) => new Int16Array(audio.buffer.slice(audio.byteOffset, audio.byteOffset + audio.byteLength));
|
|
4259
|
-
var toPcm16Bytes = (
|
|
4396
|
+
var toPcm16Bytes = (samples2) => new Uint8Array(samples2.buffer.slice(samples2.byteOffset, samples2.byteOffset + samples2.byteLength));
|
|
4260
4397
|
var createSilenceBytes = (sampleRateHz, durationMs) => new Uint8Array(Math.max(2, Math.round(sampleRateHz * 2 * durationMs / 1000)));
|
|
4261
4398
|
var concatAudioChunks = (chunks) => {
|
|
4262
4399
|
const totalByteLength = chunks.reduce((sum, chunk) => sum + chunk.byteLength, 0);
|
|
@@ -4268,20 +4405,20 @@ var concatAudioChunks = (chunks) => {
|
|
|
4268
4405
|
}
|
|
4269
4406
|
return output;
|
|
4270
4407
|
};
|
|
4271
|
-
var resamplePcm16Mono = (
|
|
4272
|
-
if (sourceRate === targetRate ||
|
|
4273
|
-
return
|
|
4408
|
+
var resamplePcm16Mono = (samples2, sourceRate, targetRate) => {
|
|
4409
|
+
if (sourceRate === targetRate || samples2.length === 0) {
|
|
4410
|
+
return samples2;
|
|
4274
4411
|
}
|
|
4275
4412
|
const ratio = targetRate / sourceRate;
|
|
4276
|
-
const targetLength = Math.max(1, Math.round(
|
|
4413
|
+
const targetLength = Math.max(1, Math.round(samples2.length * ratio));
|
|
4277
4414
|
const output = new Int16Array(targetLength);
|
|
4278
4415
|
for (let index = 0;index < targetLength; index += 1) {
|
|
4279
4416
|
const sourceIndex = index / ratio;
|
|
4280
4417
|
const previousIndex = Math.floor(sourceIndex);
|
|
4281
|
-
const nextIndex = Math.min(previousIndex + 1,
|
|
4418
|
+
const nextIndex = Math.min(previousIndex + 1, samples2.length - 1);
|
|
4282
4419
|
const fraction = sourceIndex - previousIndex;
|
|
4283
|
-
const previous =
|
|
4284
|
-
const next =
|
|
4420
|
+
const previous = samples2[previousIndex] ?? 0;
|
|
4421
|
+
const next = samples2[nextIndex] ?? previous;
|
|
4285
4422
|
output[index] = clampSample2(previous + (next - previous) * fraction);
|
|
4286
4423
|
}
|
|
4287
4424
|
return output;
|
|
@@ -4559,6 +4696,21 @@ var createVoiceIOProviderFailureSimulator = (options) => {
|
|
|
4559
4696
|
run
|
|
4560
4697
|
};
|
|
4561
4698
|
};
|
|
4699
|
+
// src/testing/outcomes.ts
|
|
4700
|
+
var summarizeVoiceBenchmarkOutcomes = (fixtures, costs) => {
|
|
4701
|
+
const critical = fixtures.map((fixture) => fixture.criticalFields).filter((value) => value !== undefined);
|
|
4702
|
+
const passingFixtureCount = fixtures.filter((fixture) => fixture.passes).length;
|
|
4703
|
+
const requiredFields = critical.flatMap((value) => value.fields.filter((field) => field.required));
|
|
4704
|
+
const completeRequiredProfileRate = critical.length > 0 ? critical.filter((value) => value.passesRequired).length / critical.length : 1;
|
|
4705
|
+
return {
|
|
4706
|
+
completeRequiredProfileRate,
|
|
4707
|
+
costPerPassingFixture: costs && passingFixtureCount > 0 ? costs.total / passingFixtureCount : undefined,
|
|
4708
|
+
fixtureCount: fixtures.length,
|
|
4709
|
+
passingFixtureCount,
|
|
4710
|
+
requiredFieldAccuracy: requiredFields.length > 0 ? requiredFields.filter((field) => field.matched).length / requiredFields.length : 1,
|
|
4711
|
+
totalCost: costs?.total
|
|
4712
|
+
};
|
|
4713
|
+
};
|
|
4562
4714
|
// src/core/debugTiming.ts
|
|
4563
4715
|
var timingEnabled = () => process.env.ABSOLUTEJS_VOICE_TIMING === "1" || process.env.ABSOLUTEJS_VOICE_TIMING === "true";
|
|
4564
4716
|
var emitTiming = (sessionId, stage, elapsedMs, detail) => {
|
|
@@ -5840,6 +5992,22 @@ var createVoiceProviderFailureSimulator = (options) => {
|
|
|
5840
5992
|
run
|
|
5841
5993
|
};
|
|
5842
5994
|
};
|
|
5995
|
+
// src/testing/provenance.ts
|
|
5996
|
+
import { createHash } from "crypto";
|
|
5997
|
+
var sha256Bytes = (value) => createHash("sha256").update(value).digest("hex");
|
|
5998
|
+
var stableBenchmarkJson = (value) => {
|
|
5999
|
+
if (Array.isArray(value))
|
|
6000
|
+
return `[${value.map(stableBenchmarkJson).join(",")}]`;
|
|
6001
|
+
if (value && typeof value === "object") {
|
|
6002
|
+
return `{${Object.entries(value).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableBenchmarkJson(entry)}`).join(",")}}`;
|
|
6003
|
+
}
|
|
6004
|
+
return JSON.stringify(value);
|
|
6005
|
+
};
|
|
6006
|
+
var buildVoiceBenchmarkArtifact = (manifest, report) => {
|
|
6007
|
+
const payload = { manifest, report };
|
|
6008
|
+
return { ...payload, artifactSha256: sha256Bytes(stableBenchmarkJson(payload)) };
|
|
6009
|
+
};
|
|
6010
|
+
var verifyVoiceBenchmarkArtifact = (artifact) => artifact.artifactSha256 === sha256Bytes(stableBenchmarkJson({ manifest: artifact.manifest, report: artifact.report }));
|
|
5843
6011
|
// src/core/memoryStore.ts
|
|
5844
6012
|
var createVoiceMemoryStore = () => {
|
|
5845
6013
|
const sessions = new Map;
|
|
@@ -6025,7 +6193,7 @@ var createVoiceBackchannelDriver = (options) => {
|
|
|
6025
6193
|
};
|
|
6026
6194
|
|
|
6027
6195
|
// src/core/handoff.ts
|
|
6028
|
-
var toHex = (
|
|
6196
|
+
var toHex = (bytes2) => Array.from(bytes2, (byte) => byte.toString(16).padStart(2, "0")).join("");
|
|
6029
6197
|
var signHandoffBody = async (input) => {
|
|
6030
6198
|
const encoder = new TextEncoder;
|
|
6031
6199
|
const key = await crypto.subtle.importKey("raw", encoder.encode(input.secret), {
|
|
@@ -7024,7 +7192,7 @@ var createVoiceSession = (options) => {
|
|
|
7024
7192
|
};
|
|
7025
7193
|
const recordingFormats = {};
|
|
7026
7194
|
let recordingPersisted = false;
|
|
7027
|
-
const captureRecordingChunk = (channel,
|
|
7195
|
+
const captureRecordingChunk = (channel, bytes2, format) => {
|
|
7028
7196
|
if (!recordingConfig || recordingPersisted) {
|
|
7029
7197
|
return;
|
|
7030
7198
|
}
|
|
@@ -7039,7 +7207,7 @@ var createVoiceSession = (options) => {
|
|
|
7039
7207
|
return;
|
|
7040
7208
|
}
|
|
7041
7209
|
const remaining = recordingMaxBytes - currentTotal;
|
|
7042
|
-
const slice =
|
|
7210
|
+
const slice = bytes2.byteLength <= remaining ? bytes2 : bytes2.subarray(0, remaining);
|
|
7043
7211
|
recordingBuffers[channel].push(new Uint8Array(slice));
|
|
7044
7212
|
recordingByteTotals[channel] += slice.byteLength;
|
|
7045
7213
|
recordingFormats[channel] = format;
|
|
@@ -10803,6 +10971,63 @@ var summarizeVoiceSessionBenchmarkSeries = (input) => {
|
|
|
10803
10971
|
}
|
|
10804
10972
|
};
|
|
10805
10973
|
};
|
|
10974
|
+
// src/testing/statistics.ts
|
|
10975
|
+
var mean = (values) => values.length === 0 ? 0 : values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
10976
|
+
var aggregateTranscriptAccuracy = (values) => {
|
|
10977
|
+
const alignments = values.map((value) => value.alignment).filter(Boolean);
|
|
10978
|
+
const sum = (key) => alignments.reduce((total, alignment) => total + alignment[key], 0);
|
|
10979
|
+
const referenceWordCount = sum("referenceWordCount");
|
|
10980
|
+
const errors = sum("substitutions") + sum("deletions") + sum("insertions");
|
|
10981
|
+
return {
|
|
10982
|
+
correct: sum("correct"),
|
|
10983
|
+
deletions: sum("deletions"),
|
|
10984
|
+
insertions: sum("insertions"),
|
|
10985
|
+
macroWordErrorRate: mean(values.map((value) => value.wordErrorRate)),
|
|
10986
|
+
microWordErrorRate: referenceWordCount > 0 ? errors / referenceWordCount : 0,
|
|
10987
|
+
referenceWordCount,
|
|
10988
|
+
sentenceErrorRate: values.length > 0 ? alignments.filter((alignment) => alignment.sentenceError).length / values.length : 0,
|
|
10989
|
+
substitutions: sum("substitutions")
|
|
10990
|
+
};
|
|
10991
|
+
};
|
|
10992
|
+
var seededRandom = (seed) => {
|
|
10993
|
+
let state = seed >>> 0;
|
|
10994
|
+
return () => {
|
|
10995
|
+
state = state * 1664525 + 1013904223 >>> 0;
|
|
10996
|
+
return state / 4294967296;
|
|
10997
|
+
};
|
|
10998
|
+
};
|
|
10999
|
+
var comparePairedMetrics = (baseline, candidate, options = {}) => {
|
|
11000
|
+
if (baseline.length !== candidate.length || baseline.length === 0) {
|
|
11001
|
+
throw new Error("Paired comparisons require equal, non-empty samples.");
|
|
11002
|
+
}
|
|
11003
|
+
const samples2 = options.samples ?? 1e4;
|
|
11004
|
+
const confidenceLevel = options.confidenceLevel ?? 0.95;
|
|
11005
|
+
const random = seededRandom(options.seed ?? 20260722);
|
|
11006
|
+
const deltas = [];
|
|
11007
|
+
for (let sample = 0;sample < samples2; sample += 1) {
|
|
11008
|
+
const selected = [];
|
|
11009
|
+
for (let index = 0;index < baseline.length; index += 1) {
|
|
11010
|
+
const selectedIndex = Math.floor(random() * baseline.length);
|
|
11011
|
+
selected.push(candidate[selectedIndex] - baseline[selectedIndex]);
|
|
11012
|
+
}
|
|
11013
|
+
deltas.push(mean(selected));
|
|
11014
|
+
}
|
|
11015
|
+
deltas.sort((left, right) => left - right);
|
|
11016
|
+
const alpha = (1 - confidenceLevel) / 2;
|
|
11017
|
+
const percentile = (value) => deltas[Math.min(deltas.length - 1, Math.floor(value * deltas.length))] ?? 0;
|
|
11018
|
+
return {
|
|
11019
|
+
baselineMean: mean(baseline),
|
|
11020
|
+
candidateMean: mean(candidate),
|
|
11021
|
+
delta: mean(candidate) - mean(baseline),
|
|
11022
|
+
deltaConfidenceInterval: {
|
|
11023
|
+
confidenceLevel,
|
|
11024
|
+
high: percentile(1 - alpha),
|
|
11025
|
+
low: percentile(alpha),
|
|
11026
|
+
samples: samples2
|
|
11027
|
+
},
|
|
11028
|
+
probabilityCandidateIsBetter: deltas.filter((delta) => delta < 0).length / deltas.length
|
|
11029
|
+
};
|
|
11030
|
+
};
|
|
10806
11031
|
// src/core/operationsRecord.ts
|
|
10807
11032
|
import { Elysia as Elysia4 } from "elysia";
|
|
10808
11033
|
import {
|
|
@@ -11165,7 +11390,7 @@ var sleep2 = async (delayMs) => {
|
|
|
11165
11390
|
}
|
|
11166
11391
|
await new Promise((resolve2) => setTimeout(resolve2, delayMs));
|
|
11167
11392
|
};
|
|
11168
|
-
var toHex2 = (
|
|
11393
|
+
var toHex2 = (bytes2) => Array.from(bytes2, (byte) => byte.toString(16).padStart(2, "0")).join("");
|
|
11169
11394
|
var signVoiceTraceSinkBody = async (input) => {
|
|
11170
11395
|
const encoder = new TextEncoder;
|
|
11171
11396
|
const key = await crypto.subtle.importKey("raw", encoder.encode(input.secret), {
|
|
@@ -13782,7 +14007,7 @@ var flattenPayload = (value) => {
|
|
|
13782
14007
|
...isRecord(data?.payload) ? data.payload : undefined
|
|
13783
14008
|
};
|
|
13784
14009
|
};
|
|
13785
|
-
var toBase64 = (
|
|
14010
|
+
var toBase64 = (bytes2) => Buffer.from(new Uint8Array(bytes2)).toString("base64");
|
|
13786
14011
|
var timingSafeEqual = (left, right) => {
|
|
13787
14012
|
const encoder = new TextEncoder;
|
|
13788
14013
|
const leftBytes = encoder.encode(left);
|
|
@@ -14630,37 +14855,37 @@ var decodeMulawSample = (value) => {
|
|
|
14630
14855
|
sample -= MULAW_BIAS;
|
|
14631
14856
|
return sign ? -sample : sample;
|
|
14632
14857
|
};
|
|
14633
|
-
var int16ArrayToBytes = (
|
|
14634
|
-
const output = new Uint8Array(
|
|
14858
|
+
var int16ArrayToBytes = (samples2) => {
|
|
14859
|
+
const output = new Uint8Array(samples2.length * 2);
|
|
14635
14860
|
const view = new DataView(output.buffer);
|
|
14636
|
-
for (let index = 0;index <
|
|
14637
|
-
view.setInt16(index * 2,
|
|
14861
|
+
for (let index = 0;index < samples2.length; index += 1) {
|
|
14862
|
+
view.setInt16(index * 2, samples2[index] ?? 0, true);
|
|
14638
14863
|
}
|
|
14639
14864
|
return output;
|
|
14640
14865
|
};
|
|
14641
|
-
var bytesToInt16Array = (
|
|
14642
|
-
const sampleCount = Math.floor(
|
|
14866
|
+
var bytesToInt16Array = (bytes2) => {
|
|
14867
|
+
const sampleCount = Math.floor(bytes2.byteLength / 2);
|
|
14643
14868
|
const output = new Int16Array(sampleCount);
|
|
14644
|
-
const view = new DataView(
|
|
14869
|
+
const view = new DataView(bytes2.buffer, bytes2.byteOffset, bytes2.byteLength);
|
|
14645
14870
|
for (let index = 0;index < sampleCount; index += 1) {
|
|
14646
14871
|
output[index] = view.getInt16(index * 2, true);
|
|
14647
14872
|
}
|
|
14648
14873
|
return output;
|
|
14649
14874
|
};
|
|
14650
14875
|
var decodeTwilioMulawBase64 = (payload) => {
|
|
14651
|
-
const
|
|
14652
|
-
const
|
|
14653
|
-
for (let index = 0;index <
|
|
14654
|
-
|
|
14876
|
+
const bytes2 = Uint8Array.from(Buffer3.from(payload, "base64"));
|
|
14877
|
+
const samples2 = new Int16Array(bytes2.length);
|
|
14878
|
+
for (let index = 0;index < bytes2.length; index += 1) {
|
|
14879
|
+
samples2[index] = decodeMulawSample(bytes2[index] ?? 0);
|
|
14655
14880
|
}
|
|
14656
|
-
return
|
|
14881
|
+
return samples2;
|
|
14657
14882
|
};
|
|
14658
|
-
var encodeTwilioMulawBase64 = (
|
|
14659
|
-
const
|
|
14660
|
-
for (let index = 0;index <
|
|
14661
|
-
|
|
14883
|
+
var encodeTwilioMulawBase64 = (samples2) => {
|
|
14884
|
+
const bytes2 = new Uint8Array(samples2.length);
|
|
14885
|
+
for (let index = 0;index < samples2.length; index += 1) {
|
|
14886
|
+
bytes2[index] = encodeMulawSample(samples2[index] ?? 0);
|
|
14662
14887
|
}
|
|
14663
|
-
return Buffer3.from(
|
|
14888
|
+
return Buffer3.from(bytes2).toString("base64");
|
|
14664
14889
|
};
|
|
14665
14890
|
var transcodePCMToTwilioOutboundPayload = (chunk, format) => {
|
|
14666
14891
|
if (format.container === "raw" && format.encoding === "mulaw" && format.channels === 1 && format.sampleRateHz === TWILIO_MULAW_SAMPLE_RATE) {
|
|
@@ -15783,13 +16008,17 @@ var summarizeTTSBenchmark = (adapterId, fixtures) => {
|
|
|
15783
16008
|
};
|
|
15784
16009
|
export {
|
|
15785
16010
|
withVoiceCallReviewId,
|
|
16011
|
+
verifyVoiceBenchmarkArtifact,
|
|
15786
16012
|
summarizeVoiceTelephonyBenchmark,
|
|
15787
16013
|
summarizeVoiceSessionBenchmarkSeries,
|
|
15788
16014
|
summarizeVoiceSessionBenchmark,
|
|
15789
16015
|
summarizeVoiceDuplexBenchmark,
|
|
16016
|
+
summarizeVoiceBenchmarkOutcomes,
|
|
15790
16017
|
summarizeTTSBenchmark,
|
|
15791
16018
|
summarizeSTTBenchmarkSeries,
|
|
15792
16019
|
summarizeSTTBenchmark,
|
|
16020
|
+
stableBenchmarkJson,
|
|
16021
|
+
sha256Bytes,
|
|
15793
16022
|
scoreVoiceCriticalFields,
|
|
15794
16023
|
scoreTranscriptAccuracy,
|
|
15795
16024
|
scoreCorrectedExpectedTerms,
|
|
@@ -15820,6 +16049,7 @@ export {
|
|
|
15820
16049
|
getDefaultTTSBenchmarkFixtures,
|
|
15821
16050
|
evaluateVoiceSTTRouting,
|
|
15822
16051
|
evaluateSTTBenchmarkAcceptance,
|
|
16052
|
+
evaluateSTTAdapterConformance,
|
|
15823
16053
|
createVoiceProviderFailureSimulator,
|
|
15824
16054
|
createVoiceIOProviderFailureSimulator,
|
|
15825
16055
|
createVoiceCallReviewRecorder,
|
|
@@ -15830,13 +16060,19 @@ export {
|
|
|
15830
16060
|
createCodeSwitchBenchmarkCorrectionHandler,
|
|
15831
16061
|
createBenchmarkCorrectionHandler,
|
|
15832
16062
|
compareSTTBenchmarks,
|
|
16063
|
+
comparePairedMetrics,
|
|
15833
16064
|
calibrateVoiceConfidence,
|
|
16065
|
+
buildVoiceBenchmarkArtifact,
|
|
16066
|
+
buildVoiceAudioMatrix,
|
|
15834
16067
|
buildSessionCorrectionAudit,
|
|
15835
16068
|
buildFixturePhraseHints,
|
|
15836
16069
|
buildCorrectionBenchmarkAudit,
|
|
15837
16070
|
buildCodeSwitchBenchmarkPhraseHints,
|
|
15838
16071
|
buildCodeSwitchBenchmarkLexicon,
|
|
16072
|
+
applyVoiceAudioCondition,
|
|
15839
16073
|
applyLexiconCorrectedBenchmarkReport,
|
|
15840
16074
|
applyExperimentalBenchmarkReport,
|
|
15841
|
-
applyCorrectedBenchmarkReport
|
|
16075
|
+
applyCorrectedBenchmarkReport,
|
|
16076
|
+
alignTranscriptWords,
|
|
16077
|
+
aggregateTranscriptAccuracy
|
|
15842
16078
|
};
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import type { VoiceSTTBenchmarkFixtureResult } from "./benchmark";
|
|
2
|
+
export type VoiceBenchmarkOutcomeSummary = {
|
|
3
|
+
completeRequiredProfileRate: number;
|
|
4
|
+
costPerPassingFixture?: number;
|
|
5
|
+
fixtureCount: number;
|
|
6
|
+
passingFixtureCount: number;
|
|
7
|
+
requiredFieldAccuracy: number;
|
|
8
|
+
totalCost?: number;
|
|
9
|
+
};
|
|
10
|
+
export declare const summarizeVoiceBenchmarkOutcomes: (fixtures: VoiceSTTBenchmarkFixtureResult[], costs?: {
|
|
11
|
+
total: number;
|
|
12
|
+
}) => VoiceBenchmarkOutcomeSummary;
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
export type VoiceBenchmarkPromptTrack = "unprompted" | "production-context" | "oracle-seeded";
|
|
2
|
+
export type VoiceCorpusSplit = "development" | "public-test" | "private-held-out";
|
|
3
|
+
export type VoiceCorpusFixtureProvenance = {
|
|
4
|
+
audioSha256: string;
|
|
5
|
+
consent?: string;
|
|
6
|
+
fixtureId: string;
|
|
7
|
+
license: string;
|
|
8
|
+
licenseClass: "permissive" | "noncommercial" | "private";
|
|
9
|
+
source: string;
|
|
10
|
+
split: VoiceCorpusSplit;
|
|
11
|
+
};
|
|
12
|
+
export type VoiceBenchmarkRunManifest = {
|
|
13
|
+
adapter: {
|
|
14
|
+
id: string;
|
|
15
|
+
model?: string;
|
|
16
|
+
provider?: string;
|
|
17
|
+
version?: string;
|
|
18
|
+
};
|
|
19
|
+
corpus: {
|
|
20
|
+
fixtures: VoiceCorpusFixtureProvenance[];
|
|
21
|
+
manifestSha256: string;
|
|
22
|
+
name: string;
|
|
23
|
+
version: string;
|
|
24
|
+
};
|
|
25
|
+
createdAt: string;
|
|
26
|
+
environment: Record<string, string | number | boolean>;
|
|
27
|
+
git: Record<string, string>;
|
|
28
|
+
preprocessing: Record<string, unknown>;
|
|
29
|
+
pricing?: Record<string, number>;
|
|
30
|
+
promptTrack: VoiceBenchmarkPromptTrack;
|
|
31
|
+
seed: number;
|
|
32
|
+
};
|
|
33
|
+
export declare const sha256Bytes: (value: Uint8Array | string) => string;
|
|
34
|
+
export declare const stableBenchmarkJson: (value: unknown) => string;
|
|
35
|
+
export declare const buildVoiceBenchmarkArtifact: <T>(manifest: VoiceBenchmarkRunManifest, report: T) => {
|
|
36
|
+
artifactSha256: string;
|
|
37
|
+
manifest: VoiceBenchmarkRunManifest;
|
|
38
|
+
report: T;
|
|
39
|
+
};
|
|
40
|
+
export declare const verifyVoiceBenchmarkArtifact: (artifact: {
|
|
41
|
+
artifactSha256: string;
|
|
42
|
+
manifest: VoiceBenchmarkRunManifest;
|
|
43
|
+
report: unknown;
|
|
44
|
+
}) => boolean;
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import type { VoiceTranscriptAccuracy } from "./accuracy";
|
|
2
|
+
export type VoiceAggregateErrorMetrics = {
|
|
3
|
+
correct: number;
|
|
4
|
+
deletions: number;
|
|
5
|
+
insertions: number;
|
|
6
|
+
macroWordErrorRate: number;
|
|
7
|
+
microWordErrorRate: number;
|
|
8
|
+
referenceWordCount: number;
|
|
9
|
+
sentenceErrorRate: number;
|
|
10
|
+
substitutions: number;
|
|
11
|
+
};
|
|
12
|
+
export type VoiceConfidenceInterval = {
|
|
13
|
+
confidenceLevel: number;
|
|
14
|
+
high: number;
|
|
15
|
+
low: number;
|
|
16
|
+
samples: number;
|
|
17
|
+
};
|
|
18
|
+
export type VoicePairedBootstrapComparison = {
|
|
19
|
+
baselineMean: number;
|
|
20
|
+
candidateMean: number;
|
|
21
|
+
delta: number;
|
|
22
|
+
deltaConfidenceInterval: VoiceConfidenceInterval;
|
|
23
|
+
probabilityCandidateIsBetter: number;
|
|
24
|
+
};
|
|
25
|
+
export declare const aggregateTranscriptAccuracy: (values: VoiceTranscriptAccuracy[]) => VoiceAggregateErrorMetrics;
|
|
26
|
+
export declare const comparePairedMetrics: (baseline: number[], candidate: number[], options?: {
|
|
27
|
+
confidenceLevel?: number;
|
|
28
|
+
samples?: number;
|
|
29
|
+
seed?: number;
|
|
30
|
+
}) => VoicePairedBootstrapComparison;
|