@absolutejs/voice 0.0.22-beta.637 → 0.0.22-beta.638
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +92 -1
- package/dist/testing/accuracy.d.ts +17 -0
- package/dist/testing/audioMatrix.d.ts +22 -0
- package/dist/testing/benchmark.d.ts +2 -0
- package/dist/testing/conformance.d.ts +11 -0
- package/dist/testing/fixtures.d.ts +1 -0
- package/dist/testing/index.d.ts +5 -0
- package/dist/testing/index.js +270 -51
- package/dist/testing/outcomes.d.ts +12 -0
- package/dist/testing/provenance.d.ts +44 -0
- package/dist/testing/statistics.d.ts +30 -0
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -49422,18 +49422,72 @@ var levenshteinDistance2 = (left, right) => {
|
|
|
49422
49422
|
}
|
|
49423
49423
|
return previous[right.length];
|
|
49424
49424
|
};
|
|
49425
|
+
var alignTranscriptWords = (actualWords, expectedWords) => {
|
|
49426
|
+
const rows = expectedWords.length + 1;
|
|
49427
|
+
const columns = actualWords.length + 1;
|
|
49428
|
+
const costs = Array.from({ length: rows }, () => new Array(columns).fill(0));
|
|
49429
|
+
for (let row2 = 0;row2 < rows; row2 += 1)
|
|
49430
|
+
costs[row2][0] = row2;
|
|
49431
|
+
for (let column2 = 0;column2 < columns; column2 += 1)
|
|
49432
|
+
costs[0][column2] = column2;
|
|
49433
|
+
for (let row2 = 1;row2 < rows; row2 += 1) {
|
|
49434
|
+
for (let column2 = 1;column2 < columns; column2 += 1) {
|
|
49435
|
+
const substitution = costs[row2 - 1][column2 - 1] + (expectedWords[row2 - 1] === actualWords[column2 - 1] ? 0 : 1);
|
|
49436
|
+
costs[row2][column2] = Math.min(substitution, costs[row2 - 1][column2] + 1, costs[row2][column2 - 1] + 1);
|
|
49437
|
+
}
|
|
49438
|
+
}
|
|
49439
|
+
const operations = [];
|
|
49440
|
+
let row = expectedWords.length;
|
|
49441
|
+
let column = actualWords.length;
|
|
49442
|
+
while (row > 0 || column > 0) {
|
|
49443
|
+
const expected = expectedWords[row - 1];
|
|
49444
|
+
const actual = actualWords[column - 1];
|
|
49445
|
+
if (row > 0 && column > 0 && expected === actual && costs[row][column] === costs[row - 1][column - 1]) {
|
|
49446
|
+
operations.push({ actual, expected, type: "correct" });
|
|
49447
|
+
row -= 1;
|
|
49448
|
+
column -= 1;
|
|
49449
|
+
} else if (row > 0 && column > 0 && costs[row][column] === costs[row - 1][column - 1] + 1) {
|
|
49450
|
+
operations.push({ actual, expected, type: "substitution" });
|
|
49451
|
+
row -= 1;
|
|
49452
|
+
column -= 1;
|
|
49453
|
+
} else if (row > 0 && costs[row][column] === costs[row - 1][column] + 1) {
|
|
49454
|
+
operations.push({ expected, type: "deletion" });
|
|
49455
|
+
row -= 1;
|
|
49456
|
+
} else {
|
|
49457
|
+
operations.push({ actual, type: "insertion" });
|
|
49458
|
+
column -= 1;
|
|
49459
|
+
}
|
|
49460
|
+
}
|
|
49461
|
+
operations.reverse();
|
|
49462
|
+
const count = (type) => operations.filter((operation) => operation.type === type).length;
|
|
49463
|
+
const substitutions = count("substitution");
|
|
49464
|
+
const deletions = count("deletion");
|
|
49465
|
+
const insertions = count("insertion");
|
|
49466
|
+
return {
|
|
49467
|
+
correct: count("correct"),
|
|
49468
|
+
deletions,
|
|
49469
|
+
hypothesisWordCount: actualWords.length,
|
|
49470
|
+
insertions,
|
|
49471
|
+
operations,
|
|
49472
|
+
referenceWordCount: expectedWords.length,
|
|
49473
|
+
sentenceError: substitutions + deletions + insertions > 0,
|
|
49474
|
+
substitutions
|
|
49475
|
+
};
|
|
49476
|
+
};
|
|
49425
49477
|
var mergeFinalTranscriptText = (transcripts) => buildTurnText(transcripts.filter((transcript) => transcript.isFinal), "");
|
|
49426
49478
|
var scoreTranscriptAccuracy = (actualText, expectedText, threshold = 0.35) => {
|
|
49427
49479
|
const normalizedActual = normalizeAccuracyText(actualText);
|
|
49428
49480
|
const normalizedExpected = normalizeAccuracyText(expectedText);
|
|
49429
49481
|
const actualWords = normalizedActual ? normalizedActual.split(" ") : [];
|
|
49430
49482
|
const expectedWords = normalizedExpected ? normalizedExpected.split(" ") : [];
|
|
49431
|
-
const
|
|
49483
|
+
const alignment = alignTranscriptWords(actualWords, expectedWords);
|
|
49484
|
+
const wordDistance = alignment.substitutions + alignment.deletions + alignment.insertions;
|
|
49432
49485
|
const charDistance = levenshteinDistance2(Array.from(normalizedActual), Array.from(normalizedExpected));
|
|
49433
49486
|
const wordErrorRate = expectedWords.length > 0 ? wordDistance / expectedWords.length : 0;
|
|
49434
49487
|
const charErrorRate = normalizedExpected.length > 0 ? charDistance / normalizedExpected.length : 0;
|
|
49435
49488
|
return {
|
|
49436
49489
|
actualText: normalizedActual,
|
|
49490
|
+
alignment,
|
|
49437
49491
|
charDistance,
|
|
49438
49492
|
charErrorRate,
|
|
49439
49493
|
expectedText: normalizedExpected,
|
|
@@ -49641,6 +49695,42 @@ var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
|
|
|
49641
49695
|
};
|
|
49642
49696
|
};
|
|
49643
49697
|
|
|
49698
|
+
// src/testing/conformance.ts
|
|
49699
|
+
var evaluateSTTAdapterConformance = (result) => {
|
|
49700
|
+
const transcripts = [
|
|
49701
|
+
...result.partialEvents.map((event) => event.transcript),
|
|
49702
|
+
...result.finalEvents.map((event) => event.transcript)
|
|
49703
|
+
];
|
|
49704
|
+
const checks = [
|
|
49705
|
+
{
|
|
49706
|
+
detail: "The adapter emitted no error events for a valid fixture.",
|
|
49707
|
+
id: "no-errors",
|
|
49708
|
+
passed: result.errorEvents.length === 0
|
|
49709
|
+
},
|
|
49710
|
+
{
|
|
49711
|
+
detail: "Every transcript has a stable id and finite start time.",
|
|
49712
|
+
id: "transcript-identity",
|
|
49713
|
+
passed: transcripts.every((transcript) => transcript.id.trim().length > 0 && Number.isFinite(transcript.startedAtMs))
|
|
49714
|
+
},
|
|
49715
|
+
{
|
|
49716
|
+
detail: "Events marked final contain final transcripts.",
|
|
49717
|
+
id: "final-semantics",
|
|
49718
|
+
passed: result.finalEvents.every((event) => event.transcript.isFinal)
|
|
49719
|
+
},
|
|
49720
|
+
{
|
|
49721
|
+
detail: "Events marked partial contain non-final transcripts.",
|
|
49722
|
+
id: "partial-semantics",
|
|
49723
|
+
passed: result.partialEvents.every((event) => !event.transcript.isFinal)
|
|
49724
|
+
},
|
|
49725
|
+
{
|
|
49726
|
+
detail: "The assembled transcript is non-empty when finals were emitted.",
|
|
49727
|
+
id: "assembly",
|
|
49728
|
+
passed: result.finalEvents.length === 0 || result.finalText.trim().length > 0
|
|
49729
|
+
}
|
|
49730
|
+
];
|
|
49731
|
+
return { checks, passed: checks.every((check) => check.passed) };
|
|
49732
|
+
};
|
|
49733
|
+
|
|
49644
49734
|
// src/testing/benchmark.ts
|
|
49645
49735
|
var resolveFixtureEnvironment = (fixture) => {
|
|
49646
49736
|
const tags = new Set(fixture.tags ?? []);
|
|
@@ -49885,6 +49975,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
49885
49975
|
return {
|
|
49886
49976
|
accuracy: result.accuracy,
|
|
49887
49977
|
closeCount: result.closeEvents.length,
|
|
49978
|
+
conformance: evaluateSTTAdapterConformance(result),
|
|
49888
49979
|
criticalFields,
|
|
49889
49980
|
difficulty: fixture.difficulty,
|
|
49890
49981
|
elapsedMs,
|
|
@@ -8,6 +8,23 @@ export type VoiceTranscriptAccuracy = {
|
|
|
8
8
|
threshold: number;
|
|
9
9
|
wordDistance: number;
|
|
10
10
|
wordErrorRate: number;
|
|
11
|
+
alignment?: VoiceWordAlignment;
|
|
11
12
|
};
|
|
13
|
+
export type VoiceWordAlignmentOperation = {
|
|
14
|
+
actual?: string;
|
|
15
|
+
expected?: string;
|
|
16
|
+
type: "correct" | "substitution" | "deletion" | "insertion";
|
|
17
|
+
};
|
|
18
|
+
export type VoiceWordAlignment = {
|
|
19
|
+
correct: number;
|
|
20
|
+
deletions: number;
|
|
21
|
+
hypothesisWordCount: number;
|
|
22
|
+
insertions: number;
|
|
23
|
+
operations: VoiceWordAlignmentOperation[];
|
|
24
|
+
referenceWordCount: number;
|
|
25
|
+
sentenceError: boolean;
|
|
26
|
+
substitutions: number;
|
|
27
|
+
};
|
|
28
|
+
export declare const alignTranscriptWords: (actualWords: string[], expectedWords: string[]) => VoiceWordAlignment;
|
|
12
29
|
export declare const mergeFinalTranscriptText: (transcripts: Transcript[]) => string;
|
|
13
30
|
export declare const scoreTranscriptAccuracy: (actualText: string, expectedText: string, threshold?: number) => VoiceTranscriptAccuracy;
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import type { VoiceTestFixture } from "./fixtures";
|
|
2
|
+
export type VoiceAudioCondition = {
|
|
3
|
+
id: string;
|
|
4
|
+
type: "gain";
|
|
5
|
+
gain: number;
|
|
6
|
+
} | {
|
|
7
|
+
id: string;
|
|
8
|
+
type: "clip";
|
|
9
|
+
ceiling: number;
|
|
10
|
+
} | {
|
|
11
|
+
id: string;
|
|
12
|
+
type: "drop-chunks";
|
|
13
|
+
every: number;
|
|
14
|
+
chunkDurationMs: number;
|
|
15
|
+
} | {
|
|
16
|
+
id: string;
|
|
17
|
+
type: "noise";
|
|
18
|
+
seed: number;
|
|
19
|
+
snrDb: number;
|
|
20
|
+
};
|
|
21
|
+
export declare const applyVoiceAudioCondition: (fixture: VoiceTestFixture, condition: VoiceAudioCondition) => VoiceTestFixture;
|
|
22
|
+
export declare const buildVoiceAudioMatrix: (fixtures: VoiceTestFixture[], conditions: VoiceAudioCondition[]) => VoiceTestFixture[];
|
|
@@ -3,6 +3,7 @@ import { type VoiceSTTAdapterHarnessOptions, type VoiceSTTAdapterHarnessResult }
|
|
|
3
3
|
import type { VoiceTestFixture } from "./fixtures";
|
|
4
4
|
import { type VoiceConfidenceCalibrationReport } from "./confidenceCalibration";
|
|
5
5
|
import { type VoiceCriticalFieldAccuracy } from "./criticalFields";
|
|
6
|
+
import { type VoiceSTTConformanceReport } from "./conformance";
|
|
6
7
|
export type VoiceExpectedTermAccuracy = {
|
|
7
8
|
allMatched: boolean;
|
|
8
9
|
expectedTerms: string[];
|
|
@@ -22,6 +23,7 @@ export type VoiceSpeakerTurnAccuracy = {
|
|
|
22
23
|
export type VoiceSTTBenchmarkFixtureResult = {
|
|
23
24
|
accuracy: VoiceSTTAdapterHarnessResult["accuracy"];
|
|
24
25
|
closeCount: number;
|
|
26
|
+
conformance?: VoiceSTTConformanceReport;
|
|
25
27
|
difficulty?: VoiceTestFixture["difficulty"];
|
|
26
28
|
elapsedMs: number;
|
|
27
29
|
endOfTurnCount: number;
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import type { VoiceSTTAdapterHarnessResult } from "./stt";
|
|
2
|
+
export type VoiceSTTConformanceCheck = {
|
|
3
|
+
detail: string;
|
|
4
|
+
id: string;
|
|
5
|
+
passed: boolean;
|
|
6
|
+
};
|
|
7
|
+
export type VoiceSTTConformanceReport = {
|
|
8
|
+
checks: VoiceSTTConformanceCheck[];
|
|
9
|
+
passed: boolean;
|
|
10
|
+
};
|
|
11
|
+
export declare const evaluateSTTAdapterConformance: (result: VoiceSTTAdapterHarnessResult) => VoiceSTTConformanceReport;
|
|
@@ -15,6 +15,7 @@ export type VoiceTestFixtureManifestEntry = {
|
|
|
15
15
|
tags?: string[];
|
|
16
16
|
tailPaddingMs?: number;
|
|
17
17
|
format?: Partial<AudioFormat>;
|
|
18
|
+
provenance?: Omit<import("./provenance").VoiceCorpusFixtureProvenance, "audioSha256" | "fixtureId">;
|
|
18
19
|
};
|
|
19
20
|
export type VoiceTestFixture = Omit<VoiceTestFixtureManifestEntry, "audioPath"> & {
|
|
20
21
|
audio: Uint8Array;
|
package/dist/testing/index.d.ts
CHANGED
|
@@ -1,16 +1,21 @@
|
|
|
1
1
|
export * from "./accuracy";
|
|
2
|
+
export * from "./audioMatrix";
|
|
2
3
|
export * from "./benchmark";
|
|
3
4
|
export * from "./confidenceCalibration";
|
|
5
|
+
export * from "./conformance";
|
|
4
6
|
export * from "./criticalFields";
|
|
5
7
|
export * from "./corrected";
|
|
6
8
|
export * from "./duplex";
|
|
7
9
|
export * from "./fixtures";
|
|
8
10
|
export * from "./ioProviderSimulator";
|
|
11
|
+
export * from "./outcomes";
|
|
9
12
|
export * from "./providerSimulator";
|
|
13
|
+
export * from "./provenance";
|
|
10
14
|
export * from "./resilience";
|
|
11
15
|
export * from "./review";
|
|
12
16
|
export * from "./routingBenchmark";
|
|
13
17
|
export * from "./sessionBenchmark";
|
|
14
18
|
export * from "./stt";
|
|
19
|
+
export * from "./statistics";
|
|
15
20
|
export * from "./telephony";
|
|
16
21
|
export * from "./tts";
|
package/dist/testing/index.js
CHANGED
|
@@ -227,18 +227,72 @@ var levenshteinDistance = (left, right) => {
|
|
|
227
227
|
}
|
|
228
228
|
return previous[right.length];
|
|
229
229
|
};
|
|
230
|
+
var alignTranscriptWords = (actualWords, expectedWords) => {
|
|
231
|
+
const rows = expectedWords.length + 1;
|
|
232
|
+
const columns = actualWords.length + 1;
|
|
233
|
+
const costs = Array.from({ length: rows }, () => new Array(columns).fill(0));
|
|
234
|
+
for (let row2 = 0;row2 < rows; row2 += 1)
|
|
235
|
+
costs[row2][0] = row2;
|
|
236
|
+
for (let column2 = 0;column2 < columns; column2 += 1)
|
|
237
|
+
costs[0][column2] = column2;
|
|
238
|
+
for (let row2 = 1;row2 < rows; row2 += 1) {
|
|
239
|
+
for (let column2 = 1;column2 < columns; column2 += 1) {
|
|
240
|
+
const substitution = costs[row2 - 1][column2 - 1] + (expectedWords[row2 - 1] === actualWords[column2 - 1] ? 0 : 1);
|
|
241
|
+
costs[row2][column2] = Math.min(substitution, costs[row2 - 1][column2] + 1, costs[row2][column2 - 1] + 1);
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
const operations = [];
|
|
245
|
+
let row = expectedWords.length;
|
|
246
|
+
let column = actualWords.length;
|
|
247
|
+
while (row > 0 || column > 0) {
|
|
248
|
+
const expected = expectedWords[row - 1];
|
|
249
|
+
const actual = actualWords[column - 1];
|
|
250
|
+
if (row > 0 && column > 0 && expected === actual && costs[row][column] === costs[row - 1][column - 1]) {
|
|
251
|
+
operations.push({ actual, expected, type: "correct" });
|
|
252
|
+
row -= 1;
|
|
253
|
+
column -= 1;
|
|
254
|
+
} else if (row > 0 && column > 0 && costs[row][column] === costs[row - 1][column - 1] + 1) {
|
|
255
|
+
operations.push({ actual, expected, type: "substitution" });
|
|
256
|
+
row -= 1;
|
|
257
|
+
column -= 1;
|
|
258
|
+
} else if (row > 0 && costs[row][column] === costs[row - 1][column] + 1) {
|
|
259
|
+
operations.push({ expected, type: "deletion" });
|
|
260
|
+
row -= 1;
|
|
261
|
+
} else {
|
|
262
|
+
operations.push({ actual, type: "insertion" });
|
|
263
|
+
column -= 1;
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
operations.reverse();
|
|
267
|
+
const count = (type) => operations.filter((operation) => operation.type === type).length;
|
|
268
|
+
const substitutions = count("substitution");
|
|
269
|
+
const deletions = count("deletion");
|
|
270
|
+
const insertions = count("insertion");
|
|
271
|
+
return {
|
|
272
|
+
correct: count("correct"),
|
|
273
|
+
deletions,
|
|
274
|
+
hypothesisWordCount: actualWords.length,
|
|
275
|
+
insertions,
|
|
276
|
+
operations,
|
|
277
|
+
referenceWordCount: expectedWords.length,
|
|
278
|
+
sentenceError: substitutions + deletions + insertions > 0,
|
|
279
|
+
substitutions
|
|
280
|
+
};
|
|
281
|
+
};
|
|
230
282
|
var mergeFinalTranscriptText = (transcripts) => buildTurnText(transcripts.filter((transcript) => transcript.isFinal), "");
|
|
231
283
|
var scoreTranscriptAccuracy = (actualText, expectedText, threshold = 0.35) => {
|
|
232
284
|
const normalizedActual = normalizeAccuracyText(actualText);
|
|
233
285
|
const normalizedExpected = normalizeAccuracyText(expectedText);
|
|
234
286
|
const actualWords = normalizedActual ? normalizedActual.split(" ") : [];
|
|
235
287
|
const expectedWords = normalizedExpected ? normalizedExpected.split(" ") : [];
|
|
236
|
-
const
|
|
288
|
+
const alignment = alignTranscriptWords(actualWords, expectedWords);
|
|
289
|
+
const wordDistance = alignment.substitutions + alignment.deletions + alignment.insertions;
|
|
237
290
|
const charDistance = levenshteinDistance(Array.from(normalizedActual), Array.from(normalizedExpected));
|
|
238
291
|
const wordErrorRate = expectedWords.length > 0 ? wordDistance / expectedWords.length : 0;
|
|
239
292
|
const charErrorRate = normalizedExpected.length > 0 ? charDistance / normalizedExpected.length : 0;
|
|
240
293
|
return {
|
|
241
294
|
actualText: normalizedActual,
|
|
295
|
+
alignment,
|
|
242
296
|
charDistance,
|
|
243
297
|
charErrorRate,
|
|
244
298
|
expectedText: normalizedExpected,
|
|
@@ -248,6 +302,35 @@ var scoreTranscriptAccuracy = (actualText, expectedText, threshold = 0.35) => {
|
|
|
248
302
|
wordErrorRate
|
|
249
303
|
};
|
|
250
304
|
};
|
|
305
|
+
// src/testing/audioMatrix.ts
|
|
306
|
+
var clamp = (value) => Math.max(-32768, Math.min(32767, Math.round(value)));
|
|
307
|
+
var samples = (audio) => new Int16Array(audio.buffer.slice(audio.byteOffset, audio.byteOffset + audio.byteLength));
|
|
308
|
+
var bytes = (audio) => new Uint8Array(audio.buffer);
|
|
309
|
+
var applyVoiceAudioCondition = (fixture, condition) => {
|
|
310
|
+
const input = samples(fixture.audio);
|
|
311
|
+
const output = new Int16Array(input);
|
|
312
|
+
if (condition.type === "gain") {
|
|
313
|
+
for (let index = 0;index < output.length; index += 1)
|
|
314
|
+
output[index] = clamp(output[index] * condition.gain);
|
|
315
|
+
} else if (condition.type === "clip") {
|
|
316
|
+
const ceiling = Math.round(32767 * condition.ceiling);
|
|
317
|
+
for (let index = 0;index < output.length; index += 1)
|
|
318
|
+
output[index] = Math.max(-ceiling, Math.min(ceiling, output[index]));
|
|
319
|
+
} else if (condition.type === "drop-chunks") {
|
|
320
|
+
const chunkSamples = Math.max(1, Math.round(fixture.format.sampleRateHz * condition.chunkDurationMs / 1000));
|
|
321
|
+
for (let start = chunkSamples * (condition.every - 1);start < output.length; start += chunkSamples * condition.every)
|
|
322
|
+
output.fill(0, start, Math.min(output.length, start + chunkSamples));
|
|
323
|
+
} else {
|
|
324
|
+
let state = condition.seed >>> 0;
|
|
325
|
+
const random = () => (state = state * 1664525 + 1013904223 >>> 0) / 4294967296 * 2 - 1;
|
|
326
|
+
const signalPower = output.reduce((sum, value) => sum + value * value, 0) / Math.max(1, output.length);
|
|
327
|
+
const noiseRms = Math.sqrt(signalPower / 10 ** (condition.snrDb / 10));
|
|
328
|
+
for (let index = 0;index < output.length; index += 1)
|
|
329
|
+
output[index] = clamp(output[index] + random() * Math.sqrt(3) * noiseRms);
|
|
330
|
+
}
|
|
331
|
+
return { ...fixture, audio: bytes(output), id: `${fixture.id}--${condition.id}`, tags: [...fixture.tags ?? [], "conditioned", condition.id] };
|
|
332
|
+
};
|
|
333
|
+
var buildVoiceAudioMatrix = (fixtures, conditions) => fixtures.flatMap((fixture) => conditions.map((condition) => applyVoiceAudioCondition(fixture, condition)));
|
|
251
334
|
// src/testing/stt.ts
|
|
252
335
|
var chunkAudio = (audio, bytesPerChunk) => {
|
|
253
336
|
const chunks = [];
|
|
@@ -369,7 +452,7 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
|
|
|
369
452
|
|
|
370
453
|
// src/testing/confidenceCalibration.ts
|
|
371
454
|
var clampConfidence = (value) => Math.max(0, Math.min(1, value));
|
|
372
|
-
var calibrateVoiceConfidence = (
|
|
455
|
+
var calibrateVoiceConfidence = (samples2, binCount = 10) => {
|
|
373
456
|
const safeBinCount = Math.max(1, Math.round(binCount));
|
|
374
457
|
const bins = Array.from({ length: safeBinCount }, (_, index) => {
|
|
375
458
|
const lowerBound = index / safeBinCount;
|
|
@@ -382,7 +465,7 @@ var calibrateVoiceConfidence = (samples, binCount = 10) => {
|
|
|
382
465
|
};
|
|
383
466
|
});
|
|
384
467
|
let brierTotal = 0;
|
|
385
|
-
for (const sample of
|
|
468
|
+
for (const sample of samples2) {
|
|
386
469
|
const confidence = clampConfidence(sample.confidence);
|
|
387
470
|
const binIndex = Math.min(safeBinCount - 1, Math.floor(confidence * safeBinCount));
|
|
388
471
|
const bin = bins[binIndex];
|
|
@@ -397,13 +480,13 @@ var calibrateVoiceConfidence = (samples, binCount = 10) => {
|
|
|
397
480
|
continue;
|
|
398
481
|
bin.averageConfidence /= bin.count;
|
|
399
482
|
bin.accuracy /= bin.count;
|
|
400
|
-
expectedCalibrationError += bin.count / Math.max(1,
|
|
483
|
+
expectedCalibrationError += bin.count / Math.max(1, samples2.length) * Math.abs(bin.accuracy - bin.averageConfidence);
|
|
401
484
|
}
|
|
402
485
|
return {
|
|
403
486
|
bins,
|
|
404
|
-
brierScore:
|
|
487
|
+
brierScore: samples2.length > 0 ? brierTotal / samples2.length : 0,
|
|
405
488
|
expectedCalibrationError,
|
|
406
|
-
sampleCount:
|
|
489
|
+
sampleCount: samples2.length
|
|
407
490
|
};
|
|
408
491
|
};
|
|
409
492
|
|
|
@@ -630,6 +713,42 @@ var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
|
|
|
630
713
|
};
|
|
631
714
|
};
|
|
632
715
|
|
|
716
|
+
// src/testing/conformance.ts
|
|
717
|
+
var evaluateSTTAdapterConformance = (result) => {
|
|
718
|
+
const transcripts = [
|
|
719
|
+
...result.partialEvents.map((event) => event.transcript),
|
|
720
|
+
...result.finalEvents.map((event) => event.transcript)
|
|
721
|
+
];
|
|
722
|
+
const checks = [
|
|
723
|
+
{
|
|
724
|
+
detail: "The adapter emitted no error events for a valid fixture.",
|
|
725
|
+
id: "no-errors",
|
|
726
|
+
passed: result.errorEvents.length === 0
|
|
727
|
+
},
|
|
728
|
+
{
|
|
729
|
+
detail: "Every transcript has a stable id and finite start time.",
|
|
730
|
+
id: "transcript-identity",
|
|
731
|
+
passed: transcripts.every((transcript) => transcript.id.trim().length > 0 && Number.isFinite(transcript.startedAtMs))
|
|
732
|
+
},
|
|
733
|
+
{
|
|
734
|
+
detail: "Events marked final contain final transcripts.",
|
|
735
|
+
id: "final-semantics",
|
|
736
|
+
passed: result.finalEvents.every((event) => event.transcript.isFinal)
|
|
737
|
+
},
|
|
738
|
+
{
|
|
739
|
+
detail: "Events marked partial contain non-final transcripts.",
|
|
740
|
+
id: "partial-semantics",
|
|
741
|
+
passed: result.partialEvents.every((event) => !event.transcript.isFinal)
|
|
742
|
+
},
|
|
743
|
+
{
|
|
744
|
+
detail: "The assembled transcript is non-empty when finals were emitted.",
|
|
745
|
+
id: "assembly",
|
|
746
|
+
passed: result.finalEvents.length === 0 || result.finalText.trim().length > 0
|
|
747
|
+
}
|
|
748
|
+
];
|
|
749
|
+
return { checks, passed: checks.every((check) => check.passed) };
|
|
750
|
+
};
|
|
751
|
+
|
|
633
752
|
// src/testing/benchmark.ts
|
|
634
753
|
var resolveFixtureEnvironment = (fixture) => {
|
|
635
754
|
const tags = new Set(fixture.tags ?? []);
|
|
@@ -874,6 +993,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
874
993
|
return {
|
|
875
994
|
accuracy: result.accuracy,
|
|
876
995
|
closeCount: result.closeEvents.length,
|
|
996
|
+
conformance: evaluateSTTAdapterConformance(result),
|
|
877
997
|
criticalFields,
|
|
878
998
|
difficulty: fixture.difficulty,
|
|
879
999
|
elapsedMs,
|
|
@@ -2057,18 +2177,18 @@ var decodePCM16LEChunk = (audioContext, chunk) => {
|
|
|
2057
2177
|
if (format.container !== "raw" || format.encoding !== "pcm_s16le") {
|
|
2058
2178
|
throw new Error(`Unsupported assistant audio format: ${format.container}/${format.encoding}`);
|
|
2059
2179
|
}
|
|
2060
|
-
const
|
|
2180
|
+
const bytes2 = chunk.chunk;
|
|
2061
2181
|
const channels = Math.max(1, format.channels);
|
|
2062
|
-
const sampleCount = Math.floor(
|
|
2182
|
+
const sampleCount = Math.floor(bytes2.byteLength / 2);
|
|
2063
2183
|
const frameCount = Math.max(1, Math.floor(sampleCount / channels));
|
|
2064
2184
|
const audioBuffer = audioContext.createBuffer(channels, frameCount, format.sampleRateHz);
|
|
2065
|
-
const view = new DataView(
|
|
2185
|
+
const view = new DataView(bytes2.buffer, bytes2.byteOffset, bytes2.byteLength);
|
|
2066
2186
|
for (let channelIndex = 0;channelIndex < channels; channelIndex += 1) {
|
|
2067
2187
|
const channelData = audioBuffer.getChannelData(channelIndex);
|
|
2068
2188
|
for (let frameIndex = 0;frameIndex < frameCount; frameIndex += 1) {
|
|
2069
2189
|
const sampleIndex = frameIndex * channels + channelIndex;
|
|
2070
2190
|
const sampleOffset = sampleIndex * 2;
|
|
2071
|
-
if (sampleOffset + 1 >=
|
|
2191
|
+
if (sampleOffset + 1 >= bytes2.byteLength) {
|
|
2072
2192
|
channelData[frameIndex] = 0;
|
|
2073
2193
|
continue;
|
|
2074
2194
|
}
|
|
@@ -2570,20 +2690,20 @@ var floatTo16BitPCM = (input) => {
|
|
|
2570
2690
|
return new Uint8Array(output.buffer);
|
|
2571
2691
|
};
|
|
2572
2692
|
var getPcmLevel = (audio) => {
|
|
2573
|
-
const
|
|
2574
|
-
if (
|
|
2693
|
+
const bytes2 = audio instanceof Uint8Array ? audio : new Uint8Array(audio);
|
|
2694
|
+
if (bytes2.byteLength < 2) {
|
|
2575
2695
|
return 0;
|
|
2576
2696
|
}
|
|
2577
|
-
const
|
|
2578
|
-
if (
|
|
2697
|
+
const samples2 = new Int16Array(bytes2.buffer, bytes2.byteOffset, Math.floor(bytes2.byteLength / 2));
|
|
2698
|
+
if (samples2.length === 0) {
|
|
2579
2699
|
return 0;
|
|
2580
2700
|
}
|
|
2581
2701
|
let sumSquares = 0;
|
|
2582
|
-
for (const sample of
|
|
2702
|
+
for (const sample of samples2) {
|
|
2583
2703
|
const normalized = sample / 32768;
|
|
2584
2704
|
sumSquares += normalized * normalized;
|
|
2585
2705
|
}
|
|
2586
|
-
return Math.min(1, Math.max(0, Math.sqrt(sumSquares /
|
|
2706
|
+
return Math.min(1, Math.max(0, Math.sqrt(sumSquares / samples2.length) * 5.5));
|
|
2587
2707
|
};
|
|
2588
2708
|
var downsampleBuffer = (input, sourceRate, targetRate) => {
|
|
2589
2709
|
if (sourceRate === targetRate) {
|
|
@@ -3461,16 +3581,16 @@ var toInt16Array = (audio) => {
|
|
|
3461
3581
|
}
|
|
3462
3582
|
return new Int16Array(audio.buffer, audio.byteOffset, Math.floor(audio.byteLength / 2));
|
|
3463
3583
|
};
|
|
3464
|
-
var computeRms = (
|
|
3465
|
-
if (
|
|
3584
|
+
var computeRms = (samples2) => {
|
|
3585
|
+
if (samples2.length === 0) {
|
|
3466
3586
|
return 0;
|
|
3467
3587
|
}
|
|
3468
3588
|
let sumSquares = 0;
|
|
3469
|
-
for (const sample of
|
|
3589
|
+
for (const sample of samples2) {
|
|
3470
3590
|
const normalized = sample / 32768;
|
|
3471
3591
|
sumSquares += normalized * normalized;
|
|
3472
3592
|
}
|
|
3473
|
-
return Math.sqrt(sumSquares /
|
|
3593
|
+
return Math.sqrt(sumSquares / samples2.length);
|
|
3474
3594
|
};
|
|
3475
3595
|
var conditionAudioChunk = (audio, config) => {
|
|
3476
3596
|
if (!config) {
|
|
@@ -4273,7 +4393,7 @@ var resolveVoiceFixtureDirectories = async (input) => {
|
|
|
4273
4393
|
};
|
|
4274
4394
|
var clampSample2 = (value) => Math.max(-32768, Math.min(32767, Math.round(value)));
|
|
4275
4395
|
var toPcm16Samples = (audio) => new Int16Array(audio.buffer.slice(audio.byteOffset, audio.byteOffset + audio.byteLength));
|
|
4276
|
-
var toPcm16Bytes = (
|
|
4396
|
+
var toPcm16Bytes = (samples2) => new Uint8Array(samples2.buffer.slice(samples2.byteOffset, samples2.byteOffset + samples2.byteLength));
|
|
4277
4397
|
var createSilenceBytes = (sampleRateHz, durationMs) => new Uint8Array(Math.max(2, Math.round(sampleRateHz * 2 * durationMs / 1000)));
|
|
4278
4398
|
var concatAudioChunks = (chunks) => {
|
|
4279
4399
|
const totalByteLength = chunks.reduce((sum, chunk) => sum + chunk.byteLength, 0);
|
|
@@ -4285,20 +4405,20 @@ var concatAudioChunks = (chunks) => {
|
|
|
4285
4405
|
}
|
|
4286
4406
|
return output;
|
|
4287
4407
|
};
|
|
4288
|
-
var resamplePcm16Mono = (
|
|
4289
|
-
if (sourceRate === targetRate ||
|
|
4290
|
-
return
|
|
4408
|
+
var resamplePcm16Mono = (samples2, sourceRate, targetRate) => {
|
|
4409
|
+
if (sourceRate === targetRate || samples2.length === 0) {
|
|
4410
|
+
return samples2;
|
|
4291
4411
|
}
|
|
4292
4412
|
const ratio = targetRate / sourceRate;
|
|
4293
|
-
const targetLength = Math.max(1, Math.round(
|
|
4413
|
+
const targetLength = Math.max(1, Math.round(samples2.length * ratio));
|
|
4294
4414
|
const output = new Int16Array(targetLength);
|
|
4295
4415
|
for (let index = 0;index < targetLength; index += 1) {
|
|
4296
4416
|
const sourceIndex = index / ratio;
|
|
4297
4417
|
const previousIndex = Math.floor(sourceIndex);
|
|
4298
|
-
const nextIndex = Math.min(previousIndex + 1,
|
|
4418
|
+
const nextIndex = Math.min(previousIndex + 1, samples2.length - 1);
|
|
4299
4419
|
const fraction = sourceIndex - previousIndex;
|
|
4300
|
-
const previous =
|
|
4301
|
-
const next =
|
|
4420
|
+
const previous = samples2[previousIndex] ?? 0;
|
|
4421
|
+
const next = samples2[nextIndex] ?? previous;
|
|
4302
4422
|
output[index] = clampSample2(previous + (next - previous) * fraction);
|
|
4303
4423
|
}
|
|
4304
4424
|
return output;
|
|
@@ -4576,6 +4696,21 @@ var createVoiceIOProviderFailureSimulator = (options) => {
|
|
|
4576
4696
|
run
|
|
4577
4697
|
};
|
|
4578
4698
|
};
|
|
4699
|
+
// src/testing/outcomes.ts
|
|
4700
|
+
var summarizeVoiceBenchmarkOutcomes = (fixtures, costs) => {
|
|
4701
|
+
const critical = fixtures.map((fixture) => fixture.criticalFields).filter((value) => value !== undefined);
|
|
4702
|
+
const passingFixtureCount = fixtures.filter((fixture) => fixture.passes).length;
|
|
4703
|
+
const requiredFields = critical.flatMap((value) => value.fields.filter((field) => field.required));
|
|
4704
|
+
const completeRequiredProfileRate = critical.length > 0 ? critical.filter((value) => value.passesRequired).length / critical.length : 1;
|
|
4705
|
+
return {
|
|
4706
|
+
completeRequiredProfileRate,
|
|
4707
|
+
costPerPassingFixture: costs && passingFixtureCount > 0 ? costs.total / passingFixtureCount : undefined,
|
|
4708
|
+
fixtureCount: fixtures.length,
|
|
4709
|
+
passingFixtureCount,
|
|
4710
|
+
requiredFieldAccuracy: requiredFields.length > 0 ? requiredFields.filter((field) => field.matched).length / requiredFields.length : 1,
|
|
4711
|
+
totalCost: costs?.total
|
|
4712
|
+
};
|
|
4713
|
+
};
|
|
4579
4714
|
// src/core/debugTiming.ts
|
|
4580
4715
|
var timingEnabled = () => process.env.ABSOLUTEJS_VOICE_TIMING === "1" || process.env.ABSOLUTEJS_VOICE_TIMING === "true";
|
|
4581
4716
|
var emitTiming = (sessionId, stage, elapsedMs, detail) => {
|
|
@@ -5857,6 +5992,22 @@ var createVoiceProviderFailureSimulator = (options) => {
|
|
|
5857
5992
|
run
|
|
5858
5993
|
};
|
|
5859
5994
|
};
|
|
5995
|
+
// src/testing/provenance.ts
|
|
5996
|
+
import { createHash } from "crypto";
|
|
5997
|
+
var sha256Bytes = (value) => createHash("sha256").update(value).digest("hex");
|
|
5998
|
+
var stableBenchmarkJson = (value) => {
|
|
5999
|
+
if (Array.isArray(value))
|
|
6000
|
+
return `[${value.map(stableBenchmarkJson).join(",")}]`;
|
|
6001
|
+
if (value && typeof value === "object") {
|
|
6002
|
+
return `{${Object.entries(value).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableBenchmarkJson(entry)}`).join(",")}}`;
|
|
6003
|
+
}
|
|
6004
|
+
return JSON.stringify(value);
|
|
6005
|
+
};
|
|
6006
|
+
var buildVoiceBenchmarkArtifact = (manifest, report) => {
|
|
6007
|
+
const payload = { manifest, report };
|
|
6008
|
+
return { ...payload, artifactSha256: sha256Bytes(stableBenchmarkJson(payload)) };
|
|
6009
|
+
};
|
|
6010
|
+
var verifyVoiceBenchmarkArtifact = (artifact) => artifact.artifactSha256 === sha256Bytes(stableBenchmarkJson({ manifest: artifact.manifest, report: artifact.report }));
|
|
5860
6011
|
// src/core/memoryStore.ts
|
|
5861
6012
|
var createVoiceMemoryStore = () => {
|
|
5862
6013
|
const sessions = new Map;
|
|
@@ -6042,7 +6193,7 @@ var createVoiceBackchannelDriver = (options) => {
|
|
|
6042
6193
|
};
|
|
6043
6194
|
|
|
6044
6195
|
// src/core/handoff.ts
|
|
6045
|
-
var toHex = (
|
|
6196
|
+
var toHex = (bytes2) => Array.from(bytes2, (byte) => byte.toString(16).padStart(2, "0")).join("");
|
|
6046
6197
|
var signHandoffBody = async (input) => {
|
|
6047
6198
|
const encoder = new TextEncoder;
|
|
6048
6199
|
const key = await crypto.subtle.importKey("raw", encoder.encode(input.secret), {
|
|
@@ -7041,7 +7192,7 @@ var createVoiceSession = (options) => {
|
|
|
7041
7192
|
};
|
|
7042
7193
|
const recordingFormats = {};
|
|
7043
7194
|
let recordingPersisted = false;
|
|
7044
|
-
const captureRecordingChunk = (channel,
|
|
7195
|
+
const captureRecordingChunk = (channel, bytes2, format) => {
|
|
7045
7196
|
if (!recordingConfig || recordingPersisted) {
|
|
7046
7197
|
return;
|
|
7047
7198
|
}
|
|
@@ -7056,7 +7207,7 @@ var createVoiceSession = (options) => {
|
|
|
7056
7207
|
return;
|
|
7057
7208
|
}
|
|
7058
7209
|
const remaining = recordingMaxBytes - currentTotal;
|
|
7059
|
-
const slice =
|
|
7210
|
+
const slice = bytes2.byteLength <= remaining ? bytes2 : bytes2.subarray(0, remaining);
|
|
7060
7211
|
recordingBuffers[channel].push(new Uint8Array(slice));
|
|
7061
7212
|
recordingByteTotals[channel] += slice.byteLength;
|
|
7062
7213
|
recordingFormats[channel] = format;
|
|
@@ -10820,6 +10971,63 @@ var summarizeVoiceSessionBenchmarkSeries = (input) => {
|
|
|
10820
10971
|
}
|
|
10821
10972
|
};
|
|
10822
10973
|
};
|
|
10974
|
+
// src/testing/statistics.ts
|
|
10975
|
+
var mean = (values) => values.length === 0 ? 0 : values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
10976
|
+
var aggregateTranscriptAccuracy = (values) => {
|
|
10977
|
+
const alignments = values.map((value) => value.alignment).filter(Boolean);
|
|
10978
|
+
const sum = (key) => alignments.reduce((total, alignment) => total + alignment[key], 0);
|
|
10979
|
+
const referenceWordCount = sum("referenceWordCount");
|
|
10980
|
+
const errors = sum("substitutions") + sum("deletions") + sum("insertions");
|
|
10981
|
+
return {
|
|
10982
|
+
correct: sum("correct"),
|
|
10983
|
+
deletions: sum("deletions"),
|
|
10984
|
+
insertions: sum("insertions"),
|
|
10985
|
+
macroWordErrorRate: mean(values.map((value) => value.wordErrorRate)),
|
|
10986
|
+
microWordErrorRate: referenceWordCount > 0 ? errors / referenceWordCount : 0,
|
|
10987
|
+
referenceWordCount,
|
|
10988
|
+
sentenceErrorRate: values.length > 0 ? alignments.filter((alignment) => alignment.sentenceError).length / values.length : 0,
|
|
10989
|
+
substitutions: sum("substitutions")
|
|
10990
|
+
};
|
|
10991
|
+
};
|
|
10992
|
+
var seededRandom = (seed) => {
|
|
10993
|
+
let state = seed >>> 0;
|
|
10994
|
+
return () => {
|
|
10995
|
+
state = state * 1664525 + 1013904223 >>> 0;
|
|
10996
|
+
return state / 4294967296;
|
|
10997
|
+
};
|
|
10998
|
+
};
|
|
10999
|
+
var comparePairedMetrics = (baseline, candidate, options = {}) => {
|
|
11000
|
+
if (baseline.length !== candidate.length || baseline.length === 0) {
|
|
11001
|
+
throw new Error("Paired comparisons require equal, non-empty samples.");
|
|
11002
|
+
}
|
|
11003
|
+
const samples2 = options.samples ?? 1e4;
|
|
11004
|
+
const confidenceLevel = options.confidenceLevel ?? 0.95;
|
|
11005
|
+
const random = seededRandom(options.seed ?? 20260722);
|
|
11006
|
+
const deltas = [];
|
|
11007
|
+
for (let sample = 0;sample < samples2; sample += 1) {
|
|
11008
|
+
const selected = [];
|
|
11009
|
+
for (let index = 0;index < baseline.length; index += 1) {
|
|
11010
|
+
const selectedIndex = Math.floor(random() * baseline.length);
|
|
11011
|
+
selected.push(candidate[selectedIndex] - baseline[selectedIndex]);
|
|
11012
|
+
}
|
|
11013
|
+
deltas.push(mean(selected));
|
|
11014
|
+
}
|
|
11015
|
+
deltas.sort((left, right) => left - right);
|
|
11016
|
+
const alpha = (1 - confidenceLevel) / 2;
|
|
11017
|
+
const percentile = (value) => deltas[Math.min(deltas.length - 1, Math.floor(value * deltas.length))] ?? 0;
|
|
11018
|
+
return {
|
|
11019
|
+
baselineMean: mean(baseline),
|
|
11020
|
+
candidateMean: mean(candidate),
|
|
11021
|
+
delta: mean(candidate) - mean(baseline),
|
|
11022
|
+
deltaConfidenceInterval: {
|
|
11023
|
+
confidenceLevel,
|
|
11024
|
+
high: percentile(1 - alpha),
|
|
11025
|
+
low: percentile(alpha),
|
|
11026
|
+
samples: samples2
|
|
11027
|
+
},
|
|
11028
|
+
probabilityCandidateIsBetter: deltas.filter((delta) => delta < 0).length / deltas.length
|
|
11029
|
+
};
|
|
11030
|
+
};
|
|
10823
11031
|
// src/core/operationsRecord.ts
|
|
10824
11032
|
import { Elysia as Elysia4 } from "elysia";
|
|
10825
11033
|
import {
|
|
@@ -11182,7 +11390,7 @@ var sleep2 = async (delayMs) => {
|
|
|
11182
11390
|
}
|
|
11183
11391
|
await new Promise((resolve2) => setTimeout(resolve2, delayMs));
|
|
11184
11392
|
};
|
|
11185
|
-
var toHex2 = (
|
|
11393
|
+
var toHex2 = (bytes2) => Array.from(bytes2, (byte) => byte.toString(16).padStart(2, "0")).join("");
|
|
11186
11394
|
var signVoiceTraceSinkBody = async (input) => {
|
|
11187
11395
|
const encoder = new TextEncoder;
|
|
11188
11396
|
const key = await crypto.subtle.importKey("raw", encoder.encode(input.secret), {
|
|
@@ -13799,7 +14007,7 @@ var flattenPayload = (value) => {
|
|
|
13799
14007
|
...isRecord(data?.payload) ? data.payload : undefined
|
|
13800
14008
|
};
|
|
13801
14009
|
};
|
|
13802
|
-
var toBase64 = (
|
|
14010
|
+
var toBase64 = (bytes2) => Buffer.from(new Uint8Array(bytes2)).toString("base64");
|
|
13803
14011
|
var timingSafeEqual = (left, right) => {
|
|
13804
14012
|
const encoder = new TextEncoder;
|
|
13805
14013
|
const leftBytes = encoder.encode(left);
|
|
@@ -14647,37 +14855,37 @@ var decodeMulawSample = (value) => {
|
|
|
14647
14855
|
sample -= MULAW_BIAS;
|
|
14648
14856
|
return sign ? -sample : sample;
|
|
14649
14857
|
};
|
|
14650
|
-
var int16ArrayToBytes = (
|
|
14651
|
-
const output = new Uint8Array(
|
|
14858
|
+
var int16ArrayToBytes = (samples2) => {
|
|
14859
|
+
const output = new Uint8Array(samples2.length * 2);
|
|
14652
14860
|
const view = new DataView(output.buffer);
|
|
14653
|
-
for (let index = 0;index <
|
|
14654
|
-
view.setInt16(index * 2,
|
|
14861
|
+
for (let index = 0;index < samples2.length; index += 1) {
|
|
14862
|
+
view.setInt16(index * 2, samples2[index] ?? 0, true);
|
|
14655
14863
|
}
|
|
14656
14864
|
return output;
|
|
14657
14865
|
};
|
|
14658
|
-
var bytesToInt16Array = (
|
|
14659
|
-
const sampleCount = Math.floor(
|
|
14866
|
+
var bytesToInt16Array = (bytes2) => {
|
|
14867
|
+
const sampleCount = Math.floor(bytes2.byteLength / 2);
|
|
14660
14868
|
const output = new Int16Array(sampleCount);
|
|
14661
|
-
const view = new DataView(
|
|
14869
|
+
const view = new DataView(bytes2.buffer, bytes2.byteOffset, bytes2.byteLength);
|
|
14662
14870
|
for (let index = 0;index < sampleCount; index += 1) {
|
|
14663
14871
|
output[index] = view.getInt16(index * 2, true);
|
|
14664
14872
|
}
|
|
14665
14873
|
return output;
|
|
14666
14874
|
};
|
|
14667
14875
|
var decodeTwilioMulawBase64 = (payload) => {
|
|
14668
|
-
const
|
|
14669
|
-
const
|
|
14670
|
-
for (let index = 0;index <
|
|
14671
|
-
|
|
14876
|
+
const bytes2 = Uint8Array.from(Buffer3.from(payload, "base64"));
|
|
14877
|
+
const samples2 = new Int16Array(bytes2.length);
|
|
14878
|
+
for (let index = 0;index < bytes2.length; index += 1) {
|
|
14879
|
+
samples2[index] = decodeMulawSample(bytes2[index] ?? 0);
|
|
14672
14880
|
}
|
|
14673
|
-
return
|
|
14881
|
+
return samples2;
|
|
14674
14882
|
};
|
|
14675
|
-
var encodeTwilioMulawBase64 = (
|
|
14676
|
-
const
|
|
14677
|
-
for (let index = 0;index <
|
|
14678
|
-
|
|
14883
|
+
var encodeTwilioMulawBase64 = (samples2) => {
|
|
14884
|
+
const bytes2 = new Uint8Array(samples2.length);
|
|
14885
|
+
for (let index = 0;index < samples2.length; index += 1) {
|
|
14886
|
+
bytes2[index] = encodeMulawSample(samples2[index] ?? 0);
|
|
14679
14887
|
}
|
|
14680
|
-
return Buffer3.from(
|
|
14888
|
+
return Buffer3.from(bytes2).toString("base64");
|
|
14681
14889
|
};
|
|
14682
14890
|
var transcodePCMToTwilioOutboundPayload = (chunk, format) => {
|
|
14683
14891
|
if (format.container === "raw" && format.encoding === "mulaw" && format.channels === 1 && format.sampleRateHz === TWILIO_MULAW_SAMPLE_RATE) {
|
|
@@ -15800,13 +16008,17 @@ var summarizeTTSBenchmark = (adapterId, fixtures) => {
|
|
|
15800
16008
|
};
|
|
15801
16009
|
export {
|
|
15802
16010
|
withVoiceCallReviewId,
|
|
16011
|
+
verifyVoiceBenchmarkArtifact,
|
|
15803
16012
|
summarizeVoiceTelephonyBenchmark,
|
|
15804
16013
|
summarizeVoiceSessionBenchmarkSeries,
|
|
15805
16014
|
summarizeVoiceSessionBenchmark,
|
|
15806
16015
|
summarizeVoiceDuplexBenchmark,
|
|
16016
|
+
summarizeVoiceBenchmarkOutcomes,
|
|
15807
16017
|
summarizeTTSBenchmark,
|
|
15808
16018
|
summarizeSTTBenchmarkSeries,
|
|
15809
16019
|
summarizeSTTBenchmark,
|
|
16020
|
+
stableBenchmarkJson,
|
|
16021
|
+
sha256Bytes,
|
|
15810
16022
|
scoreVoiceCriticalFields,
|
|
15811
16023
|
scoreTranscriptAccuracy,
|
|
15812
16024
|
scoreCorrectedExpectedTerms,
|
|
@@ -15837,6 +16049,7 @@ export {
|
|
|
15837
16049
|
getDefaultTTSBenchmarkFixtures,
|
|
15838
16050
|
evaluateVoiceSTTRouting,
|
|
15839
16051
|
evaluateSTTBenchmarkAcceptance,
|
|
16052
|
+
evaluateSTTAdapterConformance,
|
|
15840
16053
|
createVoiceProviderFailureSimulator,
|
|
15841
16054
|
createVoiceIOProviderFailureSimulator,
|
|
15842
16055
|
createVoiceCallReviewRecorder,
|
|
@@ -15847,13 +16060,19 @@ export {
|
|
|
15847
16060
|
createCodeSwitchBenchmarkCorrectionHandler,
|
|
15848
16061
|
createBenchmarkCorrectionHandler,
|
|
15849
16062
|
compareSTTBenchmarks,
|
|
16063
|
+
comparePairedMetrics,
|
|
15850
16064
|
calibrateVoiceConfidence,
|
|
16065
|
+
buildVoiceBenchmarkArtifact,
|
|
16066
|
+
buildVoiceAudioMatrix,
|
|
15851
16067
|
buildSessionCorrectionAudit,
|
|
15852
16068
|
buildFixturePhraseHints,
|
|
15853
16069
|
buildCorrectionBenchmarkAudit,
|
|
15854
16070
|
buildCodeSwitchBenchmarkPhraseHints,
|
|
15855
16071
|
buildCodeSwitchBenchmarkLexicon,
|
|
16072
|
+
applyVoiceAudioCondition,
|
|
15856
16073
|
applyLexiconCorrectedBenchmarkReport,
|
|
15857
16074
|
applyExperimentalBenchmarkReport,
|
|
15858
|
-
applyCorrectedBenchmarkReport
|
|
16075
|
+
applyCorrectedBenchmarkReport,
|
|
16076
|
+
alignTranscriptWords,
|
|
16077
|
+
aggregateTranscriptAccuracy
|
|
15859
16078
|
};
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import type { VoiceSTTBenchmarkFixtureResult } from "./benchmark";
|
|
2
|
+
export type VoiceBenchmarkOutcomeSummary = {
|
|
3
|
+
completeRequiredProfileRate: number;
|
|
4
|
+
costPerPassingFixture?: number;
|
|
5
|
+
fixtureCount: number;
|
|
6
|
+
passingFixtureCount: number;
|
|
7
|
+
requiredFieldAccuracy: number;
|
|
8
|
+
totalCost?: number;
|
|
9
|
+
};
|
|
10
|
+
export declare const summarizeVoiceBenchmarkOutcomes: (fixtures: VoiceSTTBenchmarkFixtureResult[], costs?: {
|
|
11
|
+
total: number;
|
|
12
|
+
}) => VoiceBenchmarkOutcomeSummary;
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
export type VoiceBenchmarkPromptTrack = "unprompted" | "production-context" | "oracle-seeded";
|
|
2
|
+
export type VoiceCorpusSplit = "development" | "public-test" | "private-held-out";
|
|
3
|
+
export type VoiceCorpusFixtureProvenance = {
|
|
4
|
+
audioSha256: string;
|
|
5
|
+
consent?: string;
|
|
6
|
+
fixtureId: string;
|
|
7
|
+
license: string;
|
|
8
|
+
licenseClass: "permissive" | "noncommercial" | "private";
|
|
9
|
+
source: string;
|
|
10
|
+
split: VoiceCorpusSplit;
|
|
11
|
+
};
|
|
12
|
+
export type VoiceBenchmarkRunManifest = {
|
|
13
|
+
adapter: {
|
|
14
|
+
id: string;
|
|
15
|
+
model?: string;
|
|
16
|
+
provider?: string;
|
|
17
|
+
version?: string;
|
|
18
|
+
};
|
|
19
|
+
corpus: {
|
|
20
|
+
fixtures: VoiceCorpusFixtureProvenance[];
|
|
21
|
+
manifestSha256: string;
|
|
22
|
+
name: string;
|
|
23
|
+
version: string;
|
|
24
|
+
};
|
|
25
|
+
createdAt: string;
|
|
26
|
+
environment: Record<string, string | number | boolean>;
|
|
27
|
+
git: Record<string, string>;
|
|
28
|
+
preprocessing: Record<string, unknown>;
|
|
29
|
+
pricing?: Record<string, number>;
|
|
30
|
+
promptTrack: VoiceBenchmarkPromptTrack;
|
|
31
|
+
seed: number;
|
|
32
|
+
};
|
|
33
|
+
export declare const sha256Bytes: (value: Uint8Array | string) => string;
|
|
34
|
+
export declare const stableBenchmarkJson: (value: unknown) => string;
|
|
35
|
+
export declare const buildVoiceBenchmarkArtifact: <T>(manifest: VoiceBenchmarkRunManifest, report: T) => {
|
|
36
|
+
artifactSha256: string;
|
|
37
|
+
manifest: VoiceBenchmarkRunManifest;
|
|
38
|
+
report: T;
|
|
39
|
+
};
|
|
40
|
+
export declare const verifyVoiceBenchmarkArtifact: (artifact: {
|
|
41
|
+
artifactSha256: string;
|
|
42
|
+
manifest: VoiceBenchmarkRunManifest;
|
|
43
|
+
report: unknown;
|
|
44
|
+
}) => boolean;
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import type { VoiceTranscriptAccuracy } from "./accuracy";
|
|
2
|
+
export type VoiceAggregateErrorMetrics = {
|
|
3
|
+
correct: number;
|
|
4
|
+
deletions: number;
|
|
5
|
+
insertions: number;
|
|
6
|
+
macroWordErrorRate: number;
|
|
7
|
+
microWordErrorRate: number;
|
|
8
|
+
referenceWordCount: number;
|
|
9
|
+
sentenceErrorRate: number;
|
|
10
|
+
substitutions: number;
|
|
11
|
+
};
|
|
12
|
+
export type VoiceConfidenceInterval = {
|
|
13
|
+
confidenceLevel: number;
|
|
14
|
+
high: number;
|
|
15
|
+
low: number;
|
|
16
|
+
samples: number;
|
|
17
|
+
};
|
|
18
|
+
export type VoicePairedBootstrapComparison = {
|
|
19
|
+
baselineMean: number;
|
|
20
|
+
candidateMean: number;
|
|
21
|
+
delta: number;
|
|
22
|
+
deltaConfidenceInterval: VoiceConfidenceInterval;
|
|
23
|
+
probabilityCandidateIsBetter: number;
|
|
24
|
+
};
|
|
25
|
+
export declare const aggregateTranscriptAccuracy: (values: VoiceTranscriptAccuracy[]) => VoiceAggregateErrorMetrics;
|
|
26
|
+
export declare const comparePairedMetrics: (baseline: number[], candidate: number[], options?: {
|
|
27
|
+
confidenceLevel?: number;
|
|
28
|
+
samples?: number;
|
|
29
|
+
seed?: number;
|
|
30
|
+
}) => VoicePairedBootstrapComparison;
|