@absolutejs/voice 0.0.22-beta.637 → 0.0.22-beta.639
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +95 -2
- package/dist/testing/accuracy.d.ts +17 -0
- package/dist/testing/audioMatrix.d.ts +22 -0
- package/dist/testing/benchmark.d.ts +2 -0
- package/dist/testing/conformance.d.ts +11 -0
- package/dist/testing/fixtures.d.ts +1 -0
- package/dist/testing/index.d.ts +5 -0
- package/dist/testing/index.js +273 -52
- package/dist/testing/outcomes.d.ts +12 -0
- package/dist/testing/provenance.d.ts +44 -0
- package/dist/testing/statistics.d.ts +30 -0
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -4424,10 +4424,12 @@ var createVoiceSession = (options) => {
|
|
|
4424
4424
|
if (snapshot.status === "completed" || snapshot.status === "failed" || snapshot.call?.endedAt) {
|
|
4425
4425
|
return;
|
|
4426
4426
|
}
|
|
4427
|
+
const error = `no caller progress for ${stuckCloseAfterMs}ms`;
|
|
4427
4428
|
await appendTrace({
|
|
4428
4429
|
payload: {
|
|
4429
4430
|
action: "stuck-call-close",
|
|
4430
|
-
|
|
4431
|
+
error,
|
|
4432
|
+
reason: error
|
|
4431
4433
|
},
|
|
4432
4434
|
session: snapshot,
|
|
4433
4435
|
type: "session.error"
|
|
@@ -49422,18 +49424,72 @@ var levenshteinDistance2 = (left, right) => {
|
|
|
49422
49424
|
}
|
|
49423
49425
|
return previous[right.length];
|
|
49424
49426
|
};
|
|
49427
|
+
var alignTranscriptWords = (actualWords, expectedWords) => {
|
|
49428
|
+
const rows = expectedWords.length + 1;
|
|
49429
|
+
const columns = actualWords.length + 1;
|
|
49430
|
+
const costs = Array.from({ length: rows }, () => new Array(columns).fill(0));
|
|
49431
|
+
for (let row2 = 0;row2 < rows; row2 += 1)
|
|
49432
|
+
costs[row2][0] = row2;
|
|
49433
|
+
for (let column2 = 0;column2 < columns; column2 += 1)
|
|
49434
|
+
costs[0][column2] = column2;
|
|
49435
|
+
for (let row2 = 1;row2 < rows; row2 += 1) {
|
|
49436
|
+
for (let column2 = 1;column2 < columns; column2 += 1) {
|
|
49437
|
+
const substitution = costs[row2 - 1][column2 - 1] + (expectedWords[row2 - 1] === actualWords[column2 - 1] ? 0 : 1);
|
|
49438
|
+
costs[row2][column2] = Math.min(substitution, costs[row2 - 1][column2] + 1, costs[row2][column2 - 1] + 1);
|
|
49439
|
+
}
|
|
49440
|
+
}
|
|
49441
|
+
const operations = [];
|
|
49442
|
+
let row = expectedWords.length;
|
|
49443
|
+
let column = actualWords.length;
|
|
49444
|
+
while (row > 0 || column > 0) {
|
|
49445
|
+
const expected = expectedWords[row - 1];
|
|
49446
|
+
const actual = actualWords[column - 1];
|
|
49447
|
+
if (row > 0 && column > 0 && expected === actual && costs[row][column] === costs[row - 1][column - 1]) {
|
|
49448
|
+
operations.push({ actual, expected, type: "correct" });
|
|
49449
|
+
row -= 1;
|
|
49450
|
+
column -= 1;
|
|
49451
|
+
} else if (row > 0 && column > 0 && costs[row][column] === costs[row - 1][column - 1] + 1) {
|
|
49452
|
+
operations.push({ actual, expected, type: "substitution" });
|
|
49453
|
+
row -= 1;
|
|
49454
|
+
column -= 1;
|
|
49455
|
+
} else if (row > 0 && costs[row][column] === costs[row - 1][column] + 1) {
|
|
49456
|
+
operations.push({ expected, type: "deletion" });
|
|
49457
|
+
row -= 1;
|
|
49458
|
+
} else {
|
|
49459
|
+
operations.push({ actual, type: "insertion" });
|
|
49460
|
+
column -= 1;
|
|
49461
|
+
}
|
|
49462
|
+
}
|
|
49463
|
+
operations.reverse();
|
|
49464
|
+
const count = (type) => operations.filter((operation) => operation.type === type).length;
|
|
49465
|
+
const substitutions = count("substitution");
|
|
49466
|
+
const deletions = count("deletion");
|
|
49467
|
+
const insertions = count("insertion");
|
|
49468
|
+
return {
|
|
49469
|
+
correct: count("correct"),
|
|
49470
|
+
deletions,
|
|
49471
|
+
hypothesisWordCount: actualWords.length,
|
|
49472
|
+
insertions,
|
|
49473
|
+
operations,
|
|
49474
|
+
referenceWordCount: expectedWords.length,
|
|
49475
|
+
sentenceError: substitutions + deletions + insertions > 0,
|
|
49476
|
+
substitutions
|
|
49477
|
+
};
|
|
49478
|
+
};
|
|
49425
49479
|
var mergeFinalTranscriptText = (transcripts) => buildTurnText(transcripts.filter((transcript) => transcript.isFinal), "");
|
|
49426
49480
|
var scoreTranscriptAccuracy = (actualText, expectedText, threshold = 0.35) => {
|
|
49427
49481
|
const normalizedActual = normalizeAccuracyText(actualText);
|
|
49428
49482
|
const normalizedExpected = normalizeAccuracyText(expectedText);
|
|
49429
49483
|
const actualWords = normalizedActual ? normalizedActual.split(" ") : [];
|
|
49430
49484
|
const expectedWords = normalizedExpected ? normalizedExpected.split(" ") : [];
|
|
49431
|
-
const
|
|
49485
|
+
const alignment = alignTranscriptWords(actualWords, expectedWords);
|
|
49486
|
+
const wordDistance = alignment.substitutions + alignment.deletions + alignment.insertions;
|
|
49432
49487
|
const charDistance = levenshteinDistance2(Array.from(normalizedActual), Array.from(normalizedExpected));
|
|
49433
49488
|
const wordErrorRate = expectedWords.length > 0 ? wordDistance / expectedWords.length : 0;
|
|
49434
49489
|
const charErrorRate = normalizedExpected.length > 0 ? charDistance / normalizedExpected.length : 0;
|
|
49435
49490
|
return {
|
|
49436
49491
|
actualText: normalizedActual,
|
|
49492
|
+
alignment,
|
|
49437
49493
|
charDistance,
|
|
49438
49494
|
charErrorRate,
|
|
49439
49495
|
expectedText: normalizedExpected,
|
|
@@ -49641,6 +49697,42 @@ var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
|
|
|
49641
49697
|
};
|
|
49642
49698
|
};
|
|
49643
49699
|
|
|
49700
|
+
// src/testing/conformance.ts
|
|
49701
|
+
var evaluateSTTAdapterConformance = (result) => {
|
|
49702
|
+
const transcripts = [
|
|
49703
|
+
...result.partialEvents.map((event) => event.transcript),
|
|
49704
|
+
...result.finalEvents.map((event) => event.transcript)
|
|
49705
|
+
];
|
|
49706
|
+
const checks = [
|
|
49707
|
+
{
|
|
49708
|
+
detail: "The adapter emitted no error events for a valid fixture.",
|
|
49709
|
+
id: "no-errors",
|
|
49710
|
+
passed: result.errorEvents.length === 0
|
|
49711
|
+
},
|
|
49712
|
+
{
|
|
49713
|
+
detail: "Every transcript has a stable id and finite start time.",
|
|
49714
|
+
id: "transcript-identity",
|
|
49715
|
+
passed: transcripts.every((transcript) => transcript.id.trim().length > 0 && Number.isFinite(transcript.startedAtMs))
|
|
49716
|
+
},
|
|
49717
|
+
{
|
|
49718
|
+
detail: "Events marked final contain final transcripts.",
|
|
49719
|
+
id: "final-semantics",
|
|
49720
|
+
passed: result.finalEvents.every((event) => event.transcript.isFinal)
|
|
49721
|
+
},
|
|
49722
|
+
{
|
|
49723
|
+
detail: "Events marked partial contain non-final transcripts.",
|
|
49724
|
+
id: "partial-semantics",
|
|
49725
|
+
passed: result.partialEvents.every((event) => !event.transcript.isFinal)
|
|
49726
|
+
},
|
|
49727
|
+
{
|
|
49728
|
+
detail: "The assembled transcript is non-empty when finals were emitted.",
|
|
49729
|
+
id: "assembly",
|
|
49730
|
+
passed: result.finalEvents.length === 0 || result.finalText.trim().length > 0
|
|
49731
|
+
}
|
|
49732
|
+
];
|
|
49733
|
+
return { checks, passed: checks.every((check) => check.passed) };
|
|
49734
|
+
};
|
|
49735
|
+
|
|
49644
49736
|
// src/testing/benchmark.ts
|
|
49645
49737
|
var resolveFixtureEnvironment = (fixture) => {
|
|
49646
49738
|
const tags = new Set(fixture.tags ?? []);
|
|
@@ -49885,6 +49977,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
49885
49977
|
return {
|
|
49886
49978
|
accuracy: result.accuracy,
|
|
49887
49979
|
closeCount: result.closeEvents.length,
|
|
49980
|
+
conformance: evaluateSTTAdapterConformance(result),
|
|
49888
49981
|
criticalFields,
|
|
49889
49982
|
difficulty: fixture.difficulty,
|
|
49890
49983
|
elapsedMs,
|
|
@@ -8,6 +8,23 @@ export type VoiceTranscriptAccuracy = {
|
|
|
8
8
|
threshold: number;
|
|
9
9
|
wordDistance: number;
|
|
10
10
|
wordErrorRate: number;
|
|
11
|
+
alignment?: VoiceWordAlignment;
|
|
11
12
|
};
|
|
13
|
+
export type VoiceWordAlignmentOperation = {
|
|
14
|
+
actual?: string;
|
|
15
|
+
expected?: string;
|
|
16
|
+
type: "correct" | "substitution" | "deletion" | "insertion";
|
|
17
|
+
};
|
|
18
|
+
export type VoiceWordAlignment = {
|
|
19
|
+
correct: number;
|
|
20
|
+
deletions: number;
|
|
21
|
+
hypothesisWordCount: number;
|
|
22
|
+
insertions: number;
|
|
23
|
+
operations: VoiceWordAlignmentOperation[];
|
|
24
|
+
referenceWordCount: number;
|
|
25
|
+
sentenceError: boolean;
|
|
26
|
+
substitutions: number;
|
|
27
|
+
};
|
|
28
|
+
export declare const alignTranscriptWords: (actualWords: string[], expectedWords: string[]) => VoiceWordAlignment;
|
|
12
29
|
export declare const mergeFinalTranscriptText: (transcripts: Transcript[]) => string;
|
|
13
30
|
export declare const scoreTranscriptAccuracy: (actualText: string, expectedText: string, threshold?: number) => VoiceTranscriptAccuracy;
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import type { VoiceTestFixture } from "./fixtures";
|
|
2
|
+
export type VoiceAudioCondition = {
|
|
3
|
+
id: string;
|
|
4
|
+
type: "gain";
|
|
5
|
+
gain: number;
|
|
6
|
+
} | {
|
|
7
|
+
id: string;
|
|
8
|
+
type: "clip";
|
|
9
|
+
ceiling: number;
|
|
10
|
+
} | {
|
|
11
|
+
id: string;
|
|
12
|
+
type: "drop-chunks";
|
|
13
|
+
every: number;
|
|
14
|
+
chunkDurationMs: number;
|
|
15
|
+
} | {
|
|
16
|
+
id: string;
|
|
17
|
+
type: "noise";
|
|
18
|
+
seed: number;
|
|
19
|
+
snrDb: number;
|
|
20
|
+
};
|
|
21
|
+
export declare const applyVoiceAudioCondition: (fixture: VoiceTestFixture, condition: VoiceAudioCondition) => VoiceTestFixture;
|
|
22
|
+
export declare const buildVoiceAudioMatrix: (fixtures: VoiceTestFixture[], conditions: VoiceAudioCondition[]) => VoiceTestFixture[];
|
|
@@ -3,6 +3,7 @@ import { type VoiceSTTAdapterHarnessOptions, type VoiceSTTAdapterHarnessResult }
|
|
|
3
3
|
import type { VoiceTestFixture } from "./fixtures";
|
|
4
4
|
import { type VoiceConfidenceCalibrationReport } from "./confidenceCalibration";
|
|
5
5
|
import { type VoiceCriticalFieldAccuracy } from "./criticalFields";
|
|
6
|
+
import { type VoiceSTTConformanceReport } from "./conformance";
|
|
6
7
|
export type VoiceExpectedTermAccuracy = {
|
|
7
8
|
allMatched: boolean;
|
|
8
9
|
expectedTerms: string[];
|
|
@@ -22,6 +23,7 @@ export type VoiceSpeakerTurnAccuracy = {
|
|
|
22
23
|
export type VoiceSTTBenchmarkFixtureResult = {
|
|
23
24
|
accuracy: VoiceSTTAdapterHarnessResult["accuracy"];
|
|
24
25
|
closeCount: number;
|
|
26
|
+
conformance?: VoiceSTTConformanceReport;
|
|
25
27
|
difficulty?: VoiceTestFixture["difficulty"];
|
|
26
28
|
elapsedMs: number;
|
|
27
29
|
endOfTurnCount: number;
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import type { VoiceSTTAdapterHarnessResult } from "./stt";
|
|
2
|
+
export type VoiceSTTConformanceCheck = {
|
|
3
|
+
detail: string;
|
|
4
|
+
id: string;
|
|
5
|
+
passed: boolean;
|
|
6
|
+
};
|
|
7
|
+
export type VoiceSTTConformanceReport = {
|
|
8
|
+
checks: VoiceSTTConformanceCheck[];
|
|
9
|
+
passed: boolean;
|
|
10
|
+
};
|
|
11
|
+
export declare const evaluateSTTAdapterConformance: (result: VoiceSTTAdapterHarnessResult) => VoiceSTTConformanceReport;
|
|
@@ -15,6 +15,7 @@ export type VoiceTestFixtureManifestEntry = {
|
|
|
15
15
|
tags?: string[];
|
|
16
16
|
tailPaddingMs?: number;
|
|
17
17
|
format?: Partial<AudioFormat>;
|
|
18
|
+
provenance?: Omit<import("./provenance").VoiceCorpusFixtureProvenance, "audioSha256" | "fixtureId">;
|
|
18
19
|
};
|
|
19
20
|
export type VoiceTestFixture = Omit<VoiceTestFixtureManifestEntry, "audioPath"> & {
|
|
20
21
|
audio: Uint8Array;
|
package/dist/testing/index.d.ts
CHANGED
|
@@ -1,16 +1,21 @@
|
|
|
1
1
|
export * from "./accuracy";
|
|
2
|
+
export * from "./audioMatrix";
|
|
2
3
|
export * from "./benchmark";
|
|
3
4
|
export * from "./confidenceCalibration";
|
|
5
|
+
export * from "./conformance";
|
|
4
6
|
export * from "./criticalFields";
|
|
5
7
|
export * from "./corrected";
|
|
6
8
|
export * from "./duplex";
|
|
7
9
|
export * from "./fixtures";
|
|
8
10
|
export * from "./ioProviderSimulator";
|
|
11
|
+
export * from "./outcomes";
|
|
9
12
|
export * from "./providerSimulator";
|
|
13
|
+
export * from "./provenance";
|
|
10
14
|
export * from "./resilience";
|
|
11
15
|
export * from "./review";
|
|
12
16
|
export * from "./routingBenchmark";
|
|
13
17
|
export * from "./sessionBenchmark";
|
|
14
18
|
export * from "./stt";
|
|
19
|
+
export * from "./statistics";
|
|
15
20
|
export * from "./telephony";
|
|
16
21
|
export * from "./tts";
|
package/dist/testing/index.js
CHANGED
|
@@ -227,18 +227,72 @@ var levenshteinDistance = (left, right) => {
|
|
|
227
227
|
}
|
|
228
228
|
return previous[right.length];
|
|
229
229
|
};
|
|
230
|
+
var alignTranscriptWords = (actualWords, expectedWords) => {
|
|
231
|
+
const rows = expectedWords.length + 1;
|
|
232
|
+
const columns = actualWords.length + 1;
|
|
233
|
+
const costs = Array.from({ length: rows }, () => new Array(columns).fill(0));
|
|
234
|
+
for (let row2 = 0;row2 < rows; row2 += 1)
|
|
235
|
+
costs[row2][0] = row2;
|
|
236
|
+
for (let column2 = 0;column2 < columns; column2 += 1)
|
|
237
|
+
costs[0][column2] = column2;
|
|
238
|
+
for (let row2 = 1;row2 < rows; row2 += 1) {
|
|
239
|
+
for (let column2 = 1;column2 < columns; column2 += 1) {
|
|
240
|
+
const substitution = costs[row2 - 1][column2 - 1] + (expectedWords[row2 - 1] === actualWords[column2 - 1] ? 0 : 1);
|
|
241
|
+
costs[row2][column2] = Math.min(substitution, costs[row2 - 1][column2] + 1, costs[row2][column2 - 1] + 1);
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
const operations = [];
|
|
245
|
+
let row = expectedWords.length;
|
|
246
|
+
let column = actualWords.length;
|
|
247
|
+
while (row > 0 || column > 0) {
|
|
248
|
+
const expected = expectedWords[row - 1];
|
|
249
|
+
const actual = actualWords[column - 1];
|
|
250
|
+
if (row > 0 && column > 0 && expected === actual && costs[row][column] === costs[row - 1][column - 1]) {
|
|
251
|
+
operations.push({ actual, expected, type: "correct" });
|
|
252
|
+
row -= 1;
|
|
253
|
+
column -= 1;
|
|
254
|
+
} else if (row > 0 && column > 0 && costs[row][column] === costs[row - 1][column - 1] + 1) {
|
|
255
|
+
operations.push({ actual, expected, type: "substitution" });
|
|
256
|
+
row -= 1;
|
|
257
|
+
column -= 1;
|
|
258
|
+
} else if (row > 0 && costs[row][column] === costs[row - 1][column] + 1) {
|
|
259
|
+
operations.push({ expected, type: "deletion" });
|
|
260
|
+
row -= 1;
|
|
261
|
+
} else {
|
|
262
|
+
operations.push({ actual, type: "insertion" });
|
|
263
|
+
column -= 1;
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
operations.reverse();
|
|
267
|
+
const count = (type) => operations.filter((operation) => operation.type === type).length;
|
|
268
|
+
const substitutions = count("substitution");
|
|
269
|
+
const deletions = count("deletion");
|
|
270
|
+
const insertions = count("insertion");
|
|
271
|
+
return {
|
|
272
|
+
correct: count("correct"),
|
|
273
|
+
deletions,
|
|
274
|
+
hypothesisWordCount: actualWords.length,
|
|
275
|
+
insertions,
|
|
276
|
+
operations,
|
|
277
|
+
referenceWordCount: expectedWords.length,
|
|
278
|
+
sentenceError: substitutions + deletions + insertions > 0,
|
|
279
|
+
substitutions
|
|
280
|
+
};
|
|
281
|
+
};
|
|
230
282
|
var mergeFinalTranscriptText = (transcripts) => buildTurnText(transcripts.filter((transcript) => transcript.isFinal), "");
|
|
231
283
|
var scoreTranscriptAccuracy = (actualText, expectedText, threshold = 0.35) => {
|
|
232
284
|
const normalizedActual = normalizeAccuracyText(actualText);
|
|
233
285
|
const normalizedExpected = normalizeAccuracyText(expectedText);
|
|
234
286
|
const actualWords = normalizedActual ? normalizedActual.split(" ") : [];
|
|
235
287
|
const expectedWords = normalizedExpected ? normalizedExpected.split(" ") : [];
|
|
236
|
-
const
|
|
288
|
+
const alignment = alignTranscriptWords(actualWords, expectedWords);
|
|
289
|
+
const wordDistance = alignment.substitutions + alignment.deletions + alignment.insertions;
|
|
237
290
|
const charDistance = levenshteinDistance(Array.from(normalizedActual), Array.from(normalizedExpected));
|
|
238
291
|
const wordErrorRate = expectedWords.length > 0 ? wordDistance / expectedWords.length : 0;
|
|
239
292
|
const charErrorRate = normalizedExpected.length > 0 ? charDistance / normalizedExpected.length : 0;
|
|
240
293
|
return {
|
|
241
294
|
actualText: normalizedActual,
|
|
295
|
+
alignment,
|
|
242
296
|
charDistance,
|
|
243
297
|
charErrorRate,
|
|
244
298
|
expectedText: normalizedExpected,
|
|
@@ -248,6 +302,35 @@ var scoreTranscriptAccuracy = (actualText, expectedText, threshold = 0.35) => {
|
|
|
248
302
|
wordErrorRate
|
|
249
303
|
};
|
|
250
304
|
};
|
|
305
|
+
// src/testing/audioMatrix.ts
|
|
306
|
+
var clamp = (value) => Math.max(-32768, Math.min(32767, Math.round(value)));
|
|
307
|
+
var samples = (audio) => new Int16Array(audio.buffer.slice(audio.byteOffset, audio.byteOffset + audio.byteLength));
|
|
308
|
+
var bytes = (audio) => new Uint8Array(audio.buffer);
|
|
309
|
+
var applyVoiceAudioCondition = (fixture, condition) => {
|
|
310
|
+
const input = samples(fixture.audio);
|
|
311
|
+
const output = new Int16Array(input);
|
|
312
|
+
if (condition.type === "gain") {
|
|
313
|
+
for (let index = 0;index < output.length; index += 1)
|
|
314
|
+
output[index] = clamp(output[index] * condition.gain);
|
|
315
|
+
} else if (condition.type === "clip") {
|
|
316
|
+
const ceiling = Math.round(32767 * condition.ceiling);
|
|
317
|
+
for (let index = 0;index < output.length; index += 1)
|
|
318
|
+
output[index] = Math.max(-ceiling, Math.min(ceiling, output[index]));
|
|
319
|
+
} else if (condition.type === "drop-chunks") {
|
|
320
|
+
const chunkSamples = Math.max(1, Math.round(fixture.format.sampleRateHz * condition.chunkDurationMs / 1000));
|
|
321
|
+
for (let start = chunkSamples * (condition.every - 1);start < output.length; start += chunkSamples * condition.every)
|
|
322
|
+
output.fill(0, start, Math.min(output.length, start + chunkSamples));
|
|
323
|
+
} else {
|
|
324
|
+
let state = condition.seed >>> 0;
|
|
325
|
+
const random = () => (state = state * 1664525 + 1013904223 >>> 0) / 4294967296 * 2 - 1;
|
|
326
|
+
const signalPower = output.reduce((sum, value) => sum + value * value, 0) / Math.max(1, output.length);
|
|
327
|
+
const noiseRms = Math.sqrt(signalPower / 10 ** (condition.snrDb / 10));
|
|
328
|
+
for (let index = 0;index < output.length; index += 1)
|
|
329
|
+
output[index] = clamp(output[index] + random() * Math.sqrt(3) * noiseRms);
|
|
330
|
+
}
|
|
331
|
+
return { ...fixture, audio: bytes(output), id: `${fixture.id}--${condition.id}`, tags: [...fixture.tags ?? [], "conditioned", condition.id] };
|
|
332
|
+
};
|
|
333
|
+
var buildVoiceAudioMatrix = (fixtures, conditions) => fixtures.flatMap((fixture) => conditions.map((condition) => applyVoiceAudioCondition(fixture, condition)));
|
|
251
334
|
// src/testing/stt.ts
|
|
252
335
|
var chunkAudio = (audio, bytesPerChunk) => {
|
|
253
336
|
const chunks = [];
|
|
@@ -369,7 +452,7 @@ var runSTTAdapterFixture = async (adapter, fixture, options = {}) => {
|
|
|
369
452
|
|
|
370
453
|
// src/testing/confidenceCalibration.ts
|
|
371
454
|
var clampConfidence = (value) => Math.max(0, Math.min(1, value));
|
|
372
|
-
var calibrateVoiceConfidence = (
|
|
455
|
+
var calibrateVoiceConfidence = (samples2, binCount = 10) => {
|
|
373
456
|
const safeBinCount = Math.max(1, Math.round(binCount));
|
|
374
457
|
const bins = Array.from({ length: safeBinCount }, (_, index) => {
|
|
375
458
|
const lowerBound = index / safeBinCount;
|
|
@@ -382,7 +465,7 @@ var calibrateVoiceConfidence = (samples, binCount = 10) => {
|
|
|
382
465
|
};
|
|
383
466
|
});
|
|
384
467
|
let brierTotal = 0;
|
|
385
|
-
for (const sample of
|
|
468
|
+
for (const sample of samples2) {
|
|
386
469
|
const confidence = clampConfidence(sample.confidence);
|
|
387
470
|
const binIndex = Math.min(safeBinCount - 1, Math.floor(confidence * safeBinCount));
|
|
388
471
|
const bin = bins[binIndex];
|
|
@@ -397,13 +480,13 @@ var calibrateVoiceConfidence = (samples, binCount = 10) => {
|
|
|
397
480
|
continue;
|
|
398
481
|
bin.averageConfidence /= bin.count;
|
|
399
482
|
bin.accuracy /= bin.count;
|
|
400
|
-
expectedCalibrationError += bin.count / Math.max(1,
|
|
483
|
+
expectedCalibrationError += bin.count / Math.max(1, samples2.length) * Math.abs(bin.accuracy - bin.averageConfidence);
|
|
401
484
|
}
|
|
402
485
|
return {
|
|
403
486
|
bins,
|
|
404
|
-
brierScore:
|
|
487
|
+
brierScore: samples2.length > 0 ? brierTotal / samples2.length : 0,
|
|
405
488
|
expectedCalibrationError,
|
|
406
|
-
sampleCount:
|
|
489
|
+
sampleCount: samples2.length
|
|
407
490
|
};
|
|
408
491
|
};
|
|
409
492
|
|
|
@@ -630,6 +713,42 @@ var scoreVoiceCriticalFields = (actualText, expectedFields = []) => {
|
|
|
630
713
|
};
|
|
631
714
|
};
|
|
632
715
|
|
|
716
|
+
// src/testing/conformance.ts
|
|
717
|
+
var evaluateSTTAdapterConformance = (result) => {
|
|
718
|
+
const transcripts = [
|
|
719
|
+
...result.partialEvents.map((event) => event.transcript),
|
|
720
|
+
...result.finalEvents.map((event) => event.transcript)
|
|
721
|
+
];
|
|
722
|
+
const checks = [
|
|
723
|
+
{
|
|
724
|
+
detail: "The adapter emitted no error events for a valid fixture.",
|
|
725
|
+
id: "no-errors",
|
|
726
|
+
passed: result.errorEvents.length === 0
|
|
727
|
+
},
|
|
728
|
+
{
|
|
729
|
+
detail: "Every transcript has a stable id and finite start time.",
|
|
730
|
+
id: "transcript-identity",
|
|
731
|
+
passed: transcripts.every((transcript) => transcript.id.trim().length > 0 && Number.isFinite(transcript.startedAtMs))
|
|
732
|
+
},
|
|
733
|
+
{
|
|
734
|
+
detail: "Events marked final contain final transcripts.",
|
|
735
|
+
id: "final-semantics",
|
|
736
|
+
passed: result.finalEvents.every((event) => event.transcript.isFinal)
|
|
737
|
+
},
|
|
738
|
+
{
|
|
739
|
+
detail: "Events marked partial contain non-final transcripts.",
|
|
740
|
+
id: "partial-semantics",
|
|
741
|
+
passed: result.partialEvents.every((event) => !event.transcript.isFinal)
|
|
742
|
+
},
|
|
743
|
+
{
|
|
744
|
+
detail: "The assembled transcript is non-empty when finals were emitted.",
|
|
745
|
+
id: "assembly",
|
|
746
|
+
passed: result.finalEvents.length === 0 || result.finalText.trim().length > 0
|
|
747
|
+
}
|
|
748
|
+
];
|
|
749
|
+
return { checks, passed: checks.every((check) => check.passed) };
|
|
750
|
+
};
|
|
751
|
+
|
|
633
752
|
// src/testing/benchmark.ts
|
|
634
753
|
var resolveFixtureEnvironment = (fixture) => {
|
|
635
754
|
const tags = new Set(fixture.tags ?? []);
|
|
@@ -874,6 +993,7 @@ var toFixtureBenchmarkResult = (fixture, result, elapsedMs) => {
|
|
|
874
993
|
return {
|
|
875
994
|
accuracy: result.accuracy,
|
|
876
995
|
closeCount: result.closeEvents.length,
|
|
996
|
+
conformance: evaluateSTTAdapterConformance(result),
|
|
877
997
|
criticalFields,
|
|
878
998
|
difficulty: fixture.difficulty,
|
|
879
999
|
elapsedMs,
|
|
@@ -2057,18 +2177,18 @@ var decodePCM16LEChunk = (audioContext, chunk) => {
|
|
|
2057
2177
|
if (format.container !== "raw" || format.encoding !== "pcm_s16le") {
|
|
2058
2178
|
throw new Error(`Unsupported assistant audio format: ${format.container}/${format.encoding}`);
|
|
2059
2179
|
}
|
|
2060
|
-
const
|
|
2180
|
+
const bytes2 = chunk.chunk;
|
|
2061
2181
|
const channels = Math.max(1, format.channels);
|
|
2062
|
-
const sampleCount = Math.floor(
|
|
2182
|
+
const sampleCount = Math.floor(bytes2.byteLength / 2);
|
|
2063
2183
|
const frameCount = Math.max(1, Math.floor(sampleCount / channels));
|
|
2064
2184
|
const audioBuffer = audioContext.createBuffer(channels, frameCount, format.sampleRateHz);
|
|
2065
|
-
const view = new DataView(
|
|
2185
|
+
const view = new DataView(bytes2.buffer, bytes2.byteOffset, bytes2.byteLength);
|
|
2066
2186
|
for (let channelIndex = 0;channelIndex < channels; channelIndex += 1) {
|
|
2067
2187
|
const channelData = audioBuffer.getChannelData(channelIndex);
|
|
2068
2188
|
for (let frameIndex = 0;frameIndex < frameCount; frameIndex += 1) {
|
|
2069
2189
|
const sampleIndex = frameIndex * channels + channelIndex;
|
|
2070
2190
|
const sampleOffset = sampleIndex * 2;
|
|
2071
|
-
if (sampleOffset + 1 >=
|
|
2191
|
+
if (sampleOffset + 1 >= bytes2.byteLength) {
|
|
2072
2192
|
channelData[frameIndex] = 0;
|
|
2073
2193
|
continue;
|
|
2074
2194
|
}
|
|
@@ -2570,20 +2690,20 @@ var floatTo16BitPCM = (input) => {
|
|
|
2570
2690
|
return new Uint8Array(output.buffer);
|
|
2571
2691
|
};
|
|
2572
2692
|
var getPcmLevel = (audio) => {
|
|
2573
|
-
const
|
|
2574
|
-
if (
|
|
2693
|
+
const bytes2 = audio instanceof Uint8Array ? audio : new Uint8Array(audio);
|
|
2694
|
+
if (bytes2.byteLength < 2) {
|
|
2575
2695
|
return 0;
|
|
2576
2696
|
}
|
|
2577
|
-
const
|
|
2578
|
-
if (
|
|
2697
|
+
const samples2 = new Int16Array(bytes2.buffer, bytes2.byteOffset, Math.floor(bytes2.byteLength / 2));
|
|
2698
|
+
if (samples2.length === 0) {
|
|
2579
2699
|
return 0;
|
|
2580
2700
|
}
|
|
2581
2701
|
let sumSquares = 0;
|
|
2582
|
-
for (const sample of
|
|
2702
|
+
for (const sample of samples2) {
|
|
2583
2703
|
const normalized = sample / 32768;
|
|
2584
2704
|
sumSquares += normalized * normalized;
|
|
2585
2705
|
}
|
|
2586
|
-
return Math.min(1, Math.max(0, Math.sqrt(sumSquares /
|
|
2706
|
+
return Math.min(1, Math.max(0, Math.sqrt(sumSquares / samples2.length) * 5.5));
|
|
2587
2707
|
};
|
|
2588
2708
|
var downsampleBuffer = (input, sourceRate, targetRate) => {
|
|
2589
2709
|
if (sourceRate === targetRate) {
|
|
@@ -3461,16 +3581,16 @@ var toInt16Array = (audio) => {
|
|
|
3461
3581
|
}
|
|
3462
3582
|
return new Int16Array(audio.buffer, audio.byteOffset, Math.floor(audio.byteLength / 2));
|
|
3463
3583
|
};
|
|
3464
|
-
var computeRms = (
|
|
3465
|
-
if (
|
|
3584
|
+
var computeRms = (samples2) => {
|
|
3585
|
+
if (samples2.length === 0) {
|
|
3466
3586
|
return 0;
|
|
3467
3587
|
}
|
|
3468
3588
|
let sumSquares = 0;
|
|
3469
|
-
for (const sample of
|
|
3589
|
+
for (const sample of samples2) {
|
|
3470
3590
|
const normalized = sample / 32768;
|
|
3471
3591
|
sumSquares += normalized * normalized;
|
|
3472
3592
|
}
|
|
3473
|
-
return Math.sqrt(sumSquares /
|
|
3593
|
+
return Math.sqrt(sumSquares / samples2.length);
|
|
3474
3594
|
};
|
|
3475
3595
|
var conditionAudioChunk = (audio, config) => {
|
|
3476
3596
|
if (!config) {
|
|
@@ -4273,7 +4393,7 @@ var resolveVoiceFixtureDirectories = async (input) => {
|
|
|
4273
4393
|
};
|
|
4274
4394
|
var clampSample2 = (value) => Math.max(-32768, Math.min(32767, Math.round(value)));
|
|
4275
4395
|
var toPcm16Samples = (audio) => new Int16Array(audio.buffer.slice(audio.byteOffset, audio.byteOffset + audio.byteLength));
|
|
4276
|
-
var toPcm16Bytes = (
|
|
4396
|
+
var toPcm16Bytes = (samples2) => new Uint8Array(samples2.buffer.slice(samples2.byteOffset, samples2.byteOffset + samples2.byteLength));
|
|
4277
4397
|
var createSilenceBytes = (sampleRateHz, durationMs) => new Uint8Array(Math.max(2, Math.round(sampleRateHz * 2 * durationMs / 1000)));
|
|
4278
4398
|
var concatAudioChunks = (chunks) => {
|
|
4279
4399
|
const totalByteLength = chunks.reduce((sum, chunk) => sum + chunk.byteLength, 0);
|
|
@@ -4285,20 +4405,20 @@ var concatAudioChunks = (chunks) => {
|
|
|
4285
4405
|
}
|
|
4286
4406
|
return output;
|
|
4287
4407
|
};
|
|
4288
|
-
var resamplePcm16Mono = (
|
|
4289
|
-
if (sourceRate === targetRate ||
|
|
4290
|
-
return
|
|
4408
|
+
var resamplePcm16Mono = (samples2, sourceRate, targetRate) => {
|
|
4409
|
+
if (sourceRate === targetRate || samples2.length === 0) {
|
|
4410
|
+
return samples2;
|
|
4291
4411
|
}
|
|
4292
4412
|
const ratio = targetRate / sourceRate;
|
|
4293
|
-
const targetLength = Math.max(1, Math.round(
|
|
4413
|
+
const targetLength = Math.max(1, Math.round(samples2.length * ratio));
|
|
4294
4414
|
const output = new Int16Array(targetLength);
|
|
4295
4415
|
for (let index = 0;index < targetLength; index += 1) {
|
|
4296
4416
|
const sourceIndex = index / ratio;
|
|
4297
4417
|
const previousIndex = Math.floor(sourceIndex);
|
|
4298
|
-
const nextIndex = Math.min(previousIndex + 1,
|
|
4418
|
+
const nextIndex = Math.min(previousIndex + 1, samples2.length - 1);
|
|
4299
4419
|
const fraction = sourceIndex - previousIndex;
|
|
4300
|
-
const previous =
|
|
4301
|
-
const next =
|
|
4420
|
+
const previous = samples2[previousIndex] ?? 0;
|
|
4421
|
+
const next = samples2[nextIndex] ?? previous;
|
|
4302
4422
|
output[index] = clampSample2(previous + (next - previous) * fraction);
|
|
4303
4423
|
}
|
|
4304
4424
|
return output;
|
|
@@ -4576,6 +4696,21 @@ var createVoiceIOProviderFailureSimulator = (options) => {
|
|
|
4576
4696
|
run
|
|
4577
4697
|
};
|
|
4578
4698
|
};
|
|
4699
|
+
// src/testing/outcomes.ts
|
|
4700
|
+
var summarizeVoiceBenchmarkOutcomes = (fixtures, costs) => {
|
|
4701
|
+
const critical = fixtures.map((fixture) => fixture.criticalFields).filter((value) => value !== undefined);
|
|
4702
|
+
const passingFixtureCount = fixtures.filter((fixture) => fixture.passes).length;
|
|
4703
|
+
const requiredFields = critical.flatMap((value) => value.fields.filter((field) => field.required));
|
|
4704
|
+
const completeRequiredProfileRate = critical.length > 0 ? critical.filter((value) => value.passesRequired).length / critical.length : 1;
|
|
4705
|
+
return {
|
|
4706
|
+
completeRequiredProfileRate,
|
|
4707
|
+
costPerPassingFixture: costs && passingFixtureCount > 0 ? costs.total / passingFixtureCount : undefined,
|
|
4708
|
+
fixtureCount: fixtures.length,
|
|
4709
|
+
passingFixtureCount,
|
|
4710
|
+
requiredFieldAccuracy: requiredFields.length > 0 ? requiredFields.filter((field) => field.matched).length / requiredFields.length : 1,
|
|
4711
|
+
totalCost: costs?.total
|
|
4712
|
+
};
|
|
4713
|
+
};
|
|
4579
4714
|
// src/core/debugTiming.ts
|
|
4580
4715
|
var timingEnabled = () => process.env.ABSOLUTEJS_VOICE_TIMING === "1" || process.env.ABSOLUTEJS_VOICE_TIMING === "true";
|
|
4581
4716
|
var emitTiming = (sessionId, stage, elapsedMs, detail) => {
|
|
@@ -5857,6 +5992,22 @@ var createVoiceProviderFailureSimulator = (options) => {
|
|
|
5857
5992
|
run
|
|
5858
5993
|
};
|
|
5859
5994
|
};
|
|
5995
|
+
// src/testing/provenance.ts
|
|
5996
|
+
import { createHash } from "crypto";
|
|
5997
|
+
var sha256Bytes = (value) => createHash("sha256").update(value).digest("hex");
|
|
5998
|
+
var stableBenchmarkJson = (value) => {
|
|
5999
|
+
if (Array.isArray(value))
|
|
6000
|
+
return `[${value.map(stableBenchmarkJson).join(",")}]`;
|
|
6001
|
+
if (value && typeof value === "object") {
|
|
6002
|
+
return `{${Object.entries(value).sort(([left], [right]) => left.localeCompare(right)).map(([key, entry]) => `${JSON.stringify(key)}:${stableBenchmarkJson(entry)}`).join(",")}}`;
|
|
6003
|
+
}
|
|
6004
|
+
return JSON.stringify(value);
|
|
6005
|
+
};
|
|
6006
|
+
var buildVoiceBenchmarkArtifact = (manifest, report) => {
|
|
6007
|
+
const payload = { manifest, report };
|
|
6008
|
+
return { ...payload, artifactSha256: sha256Bytes(stableBenchmarkJson(payload)) };
|
|
6009
|
+
};
|
|
6010
|
+
var verifyVoiceBenchmarkArtifact = (artifact) => artifact.artifactSha256 === sha256Bytes(stableBenchmarkJson({ manifest: artifact.manifest, report: artifact.report }));
|
|
5860
6011
|
// src/core/memoryStore.ts
|
|
5861
6012
|
var createVoiceMemoryStore = () => {
|
|
5862
6013
|
const sessions = new Map;
|
|
@@ -6042,7 +6193,7 @@ var createVoiceBackchannelDriver = (options) => {
|
|
|
6042
6193
|
};
|
|
6043
6194
|
|
|
6044
6195
|
// src/core/handoff.ts
|
|
6045
|
-
var toHex = (
|
|
6196
|
+
var toHex = (bytes2) => Array.from(bytes2, (byte) => byte.toString(16).padStart(2, "0")).join("");
|
|
6046
6197
|
var signHandoffBody = async (input) => {
|
|
6047
6198
|
const encoder = new TextEncoder;
|
|
6048
6199
|
const key = await crypto.subtle.importKey("raw", encoder.encode(input.secret), {
|
|
@@ -6905,10 +7056,12 @@ var createVoiceSession = (options) => {
|
|
|
6905
7056
|
if (snapshot.status === "completed" || snapshot.status === "failed" || snapshot.call?.endedAt) {
|
|
6906
7057
|
return;
|
|
6907
7058
|
}
|
|
7059
|
+
const error = `no caller progress for ${stuckCloseAfterMs}ms`;
|
|
6908
7060
|
await appendTrace({
|
|
6909
7061
|
payload: {
|
|
6910
7062
|
action: "stuck-call-close",
|
|
6911
|
-
|
|
7063
|
+
error,
|
|
7064
|
+
reason: error
|
|
6912
7065
|
},
|
|
6913
7066
|
session: snapshot,
|
|
6914
7067
|
type: "session.error"
|
|
@@ -7041,7 +7194,7 @@ var createVoiceSession = (options) => {
|
|
|
7041
7194
|
};
|
|
7042
7195
|
const recordingFormats = {};
|
|
7043
7196
|
let recordingPersisted = false;
|
|
7044
|
-
const captureRecordingChunk = (channel,
|
|
7197
|
+
const captureRecordingChunk = (channel, bytes2, format) => {
|
|
7045
7198
|
if (!recordingConfig || recordingPersisted) {
|
|
7046
7199
|
return;
|
|
7047
7200
|
}
|
|
@@ -7056,7 +7209,7 @@ var createVoiceSession = (options) => {
|
|
|
7056
7209
|
return;
|
|
7057
7210
|
}
|
|
7058
7211
|
const remaining = recordingMaxBytes - currentTotal;
|
|
7059
|
-
const slice =
|
|
7212
|
+
const slice = bytes2.byteLength <= remaining ? bytes2 : bytes2.subarray(0, remaining);
|
|
7060
7213
|
recordingBuffers[channel].push(new Uint8Array(slice));
|
|
7061
7214
|
recordingByteTotals[channel] += slice.byteLength;
|
|
7062
7215
|
recordingFormats[channel] = format;
|
|
@@ -10820,6 +10973,63 @@ var summarizeVoiceSessionBenchmarkSeries = (input) => {
|
|
|
10820
10973
|
}
|
|
10821
10974
|
};
|
|
10822
10975
|
};
|
|
10976
|
+
// src/testing/statistics.ts
|
|
10977
|
+
var mean = (values) => values.length === 0 ? 0 : values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
10978
|
+
var aggregateTranscriptAccuracy = (values) => {
|
|
10979
|
+
const alignments = values.map((value) => value.alignment).filter(Boolean);
|
|
10980
|
+
const sum = (key) => alignments.reduce((total, alignment) => total + alignment[key], 0);
|
|
10981
|
+
const referenceWordCount = sum("referenceWordCount");
|
|
10982
|
+
const errors = sum("substitutions") + sum("deletions") + sum("insertions");
|
|
10983
|
+
return {
|
|
10984
|
+
correct: sum("correct"),
|
|
10985
|
+
deletions: sum("deletions"),
|
|
10986
|
+
insertions: sum("insertions"),
|
|
10987
|
+
macroWordErrorRate: mean(values.map((value) => value.wordErrorRate)),
|
|
10988
|
+
microWordErrorRate: referenceWordCount > 0 ? errors / referenceWordCount : 0,
|
|
10989
|
+
referenceWordCount,
|
|
10990
|
+
sentenceErrorRate: values.length > 0 ? alignments.filter((alignment) => alignment.sentenceError).length / values.length : 0,
|
|
10991
|
+
substitutions: sum("substitutions")
|
|
10992
|
+
};
|
|
10993
|
+
};
|
|
10994
|
+
var seededRandom = (seed) => {
|
|
10995
|
+
let state = seed >>> 0;
|
|
10996
|
+
return () => {
|
|
10997
|
+
state = state * 1664525 + 1013904223 >>> 0;
|
|
10998
|
+
return state / 4294967296;
|
|
10999
|
+
};
|
|
11000
|
+
};
|
|
11001
|
+
var comparePairedMetrics = (baseline, candidate, options = {}) => {
|
|
11002
|
+
if (baseline.length !== candidate.length || baseline.length === 0) {
|
|
11003
|
+
throw new Error("Paired comparisons require equal, non-empty samples.");
|
|
11004
|
+
}
|
|
11005
|
+
const samples2 = options.samples ?? 1e4;
|
|
11006
|
+
const confidenceLevel = options.confidenceLevel ?? 0.95;
|
|
11007
|
+
const random = seededRandom(options.seed ?? 20260722);
|
|
11008
|
+
const deltas = [];
|
|
11009
|
+
for (let sample = 0;sample < samples2; sample += 1) {
|
|
11010
|
+
const selected = [];
|
|
11011
|
+
for (let index = 0;index < baseline.length; index += 1) {
|
|
11012
|
+
const selectedIndex = Math.floor(random() * baseline.length);
|
|
11013
|
+
selected.push(candidate[selectedIndex] - baseline[selectedIndex]);
|
|
11014
|
+
}
|
|
11015
|
+
deltas.push(mean(selected));
|
|
11016
|
+
}
|
|
11017
|
+
deltas.sort((left, right) => left - right);
|
|
11018
|
+
const alpha = (1 - confidenceLevel) / 2;
|
|
11019
|
+
const percentile = (value) => deltas[Math.min(deltas.length - 1, Math.floor(value * deltas.length))] ?? 0;
|
|
11020
|
+
return {
|
|
11021
|
+
baselineMean: mean(baseline),
|
|
11022
|
+
candidateMean: mean(candidate),
|
|
11023
|
+
delta: mean(candidate) - mean(baseline),
|
|
11024
|
+
deltaConfidenceInterval: {
|
|
11025
|
+
confidenceLevel,
|
|
11026
|
+
high: percentile(1 - alpha),
|
|
11027
|
+
low: percentile(alpha),
|
|
11028
|
+
samples: samples2
|
|
11029
|
+
},
|
|
11030
|
+
probabilityCandidateIsBetter: deltas.filter((delta) => delta < 0).length / deltas.length
|
|
11031
|
+
};
|
|
11032
|
+
};
|
|
10823
11033
|
// src/core/operationsRecord.ts
|
|
10824
11034
|
import { Elysia as Elysia4 } from "elysia";
|
|
10825
11035
|
import {
|
|
@@ -11182,7 +11392,7 @@ var sleep2 = async (delayMs) => {
|
|
|
11182
11392
|
}
|
|
11183
11393
|
await new Promise((resolve2) => setTimeout(resolve2, delayMs));
|
|
11184
11394
|
};
|
|
11185
|
-
var toHex2 = (
|
|
11395
|
+
var toHex2 = (bytes2) => Array.from(bytes2, (byte) => byte.toString(16).padStart(2, "0")).join("");
|
|
11186
11396
|
var signVoiceTraceSinkBody = async (input) => {
|
|
11187
11397
|
const encoder = new TextEncoder;
|
|
11188
11398
|
const key = await crypto.subtle.importKey("raw", encoder.encode(input.secret), {
|
|
@@ -13799,7 +14009,7 @@ var flattenPayload = (value) => {
|
|
|
13799
14009
|
...isRecord(data?.payload) ? data.payload : undefined
|
|
13800
14010
|
};
|
|
13801
14011
|
};
|
|
13802
|
-
var toBase64 = (
|
|
14012
|
+
var toBase64 = (bytes2) => Buffer.from(new Uint8Array(bytes2)).toString("base64");
|
|
13803
14013
|
var timingSafeEqual = (left, right) => {
|
|
13804
14014
|
const encoder = new TextEncoder;
|
|
13805
14015
|
const leftBytes = encoder.encode(left);
|
|
@@ -14647,37 +14857,37 @@ var decodeMulawSample = (value) => {
|
|
|
14647
14857
|
sample -= MULAW_BIAS;
|
|
14648
14858
|
return sign ? -sample : sample;
|
|
14649
14859
|
};
|
|
14650
|
-
var int16ArrayToBytes = (
|
|
14651
|
-
const output = new Uint8Array(
|
|
14860
|
+
var int16ArrayToBytes = (samples2) => {
|
|
14861
|
+
const output = new Uint8Array(samples2.length * 2);
|
|
14652
14862
|
const view = new DataView(output.buffer);
|
|
14653
|
-
for (let index = 0;index <
|
|
14654
|
-
view.setInt16(index * 2,
|
|
14863
|
+
for (let index = 0;index < samples2.length; index += 1) {
|
|
14864
|
+
view.setInt16(index * 2, samples2[index] ?? 0, true);
|
|
14655
14865
|
}
|
|
14656
14866
|
return output;
|
|
14657
14867
|
};
|
|
14658
|
-
var bytesToInt16Array = (
|
|
14659
|
-
const sampleCount = Math.floor(
|
|
14868
|
+
var bytesToInt16Array = (bytes2) => {
|
|
14869
|
+
const sampleCount = Math.floor(bytes2.byteLength / 2);
|
|
14660
14870
|
const output = new Int16Array(sampleCount);
|
|
14661
|
-
const view = new DataView(
|
|
14871
|
+
const view = new DataView(bytes2.buffer, bytes2.byteOffset, bytes2.byteLength);
|
|
14662
14872
|
for (let index = 0;index < sampleCount; index += 1) {
|
|
14663
14873
|
output[index] = view.getInt16(index * 2, true);
|
|
14664
14874
|
}
|
|
14665
14875
|
return output;
|
|
14666
14876
|
};
|
|
14667
14877
|
var decodeTwilioMulawBase64 = (payload) => {
|
|
14668
|
-
const
|
|
14669
|
-
const
|
|
14670
|
-
for (let index = 0;index <
|
|
14671
|
-
|
|
14878
|
+
const bytes2 = Uint8Array.from(Buffer3.from(payload, "base64"));
|
|
14879
|
+
const samples2 = new Int16Array(bytes2.length);
|
|
14880
|
+
for (let index = 0;index < bytes2.length; index += 1) {
|
|
14881
|
+
samples2[index] = decodeMulawSample(bytes2[index] ?? 0);
|
|
14672
14882
|
}
|
|
14673
|
-
return
|
|
14883
|
+
return samples2;
|
|
14674
14884
|
};
|
|
14675
|
-
var encodeTwilioMulawBase64 = (
|
|
14676
|
-
const
|
|
14677
|
-
for (let index = 0;index <
|
|
14678
|
-
|
|
14885
|
+
var encodeTwilioMulawBase64 = (samples2) => {
|
|
14886
|
+
const bytes2 = new Uint8Array(samples2.length);
|
|
14887
|
+
for (let index = 0;index < samples2.length; index += 1) {
|
|
14888
|
+
bytes2[index] = encodeMulawSample(samples2[index] ?? 0);
|
|
14679
14889
|
}
|
|
14680
|
-
return Buffer3.from(
|
|
14890
|
+
return Buffer3.from(bytes2).toString("base64");
|
|
14681
14891
|
};
|
|
14682
14892
|
var transcodePCMToTwilioOutboundPayload = (chunk, format) => {
|
|
14683
14893
|
if (format.container === "raw" && format.encoding === "mulaw" && format.channels === 1 && format.sampleRateHz === TWILIO_MULAW_SAMPLE_RATE) {
|
|
@@ -15800,13 +16010,17 @@ var summarizeTTSBenchmark = (adapterId, fixtures) => {
|
|
|
15800
16010
|
};
|
|
15801
16011
|
export {
|
|
15802
16012
|
withVoiceCallReviewId,
|
|
16013
|
+
verifyVoiceBenchmarkArtifact,
|
|
15803
16014
|
summarizeVoiceTelephonyBenchmark,
|
|
15804
16015
|
summarizeVoiceSessionBenchmarkSeries,
|
|
15805
16016
|
summarizeVoiceSessionBenchmark,
|
|
15806
16017
|
summarizeVoiceDuplexBenchmark,
|
|
16018
|
+
summarizeVoiceBenchmarkOutcomes,
|
|
15807
16019
|
summarizeTTSBenchmark,
|
|
15808
16020
|
summarizeSTTBenchmarkSeries,
|
|
15809
16021
|
summarizeSTTBenchmark,
|
|
16022
|
+
stableBenchmarkJson,
|
|
16023
|
+
sha256Bytes,
|
|
15810
16024
|
scoreVoiceCriticalFields,
|
|
15811
16025
|
scoreTranscriptAccuracy,
|
|
15812
16026
|
scoreCorrectedExpectedTerms,
|
|
@@ -15837,6 +16051,7 @@ export {
|
|
|
15837
16051
|
getDefaultTTSBenchmarkFixtures,
|
|
15838
16052
|
evaluateVoiceSTTRouting,
|
|
15839
16053
|
evaluateSTTBenchmarkAcceptance,
|
|
16054
|
+
evaluateSTTAdapterConformance,
|
|
15840
16055
|
createVoiceProviderFailureSimulator,
|
|
15841
16056
|
createVoiceIOProviderFailureSimulator,
|
|
15842
16057
|
createVoiceCallReviewRecorder,
|
|
@@ -15847,13 +16062,19 @@ export {
|
|
|
15847
16062
|
createCodeSwitchBenchmarkCorrectionHandler,
|
|
15848
16063
|
createBenchmarkCorrectionHandler,
|
|
15849
16064
|
compareSTTBenchmarks,
|
|
16065
|
+
comparePairedMetrics,
|
|
15850
16066
|
calibrateVoiceConfidence,
|
|
16067
|
+
buildVoiceBenchmarkArtifact,
|
|
16068
|
+
buildVoiceAudioMatrix,
|
|
15851
16069
|
buildSessionCorrectionAudit,
|
|
15852
16070
|
buildFixturePhraseHints,
|
|
15853
16071
|
buildCorrectionBenchmarkAudit,
|
|
15854
16072
|
buildCodeSwitchBenchmarkPhraseHints,
|
|
15855
16073
|
buildCodeSwitchBenchmarkLexicon,
|
|
16074
|
+
applyVoiceAudioCondition,
|
|
15856
16075
|
applyLexiconCorrectedBenchmarkReport,
|
|
15857
16076
|
applyExperimentalBenchmarkReport,
|
|
15858
|
-
applyCorrectedBenchmarkReport
|
|
16077
|
+
applyCorrectedBenchmarkReport,
|
|
16078
|
+
alignTranscriptWords,
|
|
16079
|
+
aggregateTranscriptAccuracy
|
|
15859
16080
|
};
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import type { VoiceSTTBenchmarkFixtureResult } from "./benchmark";
|
|
2
|
+
export type VoiceBenchmarkOutcomeSummary = {
|
|
3
|
+
completeRequiredProfileRate: number;
|
|
4
|
+
costPerPassingFixture?: number;
|
|
5
|
+
fixtureCount: number;
|
|
6
|
+
passingFixtureCount: number;
|
|
7
|
+
requiredFieldAccuracy: number;
|
|
8
|
+
totalCost?: number;
|
|
9
|
+
};
|
|
10
|
+
export declare const summarizeVoiceBenchmarkOutcomes: (fixtures: VoiceSTTBenchmarkFixtureResult[], costs?: {
|
|
11
|
+
total: number;
|
|
12
|
+
}) => VoiceBenchmarkOutcomeSummary;
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
export type VoiceBenchmarkPromptTrack = "unprompted" | "production-context" | "oracle-seeded";
|
|
2
|
+
export type VoiceCorpusSplit = "development" | "public-test" | "private-held-out";
|
|
3
|
+
export type VoiceCorpusFixtureProvenance = {
|
|
4
|
+
audioSha256: string;
|
|
5
|
+
consent?: string;
|
|
6
|
+
fixtureId: string;
|
|
7
|
+
license: string;
|
|
8
|
+
licenseClass: "permissive" | "noncommercial" | "private";
|
|
9
|
+
source: string;
|
|
10
|
+
split: VoiceCorpusSplit;
|
|
11
|
+
};
|
|
12
|
+
export type VoiceBenchmarkRunManifest = {
|
|
13
|
+
adapter: {
|
|
14
|
+
id: string;
|
|
15
|
+
model?: string;
|
|
16
|
+
provider?: string;
|
|
17
|
+
version?: string;
|
|
18
|
+
};
|
|
19
|
+
corpus: {
|
|
20
|
+
fixtures: VoiceCorpusFixtureProvenance[];
|
|
21
|
+
manifestSha256: string;
|
|
22
|
+
name: string;
|
|
23
|
+
version: string;
|
|
24
|
+
};
|
|
25
|
+
createdAt: string;
|
|
26
|
+
environment: Record<string, string | number | boolean>;
|
|
27
|
+
git: Record<string, string>;
|
|
28
|
+
preprocessing: Record<string, unknown>;
|
|
29
|
+
pricing?: Record<string, number>;
|
|
30
|
+
promptTrack: VoiceBenchmarkPromptTrack;
|
|
31
|
+
seed: number;
|
|
32
|
+
};
|
|
33
|
+
export declare const sha256Bytes: (value: Uint8Array | string) => string;
|
|
34
|
+
export declare const stableBenchmarkJson: (value: unknown) => string;
|
|
35
|
+
export declare const buildVoiceBenchmarkArtifact: <T>(manifest: VoiceBenchmarkRunManifest, report: T) => {
|
|
36
|
+
artifactSha256: string;
|
|
37
|
+
manifest: VoiceBenchmarkRunManifest;
|
|
38
|
+
report: T;
|
|
39
|
+
};
|
|
40
|
+
export declare const verifyVoiceBenchmarkArtifact: (artifact: {
|
|
41
|
+
artifactSha256: string;
|
|
42
|
+
manifest: VoiceBenchmarkRunManifest;
|
|
43
|
+
report: unknown;
|
|
44
|
+
}) => boolean;
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import type { VoiceTranscriptAccuracy } from "./accuracy";
|
|
2
|
+
export type VoiceAggregateErrorMetrics = {
|
|
3
|
+
correct: number;
|
|
4
|
+
deletions: number;
|
|
5
|
+
insertions: number;
|
|
6
|
+
macroWordErrorRate: number;
|
|
7
|
+
microWordErrorRate: number;
|
|
8
|
+
referenceWordCount: number;
|
|
9
|
+
sentenceErrorRate: number;
|
|
10
|
+
substitutions: number;
|
|
11
|
+
};
|
|
12
|
+
export type VoiceConfidenceInterval = {
|
|
13
|
+
confidenceLevel: number;
|
|
14
|
+
high: number;
|
|
15
|
+
low: number;
|
|
16
|
+
samples: number;
|
|
17
|
+
};
|
|
18
|
+
export type VoicePairedBootstrapComparison = {
|
|
19
|
+
baselineMean: number;
|
|
20
|
+
candidateMean: number;
|
|
21
|
+
delta: number;
|
|
22
|
+
deltaConfidenceInterval: VoiceConfidenceInterval;
|
|
23
|
+
probabilityCandidateIsBetter: number;
|
|
24
|
+
};
|
|
25
|
+
export declare const aggregateTranscriptAccuracy: (values: VoiceTranscriptAccuracy[]) => VoiceAggregateErrorMetrics;
|
|
26
|
+
export declare const comparePairedMetrics: (baseline: number[], candidate: number[], options?: {
|
|
27
|
+
confidenceLevel?: number;
|
|
28
|
+
samples?: number;
|
|
29
|
+
seed?: number;
|
|
30
|
+
}) => VoicePairedBootstrapComparison;
|