@alexkroman1/aai-ui 1.16.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/_module-url-C_4gRVL0.js +15 -0
- package/dist/audio.d.ts +85 -11
- package/dist/audio.js +112 -83
- package/dist/chat-view-C1oJxsWz.js +129 -0
- package/dist/client-config.d.ts +6 -6
- package/dist/components/chat-view.d.ts +1 -8
- package/dist/components/chat-view.js +2 -2
- package/dist/components/console-shell.d.ts +37 -0
- package/dist/components/controls.js +1 -1
- package/dist/components/url-chips.d.ts +1 -4
- package/dist/context.d.ts +1 -1
- package/dist/context.js +1 -4
- package/dist/{controls-BbZcmnJf.js → controls-DV368uhb.js} +2 -5
- package/dist/default-client/assets/_module-url-BX0RuRU2.js +1 -0
- package/dist/default-client/assets/audio-BGHiiDY_.js +1 -0
- package/dist/default-client/assets/capture-processor-BLPsqnGl.js +90 -0
- package/dist/default-client/assets/{index-BunwIXSP.js → index-DXf7-M0Q.js} +156 -36
- package/dist/default-client/assets/index-Dv-Q5VRL.css +2 -0
- package/dist/default-client/assets/playback-processor-B_T_1KQP.js +278 -0
- package/dist/default-client/index.html +2 -2
- package/dist/define-client-Cyf1jqAl.js +150 -0
- package/dist/define-client.d.ts +0 -9
- package/dist/define-client.js +1 -1
- package/dist/index.d.ts +2 -5
- package/dist/index.js +7 -5
- package/dist/{session-core-B64kau_v.js → session-core-BM3WHeoY.js} +36 -128
- package/dist/session-core-messages.d.ts +0 -4
- package/dist/session-core-types.d.ts +1 -27
- package/dist/session-core.js +1 -1
- package/dist/types.d.ts +24 -1
- package/dist/types.js +32 -2
- package/dist/worklets/_module-url.d.ts +10 -0
- package/dist/worklets/capture-processor.d.ts +3 -3
- package/dist/worklets/capture-processor.js +35 -52
- package/dist/worklets/playback-processor.d.ts +3 -3
- package/dist/worklets/playback-processor.js +149 -26
- package/package.json +2 -2
- package/dist/chat-view-gi6FccZq.js +0 -193
- package/dist/components/sync-chat-view.d.ts +0 -18
- package/dist/components/text-controls.d.ts +0 -14
- package/dist/default-client/assets/audio-DNDZgEZp.js +0 -1
- package/dist/default-client/assets/capture-processor-UlKEKyIW.js +0 -108
- package/dist/default-client/assets/index-BogmeUln.css +0 -2
- package/dist/default-client/assets/playback-processor-C5HVRVbu.js +0 -156
- package/dist/define-client-DdpijqAu.js +0 -906
- package/dist/session-core-upload.d.ts +0 -16
- package/dist/sync-mic.d.ts +0 -93
- package/dist/sync-session.d.ts +0 -49
- package/dist/sync-vad.d.ts +0 -54
|
@@ -1,8 +1,7 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { MIC_SEND_MAX_BUFFERED_BYTES } from "./types.js";
|
|
2
|
+
import { DEFAULT_MAX_HISTORY, WS_OPEN, errorMessage, safeJsonParse } from "@alexkroman1/aai";
|
|
2
3
|
import { ServerMessageSchema, lenientParse } from "@alexkroman1/aai/protocol";
|
|
3
|
-
import { DEFAULT_MAX_HISTORY, FILE_UPLOAD_CHUNK_BYTES, WS_OPEN, errorMessage, safeJsonParse } from "@alexkroman1/aai";
|
|
4
4
|
import ReconnectingWebSocket from "partysocket/ws";
|
|
5
|
-
import { MAX_SYNC_AUDIO_SECONDS } from "@alexkroman1/aai/stt";
|
|
6
5
|
//#region session-core-audio-setup.ts
|
|
7
6
|
/**
|
|
8
7
|
* Audio-path initialization for the voice session core.
|
|
@@ -36,6 +35,7 @@ async function initAudioCapture(conn, msg, deps, fatal) {
|
|
|
36
35
|
const gen = conn.generation;
|
|
37
36
|
const stale = () => conn.generation !== gen || !conn.ws || conn.ws.readyState !== WS_OPEN;
|
|
38
37
|
const reportAudioFailure = (message) => {
|
|
38
|
+
deps.cleanupAudio();
|
|
39
39
|
if (fatal) deps.updateState({
|
|
40
40
|
state: "error",
|
|
41
41
|
error: {
|
|
@@ -45,16 +45,13 @@ async function initAudioCapture(conn, msg, deps, fatal) {
|
|
|
45
45
|
running: false,
|
|
46
46
|
recording: false
|
|
47
47
|
});
|
|
48
|
-
else {
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
recording: false
|
|
56
|
-
});
|
|
57
|
-
}
|
|
48
|
+
else deps.updateState({
|
|
49
|
+
error: {
|
|
50
|
+
code: "audio",
|
|
51
|
+
message
|
|
52
|
+
},
|
|
53
|
+
recording: false
|
|
54
|
+
});
|
|
58
55
|
};
|
|
59
56
|
try {
|
|
60
57
|
const [{ createVoiceIO }, captureWorklet, playbackWorklet] = await Promise.all([
|
|
@@ -74,6 +71,12 @@ async function initAudioCapture(conn, msg, deps, fatal) {
|
|
|
74
71
|
console.debug("[aai-ui] sendAudio dropped: connection closed");
|
|
75
72
|
}
|
|
76
73
|
},
|
|
74
|
+
onPlaybackStats: (stats) => {
|
|
75
|
+
console.warn("[aai-ui] playback concealed a gap in this turn", stats);
|
|
76
|
+
},
|
|
77
|
+
onMicSilent: () => {
|
|
78
|
+
console.warn("[aai-ui] microphone is delivering only silence — check the selected input device");
|
|
79
|
+
},
|
|
77
80
|
onError: (err) => {
|
|
78
81
|
if (conn.generation !== gen) return;
|
|
79
82
|
reportAudioFailure(err.message);
|
|
@@ -149,7 +152,7 @@ function appendCapped(list, item, cap) {
|
|
|
149
152
|
* dedup) that previously lived as closure locals in `createSessionCore`.
|
|
150
153
|
*/
|
|
151
154
|
function createMessageHandlers(deps) {
|
|
152
|
-
const { getSnapshot, updateState, conn,
|
|
155
|
+
const { getSnapshot, updateState, conn, cleanupAudio } = deps;
|
|
153
156
|
/** Incremented on each turn boundary -- stale async callbacks compare against this. */
|
|
154
157
|
let handlerGeneration = 0;
|
|
155
158
|
/** Monotonically increasing counter for custom events -- used by useEvent to deduplicate. */
|
|
@@ -207,6 +210,7 @@ function createMessageHandlers(deps) {
|
|
|
207
210
|
} });
|
|
208
211
|
else {
|
|
209
212
|
cleanupAudio();
|
|
213
|
+
conn.generation++;
|
|
210
214
|
updateState({
|
|
211
215
|
state: "error",
|
|
212
216
|
error: {
|
|
@@ -274,7 +278,6 @@ function createMessageHandlers(deps) {
|
|
|
274
278
|
break;
|
|
275
279
|
case "reset":
|
|
276
280
|
handlerGeneration++;
|
|
277
|
-
discardUpload();
|
|
278
281
|
conn.voiceIO?.flush();
|
|
279
282
|
updateState({
|
|
280
283
|
...CLEARED_SESSION_STATE,
|
|
@@ -347,7 +350,6 @@ function createMessageHandlers(deps) {
|
|
|
347
350
|
if (msg.type === "config") return {
|
|
348
351
|
sampleRate: msg.sampleRate,
|
|
349
352
|
ttsSampleRate: msg.ttsSampleRate,
|
|
350
|
-
audioOut: msg.audioOut,
|
|
351
353
|
sid: msg.sessionId
|
|
352
354
|
};
|
|
353
355
|
if (msg.type === "audio_done") {
|
|
@@ -398,84 +400,6 @@ function reconnectPending(socket) {
|
|
|
398
400
|
return socket instanceof ReconnectingWebSocket && socket.shouldReconnect && socket.retryCount < RECONNECT_OPTIONS.maxRetries;
|
|
399
401
|
}
|
|
400
402
|
//#endregion
|
|
401
|
-
//#region session-core-upload.ts
|
|
402
|
-
/**
|
|
403
|
-
* File-upload transcription for the session core (`sendAudioFile`).
|
|
404
|
-
*
|
|
405
|
-
* Split out of `session-core.ts`: this module owns the decode → frame →
|
|
406
|
-
* reliable-send pipeline for uploaded audio, plus the two guards that keep
|
|
407
|
-
* it exclusive — an in-flight lock (no second upload, no mic start
|
|
408
|
-
* mid-upload) and an epoch that in-flight sends compare against so a
|
|
409
|
-
* reset/close/reconnect abandons them instead of streaming a stale clip
|
|
410
|
-
* into the fresh session.
|
|
411
|
-
*/
|
|
412
|
-
function createUploadSender(deps) {
|
|
413
|
-
const { conn, getSnapshot, sendJson } = deps;
|
|
414
|
-
/** True while a `sendAudioFile` upload is decoding or streaming. Blocks the
|
|
415
|
-
* mic (and a second upload) so file bytes never interleave with live audio. */
|
|
416
|
-
let uploadInFlight = false;
|
|
417
|
-
/** Bumped whenever conversation/connection state is discarded (reset, close,
|
|
418
|
-
* reconnect). An in-flight upload compares against it between chunks and
|
|
419
|
-
* aborts instead of streaming a stale clip into the fresh session. */
|
|
420
|
-
let uploadEpoch = 0;
|
|
421
|
-
/** Stream `bytes` to the socket in chunks, waiting out backpressure.
|
|
422
|
-
* Unlike live mic frames (dropped under backpressure — stale speech is
|
|
423
|
-
* worthless), file audio must arrive completely. Aborts if the connection
|
|
424
|
-
* closes or the session is reset (`epoch` moves on) between chunks. */
|
|
425
|
-
async function sendBytesReliably(bytes, epoch) {
|
|
426
|
-
for (let i = 0; i < bytes.byteLength; i += FILE_UPLOAD_CHUNK_BYTES) {
|
|
427
|
-
while (epoch === uploadEpoch && conn.ws && conn.ws.readyState === WS_OPEN && conn.ws.bufferedAmount > MIC_SEND_MAX_BUFFERED_BYTES) await new Promise((r) => setTimeout(r, FILE_SEND_BACKOFF_MS));
|
|
428
|
-
if (epoch !== uploadEpoch) throw new Error("sendAudioFile: session was reset mid-send");
|
|
429
|
-
if (!conn.ws || conn.ws.readyState !== WS_OPEN) throw new Error("sendAudioFile: connection closed mid-send");
|
|
430
|
-
conn.ws.send(bytes.subarray(i, i + FILE_UPLOAD_CHUNK_BYTES));
|
|
431
|
-
}
|
|
432
|
-
}
|
|
433
|
-
/** Validate that an upload may start; returns the session's ready config. */
|
|
434
|
-
function assertUploadReady() {
|
|
435
|
-
const cfg = conn.readyConfig;
|
|
436
|
-
if (!(cfg && conn.ws) || conn.ws.readyState !== WS_OPEN) throw new Error("sendAudioFile: session is not connected");
|
|
437
|
-
const snap = getSnapshot();
|
|
438
|
-
if (snap.audioOut) throw new Error("sendAudioFile is only available in text-only sessions (tts: none()) — voice sessions stream the microphone instead");
|
|
439
|
-
if (snap.recording) throw new Error("sendAudioFile: stop recording before uploading a file");
|
|
440
|
-
if (uploadInFlight) throw new Error("sendAudioFile: another upload is already in progress");
|
|
441
|
-
return cfg;
|
|
442
|
-
}
|
|
443
|
-
async function sendAudioFile(file) {
|
|
444
|
-
const cfg = assertUploadReady();
|
|
445
|
-
uploadInFlight = true;
|
|
446
|
-
const epoch = uploadEpoch;
|
|
447
|
-
try {
|
|
448
|
-
const { decodeAudioToPcm16 } = await import("./audio.js");
|
|
449
|
-
const clip = await decodeAudioToPcm16(await file.arrayBuffer(), cfg.sampleRate);
|
|
450
|
-
if (getSnapshot().recording || conn.audioSetupInFlight) throw new Error("sendAudioFile: stop recording before uploading a file");
|
|
451
|
-
if (epoch !== uploadEpoch) throw new Error("sendAudioFile: session was reset mid-send");
|
|
452
|
-
if (clip.length / cfg.sampleRate <= MAX_SYNC_AUDIO_SECONDS) {
|
|
453
|
-
const bytes = new Uint8Array(clip.buffer, clip.byteOffset, clip.byteLength);
|
|
454
|
-
sendJson({
|
|
455
|
-
type: "transcribe_file_start",
|
|
456
|
-
sampleRate: cfg.sampleRate,
|
|
457
|
-
byteLength: bytes.byteLength
|
|
458
|
-
});
|
|
459
|
-
await sendBytesReliably(bytes, epoch);
|
|
460
|
-
sendJson({ type: "transcribe_file_end" });
|
|
461
|
-
return;
|
|
462
|
-
}
|
|
463
|
-
const padded = new Int16Array(clip.length + cfg.sampleRate);
|
|
464
|
-
padded.set(clip);
|
|
465
|
-
await sendBytesReliably(new Uint8Array(padded.buffer), epoch);
|
|
466
|
-
} finally {
|
|
467
|
-
uploadInFlight = false;
|
|
468
|
-
}
|
|
469
|
-
}
|
|
470
|
-
return {
|
|
471
|
-
sendAudioFile,
|
|
472
|
-
inFlight: () => uploadInFlight,
|
|
473
|
-
discard: () => {
|
|
474
|
-
uploadEpoch++;
|
|
475
|
-
}
|
|
476
|
-
};
|
|
477
|
-
}
|
|
478
|
-
//#endregion
|
|
479
403
|
//#region session-core-url.ts
|
|
480
404
|
/** Build the session WebSocket URL from the platform URL and resume state. */
|
|
481
405
|
function buildWsUrl(platformUrl, resume, sessionId) {
|
|
@@ -518,7 +442,6 @@ function createSessionCore(options) {
|
|
|
518
442
|
contentVersion: 0,
|
|
519
443
|
started: false,
|
|
520
444
|
running: false,
|
|
521
|
-
audioOut: true,
|
|
522
445
|
recording: false,
|
|
523
446
|
apiUrl: buildWsUrl(options.platformUrl, false).toString()
|
|
524
447
|
};
|
|
@@ -564,8 +487,16 @@ function createSessionCore(options) {
|
|
|
564
487
|
};
|
|
565
488
|
let connectionController = null;
|
|
566
489
|
let hasConnected = false;
|
|
490
|
+
/**
|
|
491
|
+
* The session ID to resume: seeded from `options.resumeSessionId`, then
|
|
492
|
+
* kept current from every `config` frame. Reconnect URLs carry it as
|
|
493
|
+
* `?sessionId=<id>` so the server re-registers the SAME session id —
|
|
494
|
+
* that key is what per-session tool state (`ctx.state`) lives under, so
|
|
495
|
+
* a reconnect that omits it gets a fresh session with none of the
|
|
496
|
+
* agent's context, greeting suppression aside.
|
|
497
|
+
*/
|
|
498
|
+
let sessionId = options.resumeSessionId;
|
|
567
499
|
function cleanupAudio() {
|
|
568
|
-
upload.discard();
|
|
569
500
|
conn.audioSetupInFlight = false;
|
|
570
501
|
conn.voiceIO?.close().catch(() => {});
|
|
571
502
|
conn.voiceIO = null;
|
|
@@ -583,16 +514,10 @@ function createSessionCore(options) {
|
|
|
583
514
|
if (conn.ws.bufferedAmount > MIC_SEND_MAX_BUFFERED_BYTES) return;
|
|
584
515
|
conn.ws.send(bytes);
|
|
585
516
|
}
|
|
586
|
-
const upload = createUploadSender({
|
|
587
|
-
conn,
|
|
588
|
-
getSnapshot,
|
|
589
|
-
sendJson
|
|
590
|
-
});
|
|
591
517
|
const { handleMessage, settleWhenAudioDrained } = createMessageHandlers({
|
|
592
518
|
getSnapshot,
|
|
593
519
|
updateState,
|
|
594
520
|
conn,
|
|
595
|
-
discardUpload: upload.discard,
|
|
596
521
|
cleanupAudio
|
|
597
522
|
});
|
|
598
523
|
const audioDeps = {
|
|
@@ -613,20 +538,17 @@ function createSessionCore(options) {
|
|
|
613
538
|
/** React to the server's `config` message: record it, set up the audio
|
|
614
539
|
* path for the session's mode, and replay history on reconnect. */
|
|
615
540
|
function onServerConfig(config) {
|
|
616
|
-
if (config.sid)
|
|
541
|
+
if (config.sid) {
|
|
542
|
+
sessionId = config.sid;
|
|
543
|
+
options.onSessionId?.(config.sid);
|
|
544
|
+
}
|
|
617
545
|
const isReconnect = hasConnected;
|
|
618
546
|
hasConnected = true;
|
|
619
547
|
conn.readyConfig = {
|
|
620
548
|
sampleRate: config.sampleRate,
|
|
621
549
|
ttsSampleRate: config.ttsSampleRate
|
|
622
550
|
};
|
|
623
|
-
|
|
624
|
-
updateState({ audioOut });
|
|
625
|
-
if (audioOut) initAudioCapture(conn, config, audioDeps, true);
|
|
626
|
-
else {
|
|
627
|
-
sendJson({ type: "audio_ready" });
|
|
628
|
-
updateState({ state: "listening" });
|
|
629
|
-
}
|
|
551
|
+
initAudioCapture(conn, config, audioDeps, true);
|
|
630
552
|
if (isReconnect && currentSnapshot.messages.length > 0) sendJson({
|
|
631
553
|
type: "history",
|
|
632
554
|
messages: currentSnapshot.messages.map((m) => ({
|
|
@@ -639,11 +561,12 @@ function createSessionCore(options) {
|
|
|
639
561
|
* The WebSocket URL for the *next* connection attempt. Evaluated per
|
|
640
562
|
* attempt (partysocket takes it as a URL provider), so once the first
|
|
641
563
|
* `config` arrives, every reconnect — automatic or explicit — carries
|
|
642
|
-
*
|
|
564
|
+
* `?sessionId=<id>` and the server resumes the SAME session (id, tool
|
|
565
|
+
* state) instead of minting a new one. `resume=1` remains only as the
|
|
566
|
+
* greeting-suppression fallback for a server whose config carried no id.
|
|
643
567
|
*/
|
|
644
568
|
function currentWsUrl() {
|
|
645
|
-
|
|
646
|
-
return buildWsUrl(options.platformUrl, hasConnected, resumeId).toString();
|
|
569
|
+
return buildWsUrl(options.platformUrl, hasConnected, sessionId).toString();
|
|
647
570
|
}
|
|
648
571
|
/** Open a socket: an injected constructor as-is (tests), or partysocket's
|
|
649
572
|
* reconnecting WebSocket — same interface, plus reconnect-on-close. */
|
|
@@ -723,7 +646,6 @@ function createSessionCore(options) {
|
|
|
723
646
|
sendJson({ type: "cancel" });
|
|
724
647
|
}
|
|
725
648
|
function reset() {
|
|
726
|
-
upload.discard();
|
|
727
649
|
conn.voiceIO?.flush();
|
|
728
650
|
if (conn.ws && conn.ws.readyState === WS_OPEN) {
|
|
729
651
|
sendJson({ type: "reset" });
|
|
@@ -742,17 +664,6 @@ function createSessionCore(options) {
|
|
|
742
664
|
recording: false
|
|
743
665
|
});
|
|
744
666
|
}
|
|
745
|
-
function startRecording() {
|
|
746
|
-
if (currentSnapshot.audioOut || currentSnapshot.recording || conn.audioSetupInFlight || upload.inFlight()) return;
|
|
747
|
-
const cfg = conn.readyConfig;
|
|
748
|
-
if (!(cfg && conn.ws) || conn.ws.readyState !== WS_OPEN) return;
|
|
749
|
-
initAudioCapture(conn, cfg, audioDeps, false);
|
|
750
|
-
}
|
|
751
|
-
function stopRecording() {
|
|
752
|
-
if (currentSnapshot.audioOut || !currentSnapshot.recording) return;
|
|
753
|
-
cleanupAudio();
|
|
754
|
-
updateState({ recording: false });
|
|
755
|
-
}
|
|
756
667
|
function start() {
|
|
757
668
|
updateState({
|
|
758
669
|
started: true,
|
|
@@ -777,9 +688,6 @@ function createSessionCore(options) {
|
|
|
777
688
|
disconnect,
|
|
778
689
|
start,
|
|
779
690
|
toggle,
|
|
780
|
-
startRecording,
|
|
781
|
-
stopRecording,
|
|
782
|
-
sendAudioFile: upload.sendAudioFile,
|
|
783
691
|
[Symbol.dispose]() {
|
|
784
692
|
disconnect();
|
|
785
693
|
}
|
|
@@ -17,8 +17,6 @@ export declare const CLEARED_SESSION_STATE: {
|
|
|
17
17
|
export type SessionConfigMessage = {
|
|
18
18
|
sampleRate: number;
|
|
19
19
|
ttsSampleRate: number;
|
|
20
|
-
/** False for text-only agents (`tts: none()`) — see ReadyConfigSchema. */
|
|
21
|
-
audioOut?: boolean | undefined;
|
|
22
20
|
sid?: string | undefined;
|
|
23
21
|
};
|
|
24
22
|
/** Dependencies the message handlers need from the owning session core. */
|
|
@@ -26,8 +24,6 @@ type MessageHandlerDeps = {
|
|
|
26
24
|
getSnapshot: () => SessionSnapshot;
|
|
27
25
|
updateState: (partial: Partial<SessionSnapshot>) => void;
|
|
28
26
|
conn: ConnState;
|
|
29
|
-
/** Invalidate any in-flight file upload (the upload sender's `discard`). */
|
|
30
|
-
discardUpload: () => void;
|
|
31
27
|
/** Release the microphone/VoiceIO (the session core's `cleanupAudio`). */
|
|
32
28
|
cleanupAudio: () => void;
|
|
33
29
|
};
|
|
@@ -27,17 +27,7 @@ export type CustomEvent = {
|
|
|
27
27
|
*/
|
|
28
28
|
export type SessionSnapshot = {
|
|
29
29
|
readonly state: AgentState;
|
|
30
|
-
/**
|
|
31
|
-
* False when the server declared the session text-only (`tts: none()`):
|
|
32
|
-
* no audio frames will arrive and replies render as text. True until the
|
|
33
|
-
* server's `config` message says otherwise (voice is the default).
|
|
34
|
-
*/
|
|
35
|
-
readonly audioOut: boolean;
|
|
36
|
-
/**
|
|
37
|
-
* True while the microphone is live and streaming to the server. Voice
|
|
38
|
-
* sessions record for their whole lifetime; text-only sessions toggle this
|
|
39
|
-
* via `startRecording()` / `stopRecording()` (the record button).
|
|
40
|
-
*/
|
|
30
|
+
/** True while the microphone is live and streaming to the server. */
|
|
41
31
|
readonly recording: boolean;
|
|
42
32
|
/**
|
|
43
33
|
* The WebSocket URL a program can connect to directly (the same endpoint
|
|
@@ -94,22 +84,6 @@ export type SessionCore = {
|
|
|
94
84
|
start(): void;
|
|
95
85
|
/** Toggle between connected and disconnected states. */
|
|
96
86
|
toggle(): void;
|
|
97
|
-
/**
|
|
98
|
-
* Start streaming microphone audio (text-only sessions). Requests mic
|
|
99
|
-
* access on first use. No-op in voice sessions, where the mic is always on.
|
|
100
|
-
*/
|
|
101
|
-
startRecording(): void;
|
|
102
|
-
/** Stop streaming microphone audio (text-only sessions). No-op in voice sessions. */
|
|
103
|
-
stopRecording(): void;
|
|
104
|
-
/**
|
|
105
|
-
* Decode an audio file (any format the browser can decode), resample it to
|
|
106
|
-
* the session's STT rate, and stream it to the server for transcription.
|
|
107
|
-
* Text-only sessions (`tts: none()`) only — voice sessions stream the
|
|
108
|
-
* microphone instead and reject. Also rejects when no session is connected,
|
|
109
|
-
* the mic is recording, another upload is in flight, or the file cannot be
|
|
110
|
-
* decoded. Resolves once the audio has been handed to the socket.
|
|
111
|
-
*/
|
|
112
|
-
sendAudioFile(file: Blob): Promise<void>;
|
|
113
87
|
/** Alias for `disconnect` for use with `using`. */
|
|
114
88
|
[Symbol.dispose](): void;
|
|
115
89
|
};
|
package/dist/session-core.js
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
import { t as createSessionCore } from "./session-core-
|
|
1
|
+
import { t as createSessionCore } from "./session-core-BM3WHeoY.js";
|
|
2
2
|
export { createSessionCore };
|
package/dist/types.d.ts
CHANGED
|
@@ -1,5 +1,28 @@
|
|
|
1
1
|
import type { SessionErrorCode } from "@alexkroman1/aai/protocol";
|
|
2
|
-
export {
|
|
2
|
+
export { CAPTURE_STOP_ACK_TIMEOUT_MS, DEFAULT_STT_SAMPLE_RATE, MIC_BUFFER_SECONDS, MIC_SEND_MAX_BUFFERED_BYTES, MIC_SILENCE_PROBE_MS, PLAYBACK_BUFFER_SECONDS, PLAYBACK_CONCEAL_FADE_MS, PLAYBACK_CONCEAL_FLOOR, PLAYBACK_DONE_MAX_WAIT_MS, PLAYBACK_DONE_POLL_MS, PLAYBACK_JITTER_MS, PLAYBACK_REFILL_MS, } from "@alexkroman1/aai";
|
|
3
|
+
/**
|
|
4
|
+
* `getUserMedia` audio constraints for every capture path in this package.
|
|
5
|
+
*
|
|
6
|
+
* Defined once because four copies of this object drifted apart trivially, and
|
|
7
|
+
* the flags are not cosmetic — each one rewrites the signal before STT (and
|
|
8
|
+
* before the sync path's energy VAD) ever sees it:
|
|
9
|
+
*
|
|
10
|
+
* - **`autoGainControl: false`** — AGC continuously retargets level, which
|
|
11
|
+
* means riding the noise floor up through silence. An energy VAD calibrated
|
|
12
|
+
* against a moving floor is calibrated against nothing.
|
|
13
|
+
* - **`noiseSuppression: false`** / **`voiceIsolation: false`** — both discard
|
|
14
|
+
* signal to make speech sound cleaner to a human, and both can gate a quiet
|
|
15
|
+
* room to *exact* zeros, which is also what a dead microphone looks like
|
|
16
|
+
* (see `MIC_SILENCE_PROBE_MS`).
|
|
17
|
+
* - **`echoCancellation: true`** — this one stays on. The mic is open while
|
|
18
|
+
* the agent speaks (barge-in needs it), so without AEC the agent hears
|
|
19
|
+
* itself and interrupts its own reply.
|
|
20
|
+
*
|
|
21
|
+
* Cast because `voiceIsolation` is newer than TypeScript's DOM lib.
|
|
22
|
+
*
|
|
23
|
+
* @public
|
|
24
|
+
*/
|
|
25
|
+
export declare const VOICE_CAPTURE_CONSTRAINTS: MediaTrackConstraints;
|
|
3
26
|
/**
|
|
4
27
|
* Current state of the voice agent session.
|
|
5
28
|
*
|
package/dist/types.js
CHANGED
|
@@ -1,2 +1,32 @@
|
|
|
1
|
-
import {
|
|
2
|
-
|
|
1
|
+
import { CAPTURE_STOP_ACK_TIMEOUT_MS, DEFAULT_STT_SAMPLE_RATE, MIC_BUFFER_SECONDS, MIC_SEND_MAX_BUFFERED_BYTES, MIC_SILENCE_PROBE_MS, PLAYBACK_BUFFER_SECONDS, PLAYBACK_CONCEAL_FADE_MS, PLAYBACK_CONCEAL_FLOOR, PLAYBACK_DONE_MAX_WAIT_MS, PLAYBACK_DONE_POLL_MS, PLAYBACK_JITTER_MS, PLAYBACK_REFILL_MS } from "@alexkroman1/aai";
|
|
2
|
+
//#region types.ts
|
|
3
|
+
/**
|
|
4
|
+
* `getUserMedia` audio constraints for every capture path in this package.
|
|
5
|
+
*
|
|
6
|
+
* Defined once because four copies of this object drifted apart trivially, and
|
|
7
|
+
* the flags are not cosmetic — each one rewrites the signal before STT (and
|
|
8
|
+
* before the sync path's energy VAD) ever sees it:
|
|
9
|
+
*
|
|
10
|
+
* - **`autoGainControl: false`** — AGC continuously retargets level, which
|
|
11
|
+
* means riding the noise floor up through silence. An energy VAD calibrated
|
|
12
|
+
* against a moving floor is calibrated against nothing.
|
|
13
|
+
* - **`noiseSuppression: false`** / **`voiceIsolation: false`** — both discard
|
|
14
|
+
* signal to make speech sound cleaner to a human, and both can gate a quiet
|
|
15
|
+
* room to *exact* zeros, which is also what a dead microphone looks like
|
|
16
|
+
* (see `MIC_SILENCE_PROBE_MS`).
|
|
17
|
+
* - **`echoCancellation: true`** — this one stays on. The mic is open while
|
|
18
|
+
* the agent speaks (barge-in needs it), so without AEC the agent hears
|
|
19
|
+
* itself and interrupts its own reply.
|
|
20
|
+
*
|
|
21
|
+
* Cast because `voiceIsolation` is newer than TypeScript's DOM lib.
|
|
22
|
+
*
|
|
23
|
+
* @public
|
|
24
|
+
*/
|
|
25
|
+
const VOICE_CAPTURE_CONSTRAINTS = {
|
|
26
|
+
echoCancellation: true,
|
|
27
|
+
noiseSuppression: false,
|
|
28
|
+
autoGainControl: false,
|
|
29
|
+
voiceIsolation: false
|
|
30
|
+
};
|
|
31
|
+
//#endregion
|
|
32
|
+
export { CAPTURE_STOP_ACK_TIMEOUT_MS, DEFAULT_STT_SAMPLE_RATE, MIC_BUFFER_SECONDS, MIC_SEND_MAX_BUFFERED_BYTES, MIC_SILENCE_PROBE_MS, PLAYBACK_BUFFER_SECONDS, PLAYBACK_CONCEAL_FADE_MS, PLAYBACK_CONCEAL_FLOOR, PLAYBACK_DONE_MAX_WAIT_MS, PLAYBACK_DONE_POLL_MS, PLAYBACK_JITTER_MS, PLAYBACK_REFILL_MS, VOICE_CAPTURE_CONSTRAINTS };
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Blob-URL module for an inlined AudioWorklet processor source.
|
|
3
|
+
*
|
|
4
|
+
* A blob URL rather than a data URI because the agent page's CSP allows
|
|
5
|
+
* `script-src blob:` but not `data:` — a data-URI module fails `addModule`
|
|
6
|
+
* with the opaque "Unable to load a worklet's module". Every inline worklet
|
|
7
|
+
* in this package must load through this helper so that constraint lives in
|
|
8
|
+
* one place.
|
|
9
|
+
*/
|
|
10
|
+
export declare function workletModuleUrl(source: string): string;
|
|
@@ -1,4 +1,4 @@
|
|
|
1
1
|
/** Raw worklet source — exported so tests can evaluate the processor directly. */
|
|
2
|
-
export declare const captureProcessorSource = "\nclass CaptureProcessor extends AudioWorkletProcessor {\n constructor(options) {\n super();\n this.recording = false;\n const opts = options.processorOptions || {};\n
|
|
3
|
-
declare const
|
|
4
|
-
export default
|
|
2
|
+
export declare const captureProcessorSource = "\nclass CaptureProcessor extends AudioWorkletProcessor {\n constructor(options) {\n super();\n this.recording = false;\n const opts = options.processorOptions || {};\n // The context runs at the STT rate, so this is both the input and the\n // output rate \u2014 there is nothing to convert.\n this.rate = opts.sampleRate || sampleRate;\n // Int16 accumulation buffer: flushed to the main thread as one transferred\n // ArrayBuffer once ~bufferSeconds of samples are batched. Sized 2x the\n // flush target so a whole render quantum always fits before flushing.\n this.targetSamples = Math.max(1, Math.round(this.rate * (opts.bufferSeconds || 0.1)));\n this.pending = new Int16Array(this.targetSamples * 2);\n this.pendingLen = 0;\n // Dead-mic probe: samples left to inspect before concluding the device\n // delivers nothing but digital silence. Only consumed while recording, so\n // the cost disappears after the window (or after the first real sample).\n this.probeSamplesLeft = Math.round(\n (this.rate * (opts.silenceProbeMs ?? 1500)) / 1000,\n );\n this.port.onmessage = (e) => {\n if (e.data.event === 'start') this.recording = true;\n else if (e.data.event === 'stop') {\n // Final flush so the tail of speech isn't dropped on close, then ack\n // so the host knows the tail chunk (if any) has been posted and it is\n // safe to tear the context down.\n this.flush();\n this.recording = false;\n this.port.postMessage({ event: 'stopped' });\n }\n };\n }\n\n // Convert Float32 -> Int16 and append to the pending batch. Writes through\n // an Int16Array directly (assignment truncates like DataView.setInt16).\n accumulate(samples) {\n let buf = this.pending;\n if (this.pendingLen + samples.length > buf.length) {\n // Defensive: only reachable if a render quantum outproduces the 1x\n // headroom above the flush target (never with 128-sample quanta).\n const grown = new Int16Array((this.pendingLen + samples.length) * 2);\n grown.set(buf.subarray(0, this.pendingLen));\n this.pending = grown;\n buf = grown;\n }\n for (let i = 0; i < samples.length; i++) {\n const s = Math.max(-1, Math.min(1, samples[i]));\n buf[this.pendingLen++] = s < 0 ? s * 0x8000 : s * 0x7fff;\n }\n }\n\n // Post the batched samples as one transferred ArrayBuffer and reset.\n flush() {\n if (this.pendingLen === 0) return;\n const buffer = this.pending.buffer.slice(0, this.pendingLen * 2);\n this.pendingLen = 0;\n this.port.postMessage({ event: 'chunk', buffer }, [buffer]);\n }\n\n // Watch the first window of input for any nonzero sample. One is enough to\n // prove the device is live \u2014 a real mic in a quiet room still carries a\n // noise floor, so all-zeros means muted, wrong input, or no input at all.\n probeForSilence(channel) {\n for (let i = 0; i < channel.length; i++) {\n if (channel[i] !== 0) {\n this.probeSamplesLeft = 0;\n return;\n }\n }\n this.probeSamplesLeft -= channel.length;\n if (this.probeSamplesLeft <= 0) {\n this.port.postMessage({ event: 'silent' });\n }\n }\n\n process(inputs) {\n const input = inputs[0];\n if (!input || !input[0] || !this.recording) return true;\n\n if (this.probeSamplesLeft > 0) this.probeForSilence(input[0]);\n\n this.accumulate(input[0]);\n if (this.pendingLen >= this.targetSamples) this.flush();\n return true;\n }\n}\n\nregisterProcessor('capture-processor', CaptureProcessor);\n";
|
|
3
|
+
declare const _default: string;
|
|
4
|
+
export default _default;
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { MIC_BUFFER_SECONDS, MIC_SILENCE_PROBE_MS } from "../types.js";
|
|
2
|
+
import { t as workletModuleUrl } from "../_module-url-C_4gRVL0.js";
|
|
1
3
|
//#region worklets/capture-processor.ts
|
|
2
4
|
const CaptureProcessorWorklet = `
|
|
3
5
|
class CaptureProcessor extends AudioWorkletProcessor {
|
|
@@ -5,27 +7,21 @@ class CaptureProcessor extends AudioWorkletProcessor {
|
|
|
5
7
|
super();
|
|
6
8
|
this.recording = false;
|
|
7
9
|
const opts = options.processorOptions || {};
|
|
8
|
-
this
|
|
9
|
-
|
|
10
|
-
this.
|
|
11
|
-
this.needsResample = this.fromRate !== this.toRate;
|
|
12
|
-
// Streaming-resampler state carried across process() blocks so the output
|
|
13
|
-
// clock doesn't drift and there's no discontinuity at 128-sample block
|
|
14
|
-
// boundaries. \`prev\` is the previous block's last input sample (used to
|
|
15
|
-
// interpolate across the boundary); \`pos\` is the fractional read position,
|
|
16
|
-
// in input samples, relative to an extended [prev, ...input] frame. Start
|
|
17
|
-
// at 1 so the first output is input[0] (not the bogus initial prev).
|
|
18
|
-
this.prev = 0;
|
|
19
|
-
this.pos = 1;
|
|
20
|
-
// Output buffer reused across process() calls -- the render quantum is a
|
|
21
|
-
// fixed size, so allocating per call would churn the realtime audio thread.
|
|
22
|
-
this.resampleBuf = null;
|
|
10
|
+
// The context runs at the STT rate, so this is both the input and the
|
|
11
|
+
// output rate — there is nothing to convert.
|
|
12
|
+
this.rate = opts.sampleRate || sampleRate;
|
|
23
13
|
// Int16 accumulation buffer: flushed to the main thread as one transferred
|
|
24
|
-
// ArrayBuffer once ~bufferSeconds of
|
|
25
|
-
//
|
|
26
|
-
this.targetSamples = Math.max(1, Math.round(this.
|
|
14
|
+
// ArrayBuffer once ~bufferSeconds of samples are batched. Sized 2x the
|
|
15
|
+
// flush target so a whole render quantum always fits before flushing.
|
|
16
|
+
this.targetSamples = Math.max(1, Math.round(this.rate * (opts.bufferSeconds || ${MIC_BUFFER_SECONDS})));
|
|
27
17
|
this.pending = new Int16Array(this.targetSamples * 2);
|
|
28
18
|
this.pendingLen = 0;
|
|
19
|
+
// Dead-mic probe: samples left to inspect before concluding the device
|
|
20
|
+
// delivers nothing but digital silence. Only consumed while recording, so
|
|
21
|
+
// the cost disappears after the window (or after the first real sample).
|
|
22
|
+
this.probeSamplesLeft = Math.round(
|
|
23
|
+
(this.rate * (opts.silenceProbeMs ?? ${MIC_SILENCE_PROBE_MS})) / 1000,
|
|
24
|
+
);
|
|
29
25
|
this.port.onmessage = (e) => {
|
|
30
26
|
if (e.data.event === 'start') this.recording = true;
|
|
31
27
|
else if (e.data.event === 'stop') {
|
|
@@ -39,35 +35,6 @@ class CaptureProcessor extends AudioWorkletProcessor {
|
|
|
39
35
|
};
|
|
40
36
|
}
|
|
41
37
|
|
|
42
|
-
resample(input) {
|
|
43
|
-
const ratio = this.ratio;
|
|
44
|
-
const n = input.length;
|
|
45
|
-
if (n === 0) return new Float32Array(0);
|
|
46
|
-
// At most ceil(n / ratio) + 1 outputs per block (pos starts in [0, ratio)).
|
|
47
|
-
const maxLen = Math.ceil(n / ratio) + 1;
|
|
48
|
-
if (this.resampleBuf === null || this.resampleBuf.length < maxLen) {
|
|
49
|
-
this.resampleBuf = new Float32Array(maxLen);
|
|
50
|
-
}
|
|
51
|
-
const out = this.resampleBuf;
|
|
52
|
-
// Extended frame: index 0 = prev (previous block's last sample),
|
|
53
|
-
// index k>=1 = input[k-1]. Interpolate at fractional positions stepping by
|
|
54
|
-
// ratio, carrying \`pos\` across calls to preserve the sample clock.
|
|
55
|
-
let count = 0;
|
|
56
|
-
let pos = this.pos;
|
|
57
|
-
while (pos < n) {
|
|
58
|
-
const idx = pos | 0;
|
|
59
|
-
const frac = pos - idx;
|
|
60
|
-
const a = idx === 0 ? this.prev : input[idx - 1];
|
|
61
|
-
const b = input[idx];
|
|
62
|
-
out[count++] = a + frac * (b - a);
|
|
63
|
-
pos += ratio;
|
|
64
|
-
}
|
|
65
|
-
// Shift the origin to this block's last sample for the next call.
|
|
66
|
-
this.prev = input[n - 1];
|
|
67
|
-
this.pos = pos - n;
|
|
68
|
-
return out.subarray(0, count);
|
|
69
|
-
}
|
|
70
|
-
|
|
71
38
|
// Convert Float32 -> Int16 and append to the pending batch. Writes through
|
|
72
39
|
// an Int16Array directly (assignment truncates like DataView.setInt16).
|
|
73
40
|
accumulate(samples) {
|
|
@@ -94,12 +61,29 @@ class CaptureProcessor extends AudioWorkletProcessor {
|
|
|
94
61
|
this.port.postMessage({ event: 'chunk', buffer }, [buffer]);
|
|
95
62
|
}
|
|
96
63
|
|
|
64
|
+
// Watch the first window of input for any nonzero sample. One is enough to
|
|
65
|
+
// prove the device is live — a real mic in a quiet room still carries a
|
|
66
|
+
// noise floor, so all-zeros means muted, wrong input, or no input at all.
|
|
67
|
+
probeForSilence(channel) {
|
|
68
|
+
for (let i = 0; i < channel.length; i++) {
|
|
69
|
+
if (channel[i] !== 0) {
|
|
70
|
+
this.probeSamplesLeft = 0;
|
|
71
|
+
return;
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
this.probeSamplesLeft -= channel.length;
|
|
75
|
+
if (this.probeSamplesLeft <= 0) {
|
|
76
|
+
this.port.postMessage({ event: 'silent' });
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
|
|
97
80
|
process(inputs) {
|
|
98
81
|
const input = inputs[0];
|
|
99
82
|
if (!input || !input[0] || !this.recording) return true;
|
|
100
83
|
|
|
101
|
-
|
|
102
|
-
|
|
84
|
+
if (this.probeSamplesLeft > 0) this.probeForSilence(input[0]);
|
|
85
|
+
|
|
86
|
+
this.accumulate(input[0]);
|
|
103
87
|
if (this.pendingLen >= this.targetSamples) this.flush();
|
|
104
88
|
return true;
|
|
105
89
|
}
|
|
@@ -109,7 +93,6 @@ registerProcessor('capture-processor', CaptureProcessor);
|
|
|
109
93
|
`;
|
|
110
94
|
/** Raw worklet source — exported so tests can evaluate the processor directly. */
|
|
111
95
|
const captureProcessorSource = CaptureProcessorWorklet;
|
|
112
|
-
|
|
113
|
-
const src = URL.createObjectURL(script);
|
|
96
|
+
var capture_processor_default = workletModuleUrl(CaptureProcessorWorklet);
|
|
114
97
|
//#endregion
|
|
115
|
-
export { captureProcessorSource,
|
|
98
|
+
export { captureProcessorSource, capture_processor_default as default };
|
|
@@ -1,4 +1,4 @@
|
|
|
1
1
|
/** Raw worklet source — exported so tests can evaluate the processor directly. */
|
|
2
|
-
export declare const playbackProcessorSource = "\nclass PlaybackProcessor extends AudioWorkletProcessor {\n constructor(options) {\n super();\n const rate = options.processorOptions?.sampleRate ?? 24000;\n // Wait for ~400ms of audio before starting.\n // If 'done' arrives first (short utterance), start immediately.\n this.jitterSamples = Math.floor(rate * 0.4);\n // Float32 ring buffer \u2014 60s at the context sample rate. Allocated once for\n // the node's lifetime; per-turn state resets via resetTurn(). writePos and\n // readPos are absolute (monotonic) sample counts; the buffer is indexed\n // modulo capacity so a reply longer than 60s keeps playing instead of\n // writing past the end and going silent.\n this.capacity = rate * 60;\n this.samples = new Float32Array(this.capacity);\n // Platform endianness probe: the wire format is PCM16 little-endian, so\n // the Int16Array fast path in ingestBytes is only valid on LE hosts\n // (every shipping browser target; the DataView path is the fallback).\n this.littleEndian = new Uint8Array(new Uint16Array([1]).buffer)[0] === 1;\n this.resetTurn();\n\n this.port.onmessage = (e) => {\n const d = e.data;\n if (d.event === 'write') {\n this.ingestBytes(d.buffer);\n } else if (d.event === 'interrupt') {\n this.interrupted = true;\n } else if (d.event === 'done') {\n this.isDone = true;\n }\n };\n }\n\n // Reset per-turn state so the node is reusable across replies without\n // reallocating the sample buffer or re-instantiating the worklet.\n resetTurn() {\n this.interrupted = false;\n this.isDone = false;\n this.playing = false;\n // Carry-over byte for split samples across chunks\n this.carry = null;\n this.writePos = 0;\n this.readPos = 0;\n }\n\n // End the current turn: notify the host and rearm for the next reply.\n // Must NOT return false from process() \u2014 a processor that stops is dead\n // for good, forcing a new node (and buffer) per reply.\n // `reason` ('interrupt' | 'done') tells the host which turn boundary this\n // stop belongs to: interrupt-stops are dropped host-side (flush() already\n // settled that turn), so they can never resolve a later turn's done() early.\n stopTurn(reason) {\n this.port.postMessage({ event: 'stop', reason });\n this.resetTurn();\n }\n\n ingestBytes(uint8) {\n let bytes = uint8;\n\n if (this.carry !== null) {\n const merged = new Uint8Array(1 + bytes.length);\n merged[0] = this.carry;\n merged.set(bytes, 1);\n bytes = merged;\n this.carry = null;\n }\n\n if (bytes.length % 2 !== 0) {\n this.carry = bytes[bytes.length - 1];\n bytes = bytes.subarray(0, bytes.length - 1);\n }\n\n if (bytes.length === 0) return;\n const numSamples = bytes.length / 2;\n const cap = this.capacity;\n const samples = this.samples;\n if (this.littleEndian && (bytes.byteOffset & 1) === 0) {\n // Fast path: 2-byte-aligned LE bytes wrap directly as an Int16Array;\n // copy wrap-aware in at most two runs with no per-sample DataView call\n // or modulo. This runs on the realtime audio thread.\n const int16 = new Int16Array(bytes.buffer, bytes.byteOffset, numSamples);\n let src = 0;\n let dst = this.writePos % cap;\n while (src < numSamples) {\n const run = Math.min(numSamples - src, cap - dst);\n for (let j = 0; j < run; j++) {\n samples[dst + j] = int16[src + j] / 0x8000;\n }\n src += run;\n dst = 0;\n }\n } else {\n // Odd byte offset (or big-endian host): fall back to per-sample reads.\n const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.length);\n let dst = this.writePos % cap;\n for (let i = 0; i < numSamples; i++) {\n samples[dst] = view.getInt16(i * 2, true) / 0x8000;\n dst++;\n if (dst === cap) dst = 0;\n }\n }\n this.writePos += numSamples;\n // If the producer outran the consumer by more than the buffer holds, drop\n // the oldest unplayed audio rather than reading samples we've overwritten.\n if (this.writePos - this.readPos > this.capacity) {\n this.readPos = this.writePos - this.capacity;\n }\n }\n\n process(inputs, outputs) {\n // No output wired up yet \u2014 nothing to render this quantum. Throwing here\n // would permanently kill the processor (the node is persistent per\n // session), so guard like the capture processor does.\n if (!outputs[0] || !outputs[0][0]) return true;\n const out = outputs[0][0];\n if (this.interrupted) {\n out.fill(0);\n this.stopTurn('interrupt');\n return true;\n }\n\n const avail = this.writePos - this.readPos;\n\n // Wait for jitter buffer to fill, unless done (short utterance)\n if (!this.playing) {\n if (avail >= this.jitterSamples || this.isDone) {\n this.playing = true;\n } else {\n out.fill(0);\n return true;\n }\n }\n\n if (avail > 0) {\n const n = Math.min(avail, out.length);\n // Copy from the ring buffer, splitting across the wrap boundary.\n const start = this.readPos % this.capacity;\n const first = Math.min(n, this.capacity - start);\n out.set(this.samples.subarray(start, start + first), 0);\n if (n > first) out.set(this.samples.subarray(0, n - first), first);\n this.readPos += n;\n out.fill(0, n);\n return true;\n }\n\n // No data: output silence, end the turn only when done\n out.fill(0);\n if (this.isDone) {\n this.stopTurn('done');\n }\n return true;\n }\n}\n\nregisterProcessor('playback-processor', PlaybackProcessor);\n";
|
|
3
|
-
declare const
|
|
4
|
-
export default
|
|
2
|
+
export declare const playbackProcessorSource = "\nclass PlaybackProcessor extends AudioWorkletProcessor {\n constructor(options) {\n super();\n const opts = options.processorOptions || {};\n // The node's context IS the playback context (audio.ts asserts its rate),\n // so the worklet-global sampleRate is authoritative; the option exists\n // for the node-less test harness.\n const rate = opts.sampleRate ?? sampleRate;\n // Fill target for the start of a turn. If 'done' arrives first (short\n // utterance), start immediately instead of waiting for audio that is\n // never coming.\n this.jitterSamples = Math.floor((rate * (opts.jitterMs ?? 400)) / 1000);\n // Fill target after an underrun \u2014 see PLAYBACK_REFILL_MS.\n this.refillSamples = Math.floor((rate * (opts.refillMs ?? 200)) / 1000);\n // Concealment source: a ring of the most recently played samples, looped\n // under a decaying gain to cover a gap. Sized to the fade window, with a\n // per-sample decay that reaches the floor exactly at its end.\n this.concealCapacity = Math.max(1, Math.floor((rate * 40) / 1000));\n this.concealBuf = new Float32Array(this.concealCapacity);\n this.concealDecay = Math.exp(Math.log(0.001) / this.concealCapacity);\n // Float32 ring buffer \u2014 PLAYBACK_BUFFER_SECONDS at the context sample\n // rate. Allocated once for the node's lifetime; per-turn state resets via\n // resetTurn(). writePos and readPos are absolute (monotonic) sample\n // counts; the buffer is indexed modulo capacity so a longer reply keeps\n // playing instead of writing past the end and going silent.\n this.capacity = rate * 60;\n this.samples = new Float32Array(this.capacity);\n // Platform endianness probe: the wire format is PCM16 little-endian, so\n // the Int16Array fast path in ingestBytes is only valid on LE hosts\n // (every shipping browser target; the DataView path is the fallback).\n this.littleEndian = new Uint8Array(new Uint16Array([1]).buffer)[0] === 1;\n this.resetTurn();\n\n this.port.onmessage = (e) => {\n const d = e.data;\n if (d.event === 'write') {\n this.ingestBytes(d.buffer);\n } else if (d.event === 'interrupt') {\n // Applied eagerly, not deferred to the next process(): onmessage and\n // process() never interleave (one audio thread), and a deferred\n // interrupt would let 'write'/'done' frames for the NEXT turn\n // coalesce in ahead of it and be wiped by the reset along with the\n // cancelled turn's audio.\n this.stopTurn('interrupt');\n } else if (d.event === 'done') {\n this.isDone = true;\n }\n };\n }\n\n // Reset per-turn state so the node is reusable across replies without\n // reallocating the sample buffer or re-instantiating the worklet.\n resetTurn() {\n this.isDone = false;\n this.playing = false;\n // Whether any real audio has been rendered this turn. Separates a turn's\n // pre-roll (nothing to extrapolate from, and not a defect) from a\n // mid-turn underrun.\n this.hasPlayed = false;\n this.fillTarget = this.jitterSamples;\n // Carry-over byte for split samples across chunks\n this.carry = null;\n this.writePos = 0;\n this.readPos = 0;\n // Concealment ring state and the current fade position.\n this.concealLen = 0;\n this.concealWrite = 0;\n this.concealPos = 0;\n this.concealGain = 1;\n // Episode flags, so a multi-quantum gap counts as one event.\n this.concealing = false;\n this.concealedSilence = false;\n // Reported to the host on 'stop'. A fresh object per turn: the one just\n // posted must not be mutated by the next turn.\n this.stats = {\n concealedSamples: 0,\n silentConcealedSamples: 0,\n concealmentEvents: 0,\n silentConcealmentEvents: 0,\n };\n }\n\n // End the current turn: notify the host and rearm for the next reply.\n // Must NOT return false from process() \u2014 a processor that stops is dead\n // for good, forcing a new node (and buffer) per reply.\n // `reason` ('interrupt' | 'done') tells the host which turn boundary this\n // stop belongs to: interrupt-stops are dropped host-side (flush() already\n // settled that turn), so they can never resolve a later turn's done() early.\n stopTurn(reason) {\n this.port.postMessage({ event: 'stop', reason, stats: this.stats });\n this.resetTurn();\n }\n\n // Cover a quantum (from `start`) where real audio should have been.\n //\n // Before the turn's first samples there is nothing to extrapolate from, so\n // the gap is plain silence and counted as nothing \u2014 WebRTC likewise only\n // counts concealment once playout has begun. After that, loop the retained\n // tail under a decaying gain: a hard zero-fill is a discontinuity mid-word,\n // which is the click that makes a brief stall sound like breakage.\n coverGap(out, start) {\n if (!this.hasPlayed) {\n out.fill(0, start);\n return;\n }\n if (!this.concealing) {\n this.concealing = true;\n this.concealedSilence = false;\n this.stats.concealmentEvents++;\n }\n const total = out.length - start;\n const len = this.concealLen;\n let silent = 0;\n // Second condition: once the fade has decayed to the floor, every sample\n // the loop would emit is 0 anyway \u2014 bulk-fill instead of running 128\n // branchy iterations per quantum for the whole tail of a long stall.\n if (len === 0 || this.concealGain < 0.001) {\n out.fill(0, start);\n silent = total;\n } else {\n let g = this.concealGain;\n for (let i = start; i < out.length; i++) {\n if (g < 0.001) {\n // The fade has run out: keep counting the gap, but stop looping a\n // fragment that is now inaudible anyway.\n out[i] = 0;\n silent++;\n continue;\n }\n out[i] = this.concealBuf[this.concealPos] * g;\n this.concealPos = this.concealPos + 1 === len ? 0 : this.concealPos + 1;\n g *= this.concealDecay;\n }\n this.concealGain = g;\n }\n this.stats.concealedSamples += total;\n if (silent > 0) {\n this.stats.silentConcealedSamples += silent;\n if (!this.concealedSilence) {\n this.concealedSilence = true;\n this.stats.silentConcealmentEvents++;\n }\n }\n }\n\n // Retain the tail of a rendered quantum as the next gap's concealment\n // source, and close any episode the real audio just ended.\n rememberTail(out, n) {\n const cap = this.concealCapacity;\n const take = Math.min(n, cap);\n // Bulk copies (this runs on every cleanly rendered quantum): the tail is\n // one contiguous source run, landing in at most two ring runs.\n const tail = out.subarray(n - take, n);\n const first = Math.min(take, cap - this.concealWrite);\n this.concealBuf.set(tail.subarray(0, first), this.concealWrite);\n if (take > first) this.concealBuf.set(tail.subarray(first), 0);\n this.concealWrite = (this.concealWrite + take) % cap;\n this.concealLen = Math.min(cap, this.concealLen + take);\n // Read the loop oldest-first; once the ring is full the write cursor is\n // the oldest retained sample.\n this.concealPos = this.concealLen === cap ? this.concealWrite : 0;\n this.concealing = false;\n this.concealGain = 1;\n }\n\n ingestBytes(uint8) {\n let bytes = uint8;\n\n if (this.carry !== null) {\n const merged = new Uint8Array(1 + bytes.length);\n merged[0] = this.carry;\n merged.set(bytes, 1);\n bytes = merged;\n this.carry = null;\n }\n\n if (bytes.length % 2 !== 0) {\n this.carry = bytes[bytes.length - 1];\n bytes = bytes.subarray(0, bytes.length - 1);\n }\n\n if (bytes.length === 0) return;\n const numSamples = bytes.length / 2;\n const cap = this.capacity;\n const samples = this.samples;\n if (this.littleEndian && (bytes.byteOffset & 1) === 0) {\n // Fast path: 2-byte-aligned LE bytes wrap directly as an Int16Array;\n // copy wrap-aware in at most two runs with no per-sample DataView call\n // or modulo. This runs on the realtime audio thread.\n const int16 = new Int16Array(bytes.buffer, bytes.byteOffset, numSamples);\n let src = 0;\n let dst = this.writePos % cap;\n while (src < numSamples) {\n const run = Math.min(numSamples - src, cap - dst);\n for (let j = 0; j < run; j++) {\n samples[dst + j] = int16[src + j] / 0x8000;\n }\n src += run;\n dst = 0;\n }\n } else {\n // Odd byte offset (or big-endian host): fall back to per-sample reads.\n const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.length);\n let dst = this.writePos % cap;\n for (let i = 0; i < numSamples; i++) {\n samples[dst] = view.getInt16(i * 2, true) / 0x8000;\n dst++;\n if (dst === cap) dst = 0;\n }\n }\n this.writePos += numSamples;\n // If the producer outran the consumer by more than the buffer holds, drop\n // the oldest unplayed audio rather than reading samples we've overwritten.\n if (this.writePos - this.readPos > this.capacity) {\n this.readPos = this.writePos - this.capacity;\n }\n }\n\n process(inputs, outputs) {\n // No output wired up yet \u2014 nothing to render this quantum. Throwing here\n // would permanently kill the processor (the node is persistent per\n // session), so guard like the capture processor does.\n if (!outputs[0] || !outputs[0][0]) return true;\n const out = outputs[0][0];\n const avail = this.writePos - this.readPos;\n\n // Filling: wait for the target. 'done' short-circuits it \u2014 what is\n // buffered is all there will be, so there is nothing left to wait for.\n if (!this.playing) {\n if (avail >= this.fillTarget || this.isDone) {\n this.playing = true;\n } else {\n this.coverGap(out, 0);\n return true;\n }\n }\n\n // Underrun: this quantum cannot be filled and more audio is still coming.\n // Go back to filling (at the refill target) and cover the gap, leaving\n // readPos untouched \u2014 the fragment stays buffered and plays intact once\n // the buffer recovers, instead of being dribbled out a few samples at a\n // time for the rest of the turn.\n if (avail < out.length && !this.isDone) {\n this.playing = false;\n this.fillTarget = this.refillSamples;\n this.coverGap(out, 0);\n return true;\n }\n\n if (avail > 0) {\n const n = Math.min(avail, out.length);\n // Copy from the ring buffer, splitting across the wrap boundary.\n const start = this.readPos % this.capacity;\n const first = Math.min(n, this.capacity - start);\n out.set(this.samples.subarray(start, start + first), 0);\n if (n > first) out.set(this.samples.subarray(0, n - first), first);\n this.readPos += n;\n // Only reachable with n < out.length on the turn's final partial\n // quantum (the underrun branch above catches every other case).\n out.fill(0, n);\n this.hasPlayed = true;\n this.rememberTail(out, n);\n return true;\n }\n\n // Drained and done: end the turn. Not reachable mid-turn \u2014 an empty\n // buffer with audio still coming is the underrun branch above.\n out.fill(0);\n if (this.isDone) {\n this.stopTurn('done');\n }\n return true;\n }\n}\n\nregisterProcessor('playback-processor', PlaybackProcessor);\n";
|
|
3
|
+
declare const _default: string;
|
|
4
|
+
export default _default;
|