@alexkroman1/aai-ui 1.16.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/_module-url-C_4gRVL0.js +15 -0
  2. package/dist/audio.d.ts +85 -11
  3. package/dist/audio.js +112 -83
  4. package/dist/chat-view-C1oJxsWz.js +129 -0
  5. package/dist/client-config.d.ts +6 -6
  6. package/dist/components/chat-view.d.ts +1 -8
  7. package/dist/components/chat-view.js +2 -2
  8. package/dist/components/console-shell.d.ts +37 -0
  9. package/dist/components/controls.js +1 -1
  10. package/dist/components/url-chips.d.ts +1 -4
  11. package/dist/context.d.ts +1 -1
  12. package/dist/context.js +1 -4
  13. package/dist/{controls-BbZcmnJf.js → controls-DV368uhb.js} +2 -5
  14. package/dist/default-client/assets/_module-url-BX0RuRU2.js +1 -0
  15. package/dist/default-client/assets/audio-BGHiiDY_.js +1 -0
  16. package/dist/default-client/assets/capture-processor-BLPsqnGl.js +90 -0
  17. package/dist/default-client/assets/{index-BunwIXSP.js → index-DXf7-M0Q.js} +156 -36
  18. package/dist/default-client/assets/index-Dv-Q5VRL.css +2 -0
  19. package/dist/default-client/assets/playback-processor-B_T_1KQP.js +278 -0
  20. package/dist/default-client/index.html +2 -2
  21. package/dist/define-client-Cyf1jqAl.js +150 -0
  22. package/dist/define-client.d.ts +0 -9
  23. package/dist/define-client.js +1 -1
  24. package/dist/index.d.ts +2 -5
  25. package/dist/index.js +7 -5
  26. package/dist/{session-core-B64kau_v.js → session-core-BM3WHeoY.js} +36 -128
  27. package/dist/session-core-messages.d.ts +0 -4
  28. package/dist/session-core-types.d.ts +1 -27
  29. package/dist/session-core.js +1 -1
  30. package/dist/types.d.ts +24 -1
  31. package/dist/types.js +32 -2
  32. package/dist/worklets/_module-url.d.ts +10 -0
  33. package/dist/worklets/capture-processor.d.ts +3 -3
  34. package/dist/worklets/capture-processor.js +35 -52
  35. package/dist/worklets/playback-processor.d.ts +3 -3
  36. package/dist/worklets/playback-processor.js +149 -26
  37. package/package.json +2 -2
  38. package/dist/chat-view-gi6FccZq.js +0 -193
  39. package/dist/components/sync-chat-view.d.ts +0 -18
  40. package/dist/components/text-controls.d.ts +0 -14
  41. package/dist/default-client/assets/audio-DNDZgEZp.js +0 -1
  42. package/dist/default-client/assets/capture-processor-UlKEKyIW.js +0 -108
  43. package/dist/default-client/assets/index-BogmeUln.css +0 -2
  44. package/dist/default-client/assets/playback-processor-C5HVRVbu.js +0 -156
  45. package/dist/define-client-DdpijqAu.js +0 -906
  46. package/dist/session-core-upload.d.ts +0 -16
  47. package/dist/sync-mic.d.ts +0 -93
  48. package/dist/sync-session.d.ts +0 -49
  49. package/dist/sync-vad.d.ts +0 -54
@@ -1,8 +1,7 @@
1
- import { FILE_SEND_BACKOFF_MS, MIC_SEND_MAX_BUFFERED_BYTES } from "./types.js";
1
+ import { MIC_SEND_MAX_BUFFERED_BYTES } from "./types.js";
2
+ import { DEFAULT_MAX_HISTORY, WS_OPEN, errorMessage, safeJsonParse } from "@alexkroman1/aai";
2
3
  import { ServerMessageSchema, lenientParse } from "@alexkroman1/aai/protocol";
3
- import { DEFAULT_MAX_HISTORY, FILE_UPLOAD_CHUNK_BYTES, WS_OPEN, errorMessage, safeJsonParse } from "@alexkroman1/aai";
4
4
  import ReconnectingWebSocket from "partysocket/ws";
5
- import { MAX_SYNC_AUDIO_SECONDS } from "@alexkroman1/aai/stt";
6
5
  //#region session-core-audio-setup.ts
7
6
  /**
8
7
  * Audio-path initialization for the voice session core.
@@ -36,6 +35,7 @@ async function initAudioCapture(conn, msg, deps, fatal) {
36
35
  const gen = conn.generation;
37
36
  const stale = () => conn.generation !== gen || !conn.ws || conn.ws.readyState !== WS_OPEN;
38
37
  const reportAudioFailure = (message) => {
38
+ deps.cleanupAudio();
39
39
  if (fatal) deps.updateState({
40
40
  state: "error",
41
41
  error: {
@@ -45,16 +45,13 @@ async function initAudioCapture(conn, msg, deps, fatal) {
45
45
  running: false,
46
46
  recording: false
47
47
  });
48
- else {
49
- deps.cleanupAudio();
50
- deps.updateState({
51
- error: {
52
- code: "audio",
53
- message
54
- },
55
- recording: false
56
- });
57
- }
48
+ else deps.updateState({
49
+ error: {
50
+ code: "audio",
51
+ message
52
+ },
53
+ recording: false
54
+ });
58
55
  };
59
56
  try {
60
57
  const [{ createVoiceIO }, captureWorklet, playbackWorklet] = await Promise.all([
@@ -74,6 +71,12 @@ async function initAudioCapture(conn, msg, deps, fatal) {
74
71
  console.debug("[aai-ui] sendAudio dropped: connection closed");
75
72
  }
76
73
  },
74
+ onPlaybackStats: (stats) => {
75
+ console.warn("[aai-ui] playback concealed a gap in this turn", stats);
76
+ },
77
+ onMicSilent: () => {
78
+ console.warn("[aai-ui] microphone is delivering only silence — check the selected input device");
79
+ },
77
80
  onError: (err) => {
78
81
  if (conn.generation !== gen) return;
79
82
  reportAudioFailure(err.message);
@@ -149,7 +152,7 @@ function appendCapped(list, item, cap) {
149
152
  * dedup) that previously lived as closure locals in `createSessionCore`.
150
153
  */
151
154
  function createMessageHandlers(deps) {
152
- const { getSnapshot, updateState, conn, discardUpload, cleanupAudio } = deps;
155
+ const { getSnapshot, updateState, conn, cleanupAudio } = deps;
153
156
  /** Incremented on each turn boundary -- stale async callbacks compare against this. */
154
157
  let handlerGeneration = 0;
155
158
  /** Monotonically increasing counter for custom events -- used by useEvent to deduplicate. */
@@ -207,6 +210,7 @@ function createMessageHandlers(deps) {
207
210
  } });
208
211
  else {
209
212
  cleanupAudio();
213
+ conn.generation++;
210
214
  updateState({
211
215
  state: "error",
212
216
  error: {
@@ -274,7 +278,6 @@ function createMessageHandlers(deps) {
274
278
  break;
275
279
  case "reset":
276
280
  handlerGeneration++;
277
- discardUpload();
278
281
  conn.voiceIO?.flush();
279
282
  updateState({
280
283
  ...CLEARED_SESSION_STATE,
@@ -347,7 +350,6 @@ function createMessageHandlers(deps) {
347
350
  if (msg.type === "config") return {
348
351
  sampleRate: msg.sampleRate,
349
352
  ttsSampleRate: msg.ttsSampleRate,
350
- audioOut: msg.audioOut,
351
353
  sid: msg.sessionId
352
354
  };
353
355
  if (msg.type === "audio_done") {
@@ -398,84 +400,6 @@ function reconnectPending(socket) {
398
400
  return socket instanceof ReconnectingWebSocket && socket.shouldReconnect && socket.retryCount < RECONNECT_OPTIONS.maxRetries;
399
401
  }
400
402
  //#endregion
401
- //#region session-core-upload.ts
402
- /**
403
- * File-upload transcription for the session core (`sendAudioFile`).
404
- *
405
- * Split out of `session-core.ts`: this module owns the decode → frame →
406
- * reliable-send pipeline for uploaded audio, plus the two guards that keep
407
- * it exclusive — an in-flight lock (no second upload, no mic start
408
- * mid-upload) and an epoch that in-flight sends compare against so a
409
- * reset/close/reconnect abandons them instead of streaming a stale clip
410
- * into the fresh session.
411
- */
412
- function createUploadSender(deps) {
413
- const { conn, getSnapshot, sendJson } = deps;
414
- /** True while a `sendAudioFile` upload is decoding or streaming. Blocks the
415
- * mic (and a second upload) so file bytes never interleave with live audio. */
416
- let uploadInFlight = false;
417
- /** Bumped whenever conversation/connection state is discarded (reset, close,
418
- * reconnect). An in-flight upload compares against it between chunks and
419
- * aborts instead of streaming a stale clip into the fresh session. */
420
- let uploadEpoch = 0;
421
- /** Stream `bytes` to the socket in chunks, waiting out backpressure.
422
- * Unlike live mic frames (dropped under backpressure — stale speech is
423
- * worthless), file audio must arrive completely. Aborts if the connection
424
- * closes or the session is reset (`epoch` moves on) between chunks. */
425
- async function sendBytesReliably(bytes, epoch) {
426
- for (let i = 0; i < bytes.byteLength; i += FILE_UPLOAD_CHUNK_BYTES) {
427
- while (epoch === uploadEpoch && conn.ws && conn.ws.readyState === WS_OPEN && conn.ws.bufferedAmount > MIC_SEND_MAX_BUFFERED_BYTES) await new Promise((r) => setTimeout(r, FILE_SEND_BACKOFF_MS));
428
- if (epoch !== uploadEpoch) throw new Error("sendAudioFile: session was reset mid-send");
429
- if (!conn.ws || conn.ws.readyState !== WS_OPEN) throw new Error("sendAudioFile: connection closed mid-send");
430
- conn.ws.send(bytes.subarray(i, i + FILE_UPLOAD_CHUNK_BYTES));
431
- }
432
- }
433
- /** Validate that an upload may start; returns the session's ready config. */
434
- function assertUploadReady() {
435
- const cfg = conn.readyConfig;
436
- if (!(cfg && conn.ws) || conn.ws.readyState !== WS_OPEN) throw new Error("sendAudioFile: session is not connected");
437
- const snap = getSnapshot();
438
- if (snap.audioOut) throw new Error("sendAudioFile is only available in text-only sessions (tts: none()) — voice sessions stream the microphone instead");
439
- if (snap.recording) throw new Error("sendAudioFile: stop recording before uploading a file");
440
- if (uploadInFlight) throw new Error("sendAudioFile: another upload is already in progress");
441
- return cfg;
442
- }
443
- async function sendAudioFile(file) {
444
- const cfg = assertUploadReady();
445
- uploadInFlight = true;
446
- const epoch = uploadEpoch;
447
- try {
448
- const { decodeAudioToPcm16 } = await import("./audio.js");
449
- const clip = await decodeAudioToPcm16(await file.arrayBuffer(), cfg.sampleRate);
450
- if (getSnapshot().recording || conn.audioSetupInFlight) throw new Error("sendAudioFile: stop recording before uploading a file");
451
- if (epoch !== uploadEpoch) throw new Error("sendAudioFile: session was reset mid-send");
452
- if (clip.length / cfg.sampleRate <= MAX_SYNC_AUDIO_SECONDS) {
453
- const bytes = new Uint8Array(clip.buffer, clip.byteOffset, clip.byteLength);
454
- sendJson({
455
- type: "transcribe_file_start",
456
- sampleRate: cfg.sampleRate,
457
- byteLength: bytes.byteLength
458
- });
459
- await sendBytesReliably(bytes, epoch);
460
- sendJson({ type: "transcribe_file_end" });
461
- return;
462
- }
463
- const padded = new Int16Array(clip.length + cfg.sampleRate);
464
- padded.set(clip);
465
- await sendBytesReliably(new Uint8Array(padded.buffer), epoch);
466
- } finally {
467
- uploadInFlight = false;
468
- }
469
- }
470
- return {
471
- sendAudioFile,
472
- inFlight: () => uploadInFlight,
473
- discard: () => {
474
- uploadEpoch++;
475
- }
476
- };
477
- }
478
- //#endregion
479
403
  //#region session-core-url.ts
480
404
  /** Build the session WebSocket URL from the platform URL and resume state. */
481
405
  function buildWsUrl(platformUrl, resume, sessionId) {
@@ -518,7 +442,6 @@ function createSessionCore(options) {
518
442
  contentVersion: 0,
519
443
  started: false,
520
444
  running: false,
521
- audioOut: true,
522
445
  recording: false,
523
446
  apiUrl: buildWsUrl(options.platformUrl, false).toString()
524
447
  };
@@ -564,8 +487,16 @@ function createSessionCore(options) {
564
487
  };
565
488
  let connectionController = null;
566
489
  let hasConnected = false;
490
+ /**
491
+ * The session ID to resume: seeded from `options.resumeSessionId`, then
492
+ * kept current from every `config` frame. Reconnect URLs carry it as
493
+ * `?sessionId=<id>` so the server re-registers the SAME session id —
494
+ * that key is what per-session tool state (`ctx.state`) lives under, so
495
+ * a reconnect that omits it gets a fresh session with none of the
496
+ * agent's context, greeting suppression aside.
497
+ */
498
+ let sessionId = options.resumeSessionId;
567
499
  function cleanupAudio() {
568
- upload.discard();
569
500
  conn.audioSetupInFlight = false;
570
501
  conn.voiceIO?.close().catch(() => {});
571
502
  conn.voiceIO = null;
@@ -583,16 +514,10 @@ function createSessionCore(options) {
583
514
  if (conn.ws.bufferedAmount > MIC_SEND_MAX_BUFFERED_BYTES) return;
584
515
  conn.ws.send(bytes);
585
516
  }
586
- const upload = createUploadSender({
587
- conn,
588
- getSnapshot,
589
- sendJson
590
- });
591
517
  const { handleMessage, settleWhenAudioDrained } = createMessageHandlers({
592
518
  getSnapshot,
593
519
  updateState,
594
520
  conn,
595
- discardUpload: upload.discard,
596
521
  cleanupAudio
597
522
  });
598
523
  const audioDeps = {
@@ -613,20 +538,17 @@ function createSessionCore(options) {
613
538
  /** React to the server's `config` message: record it, set up the audio
614
539
  * path for the session's mode, and replay history on reconnect. */
615
540
  function onServerConfig(config) {
616
- if (config.sid) options.onSessionId?.(config.sid);
541
+ if (config.sid) {
542
+ sessionId = config.sid;
543
+ options.onSessionId?.(config.sid);
544
+ }
617
545
  const isReconnect = hasConnected;
618
546
  hasConnected = true;
619
547
  conn.readyConfig = {
620
548
  sampleRate: config.sampleRate,
621
549
  ttsSampleRate: config.ttsSampleRate
622
550
  };
623
- const audioOut = config.audioOut !== false;
624
- updateState({ audioOut });
625
- if (audioOut) initAudioCapture(conn, config, audioDeps, true);
626
- else {
627
- sendJson({ type: "audio_ready" });
628
- updateState({ state: "listening" });
629
- }
551
+ initAudioCapture(conn, config, audioDeps, true);
630
552
  if (isReconnect && currentSnapshot.messages.length > 0) sendJson({
631
553
  type: "history",
632
554
  messages: currentSnapshot.messages.map((m) => ({
@@ -639,11 +561,12 @@ function createSessionCore(options) {
639
561
  * The WebSocket URL for the *next* connection attempt. Evaluated per
640
562
  * attempt (partysocket takes it as a URL provider), so once the first
641
563
  * `config` arrives, every reconnect — automatic or explicit — carries
642
- * `resume=1` and the session resumes instead of starting over.
564
+ * `?sessionId=<id>` and the server resumes the SAME session (id, tool
565
+ * state) instead of minting a new one. `resume=1` remains only as the
566
+ * greeting-suppression fallback for a server whose config carried no id.
643
567
  */
644
568
  function currentWsUrl() {
645
- const resumeId = !hasConnected ? options.resumeSessionId : void 0;
646
- return buildWsUrl(options.platformUrl, hasConnected, resumeId).toString();
569
+ return buildWsUrl(options.platformUrl, hasConnected, sessionId).toString();
647
570
  }
648
571
  /** Open a socket: an injected constructor as-is (tests), or partysocket's
649
572
  * reconnecting WebSocket — same interface, plus reconnect-on-close. */
@@ -723,7 +646,6 @@ function createSessionCore(options) {
723
646
  sendJson({ type: "cancel" });
724
647
  }
725
648
  function reset() {
726
- upload.discard();
727
649
  conn.voiceIO?.flush();
728
650
  if (conn.ws && conn.ws.readyState === WS_OPEN) {
729
651
  sendJson({ type: "reset" });
@@ -742,17 +664,6 @@ function createSessionCore(options) {
742
664
  recording: false
743
665
  });
744
666
  }
745
- function startRecording() {
746
- if (currentSnapshot.audioOut || currentSnapshot.recording || conn.audioSetupInFlight || upload.inFlight()) return;
747
- const cfg = conn.readyConfig;
748
- if (!(cfg && conn.ws) || conn.ws.readyState !== WS_OPEN) return;
749
- initAudioCapture(conn, cfg, audioDeps, false);
750
- }
751
- function stopRecording() {
752
- if (currentSnapshot.audioOut || !currentSnapshot.recording) return;
753
- cleanupAudio();
754
- updateState({ recording: false });
755
- }
756
667
  function start() {
757
668
  updateState({
758
669
  started: true,
@@ -777,9 +688,6 @@ function createSessionCore(options) {
777
688
  disconnect,
778
689
  start,
779
690
  toggle,
780
- startRecording,
781
- stopRecording,
782
- sendAudioFile: upload.sendAudioFile,
783
691
  [Symbol.dispose]() {
784
692
  disconnect();
785
693
  }
@@ -17,8 +17,6 @@ export declare const CLEARED_SESSION_STATE: {
17
17
  export type SessionConfigMessage = {
18
18
  sampleRate: number;
19
19
  ttsSampleRate: number;
20
- /** False for text-only agents (`tts: none()`) — see ReadyConfigSchema. */
21
- audioOut?: boolean | undefined;
22
20
  sid?: string | undefined;
23
21
  };
24
22
  /** Dependencies the message handlers need from the owning session core. */
@@ -26,8 +24,6 @@ type MessageHandlerDeps = {
26
24
  getSnapshot: () => SessionSnapshot;
27
25
  updateState: (partial: Partial<SessionSnapshot>) => void;
28
26
  conn: ConnState;
29
- /** Invalidate any in-flight file upload (the upload sender's `discard`). */
30
- discardUpload: () => void;
31
27
  /** Release the microphone/VoiceIO (the session core's `cleanupAudio`). */
32
28
  cleanupAudio: () => void;
33
29
  };
@@ -27,17 +27,7 @@ export type CustomEvent = {
27
27
  */
28
28
  export type SessionSnapshot = {
29
29
  readonly state: AgentState;
30
- /**
31
- * False when the server declared the session text-only (`tts: none()`):
32
- * no audio frames will arrive and replies render as text. True until the
33
- * server's `config` message says otherwise (voice is the default).
34
- */
35
- readonly audioOut: boolean;
36
- /**
37
- * True while the microphone is live and streaming to the server. Voice
38
- * sessions record for their whole lifetime; text-only sessions toggle this
39
- * via `startRecording()` / `stopRecording()` (the record button).
40
- */
30
+ /** True while the microphone is live and streaming to the server. */
41
31
  readonly recording: boolean;
42
32
  /**
43
33
  * The WebSocket URL a program can connect to directly (the same endpoint
@@ -94,22 +84,6 @@ export type SessionCore = {
94
84
  start(): void;
95
85
  /** Toggle between connected and disconnected states. */
96
86
  toggle(): void;
97
- /**
98
- * Start streaming microphone audio (text-only sessions). Requests mic
99
- * access on first use. No-op in voice sessions, where the mic is always on.
100
- */
101
- startRecording(): void;
102
- /** Stop streaming microphone audio (text-only sessions). No-op in voice sessions. */
103
- stopRecording(): void;
104
- /**
105
- * Decode an audio file (any format the browser can decode), resample it to
106
- * the session's STT rate, and stream it to the server for transcription.
107
- * Text-only sessions (`tts: none()`) only — voice sessions stream the
108
- * microphone instead and reject. Also rejects when no session is connected,
109
- * the mic is recording, another upload is in flight, or the file cannot be
110
- * decoded. Resolves once the audio has been handed to the socket.
111
- */
112
- sendAudioFile(file: Blob): Promise<void>;
113
87
  /** Alias for `disconnect` for use with `using`. */
114
88
  [Symbol.dispose](): void;
115
89
  };
@@ -1,2 +1,2 @@
1
- import { t as createSessionCore } from "./session-core-B64kau_v.js";
1
+ import { t as createSessionCore } from "./session-core-BM3WHeoY.js";
2
2
  export { createSessionCore };
package/dist/types.d.ts CHANGED
@@ -1,5 +1,28 @@
1
1
  import type { SessionErrorCode } from "@alexkroman1/aai/protocol";
2
- export { FILE_SEND_BACKOFF_MS, MIC_BUFFER_SECONDS, MIC_SEND_MAX_BUFFERED_BYTES, } from "@alexkroman1/aai";
2
+ export { CAPTURE_STOP_ACK_TIMEOUT_MS, DEFAULT_STT_SAMPLE_RATE, MIC_BUFFER_SECONDS, MIC_SEND_MAX_BUFFERED_BYTES, MIC_SILENCE_PROBE_MS, PLAYBACK_BUFFER_SECONDS, PLAYBACK_CONCEAL_FADE_MS, PLAYBACK_CONCEAL_FLOOR, PLAYBACK_DONE_MAX_WAIT_MS, PLAYBACK_DONE_POLL_MS, PLAYBACK_JITTER_MS, PLAYBACK_REFILL_MS, } from "@alexkroman1/aai";
3
+ /**
4
+ * `getUserMedia` audio constraints for every capture path in this package.
5
+ *
6
+ * Defined once because four copies of this object drifted apart trivially, and
7
+ * the flags are not cosmetic — each one rewrites the signal before STT (and
8
+ * before the sync path's energy VAD) ever sees it:
9
+ *
10
+ * - **`autoGainControl: false`** — AGC continuously retargets level, which
11
+ * means riding the noise floor up through silence. An energy VAD calibrated
12
+ * against a moving floor is calibrated against nothing.
13
+ * - **`noiseSuppression: false`** / **`voiceIsolation: false`** — both discard
14
+ * signal to make speech sound cleaner to a human, and both can gate a quiet
15
+ * room to *exact* zeros, which is also what a dead microphone looks like
16
+ * (see `MIC_SILENCE_PROBE_MS`).
17
+ * - **`echoCancellation: true`** — this one stays on. The mic is open while
18
+ * the agent speaks (barge-in needs it), so without AEC the agent hears
19
+ * itself and interrupts its own reply.
20
+ *
21
+ * Cast because `voiceIsolation` is newer than TypeScript's DOM lib.
22
+ *
23
+ * @public
24
+ */
25
+ export declare const VOICE_CAPTURE_CONSTRAINTS: MediaTrackConstraints;
3
26
  /**
4
27
  * Current state of the voice agent session.
5
28
  *
package/dist/types.js CHANGED
@@ -1,2 +1,32 @@
1
- import { FILE_SEND_BACKOFF_MS, MIC_BUFFER_SECONDS, MIC_SEND_MAX_BUFFERED_BYTES } from "@alexkroman1/aai";
2
- export { FILE_SEND_BACKOFF_MS, MIC_BUFFER_SECONDS, MIC_SEND_MAX_BUFFERED_BYTES };
1
+ import { CAPTURE_STOP_ACK_TIMEOUT_MS, DEFAULT_STT_SAMPLE_RATE, MIC_BUFFER_SECONDS, MIC_SEND_MAX_BUFFERED_BYTES, MIC_SILENCE_PROBE_MS, PLAYBACK_BUFFER_SECONDS, PLAYBACK_CONCEAL_FADE_MS, PLAYBACK_CONCEAL_FLOOR, PLAYBACK_DONE_MAX_WAIT_MS, PLAYBACK_DONE_POLL_MS, PLAYBACK_JITTER_MS, PLAYBACK_REFILL_MS } from "@alexkroman1/aai";
2
+ //#region types.ts
3
+ /**
4
+ * `getUserMedia` audio constraints for every capture path in this package.
5
+ *
6
+ * Defined once because four copies of this object drifted apart trivially, and
7
+ * the flags are not cosmetic — each one rewrites the signal before STT (and
8
+ * before the sync path's energy VAD) ever sees it:
9
+ *
10
+ * - **`autoGainControl: false`** — AGC continuously retargets level, which
11
+ * means riding the noise floor up through silence. An energy VAD calibrated
12
+ * against a moving floor is calibrated against nothing.
13
+ * - **`noiseSuppression: false`** / **`voiceIsolation: false`** — both discard
14
+ * signal to make speech sound cleaner to a human, and both can gate a quiet
15
+ * room to *exact* zeros, which is also what a dead microphone looks like
16
+ * (see `MIC_SILENCE_PROBE_MS`).
17
+ * - **`echoCancellation: true`** — this one stays on. The mic is open while
18
+ * the agent speaks (barge-in needs it), so without AEC the agent hears
19
+ * itself and interrupts its own reply.
20
+ *
21
+ * Cast because `voiceIsolation` is newer than TypeScript's DOM lib.
22
+ *
23
+ * @public
24
+ */
25
+ const VOICE_CAPTURE_CONSTRAINTS = {
26
+ echoCancellation: true,
27
+ noiseSuppression: false,
28
+ autoGainControl: false,
29
+ voiceIsolation: false
30
+ };
31
+ //#endregion
32
+ export { CAPTURE_STOP_ACK_TIMEOUT_MS, DEFAULT_STT_SAMPLE_RATE, MIC_BUFFER_SECONDS, MIC_SEND_MAX_BUFFERED_BYTES, MIC_SILENCE_PROBE_MS, PLAYBACK_BUFFER_SECONDS, PLAYBACK_CONCEAL_FADE_MS, PLAYBACK_CONCEAL_FLOOR, PLAYBACK_DONE_MAX_WAIT_MS, PLAYBACK_DONE_POLL_MS, PLAYBACK_JITTER_MS, PLAYBACK_REFILL_MS, VOICE_CAPTURE_CONSTRAINTS };
@@ -0,0 +1,10 @@
1
+ /**
2
+ * Blob-URL module for an inlined AudioWorklet processor source.
3
+ *
4
+ * A blob URL rather than a data URI because the agent page's CSP allows
5
+ * `script-src blob:` but not `data:` — a data-URI module fails `addModule`
6
+ * with the opaque "Unable to load a worklet's module". Every inline worklet
7
+ * in this package must load through this helper so that constraint lives in
8
+ * one place.
9
+ */
10
+ export declare function workletModuleUrl(source: string): string;
@@ -1,4 +1,4 @@
1
1
  /** Raw worklet source — exported so tests can evaluate the processor directly. */
2
- export declare const captureProcessorSource = "\nclass CaptureProcessor extends AudioWorkletProcessor {\n constructor(options) {\n super();\n this.recording = false;\n const opts = options.processorOptions || {};\n this.fromRate = opts.contextRate || sampleRate;\n this.toRate = opts.sttSampleRate || sampleRate;\n this.ratio = this.fromRate / this.toRate;\n this.needsResample = this.fromRate !== this.toRate;\n // Streaming-resampler state carried across process() blocks so the output\n // clock doesn't drift and there's no discontinuity at 128-sample block\n // boundaries. `prev` is the previous block's last input sample (used to\n // interpolate across the boundary); `pos` is the fractional read position,\n // in input samples, relative to an extended [prev, ...input] frame. Start\n // at 1 so the first output is input[0] (not the bogus initial prev).\n this.prev = 0;\n this.pos = 1;\n // Output buffer reused across process() calls -- the render quantum is a\n // fixed size, so allocating per call would churn the realtime audio thread.\n this.resampleBuf = null;\n // Int16 accumulation buffer: flushed to the main thread as one transferred\n // ArrayBuffer once ~bufferSeconds of output samples are batched. Sized 2x\n // the flush target so a whole render quantum always fits before flushing.\n this.targetSamples = Math.max(1, Math.round(this.toRate * (opts.bufferSeconds || 0.1)));\n this.pending = new Int16Array(this.targetSamples * 2);\n this.pendingLen = 0;\n this.port.onmessage = (e) => {\n if (e.data.event === 'start') this.recording = true;\n else if (e.data.event === 'stop') {\n // Final flush so the tail of speech isn't dropped on close, then ack\n // so the host knows the tail chunk (if any) has been posted and it is\n // safe to tear the context down.\n this.flush();\n this.recording = false;\n this.port.postMessage({ event: 'stopped' });\n }\n };\n }\n\n resample(input) {\n const ratio = this.ratio;\n const n = input.length;\n if (n === 0) return new Float32Array(0);\n // At most ceil(n / ratio) + 1 outputs per block (pos starts in [0, ratio)).\n const maxLen = Math.ceil(n / ratio) + 1;\n if (this.resampleBuf === null || this.resampleBuf.length < maxLen) {\n this.resampleBuf = new Float32Array(maxLen);\n }\n const out = this.resampleBuf;\n // Extended frame: index 0 = prev (previous block's last sample),\n // index k>=1 = input[k-1]. Interpolate at fractional positions stepping by\n // ratio, carrying `pos` across calls to preserve the sample clock.\n let count = 0;\n let pos = this.pos;\n while (pos < n) {\n const idx = pos | 0;\n const frac = pos - idx;\n const a = idx === 0 ? this.prev : input[idx - 1];\n const b = input[idx];\n out[count++] = a + frac * (b - a);\n pos += ratio;\n }\n // Shift the origin to this block's last sample for the next call.\n this.prev = input[n - 1];\n this.pos = pos - n;\n return out.subarray(0, count);\n }\n\n // Convert Float32 -> Int16 and append to the pending batch. Writes through\n // an Int16Array directly (assignment truncates like DataView.setInt16).\n accumulate(samples) {\n let buf = this.pending;\n if (this.pendingLen + samples.length > buf.length) {\n // Defensive: only reachable if a render quantum outproduces the 1x\n // headroom above the flush target (never with 128-sample quanta).\n const grown = new Int16Array((this.pendingLen + samples.length) * 2);\n grown.set(buf.subarray(0, this.pendingLen));\n this.pending = grown;\n buf = grown;\n }\n for (let i = 0; i < samples.length; i++) {\n const s = Math.max(-1, Math.min(1, samples[i]));\n buf[this.pendingLen++] = s < 0 ? s * 0x8000 : s * 0x7fff;\n }\n }\n\n // Post the batched samples as one transferred ArrayBuffer and reset.\n flush() {\n if (this.pendingLen === 0) return;\n const buffer = this.pending.buffer.slice(0, this.pendingLen * 2);\n this.pendingLen = 0;\n this.port.postMessage({ event: 'chunk', buffer }, [buffer]);\n }\n\n process(inputs) {\n const input = inputs[0];\n if (!input || !input[0] || !this.recording) return true;\n\n const samples = this.needsResample ? this.resample(input[0]) : input[0];\n this.accumulate(samples);\n if (this.pendingLen >= this.targetSamples) this.flush();\n return true;\n }\n}\n\nregisterProcessor('capture-processor', CaptureProcessor);\n";
3
- declare const src: string;
4
- export default src;
2
+ export declare const captureProcessorSource = "\nclass CaptureProcessor extends AudioWorkletProcessor {\n constructor(options) {\n super();\n this.recording = false;\n const opts = options.processorOptions || {};\n // The context runs at the STT rate, so this is both the input and the\n // output rate \u2014 there is nothing to convert.\n this.rate = opts.sampleRate || sampleRate;\n // Int16 accumulation buffer: flushed to the main thread as one transferred\n // ArrayBuffer once ~bufferSeconds of samples are batched. Sized 2x the\n // flush target so a whole render quantum always fits before flushing.\n this.targetSamples = Math.max(1, Math.round(this.rate * (opts.bufferSeconds || 0.1)));\n this.pending = new Int16Array(this.targetSamples * 2);\n this.pendingLen = 0;\n // Dead-mic probe: samples left to inspect before concluding the device\n // delivers nothing but digital silence. Only consumed while recording, so\n // the cost disappears after the window (or after the first real sample).\n this.probeSamplesLeft = Math.round(\n (this.rate * (opts.silenceProbeMs ?? 1500)) / 1000,\n );\n this.port.onmessage = (e) => {\n if (e.data.event === 'start') this.recording = true;\n else if (e.data.event === 'stop') {\n // Final flush so the tail of speech isn't dropped on close, then ack\n // so the host knows the tail chunk (if any) has been posted and it is\n // safe to tear the context down.\n this.flush();\n this.recording = false;\n this.port.postMessage({ event: 'stopped' });\n }\n };\n }\n\n // Convert Float32 -> Int16 and append to the pending batch. Writes through\n // an Int16Array directly (assignment truncates like DataView.setInt16).\n accumulate(samples) {\n let buf = this.pending;\n if (this.pendingLen + samples.length > buf.length) {\n // Defensive: only reachable if a render quantum outproduces the 1x\n // headroom above the flush target (never with 128-sample quanta).\n const grown = new Int16Array((this.pendingLen + samples.length) * 2);\n grown.set(buf.subarray(0, this.pendingLen));\n this.pending = grown;\n buf = grown;\n }\n for (let i = 0; i < samples.length; i++) {\n const s = Math.max(-1, Math.min(1, samples[i]));\n buf[this.pendingLen++] = s < 0 ? s * 0x8000 : s * 0x7fff;\n }\n }\n\n // Post the batched samples as one transferred ArrayBuffer and reset.\n flush() {\n if (this.pendingLen === 0) return;\n const buffer = this.pending.buffer.slice(0, this.pendingLen * 2);\n this.pendingLen = 0;\n this.port.postMessage({ event: 'chunk', buffer }, [buffer]);\n }\n\n // Watch the first window of input for any nonzero sample. One is enough to\n // prove the device is live \u2014 a real mic in a quiet room still carries a\n // noise floor, so all-zeros means muted, wrong input, or no input at all.\n probeForSilence(channel) {\n for (let i = 0; i < channel.length; i++) {\n if (channel[i] !== 0) {\n this.probeSamplesLeft = 0;\n return;\n }\n }\n this.probeSamplesLeft -= channel.length;\n if (this.probeSamplesLeft <= 0) {\n this.port.postMessage({ event: 'silent' });\n }\n }\n\n process(inputs) {\n const input = inputs[0];\n if (!input || !input[0] || !this.recording) return true;\n\n if (this.probeSamplesLeft > 0) this.probeForSilence(input[0]);\n\n this.accumulate(input[0]);\n if (this.pendingLen >= this.targetSamples) this.flush();\n return true;\n }\n}\n\nregisterProcessor('capture-processor', CaptureProcessor);\n";
3
+ declare const _default: string;
4
+ export default _default;
@@ -1,3 +1,5 @@
1
+ import { MIC_BUFFER_SECONDS, MIC_SILENCE_PROBE_MS } from "../types.js";
2
+ import { t as workletModuleUrl } from "../_module-url-C_4gRVL0.js";
1
3
  //#region worklets/capture-processor.ts
2
4
  const CaptureProcessorWorklet = `
3
5
  class CaptureProcessor extends AudioWorkletProcessor {
@@ -5,27 +7,21 @@ class CaptureProcessor extends AudioWorkletProcessor {
5
7
  super();
6
8
  this.recording = false;
7
9
  const opts = options.processorOptions || {};
8
- this.fromRate = opts.contextRate || sampleRate;
9
- this.toRate = opts.sttSampleRate || sampleRate;
10
- this.ratio = this.fromRate / this.toRate;
11
- this.needsResample = this.fromRate !== this.toRate;
12
- // Streaming-resampler state carried across process() blocks so the output
13
- // clock doesn't drift and there's no discontinuity at 128-sample block
14
- // boundaries. \`prev\` is the previous block's last input sample (used to
15
- // interpolate across the boundary); \`pos\` is the fractional read position,
16
- // in input samples, relative to an extended [prev, ...input] frame. Start
17
- // at 1 so the first output is input[0] (not the bogus initial prev).
18
- this.prev = 0;
19
- this.pos = 1;
20
- // Output buffer reused across process() calls -- the render quantum is a
21
- // fixed size, so allocating per call would churn the realtime audio thread.
22
- this.resampleBuf = null;
10
+ // The context runs at the STT rate, so this is both the input and the
11
+ // output rate — there is nothing to convert.
12
+ this.rate = opts.sampleRate || sampleRate;
23
13
  // Int16 accumulation buffer: flushed to the main thread as one transferred
24
- // ArrayBuffer once ~bufferSeconds of output samples are batched. Sized 2x
25
- // the flush target so a whole render quantum always fits before flushing.
26
- this.targetSamples = Math.max(1, Math.round(this.toRate * (opts.bufferSeconds || 0.1)));
14
+ // ArrayBuffer once ~bufferSeconds of samples are batched. Sized 2x the
15
+ // flush target so a whole render quantum always fits before flushing.
16
+ this.targetSamples = Math.max(1, Math.round(this.rate * (opts.bufferSeconds || ${MIC_BUFFER_SECONDS})));
27
17
  this.pending = new Int16Array(this.targetSamples * 2);
28
18
  this.pendingLen = 0;
19
+ // Dead-mic probe: samples left to inspect before concluding the device
20
+ // delivers nothing but digital silence. Only consumed while recording, so
21
+ // the cost disappears after the window (or after the first real sample).
22
+ this.probeSamplesLeft = Math.round(
23
+ (this.rate * (opts.silenceProbeMs ?? ${MIC_SILENCE_PROBE_MS})) / 1000,
24
+ );
29
25
  this.port.onmessage = (e) => {
30
26
  if (e.data.event === 'start') this.recording = true;
31
27
  else if (e.data.event === 'stop') {
@@ -39,35 +35,6 @@ class CaptureProcessor extends AudioWorkletProcessor {
39
35
  };
40
36
  }
41
37
 
42
- resample(input) {
43
- const ratio = this.ratio;
44
- const n = input.length;
45
- if (n === 0) return new Float32Array(0);
46
- // At most ceil(n / ratio) + 1 outputs per block (pos starts in [0, ratio)).
47
- const maxLen = Math.ceil(n / ratio) + 1;
48
- if (this.resampleBuf === null || this.resampleBuf.length < maxLen) {
49
- this.resampleBuf = new Float32Array(maxLen);
50
- }
51
- const out = this.resampleBuf;
52
- // Extended frame: index 0 = prev (previous block's last sample),
53
- // index k>=1 = input[k-1]. Interpolate at fractional positions stepping by
54
- // ratio, carrying \`pos\` across calls to preserve the sample clock.
55
- let count = 0;
56
- let pos = this.pos;
57
- while (pos < n) {
58
- const idx = pos | 0;
59
- const frac = pos - idx;
60
- const a = idx === 0 ? this.prev : input[idx - 1];
61
- const b = input[idx];
62
- out[count++] = a + frac * (b - a);
63
- pos += ratio;
64
- }
65
- // Shift the origin to this block's last sample for the next call.
66
- this.prev = input[n - 1];
67
- this.pos = pos - n;
68
- return out.subarray(0, count);
69
- }
70
-
71
38
  // Convert Float32 -> Int16 and append to the pending batch. Writes through
72
39
  // an Int16Array directly (assignment truncates like DataView.setInt16).
73
40
  accumulate(samples) {
@@ -94,12 +61,29 @@ class CaptureProcessor extends AudioWorkletProcessor {
94
61
  this.port.postMessage({ event: 'chunk', buffer }, [buffer]);
95
62
  }
96
63
 
64
+ // Watch the first window of input for any nonzero sample. One is enough to
65
+ // prove the device is live — a real mic in a quiet room still carries a
66
+ // noise floor, so all-zeros means muted, wrong input, or no input at all.
67
+ probeForSilence(channel) {
68
+ for (let i = 0; i < channel.length; i++) {
69
+ if (channel[i] !== 0) {
70
+ this.probeSamplesLeft = 0;
71
+ return;
72
+ }
73
+ }
74
+ this.probeSamplesLeft -= channel.length;
75
+ if (this.probeSamplesLeft <= 0) {
76
+ this.port.postMessage({ event: 'silent' });
77
+ }
78
+ }
79
+
97
80
  process(inputs) {
98
81
  const input = inputs[0];
99
82
  if (!input || !input[0] || !this.recording) return true;
100
83
 
101
- const samples = this.needsResample ? this.resample(input[0]) : input[0];
102
- this.accumulate(samples);
84
+ if (this.probeSamplesLeft > 0) this.probeForSilence(input[0]);
85
+
86
+ this.accumulate(input[0]);
103
87
  if (this.pendingLen >= this.targetSamples) this.flush();
104
88
  return true;
105
89
  }
@@ -109,7 +93,6 @@ registerProcessor('capture-processor', CaptureProcessor);
109
93
  `;
110
94
  /** Raw worklet source — exported so tests can evaluate the processor directly. */
111
95
  const captureProcessorSource = CaptureProcessorWorklet;
112
- const script = new Blob([CaptureProcessorWorklet], { type: "application/javascript" });
113
- const src = URL.createObjectURL(script);
96
+ var capture_processor_default = workletModuleUrl(CaptureProcessorWorklet);
114
97
  //#endregion
115
- export { captureProcessorSource, src as default };
98
+ export { captureProcessorSource, capture_processor_default as default };
@@ -1,4 +1,4 @@
1
1
  /** Raw worklet source — exported so tests can evaluate the processor directly. */
2
- export declare const playbackProcessorSource = "\nclass PlaybackProcessor extends AudioWorkletProcessor {\n constructor(options) {\n super();\n const rate = options.processorOptions?.sampleRate ?? 24000;\n // Wait for ~400ms of audio before starting.\n // If 'done' arrives first (short utterance), start immediately.\n this.jitterSamples = Math.floor(rate * 0.4);\n // Float32 ring buffer \u2014 60s at the context sample rate. Allocated once for\n // the node's lifetime; per-turn state resets via resetTurn(). writePos and\n // readPos are absolute (monotonic) sample counts; the buffer is indexed\n // modulo capacity so a reply longer than 60s keeps playing instead of\n // writing past the end and going silent.\n this.capacity = rate * 60;\n this.samples = new Float32Array(this.capacity);\n // Platform endianness probe: the wire format is PCM16 little-endian, so\n // the Int16Array fast path in ingestBytes is only valid on LE hosts\n // (every shipping browser target; the DataView path is the fallback).\n this.littleEndian = new Uint8Array(new Uint16Array([1]).buffer)[0] === 1;\n this.resetTurn();\n\n this.port.onmessage = (e) => {\n const d = e.data;\n if (d.event === 'write') {\n this.ingestBytes(d.buffer);\n } else if (d.event === 'interrupt') {\n this.interrupted = true;\n } else if (d.event === 'done') {\n this.isDone = true;\n }\n };\n }\n\n // Reset per-turn state so the node is reusable across replies without\n // reallocating the sample buffer or re-instantiating the worklet.\n resetTurn() {\n this.interrupted = false;\n this.isDone = false;\n this.playing = false;\n // Carry-over byte for split samples across chunks\n this.carry = null;\n this.writePos = 0;\n this.readPos = 0;\n }\n\n // End the current turn: notify the host and rearm for the next reply.\n // Must NOT return false from process() \u2014 a processor that stops is dead\n // for good, forcing a new node (and buffer) per reply.\n // `reason` ('interrupt' | 'done') tells the host which turn boundary this\n // stop belongs to: interrupt-stops are dropped host-side (flush() already\n // settled that turn), so they can never resolve a later turn's done() early.\n stopTurn(reason) {\n this.port.postMessage({ event: 'stop', reason });\n this.resetTurn();\n }\n\n ingestBytes(uint8) {\n let bytes = uint8;\n\n if (this.carry !== null) {\n const merged = new Uint8Array(1 + bytes.length);\n merged[0] = this.carry;\n merged.set(bytes, 1);\n bytes = merged;\n this.carry = null;\n }\n\n if (bytes.length % 2 !== 0) {\n this.carry = bytes[bytes.length - 1];\n bytes = bytes.subarray(0, bytes.length - 1);\n }\n\n if (bytes.length === 0) return;\n const numSamples = bytes.length / 2;\n const cap = this.capacity;\n const samples = this.samples;\n if (this.littleEndian && (bytes.byteOffset & 1) === 0) {\n // Fast path: 2-byte-aligned LE bytes wrap directly as an Int16Array;\n // copy wrap-aware in at most two runs with no per-sample DataView call\n // or modulo. This runs on the realtime audio thread.\n const int16 = new Int16Array(bytes.buffer, bytes.byteOffset, numSamples);\n let src = 0;\n let dst = this.writePos % cap;\n while (src < numSamples) {\n const run = Math.min(numSamples - src, cap - dst);\n for (let j = 0; j < run; j++) {\n samples[dst + j] = int16[src + j] / 0x8000;\n }\n src += run;\n dst = 0;\n }\n } else {\n // Odd byte offset (or big-endian host): fall back to per-sample reads.\n const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.length);\n let dst = this.writePos % cap;\n for (let i = 0; i < numSamples; i++) {\n samples[dst] = view.getInt16(i * 2, true) / 0x8000;\n dst++;\n if (dst === cap) dst = 0;\n }\n }\n this.writePos += numSamples;\n // If the producer outran the consumer by more than the buffer holds, drop\n // the oldest unplayed audio rather than reading samples we've overwritten.\n if (this.writePos - this.readPos > this.capacity) {\n this.readPos = this.writePos - this.capacity;\n }\n }\n\n process(inputs, outputs) {\n // No output wired up yet \u2014 nothing to render this quantum. Throwing here\n // would permanently kill the processor (the node is persistent per\n // session), so guard like the capture processor does.\n if (!outputs[0] || !outputs[0][0]) return true;\n const out = outputs[0][0];\n if (this.interrupted) {\n out.fill(0);\n this.stopTurn('interrupt');\n return true;\n }\n\n const avail = this.writePos - this.readPos;\n\n // Wait for jitter buffer to fill, unless done (short utterance)\n if (!this.playing) {\n if (avail >= this.jitterSamples || this.isDone) {\n this.playing = true;\n } else {\n out.fill(0);\n return true;\n }\n }\n\n if (avail > 0) {\n const n = Math.min(avail, out.length);\n // Copy from the ring buffer, splitting across the wrap boundary.\n const start = this.readPos % this.capacity;\n const first = Math.min(n, this.capacity - start);\n out.set(this.samples.subarray(start, start + first), 0);\n if (n > first) out.set(this.samples.subarray(0, n - first), first);\n this.readPos += n;\n out.fill(0, n);\n return true;\n }\n\n // No data: output silence, end the turn only when done\n out.fill(0);\n if (this.isDone) {\n this.stopTurn('done');\n }\n return true;\n }\n}\n\nregisterProcessor('playback-processor', PlaybackProcessor);\n";
3
- declare const src: string;
4
- export default src;
2
+ export declare const playbackProcessorSource = "\nclass PlaybackProcessor extends AudioWorkletProcessor {\n constructor(options) {\n super();\n const opts = options.processorOptions || {};\n // The node's context IS the playback context (audio.ts asserts its rate),\n // so the worklet-global sampleRate is authoritative; the option exists\n // for the node-less test harness.\n const rate = opts.sampleRate ?? sampleRate;\n // Fill target for the start of a turn. If 'done' arrives first (short\n // utterance), start immediately instead of waiting for audio that is\n // never coming.\n this.jitterSamples = Math.floor((rate * (opts.jitterMs ?? 400)) / 1000);\n // Fill target after an underrun \u2014 see PLAYBACK_REFILL_MS.\n this.refillSamples = Math.floor((rate * (opts.refillMs ?? 200)) / 1000);\n // Concealment source: a ring of the most recently played samples, looped\n // under a decaying gain to cover a gap. Sized to the fade window, with a\n // per-sample decay that reaches the floor exactly at its end.\n this.concealCapacity = Math.max(1, Math.floor((rate * 40) / 1000));\n this.concealBuf = new Float32Array(this.concealCapacity);\n this.concealDecay = Math.exp(Math.log(0.001) / this.concealCapacity);\n // Float32 ring buffer \u2014 PLAYBACK_BUFFER_SECONDS at the context sample\n // rate. Allocated once for the node's lifetime; per-turn state resets via\n // resetTurn(). writePos and readPos are absolute (monotonic) sample\n // counts; the buffer is indexed modulo capacity so a longer reply keeps\n // playing instead of writing past the end and going silent.\n this.capacity = rate * 60;\n this.samples = new Float32Array(this.capacity);\n // Platform endianness probe: the wire format is PCM16 little-endian, so\n // the Int16Array fast path in ingestBytes is only valid on LE hosts\n // (every shipping browser target; the DataView path is the fallback).\n this.littleEndian = new Uint8Array(new Uint16Array([1]).buffer)[0] === 1;\n this.resetTurn();\n\n this.port.onmessage = (e) => {\n const d = e.data;\n if (d.event === 'write') {\n this.ingestBytes(d.buffer);\n } else if (d.event === 'interrupt') {\n // Applied eagerly, not deferred to the next process(): onmessage and\n // process() never interleave (one audio thread), and a deferred\n // interrupt would let 'write'/'done' frames for the NEXT turn\n // coalesce in ahead of it and be wiped by the reset along with the\n // cancelled turn's audio.\n this.stopTurn('interrupt');\n } else if (d.event === 'done') {\n this.isDone = true;\n }\n };\n }\n\n // Reset per-turn state so the node is reusable across replies without\n // reallocating the sample buffer or re-instantiating the worklet.\n resetTurn() {\n this.isDone = false;\n this.playing = false;\n // Whether any real audio has been rendered this turn. Separates a turn's\n // pre-roll (nothing to extrapolate from, and not a defect) from a\n // mid-turn underrun.\n this.hasPlayed = false;\n this.fillTarget = this.jitterSamples;\n // Carry-over byte for split samples across chunks\n this.carry = null;\n this.writePos = 0;\n this.readPos = 0;\n // Concealment ring state and the current fade position.\n this.concealLen = 0;\n this.concealWrite = 0;\n this.concealPos = 0;\n this.concealGain = 1;\n // Episode flags, so a multi-quantum gap counts as one event.\n this.concealing = false;\n this.concealedSilence = false;\n // Reported to the host on 'stop'. A fresh object per turn: the one just\n // posted must not be mutated by the next turn.\n this.stats = {\n concealedSamples: 0,\n silentConcealedSamples: 0,\n concealmentEvents: 0,\n silentConcealmentEvents: 0,\n };\n }\n\n // End the current turn: notify the host and rearm for the next reply.\n // Must NOT return false from process() \u2014 a processor that stops is dead\n // for good, forcing a new node (and buffer) per reply.\n // `reason` ('interrupt' | 'done') tells the host which turn boundary this\n // stop belongs to: interrupt-stops are dropped host-side (flush() already\n // settled that turn), so they can never resolve a later turn's done() early.\n stopTurn(reason) {\n this.port.postMessage({ event: 'stop', reason, stats: this.stats });\n this.resetTurn();\n }\n\n // Cover a quantum (from `start`) where real audio should have been.\n //\n // Before the turn's first samples there is nothing to extrapolate from, so\n // the gap is plain silence and counted as nothing \u2014 WebRTC likewise only\n // counts concealment once playout has begun. After that, loop the retained\n // tail under a decaying gain: a hard zero-fill is a discontinuity mid-word,\n // which is the click that makes a brief stall sound like breakage.\n coverGap(out, start) {\n if (!this.hasPlayed) {\n out.fill(0, start);\n return;\n }\n if (!this.concealing) {\n this.concealing = true;\n this.concealedSilence = false;\n this.stats.concealmentEvents++;\n }\n const total = out.length - start;\n const len = this.concealLen;\n let silent = 0;\n // Second condition: once the fade has decayed to the floor, every sample\n // the loop would emit is 0 anyway \u2014 bulk-fill instead of running 128\n // branchy iterations per quantum for the whole tail of a long stall.\n if (len === 0 || this.concealGain < 0.001) {\n out.fill(0, start);\n silent = total;\n } else {\n let g = this.concealGain;\n for (let i = start; i < out.length; i++) {\n if (g < 0.001) {\n // The fade has run out: keep counting the gap, but stop looping a\n // fragment that is now inaudible anyway.\n out[i] = 0;\n silent++;\n continue;\n }\n out[i] = this.concealBuf[this.concealPos] * g;\n this.concealPos = this.concealPos + 1 === len ? 0 : this.concealPos + 1;\n g *= this.concealDecay;\n }\n this.concealGain = g;\n }\n this.stats.concealedSamples += total;\n if (silent > 0) {\n this.stats.silentConcealedSamples += silent;\n if (!this.concealedSilence) {\n this.concealedSilence = true;\n this.stats.silentConcealmentEvents++;\n }\n }\n }\n\n // Retain the tail of a rendered quantum as the next gap's concealment\n // source, and close any episode the real audio just ended.\n rememberTail(out, n) {\n const cap = this.concealCapacity;\n const take = Math.min(n, cap);\n // Bulk copies (this runs on every cleanly rendered quantum): the tail is\n // one contiguous source run, landing in at most two ring runs.\n const tail = out.subarray(n - take, n);\n const first = Math.min(take, cap - this.concealWrite);\n this.concealBuf.set(tail.subarray(0, first), this.concealWrite);\n if (take > first) this.concealBuf.set(tail.subarray(first), 0);\n this.concealWrite = (this.concealWrite + take) % cap;\n this.concealLen = Math.min(cap, this.concealLen + take);\n // Read the loop oldest-first; once the ring is full the write cursor is\n // the oldest retained sample.\n this.concealPos = this.concealLen === cap ? this.concealWrite : 0;\n this.concealing = false;\n this.concealGain = 1;\n }\n\n ingestBytes(uint8) {\n let bytes = uint8;\n\n if (this.carry !== null) {\n const merged = new Uint8Array(1 + bytes.length);\n merged[0] = this.carry;\n merged.set(bytes, 1);\n bytes = merged;\n this.carry = null;\n }\n\n if (bytes.length % 2 !== 0) {\n this.carry = bytes[bytes.length - 1];\n bytes = bytes.subarray(0, bytes.length - 1);\n }\n\n if (bytes.length === 0) return;\n const numSamples = bytes.length / 2;\n const cap = this.capacity;\n const samples = this.samples;\n if (this.littleEndian && (bytes.byteOffset & 1) === 0) {\n // Fast path: 2-byte-aligned LE bytes wrap directly as an Int16Array;\n // copy wrap-aware in at most two runs with no per-sample DataView call\n // or modulo. This runs on the realtime audio thread.\n const int16 = new Int16Array(bytes.buffer, bytes.byteOffset, numSamples);\n let src = 0;\n let dst = this.writePos % cap;\n while (src < numSamples) {\n const run = Math.min(numSamples - src, cap - dst);\n for (let j = 0; j < run; j++) {\n samples[dst + j] = int16[src + j] / 0x8000;\n }\n src += run;\n dst = 0;\n }\n } else {\n // Odd byte offset (or big-endian host): fall back to per-sample reads.\n const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.length);\n let dst = this.writePos % cap;\n for (let i = 0; i < numSamples; i++) {\n samples[dst] = view.getInt16(i * 2, true) / 0x8000;\n dst++;\n if (dst === cap) dst = 0;\n }\n }\n this.writePos += numSamples;\n // If the producer outran the consumer by more than the buffer holds, drop\n // the oldest unplayed audio rather than reading samples we've overwritten.\n if (this.writePos - this.readPos > this.capacity) {\n this.readPos = this.writePos - this.capacity;\n }\n }\n\n process(inputs, outputs) {\n // No output wired up yet \u2014 nothing to render this quantum. Throwing here\n // would permanently kill the processor (the node is persistent per\n // session), so guard like the capture processor does.\n if (!outputs[0] || !outputs[0][0]) return true;\n const out = outputs[0][0];\n const avail = this.writePos - this.readPos;\n\n // Filling: wait for the target. 'done' short-circuits it \u2014 what is\n // buffered is all there will be, so there is nothing left to wait for.\n if (!this.playing) {\n if (avail >= this.fillTarget || this.isDone) {\n this.playing = true;\n } else {\n this.coverGap(out, 0);\n return true;\n }\n }\n\n // Underrun: this quantum cannot be filled and more audio is still coming.\n // Go back to filling (at the refill target) and cover the gap, leaving\n // readPos untouched \u2014 the fragment stays buffered and plays intact once\n // the buffer recovers, instead of being dribbled out a few samples at a\n // time for the rest of the turn.\n if (avail < out.length && !this.isDone) {\n this.playing = false;\n this.fillTarget = this.refillSamples;\n this.coverGap(out, 0);\n return true;\n }\n\n if (avail > 0) {\n const n = Math.min(avail, out.length);\n // Copy from the ring buffer, splitting across the wrap boundary.\n const start = this.readPos % this.capacity;\n const first = Math.min(n, this.capacity - start);\n out.set(this.samples.subarray(start, start + first), 0);\n if (n > first) out.set(this.samples.subarray(0, n - first), first);\n this.readPos += n;\n // Only reachable with n < out.length on the turn's final partial\n // quantum (the underrun branch above catches every other case).\n out.fill(0, n);\n this.hasPlayed = true;\n this.rememberTail(out, n);\n return true;\n }\n\n // Drained and done: end the turn. Not reachable mid-turn \u2014 an empty\n // buffer with audio still coming is the underrun branch above.\n out.fill(0);\n if (this.isDone) {\n this.stopTurn('done');\n }\n return true;\n }\n}\n\nregisterProcessor('playback-processor', PlaybackProcessor);\n";
3
+ declare const _default: string;
4
+ export default _default;