@alexkroman1/aai-ui 1.9.2 → 1.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/dist/{_colors-DJUordGv.js → _colors-DYX7XRTr.js} +8 -0
  2. package/dist/{_utils-5cs73OrA.js → _utils-CN1yVgYS.js} +2 -5
  3. package/dist/audio.d.ts +6 -0
  4. package/dist/audio.js +62 -9
  5. package/dist/{chat-view-C1XbqDk0.js → chat-view-u3yBAlig.js} +16 -14
  6. package/dist/components/_colors.d.ts +0 -4
  7. package/dist/components/chat-view.js +1 -1
  8. package/dist/components/controls.d.ts +2 -2
  9. package/dist/components/controls.js +1 -1
  10. package/dist/components/message-list.d.ts +2 -2
  11. package/dist/components/message-list.js +70 -64
  12. package/dist/components/start-screen.js +1 -1
  13. package/dist/components/text-controls.d.ts +2 -2
  14. package/dist/components/tool-call-block.d.ts +6 -2
  15. package/dist/components/tool-call-block.js +1 -1
  16. package/dist/context.d.ts +3 -11
  17. package/dist/context.js +9 -22
  18. package/dist/{controls-BngrPbOC.js → controls-B2EPUJDU.js} +12 -9
  19. package/dist/default-client/assets/audio-CSwpo0gM.js +1 -0
  20. package/dist/default-client/assets/{capture-processor-C19oBn4L.js → capture-processor-UlKEKyIW.js} +4 -1
  21. package/dist/default-client/assets/index-B2bISVbm.css +2 -0
  22. package/dist/default-client/assets/index-BTqSueyz.js +94 -0
  23. package/dist/default-client/assets/{playback-processor-BtlzAH78.js → playback-processor-C5HVRVbu.js} +11 -4
  24. package/dist/default-client/index.html +2 -2
  25. package/dist/define-client.js +3 -3
  26. package/dist/hooks.js +5 -2
  27. package/dist/index.d.ts +4 -0
  28. package/dist/index.js +354 -5
  29. package/dist/{session-core-D3NDaySY.js → session-core-KUoY-0JO.js} +251 -114
  30. package/dist/session-core-audio-setup.d.ts +36 -0
  31. package/dist/session-core-messages.d.ts +12 -0
  32. package/dist/session-core-reconnect.d.ts +19 -0
  33. package/dist/session-core-url.d.ts +2 -0
  34. package/dist/session-core.js +1 -1
  35. package/dist/sync-mic.d.ts +49 -0
  36. package/dist/sync-session.d.ts +49 -0
  37. package/dist/sync-vad.d.ts +54 -0
  38. package/dist/{tool-call-block-7f1GTG-P.js → tool-call-block-DIxpG8GM.js} +11 -6
  39. package/dist/types.d.ts +12 -14
  40. package/dist/types.js +1 -13
  41. package/dist/worklets/capture-processor.d.ts +1 -1
  42. package/dist/worklets/capture-processor.js +4 -1
  43. package/dist/worklets/playback-processor.d.ts +1 -1
  44. package/dist/worklets/playback-processor.js +11 -4
  45. package/package.json +8 -5
  46. package/dist/default-client/assets/audio-Cs-6t_Wd.js +0 -1
  47. package/dist/default-client/assets/index-Bf4ZTNcx.js +0 -73
  48. package/dist/default-client/assets/index-Bzlh9i7w.css +0 -2
@@ -1,7 +1,108 @@
1
- import "./types.js";
1
+ import { FILE_SEND_BACKOFF_MS, MIC_SEND_MAX_BUFFERED_BYTES } from "./types.js";
2
2
  import { DEFAULT_MAX_HISTORY, FILE_UPLOAD_CHUNK_BYTES, WS_OPEN, errorMessage, safeJsonParse } from "@alexkroman1/aai";
3
3
  import { ServerMessageSchema, lenientParse } from "@alexkroman1/aai/protocol";
4
+ import ReconnectingWebSocket from "partysocket/ws";
4
5
  import { MAX_SYNC_AUDIO_SECONDS } from "@alexkroman1/aai/stt";
6
+ //#region session-core-audio-setup.ts
7
+ /**
8
+ * Audio-path initialization for the voice session core.
9
+ *
10
+ * Split out of `session-core.ts`: this module owns the async
11
+ * mic-permission → worklet-registration → `VoiceIO` bring-up (and its
12
+ * staleness/failure handling), while `session-core.ts` owns the state store
13
+ * and connection lifecycle.
14
+ */
15
+ /**
16
+ * Initialize audio capture and playback after the server sends a ready config.
17
+ *
18
+ * Lifecycle: dynamically import audio modules -> request microphone access ->
19
+ * register AudioWorklet processors -> create a `VoiceIO` instance -> send
20
+ * `audio_ready` to the server -> transition state to `"listening"`.
21
+ *
22
+ * Uses the connection `generation` counter to detect if `connect()` was called
23
+ * (or a reconnect happened) while awaiting async operations; if so, the stale
24
+ * VoiceIO is closed immediately to prevent it from being assigned to a newer
25
+ * connection.
26
+ *
27
+ * `fatal` selects the failure mode: voice sessions (`config` handshake) can't
28
+ * function without the mic, so failure there sets the error state and ends
29
+ * the session; a text-only session's opt-in record button must not brick an
30
+ * otherwise healthy session, so failure there is a banner (`error` set,
31
+ * state kept) and the session stays interactive.
32
+ */
33
+ async function initAudioCapture(conn, msg, deps, fatal) {
34
+ if (conn.audioSetupInFlight) return;
35
+ conn.audioSetupInFlight = true;
36
+ const gen = conn.generation;
37
+ const stale = () => conn.generation !== gen || !conn.ws || conn.ws.readyState !== WS_OPEN;
38
+ const reportAudioFailure = (message) => {
39
+ if (fatal) deps.updateState({
40
+ state: "error",
41
+ error: {
42
+ code: "audio",
43
+ message
44
+ },
45
+ running: false,
46
+ recording: false
47
+ });
48
+ else {
49
+ deps.cleanupAudio();
50
+ deps.updateState({
51
+ error: {
52
+ code: "audio",
53
+ message
54
+ },
55
+ recording: false
56
+ });
57
+ }
58
+ };
59
+ try {
60
+ const [{ createVoiceIO }, captureWorklet, playbackWorklet] = await Promise.all([
61
+ import("./audio.js"),
62
+ import("./worklets/capture-processor.js").then((m) => m.default),
63
+ import("./worklets/playback-processor.js").then((m) => m.default)
64
+ ]);
65
+ const io = await createVoiceIO({
66
+ sttSampleRate: msg.sampleRate,
67
+ ttsSampleRate: msg.ttsSampleRate,
68
+ captureWorkletSrc: captureWorklet,
69
+ playbackWorkletSrc: playbackWorklet,
70
+ onMicData: (pcm16) => {
71
+ try {
72
+ deps.sendAudio(new Uint8Array(pcm16));
73
+ } catch {
74
+ console.debug("[aai-ui] sendAudio dropped: connection closed");
75
+ }
76
+ },
77
+ onError: (err) => {
78
+ if (conn.generation !== gen) return;
79
+ reportAudioFailure(err.message);
80
+ }
81
+ });
82
+ if (stale()) {
83
+ io.close().catch(() => {});
84
+ return;
85
+ }
86
+ conn.voiceIO?.close().catch(() => {});
87
+ conn.voiceIO = io;
88
+ if (conn.preInitAudio.length > 0) {
89
+ for (const chunk of conn.preInitAudio) io.enqueue(chunk.buffer);
90
+ conn.preInitAudio = [];
91
+ }
92
+ deps.sendJson({ type: "audio_ready" });
93
+ deps.updateState({ recording: true });
94
+ if (conn.preInitDone) {
95
+ conn.preInitDone = false;
96
+ deps.settleWhenAudioDrained(io);
97
+ } else deps.updateState({ state: "listening" });
98
+ } catch (err) {
99
+ if (stale()) return;
100
+ reportAudioFailure(`Microphone access failed: ${errorMessage(err)}`);
101
+ } finally {
102
+ if (conn.generation === gen) conn.audioSetupInFlight = false;
103
+ }
104
+ }
105
+ //#endregion
5
106
  //#region session-core-messages.ts
6
107
  /**
7
108
  * Incoming-message handling for the voice session core.
@@ -48,7 +149,7 @@ function appendCapped(list, item, cap) {
48
149
  * dedup) that previously lived as closure locals in `createSessionCore`.
49
150
  */
50
151
  function createMessageHandlers(deps) {
51
- const { getSnapshot, updateState, conn } = deps;
152
+ const { getSnapshot, updateState, conn, discardUpload, cleanupAudio } = deps;
52
153
  /** Incremented on each turn boundary -- stale async callbacks compare against this. */
53
154
  let handlerGeneration = 0;
54
155
  /** Monotonically increasing counter for custom events -- used by useEvent to deduplicate. */
@@ -93,7 +194,7 @@ function createMessageHandlers(deps) {
93
194
  function clearRecoveredError() {
94
195
  const snap = getSnapshot();
95
196
  if (snap.state === "error") updateState({
96
- state: "disconnected",
197
+ state: "listening",
97
198
  error: null
98
199
  });
99
200
  else if (snap.error !== null) updateState({ error: null });
@@ -104,14 +205,18 @@ function createMessageHandlers(deps) {
104
205
  code: e.code,
105
206
  message: e.message
106
207
  } });
107
- else updateState({
108
- state: "error",
109
- error: {
110
- code: e.code,
111
- message: e.message
112
- },
113
- running: false
114
- });
208
+ else {
209
+ cleanupAudio();
210
+ updateState({
211
+ state: "error",
212
+ error: {
213
+ code: e.code,
214
+ message: e.message
215
+ },
216
+ running: false,
217
+ recording: false
218
+ });
219
+ }
115
220
  }
116
221
  /** Single entry point for all server->client session events. */
117
222
  function handleEvent(e) {
@@ -169,6 +274,7 @@ function createMessageHandlers(deps) {
169
274
  break;
170
275
  case "reset":
171
276
  handlerGeneration++;
277
+ discardUpload();
172
278
  conn.voiceIO?.flush();
173
279
  updateState({
174
280
  ...CLEARED_SESSION_STATE,
@@ -188,25 +294,31 @@ function createMessageHandlers(deps) {
188
294
  /** Enqueue a PCM16 audio chunk for playback. Transitions state to `"speaking"` on the first chunk. */
189
295
  function playAudioChunk(chunk) {
190
296
  const snap = getSnapshot();
191
- if (snap.state === "disconnected" && snap.error !== null) return;
297
+ if (snap.state === "error" || snap.state === "disconnected" && snap.error !== null) return;
192
298
  if (snap.state !== "speaking") updateState({ state: "speaking" });
193
299
  if (conn.voiceIO) conn.voiceIO.enqueue(chunk.buffer);
194
300
  else if (conn.preInitAudio.length < MAX_PREINIT_AUDIO_CHUNKS) conn.preInitAudio.push(chunk);
195
301
  }
302
+ /** See {@link MessageHandlers.settleWhenAudioDrained}. Captures
303
+ * `handlerGeneration` so a completion (or failure) that lands after a turn
304
+ * boundary is discarded instead of overwriting the newer turn's state. */
305
+ function settleWhenAudioDrained(io) {
306
+ const gen = handlerGeneration;
307
+ io.done().then(() => {
308
+ if (handlerGeneration !== gen) return;
309
+ updateState({ state: "listening" });
310
+ }).catch((err) => {
311
+ console.warn("Audio playback done failed:", err);
312
+ });
313
+ }
196
314
  /**
197
315
  * Signal that the server has finished sending audio for this turn.
198
316
  * Waits for the audio queue to drain, then transitions state to `"listening"`.
199
317
  * Uses the `handlerGeneration` counter to discard stale completions from interrupted turns.
200
318
  */
201
319
  function playAudioDone() {
202
- const gen = handlerGeneration;
203
320
  const io = conn.voiceIO;
204
- if (io) io.done().then(() => {
205
- if (handlerGeneration !== gen) return;
206
- updateState({ state: "listening" });
207
- }).catch((err) => {
208
- console.warn("Audio playback done failed:", err);
209
- });
321
+ if (io) settleWhenAudioDrained(io);
210
322
  else {
211
323
  conn.preInitDone = true;
212
324
  updateState({ state: "listening" });
@@ -244,7 +356,46 @@ function createMessageHandlers(deps) {
244
356
  }
245
357
  handleEvent(msg);
246
358
  }
247
- return { handleMessage };
359
+ return {
360
+ handleMessage,
361
+ settleWhenAudioDrained
362
+ };
363
+ }
364
+ //#endregion
365
+ //#region session-core-reconnect.ts
366
+ /**
367
+ * Automatic reconnection for the browser session socket, built on
368
+ * partysocket's `ReconnectingWebSocket`. Kept out of `session-core.ts` so
369
+ * the state machine there reads as protocol logic, not socket plumbing.
370
+ */
371
+ /**
372
+ * Backoff for automatic reconnects after an unexpected close (partysocket):
373
+ * exponential from 1s, capped at 15s, giving up after 10 attempts. Applies
374
+ * only to the default socket implementation — an injected
375
+ * `options.WebSocket` (tests) never reconnects on its own.
376
+ */
377
+ const RECONNECT_OPTIONS = {
378
+ minReconnectionDelay: 1e3,
379
+ maxReconnectionDelay: 15e3,
380
+ reconnectionDelayGrowFactor: 2,
381
+ maxRetries: 10
382
+ };
383
+ /**
384
+ * Open partysocket's reconnecting WebSocket. The URL is a *provider*,
385
+ * re-evaluated on every attempt, so each retry picks up the current resume
386
+ * URL rather than the one the session started with.
387
+ */
388
+ function openReconnectingSocket(urlProvider) {
389
+ return new ReconnectingWebSocket(urlProvider, void 0, RECONNECT_OPTIONS);
390
+ }
391
+ /**
392
+ * True while `socket` is a reconnecting socket that will retry after the
393
+ * close event currently being handled. partysocket schedules the retry
394
+ * *before* dispatching `close`, so `retryCount` already names the attempt
395
+ * just scheduled — at `maxRetries` it has given up.
396
+ */
397
+ function reconnectPending(socket) {
398
+ return socket instanceof ReconnectingWebSocket && socket.shouldReconnect && socket.retryCount < RECONNECT_OPTIONS.maxRetries;
248
399
  }
249
400
  //#endregion
250
401
  //#region session-core-upload.ts
@@ -273,7 +424,7 @@ function createUploadSender(deps) {
273
424
  * closes or the session is reset (`epoch` moves on) between chunks. */
274
425
  async function sendBytesReliably(bytes, epoch) {
275
426
  for (let i = 0; i < bytes.byteLength; i += FILE_UPLOAD_CHUNK_BYTES) {
276
- while (epoch === uploadEpoch && conn.ws && conn.ws.readyState === WS_OPEN && conn.ws.bufferedAmount > 65536) await new Promise((r) => setTimeout(r, 50));
427
+ while (epoch === uploadEpoch && conn.ws && conn.ws.readyState === WS_OPEN && conn.ws.bufferedAmount > MIC_SEND_MAX_BUFFERED_BYTES) await new Promise((r) => setTimeout(r, FILE_SEND_BACKOFF_MS));
277
428
  if (epoch !== uploadEpoch) throw new Error("sendAudioFile: session was reset mid-send");
278
429
  if (!conn.ws || conn.ws.readyState !== WS_OPEN) throw new Error("sendAudioFile: connection closed mid-send");
279
430
  conn.ws.send(bytes.subarray(i, i + FILE_UPLOAD_CHUNK_BYTES));
@@ -325,6 +476,16 @@ function createUploadSender(deps) {
325
476
  };
326
477
  }
327
478
  //#endregion
479
+ //#region session-core-url.ts
480
+ /** Build the session WebSocket URL from the platform URL and resume state. */
481
+ function buildWsUrl(platformUrl, resume, sessionId) {
482
+ const wsUrl = new URL("websocket", platformUrl.endsWith("/") ? platformUrl : `${platformUrl}/`);
483
+ wsUrl.protocol = wsUrl.protocol === "https:" ? "wss:" : "ws:";
484
+ if (sessionId) wsUrl.searchParams.set("sessionId", sessionId);
485
+ else if (resume) wsUrl.searchParams.set("resume", "1");
486
+ return wsUrl;
487
+ }
488
+ //#endregion
328
489
  //#region session-core.ts
329
490
  /**
330
491
  * Framework-agnostic voice session core.
@@ -339,82 +500,6 @@ function createUploadSender(deps) {
339
500
  * No dependency on React, Preact, or any UI framework.
340
501
  */
341
502
  /**
342
- * Initialize audio capture and playback after the server sends a ready config.
343
- *
344
- * Lifecycle: dynamically import audio modules -> request microphone access ->
345
- * register AudioWorklet processors -> create a `VoiceIO` instance -> send
346
- * `audio_ready` to the server -> transition state to `"listening"`.
347
- *
348
- * Uses the connection `generation` counter to detect if `connect()` was called
349
- * while awaiting async operations; if so, the stale VoiceIO is closed immediately
350
- * to prevent it from being assigned to a newer connection.
351
- *
352
- * On failure (e.g. microphone permission denied, WebSocket closed mid-setup),
353
- * sets the error state and transitions to `"disconnected"`.
354
- */
355
- async function initAudioCapture(conn, msg, deps) {
356
- if (conn.audioSetupInFlight) return;
357
- conn.audioSetupInFlight = true;
358
- const gen = conn.generation;
359
- try {
360
- const [{ createVoiceIO }, captureWorklet, playbackWorklet] = await Promise.all([
361
- import("./audio.js"),
362
- import("./worklets/capture-processor.js").then((m) => m.default),
363
- import("./worklets/playback-processor.js").then((m) => m.default)
364
- ]);
365
- const io = await createVoiceIO({
366
- sttSampleRate: msg.sampleRate,
367
- ttsSampleRate: msg.ttsSampleRate,
368
- captureWorkletSrc: captureWorklet,
369
- playbackWorkletSrc: playbackWorklet,
370
- onMicData: (pcm16) => {
371
- try {
372
- deps.sendAudio(new Uint8Array(pcm16));
373
- } catch {
374
- console.debug("[aai-ui] sendAudio dropped: connection closed");
375
- }
376
- }
377
- });
378
- if (conn.generation !== gen || !conn.ws || conn.ws.readyState !== WS_OPEN) {
379
- io.close();
380
- return;
381
- }
382
- conn.voiceIO = io;
383
- if (conn.preInitAudio.length > 0) {
384
- for (const chunk of conn.preInitAudio) io.enqueue(chunk.buffer);
385
- conn.preInitAudio = [];
386
- }
387
- deps.sendJson({ type: "audio_ready" });
388
- deps.updateState({ recording: true });
389
- if (conn.preInitDone) {
390
- conn.preInitDone = false;
391
- io.done().then(() => {
392
- if (conn.generation !== gen) return;
393
- deps.updateState({ state: "listening" });
394
- }).catch(() => deps.updateState({ state: "listening" }));
395
- } else deps.updateState({ state: "listening" });
396
- } catch (err) {
397
- if (conn.generation !== gen || !conn.ws || conn.ws.readyState !== WS_OPEN) return;
398
- deps.updateState({
399
- state: "error",
400
- error: {
401
- code: "audio",
402
- message: `Microphone access failed: ${errorMessage(err)}`
403
- },
404
- running: false
405
- });
406
- } finally {
407
- conn.audioSetupInFlight = false;
408
- }
409
- }
410
- function buildWsUrl(platformUrl, resume, sessionId) {
411
- const wsUrl = new URL("websocket", platformUrl.endsWith("/") ? platformUrl : `${platformUrl}/`);
412
- wsUrl.protocol = wsUrl.protocol === "https:" ? "wss:" : "ws:";
413
- if (sessionId) wsUrl.searchParams.set("sessionId", sessionId);
414
- else if (resume) wsUrl.searchParams.set("resume", "1");
415
- return wsUrl;
416
- }
417
- /**
418
503
  * Create a framework-agnostic voice session core that connects to an AAI
419
504
  * server via WebSocket.
420
505
  *
@@ -427,7 +512,6 @@ function buildWsUrl(platformUrl, resume, sessionId) {
427
512
  * @public
428
513
  */
429
514
  function createSessionCore(options) {
430
- const WS = options.WebSocket ?? WebSocket;
431
515
  let currentSnapshot = {
432
516
  ...CLEARED_SESSION_STATE,
433
517
  state: "disconnected",
@@ -483,7 +567,7 @@ function createSessionCore(options) {
483
567
  function cleanupAudio() {
484
568
  upload.discard();
485
569
  conn.audioSetupInFlight = false;
486
- conn.voiceIO?.close();
570
+ conn.voiceIO?.close().catch(() => {});
487
571
  conn.voiceIO = null;
488
572
  conn.preInitAudio = [];
489
573
  conn.preInitDone = false;
@@ -496,24 +580,28 @@ function createSessionCore(options) {
496
580
  }
497
581
  function sendAudio(bytes) {
498
582
  if (!conn.ws || conn.ws.readyState !== WS_OPEN) return;
499
- if (conn.ws.bufferedAmount > 65536) return;
583
+ if (conn.ws.bufferedAmount > MIC_SEND_MAX_BUFFERED_BYTES) return;
500
584
  conn.ws.send(bytes);
501
585
  }
502
- const audioDeps = {
503
- sendJson,
504
- sendAudio,
505
- updateState
506
- };
507
586
  const upload = createUploadSender({
508
587
  conn,
509
588
  getSnapshot,
510
589
  sendJson
511
590
  });
512
- const { handleMessage } = createMessageHandlers({
591
+ const { handleMessage, settleWhenAudioDrained } = createMessageHandlers({
513
592
  getSnapshot,
514
593
  updateState,
515
- conn
594
+ conn,
595
+ discardUpload: upload.discard,
596
+ cleanupAudio
516
597
  });
598
+ const audioDeps = {
599
+ sendJson,
600
+ sendAudio,
601
+ updateState,
602
+ settleWhenAudioDrained,
603
+ cleanupAudio
604
+ };
517
605
  /** Abort the in-flight connection and release audio + WebSocket resources. */
518
606
  function teardownConnection() {
519
607
  connectionController?.abort();
@@ -534,7 +622,7 @@ function createSessionCore(options) {
534
622
  };
535
623
  const audioOut = config.audioOut !== false;
536
624
  updateState({ audioOut });
537
- if (audioOut) initAudioCapture(conn, config, audioDeps);
625
+ if (audioOut) initAudioCapture(conn, config, audioDeps, true);
538
626
  else {
539
627
  sendJson({ type: "audio_ready" });
540
628
  updateState({ state: "listening" });
@@ -547,7 +635,27 @@ function createSessionCore(options) {
547
635
  }))
548
636
  });
549
637
  }
638
+ /**
639
+ * The WebSocket URL for the *next* connection attempt. Evaluated per
640
+ * attempt (partysocket takes it as a URL provider), so once the first
641
+ * `config` arrives, every reconnect — automatic or explicit — carries
642
+ * `resume=1` and the session resumes instead of starting over.
643
+ */
644
+ function currentWsUrl() {
645
+ const resumeId = !hasConnected ? options.resumeSessionId : void 0;
646
+ return buildWsUrl(options.platformUrl, hasConnected, resumeId).toString();
647
+ }
648
+ /** Open a socket: an injected constructor as-is (tests), or partysocket's
649
+ * reconnecting WebSocket — same interface, plus reconnect-on-close. */
650
+ function openSocket() {
651
+ if (options.WebSocket) return new options.WebSocket(currentWsUrl());
652
+ return openReconnectingSocket(currentWsUrl);
653
+ }
550
654
  function connect(opts) {
655
+ if (opts?.signal?.aborted) {
656
+ disconnect();
657
+ return;
658
+ }
551
659
  updateState({
552
660
  state: "connecting",
553
661
  error: null
@@ -558,11 +666,10 @@ function createSessionCore(options) {
558
666
  connectionController = controller;
559
667
  const { signal: sig } = controller;
560
668
  if (opts?.signal) opts.signal.addEventListener("abort", () => disconnect(), { signal: sig });
561
- const resumeId = !hasConnected ? options.resumeSessionId : void 0;
562
- const wsUrl = buildWsUrl(options.platformUrl, hasConnected, resumeId);
563
- const socket = new WS(wsUrl.toString());
669
+ const socket = openSocket();
564
670
  socket.binaryType = "arraybuffer";
565
671
  conn.ws = socket;
672
+ let socketErrored = false;
566
673
  socket.addEventListener("open", () => {
567
674
  updateState({ state: "ready" });
568
675
  }, { signal: sig });
@@ -570,18 +677,47 @@ function createSessionCore(options) {
570
677
  const config = handleMessage(event.data);
571
678
  if (config) onServerConfig(config);
572
679
  }, { signal: sig });
680
+ socket.addEventListener("error", () => {
681
+ socketErrored = true;
682
+ }, { signal: sig });
573
683
  socket.addEventListener("close", () => {
574
684
  if (sig.aborted) return;
575
- controller.abort();
576
685
  cleanupAudio();
577
- updateState({
686
+ if (reconnectPending(socket)) {
687
+ conn.generation++;
688
+ socketErrored = false;
689
+ updateState({
690
+ state: "connecting",
691
+ recording: false
692
+ });
693
+ return;
694
+ }
695
+ controller.abort();
696
+ socket.close();
697
+ conn.ws = null;
698
+ if (socketErrored) updateState({
699
+ state: "error",
700
+ error: {
701
+ code: "connection",
702
+ message: "WebSocket connection error"
703
+ },
704
+ running: false,
705
+ recording: false
706
+ });
707
+ else if (currentSnapshot.state === "error") updateState({
708
+ running: false,
709
+ recording: false
710
+ });
711
+ else updateState({
578
712
  state: "disconnected",
713
+ error: null,
579
714
  running: false,
580
715
  recording: false
581
716
  });
582
717
  }, { signal: sig });
583
718
  }
584
719
  function cancel() {
720
+ if (!conn.ws || conn.ws.readyState !== WS_OPEN) return;
585
721
  conn.voiceIO?.flush();
586
722
  updateState({ state: "listening" });
587
723
  sendJson({ type: "cancel" });
@@ -595,6 +731,7 @@ function createSessionCore(options) {
595
731
  }
596
732
  resetState();
597
733
  disconnect();
734
+ updateState({ running: true });
598
735
  connect();
599
736
  }
600
737
  function disconnect() {
@@ -609,7 +746,7 @@ function createSessionCore(options) {
609
746
  if (currentSnapshot.audioOut || currentSnapshot.recording || conn.audioSetupInFlight || upload.inFlight()) return;
610
747
  const cfg = conn.readyConfig;
611
748
  if (!(cfg && conn.ws) || conn.ws.readyState !== WS_OPEN) return;
612
- initAudioCapture(conn, cfg, audioDeps);
749
+ initAudioCapture(conn, cfg, audioDeps, false);
613
750
  }
614
751
  function stopRecording() {
615
752
  if (currentSnapshot.audioOut || !currentSnapshot.recording) return;
@@ -0,0 +1,36 @@
1
+ import type { ClientMessage } from "@alexkroman1/aai/protocol";
2
+ import type { VoiceIO } from "./audio.ts";
3
+ import type { ConnState, SessionSnapshot } from "./session-core-types.ts";
4
+ /** Dependencies `initAudioCapture` needs from the owning session core. */
5
+ export type AudioSetupDeps = {
6
+ sendJson: (msg: ClientMessage) => void;
7
+ sendAudio: (bytes: Uint8Array) => void;
8
+ updateState: (partial: Partial<SessionSnapshot>) => void;
9
+ /** Turn-boundary-guarded drain from the message handlers — replays a
10
+ * buffered `audio_done` without stomping a barge-in's state. */
11
+ settleWhenAudioDrained: (io: VoiceIO) => void;
12
+ /** Release the mic/VoiceIO (used when the audio path dies non-fatally). */
13
+ cleanupAudio: () => void;
14
+ };
15
+ /**
16
+ * Initialize audio capture and playback after the server sends a ready config.
17
+ *
18
+ * Lifecycle: dynamically import audio modules -> request microphone access ->
19
+ * register AudioWorklet processors -> create a `VoiceIO` instance -> send
20
+ * `audio_ready` to the server -> transition state to `"listening"`.
21
+ *
22
+ * Uses the connection `generation` counter to detect if `connect()` was called
23
+ * (or a reconnect happened) while awaiting async operations; if so, the stale
24
+ * VoiceIO is closed immediately to prevent it from being assigned to a newer
25
+ * connection.
26
+ *
27
+ * `fatal` selects the failure mode: voice sessions (`config` handshake) can't
28
+ * function without the mic, so failure there sets the error state and ends
29
+ * the session; a text-only session's opt-in record button must not brick an
30
+ * otherwise healthy session, so failure there is a banner (`error` set,
31
+ * state kept) and the session stays interactive.
32
+ */
33
+ export declare function initAudioCapture(conn: ConnState, msg: {
34
+ sampleRate: number;
35
+ ttsSampleRate: number;
36
+ }, deps: AudioSetupDeps, fatal: boolean): Promise<void>;
@@ -26,6 +26,10 @@ type MessageHandlerDeps = {
26
26
  getSnapshot: () => SessionSnapshot;
27
27
  updateState: (partial: Partial<SessionSnapshot>) => void;
28
28
  conn: ConnState;
29
+ /** Invalidate any in-flight file upload (the upload sender's `discard`). */
30
+ discardUpload: () => void;
31
+ /** Release the microphone/VoiceIO (the session core's `cleanupAudio`). */
32
+ cleanupAudio: () => void;
29
33
  };
30
34
  type MessageHandlers = {
31
35
  /**
@@ -38,6 +42,14 @@ type MessageHandlers = {
38
42
  * otherwise `undefined`.
39
43
  */
40
44
  handleMessage(data: unknown): SessionConfigMessage | undefined;
45
+ /**
46
+ * Wait for `io`'s playback queue to drain, then transition to `"listening"`
47
+ * — guarded by the same turn-boundary generation the live `audio_done`
48
+ * path uses. `initAudioCapture` routes the pre-init greeting replay
49
+ * through this so a barge-in mid-greeting can't be stomped by the
50
+ * replayed completion resolving late.
51
+ */
52
+ settleWhenAudioDrained(io: NonNullable<ConnState["voiceIO"]>): void;
41
53
  };
42
54
  /**
43
55
  * Create the server→client message handlers for one session core.
@@ -0,0 +1,19 @@
1
+ /**
2
+ * Automatic reconnection for the browser session socket, built on
3
+ * partysocket's `ReconnectingWebSocket`. Kept out of `session-core.ts` so
4
+ * the state machine there reads as protocol logic, not socket plumbing.
5
+ */
6
+ import ReconnectingWebSocket from "partysocket/ws";
7
+ /**
8
+ * Open partysocket's reconnecting WebSocket. The URL is a *provider*,
9
+ * re-evaluated on every attempt, so each retry picks up the current resume
10
+ * URL rather than the one the session started with.
11
+ */
12
+ export declare function openReconnectingSocket(urlProvider: () => string): ReconnectingWebSocket;
13
+ /**
14
+ * True while `socket` is a reconnecting socket that will retry after the
15
+ * close event currently being handled. partysocket schedules the retry
16
+ * *before* dispatching `close`, so `retryCount` already names the attempt
17
+ * just scheduled — at `maxRetries` it has given up.
18
+ */
19
+ export declare function reconnectPending(socket: unknown): boolean;
@@ -0,0 +1,2 @@
1
+ /** Build the session WebSocket URL from the platform URL and resume state. */
2
+ export declare function buildWsUrl(platformUrl: string, resume: boolean, sessionId?: string): URL;
@@ -1,2 +1,2 @@
1
- import { t as createSessionCore } from "./session-core-D3NDaySY.js";
1
+ import { t as createSessionCore } from "./session-core-KUoY-0JO.js";
2
2
  export { createSessionCore };
@@ -0,0 +1,49 @@
1
+ /**
2
+ * WebRTC microphone capture for sync mode.
3
+ *
4
+ * Captures voice through `getUserMedia` with the WebRTC voice-processing
5
+ * constraints (echo cancellation, noise suppression, auto gain — the
6
+ * processing that makes the energy VAD in `sync-vad.ts` reliable), runs an
7
+ * AudioWorklet that batches raw frames to the main thread, feeds them
8
+ * through the utterance detector, and hands each completed utterance to
9
+ * the sync session as one HTTP turn. No WebSocket anywhere on the path.
10
+ *
11
+ * The worklet module ships inline as a data URI, so sync mode needs no
12
+ * separately-served processor file.
13
+ */
14
+ import type { SyncSession } from "./sync-session.ts";
15
+ import { type UtteranceDetectorOptions } from "./sync-vad.ts";
16
+ /** Default capture rate — what the STT providers expect. */
17
+ export declare const DEFAULT_SYNC_MIC_SAMPLE_RATE = 16000;
18
+ /** Data URI form of the capture processor (no served asset, no blob URL). */
19
+ export declare const CAPTURE_WORKLET_DATA_URI: string;
20
+ /** Clamp-and-convert one Float32 capture batch to PCM16. */
21
+ export declare function floatToPcm16(samples: Float32Array): Int16Array;
22
+ /** Configuration for {@link startSyncMicrophone}. */
23
+ export type SyncMicrophoneOptions = {
24
+ /** The session each completed utterance is sent through. */
25
+ session: Pick<SyncSession, "sendPcm16">;
26
+ /** Capture/STT sample rate. Defaults to {@link DEFAULT_SYNC_MIC_SAMPLE_RATE}. */
27
+ sampleRate?: number | undefined;
28
+ /** VAD tuning overrides (see {@link UtteranceDetectorOptions}). */
29
+ vad?: Omit<UtteranceDetectorOptions, "sampleRate"> | undefined;
30
+ /** Speech onset confirmed — a turn will follow once the user pauses. */
31
+ onSpeechStart?: (() => void) | undefined;
32
+ /** An utterance was endpointed and its turn dispatched. */
33
+ onSpeechEnd?: (() => void) | undefined;
34
+ /** Capture or turn failure (the mic keeps running unless stopped). */
35
+ onError?: ((err: Error) => void) | undefined;
36
+ };
37
+ /** Live microphone handle returned by {@link startSyncMicrophone}. */
38
+ export type SyncMicrophone = {
39
+ /** True while the detector is inside an utterance. */
40
+ readonly speaking: boolean;
41
+ /** Release the mic, the AudioContext, and flush a trailing utterance. */
42
+ stop(): Promise<void>;
43
+ };
44
+ /**
45
+ * Open the microphone and stream endpointed utterances into a sync session.
46
+ *
47
+ * @throws If microphone access is denied or worklet registration fails.
48
+ */
49
+ export declare function startSyncMicrophone(opts: SyncMicrophoneOptions): Promise<SyncMicrophone>;