@alexkroman1/aai-ui 1.8.2 → 1.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/dist/_colors-DJUordGv.js +19 -0
  2. package/dist/_utils-5cs73OrA.js +16 -0
  3. package/dist/_utils.d.ts +4 -0
  4. package/dist/aai-logo-B8lDmsut.js +84 -0
  5. package/dist/audio.d.ts +9 -2
  6. package/dist/audio.js +41 -24
  7. package/dist/chat-view-C1XbqDk0.js +186 -0
  8. package/dist/components/_colors.d.ts +26 -0
  9. package/dist/components/aai-logo.d.ts +7 -6
  10. package/dist/components/button.d.ts +11 -13
  11. package/dist/components/button.js +7 -16
  12. package/dist/components/chat-view.d.ts +8 -19
  13. package/dist/components/chat-view.js +2 -98
  14. package/dist/components/controls.d.ts +1 -1
  15. package/dist/components/controls.js +2 -39
  16. package/dist/components/eyebrow.d.ts +12 -0
  17. package/dist/components/message-list.d.ts +1 -1
  18. package/dist/components/message-list.js +126 -56
  19. package/dist/components/sidebar-layout.d.ts +1 -1
  20. package/dist/components/start-screen.d.ts +4 -3
  21. package/dist/components/start-screen.js +20 -14
  22. package/dist/components/text-controls.d.ts +14 -0
  23. package/dist/components/tool-call-block.d.ts +5 -3
  24. package/dist/components/tool-call-block.js +1 -1
  25. package/dist/components/url-chips.d.ts +28 -0
  26. package/dist/context.d.ts +30 -4
  27. package/dist/context.js +86 -10
  28. package/dist/controls-BngrPbOC.js +135 -0
  29. package/dist/default-client/assets/audio-Cs-6t_Wd.js +1 -0
  30. package/dist/default-client/assets/capture-processor-C19oBn4L.js +105 -0
  31. package/dist/default-client/assets/index-Bf4ZTNcx.js +73 -0
  32. package/dist/default-client/assets/index-Bzlh9i7w.css +2 -0
  33. package/dist/default-client/assets/playback-processor-BtlzAH78.js +149 -0
  34. package/dist/default-client/index.html +3 -3
  35. package/dist/define-client.d.ts +2 -2
  36. package/dist/define-client.js +18 -24
  37. package/dist/eyebrow-C6ZFuiz6.js +27 -0
  38. package/dist/hooks.js +71 -42
  39. package/dist/index.d.ts +3 -1
  40. package/dist/index.js +6 -5
  41. package/dist/session-core-D3NDaySY.js +652 -0
  42. package/dist/session-core-messages.d.ts +50 -0
  43. package/dist/session-core-types.d.ts +147 -0
  44. package/dist/session-core-upload.d.ts +16 -0
  45. package/dist/session-core.d.ts +2 -68
  46. package/dist/session-core.js +1 -455
  47. package/dist/{tool-call-block-wby_jyoY.js → tool-call-block-7f1GTG-P.js} +33 -31
  48. package/dist/types.d.ts +29 -4
  49. package/dist/types.js +10 -1
  50. package/dist/worklets/capture-processor.d.ts +2 -0
  51. package/dist/worklets/capture-processor.js +80 -25
  52. package/dist/worklets/playback-processor.d.ts +2 -0
  53. package/dist/worklets/playback-processor.js +80 -29
  54. package/package.json +18 -18
  55. package/styles.css +30 -0
  56. package/dist/_react-test-utils.d.ts +0 -86
  57. package/dist/aai-logo-BqFv6JU6.js +0 -21
  58. package/dist/default-client/assets/audio-UowwhmYo.js +0 -1
  59. package/dist/default-client/assets/capture-processor-C26lSiVr.js +0 -53
  60. package/dist/default-client/assets/index-BZqVGR2o.css +0 -2
  61. package/dist/default-client/assets/index-YW9WUhbL.js +0 -49
  62. package/dist/default-client/assets/playback-processor-dSA8Im99.js +0 -101
@@ -0,0 +1,652 @@
1
+ import "./types.js";
2
+ import { DEFAULT_MAX_HISTORY, FILE_UPLOAD_CHUNK_BYTES, WS_OPEN, errorMessage, safeJsonParse } from "@alexkroman1/aai";
3
+ import { ServerMessageSchema, lenientParse } from "@alexkroman1/aai/protocol";
4
+ import { MAX_SYNC_AUDIO_SECONDS } from "@alexkroman1/aai/stt";
5
+ //#region session-core-messages.ts
6
+ /**
7
+ * Incoming-message handling for the voice session core.
8
+ *
9
+ * Split out of `session-core.ts`: this module owns the interpretation of
10
+ * server→client frames (audio chunks + JSON {@link ServerMessage}s) and the
11
+ * turn-boundary generation counters, while `session-core.ts` owns the state
12
+ * store and connection lifecycle. The handlers read and mutate session state
13
+ * exclusively through the injected `getSnapshot`/`updateState` deps.
14
+ */
15
+ /** Cap on `customEvents` retained in the session snapshot to avoid unbounded growth. */
16
+ const MAX_CUSTOM_EVENTS = 200;
17
+ /** Cap on `messages` retained in the session snapshot; matches the host-side history cap. */
18
+ const MAX_MESSAGES = DEFAULT_MAX_HISTORY;
19
+ /** Cap on pre-init audio chunks buffered while `voiceIO` is initializing. ~100 chunks at
20
+ * typical S2S chunk sizes is well over a second of audio — far longer than init takes
21
+ * in practice, but bounded against pathological cases (mic-permission stalls). */
22
+ const MAX_PREINIT_AUDIO_CHUNKS = 100;
23
+ /**
24
+ * Snapshot fields cleared when a session's conversation state is wiped —
25
+ * shared by the initial snapshot, `resetState()`, and the server `reset` event.
26
+ * The empty arrays are safe to share: snapshot collections are never mutated
27
+ * in place, only replaced.
28
+ */
29
+ const CLEARED_SESSION_STATE = {
30
+ messages: [],
31
+ toolCalls: [],
32
+ customEvents: [],
33
+ userTranscript: null,
34
+ agentTranscript: null,
35
+ error: null
36
+ };
37
+ function appendCapped(list, item, cap) {
38
+ if (list.length < cap) return [...list, item];
39
+ const next = list.slice(list.length - cap + 1);
40
+ next.push(item);
41
+ return next;
42
+ }
43
+ /**
44
+ * Create the server→client message handlers for one session core.
45
+ *
46
+ * Encapsulates the two turn-boundary counters (`handlerGeneration` for
47
+ * discarding stale async audio completions, `customEventSeq` for event
48
+ * dedup) that previously lived as closure locals in `createSessionCore`.
49
+ */
50
+ function createMessageHandlers(deps) {
51
+ const { getSnapshot, updateState, conn } = deps;
52
+ /** Incremented on each turn boundary -- stale async callbacks compare against this. */
53
+ let handlerGeneration = 0;
54
+ /** Monotonically increasing counter for custom events -- used by useEvent to deduplicate. */
55
+ let customEventSeq = 0;
56
+ /** Monotonically increasing counter for chat messages -- stable render keys
57
+ * and tool-call anchoring that survive the sliding message window. */
58
+ let messageSeq = 0;
59
+ /** Monotonically increasing counter for tool calls -- used by the tool-call
60
+ * hooks to iterate only the unprocessed tail. */
61
+ let toolCallSeq = 0;
62
+ function appendCustomEvent(name, data) {
63
+ updateState({ customEvents: appendCapped(getSnapshot().customEvents, {
64
+ id: ++customEventSeq,
65
+ event: name,
66
+ data
67
+ }, MAX_CUSTOM_EVENTS) });
68
+ }
69
+ function handleUserTranscriptEvent(text) {
70
+ handlerGeneration++;
71
+ updateState({
72
+ userTranscript: null,
73
+ messages: appendCapped(getSnapshot().messages, {
74
+ id: ++messageSeq,
75
+ role: "user",
76
+ content: text
77
+ }, MAX_MESSAGES),
78
+ state: "thinking"
79
+ });
80
+ }
81
+ function handleAgentTranscriptEvent(text) {
82
+ updateState({
83
+ agentTranscript: null,
84
+ messages: appendCapped(getSnapshot().messages, {
85
+ id: ++messageSeq,
86
+ role: "assistant",
87
+ content: text
88
+ }, MAX_MESSAGES)
89
+ });
90
+ }
91
+ /** Clear error state when a non-error event arrives — proves the session
92
+ * is functional (e.g. audio init failed but WebSocket still works). */
93
+ function clearRecoveredError() {
94
+ const snap = getSnapshot();
95
+ if (snap.state === "error") updateState({
96
+ state: "disconnected",
97
+ error: null
98
+ });
99
+ else if (snap.error !== null) updateState({ error: null });
100
+ }
101
+ function handleErrorEvent(e) {
102
+ console.error("Agent error:", e.message);
103
+ if (e.fatal === false) updateState({ error: {
104
+ code: e.code,
105
+ message: e.message
106
+ } });
107
+ else updateState({
108
+ state: "error",
109
+ error: {
110
+ code: e.code,
111
+ message: e.message
112
+ },
113
+ running: false
114
+ });
115
+ }
116
+ /** Single entry point for all server->client session events. */
117
+ function handleEvent(e) {
118
+ if (e.type !== "error") clearRecoveredError();
119
+ switch (e.type) {
120
+ case "speech_started":
121
+ updateState({ userTranscript: "" });
122
+ break;
123
+ case "speech_stopped": break;
124
+ case "user_transcript":
125
+ handleUserTranscriptEvent(e.text);
126
+ break;
127
+ case "user_transcript_partial":
128
+ updateState({ userTranscript: e.text });
129
+ break;
130
+ case "agent_transcript":
131
+ handleAgentTranscriptEvent(e.text);
132
+ break;
133
+ case "tool_call":
134
+ updateState({ toolCalls: appendCapped(getSnapshot().toolCalls, {
135
+ callId: e.toolCallId,
136
+ name: e.toolName,
137
+ args: e.args ?? {},
138
+ status: "pending",
139
+ seq: ++toolCallSeq,
140
+ afterMessageId: getSnapshot().messages.at(-1)?.id ?? -1
141
+ }, MAX_MESSAGES) });
142
+ break;
143
+ case "tool_call_done": {
144
+ const tcs = getSnapshot().toolCalls;
145
+ const idx = tcs.findIndex((tc) => tc.callId === e.toolCallId);
146
+ if (idx !== -1) {
147
+ const updated = [...tcs];
148
+ const existing = updated[idx];
149
+ if (existing) updated[idx] = {
150
+ ...existing,
151
+ status: "done",
152
+ result: e.result
153
+ };
154
+ updateState({ toolCalls: updated });
155
+ }
156
+ break;
157
+ }
158
+ case "reply_done":
159
+ updateState({ state: "listening" });
160
+ break;
161
+ case "cancelled":
162
+ handlerGeneration++;
163
+ conn.voiceIO?.flush();
164
+ updateState({
165
+ userTranscript: null,
166
+ agentTranscript: null,
167
+ state: "listening"
168
+ });
169
+ break;
170
+ case "reset":
171
+ handlerGeneration++;
172
+ conn.voiceIO?.flush();
173
+ updateState({
174
+ ...CLEARED_SESSION_STATE,
175
+ state: "listening"
176
+ });
177
+ break;
178
+ case "custom_event":
179
+ appendCustomEvent(e.event, e.data);
180
+ break;
181
+ case "error":
182
+ handleErrorEvent(e);
183
+ break;
184
+ case "idle_timeout": break;
185
+ default: break;
186
+ }
187
+ }
188
+ /** Enqueue a PCM16 audio chunk for playback. Transitions state to `"speaking"` on the first chunk. */
189
+ function playAudioChunk(chunk) {
190
+ const snap = getSnapshot();
191
+ if (snap.state === "disconnected" && snap.error !== null) return;
192
+ if (snap.state !== "speaking") updateState({ state: "speaking" });
193
+ if (conn.voiceIO) conn.voiceIO.enqueue(chunk.buffer);
194
+ else if (conn.preInitAudio.length < MAX_PREINIT_AUDIO_CHUNKS) conn.preInitAudio.push(chunk);
195
+ }
196
+ /**
197
+ * Signal that the server has finished sending audio for this turn.
198
+ * Waits for the audio queue to drain, then transitions state to `"listening"`.
199
+ * Uses the `handlerGeneration` counter to discard stale completions from interrupted turns.
200
+ */
201
+ function playAudioDone() {
202
+ const gen = handlerGeneration;
203
+ const io = conn.voiceIO;
204
+ if (io) io.done().then(() => {
205
+ if (handlerGeneration !== gen) return;
206
+ updateState({ state: "listening" });
207
+ }).catch((err) => {
208
+ console.warn("Audio playback done failed:", err);
209
+ });
210
+ else {
211
+ conn.preInitDone = true;
212
+ updateState({ state: "listening" });
213
+ }
214
+ }
215
+ function handleMessage(data) {
216
+ if (data instanceof ArrayBuffer) {
217
+ playAudioChunk(new Uint8Array(data));
218
+ return;
219
+ }
220
+ if (typeof data !== "string") {
221
+ console.warn("session-core: non-string, non-binary frame received; dropping");
222
+ return;
223
+ }
224
+ const raw = safeJsonParse(data);
225
+ if (raw === void 0) {
226
+ console.warn("session-core: invalid JSON; dropping");
227
+ return;
228
+ }
229
+ const parsed = lenientParse(ServerMessageSchema, raw);
230
+ if (!parsed.ok) {
231
+ if (parsed.malformed) console.warn("session-core: malformed server message", parsed.error);
232
+ return;
233
+ }
234
+ const msg = parsed.data;
235
+ if (msg.type === "config") return {
236
+ sampleRate: msg.sampleRate,
237
+ ttsSampleRate: msg.ttsSampleRate,
238
+ audioOut: msg.audioOut,
239
+ sid: msg.sessionId
240
+ };
241
+ if (msg.type === "audio_done") {
242
+ playAudioDone();
243
+ return;
244
+ }
245
+ handleEvent(msg);
246
+ }
247
+ return { handleMessage };
248
+ }
249
+ //#endregion
250
+ //#region session-core-upload.ts
251
+ /**
252
+ * File-upload transcription for the session core (`sendAudioFile`).
253
+ *
254
+ * Split out of `session-core.ts`: this module owns the decode → frame →
255
+ * reliable-send pipeline for uploaded audio, plus the two guards that keep
256
+ * it exclusive — an in-flight lock (no second upload, no mic start
257
+ * mid-upload) and an epoch that in-flight sends compare against so a
258
+ * reset/close/reconnect abandons them instead of streaming a stale clip
259
+ * into the fresh session.
260
+ */
261
+ function createUploadSender(deps) {
262
+ const { conn, getSnapshot, sendJson } = deps;
263
+ /** True while a `sendAudioFile` upload is decoding or streaming. Blocks the
264
+ * mic (and a second upload) so file bytes never interleave with live audio. */
265
+ let uploadInFlight = false;
266
+ /** Bumped whenever conversation/connection state is discarded (reset, close,
267
+ * reconnect). An in-flight upload compares against it between chunks and
268
+ * aborts instead of streaming a stale clip into the fresh session. */
269
+ let uploadEpoch = 0;
270
+ /** Stream `bytes` to the socket in chunks, waiting out backpressure.
271
+ * Unlike live mic frames (dropped under backpressure — stale speech is
272
+ * worthless), file audio must arrive completely. Aborts if the connection
273
+ * closes or the session is reset (`epoch` moves on) between chunks. */
274
+ async function sendBytesReliably(bytes, epoch) {
275
+ for (let i = 0; i < bytes.byteLength; i += FILE_UPLOAD_CHUNK_BYTES) {
276
+ while (epoch === uploadEpoch && conn.ws && conn.ws.readyState === WS_OPEN && conn.ws.bufferedAmount > 65536) await new Promise((r) => setTimeout(r, 50));
277
+ if (epoch !== uploadEpoch) throw new Error("sendAudioFile: session was reset mid-send");
278
+ if (!conn.ws || conn.ws.readyState !== WS_OPEN) throw new Error("sendAudioFile: connection closed mid-send");
279
+ conn.ws.send(bytes.subarray(i, i + FILE_UPLOAD_CHUNK_BYTES));
280
+ }
281
+ }
282
+ /** Validate that an upload may start; returns the session's ready config. */
283
+ function assertUploadReady() {
284
+ const cfg = conn.readyConfig;
285
+ if (!(cfg && conn.ws) || conn.ws.readyState !== WS_OPEN) throw new Error("sendAudioFile: session is not connected");
286
+ const snap = getSnapshot();
287
+ if (snap.audioOut) throw new Error("sendAudioFile is only available in text-only sessions (tts: none()) — voice sessions stream the microphone instead");
288
+ if (snap.recording) throw new Error("sendAudioFile: stop recording before uploading a file");
289
+ if (uploadInFlight) throw new Error("sendAudioFile: another upload is already in progress");
290
+ return cfg;
291
+ }
292
+ async function sendAudioFile(file) {
293
+ const cfg = assertUploadReady();
294
+ uploadInFlight = true;
295
+ const epoch = uploadEpoch;
296
+ try {
297
+ const { decodeAudioToPcm16 } = await import("./audio.js");
298
+ const clip = await decodeAudioToPcm16(await file.arrayBuffer(), cfg.sampleRate);
299
+ if (getSnapshot().recording || conn.audioSetupInFlight) throw new Error("sendAudioFile: stop recording before uploading a file");
300
+ if (epoch !== uploadEpoch) throw new Error("sendAudioFile: session was reset mid-send");
301
+ if (clip.length / cfg.sampleRate <= MAX_SYNC_AUDIO_SECONDS) {
302
+ const bytes = new Uint8Array(clip.buffer, clip.byteOffset, clip.byteLength);
303
+ sendJson({
304
+ type: "transcribe_file_start",
305
+ sampleRate: cfg.sampleRate,
306
+ byteLength: bytes.byteLength
307
+ });
308
+ await sendBytesReliably(bytes, epoch);
309
+ sendJson({ type: "transcribe_file_end" });
310
+ return;
311
+ }
312
+ const padded = new Int16Array(clip.length + cfg.sampleRate);
313
+ padded.set(clip);
314
+ await sendBytesReliably(new Uint8Array(padded.buffer), epoch);
315
+ } finally {
316
+ uploadInFlight = false;
317
+ }
318
+ }
319
+ return {
320
+ sendAudioFile,
321
+ inFlight: () => uploadInFlight,
322
+ discard: () => {
323
+ uploadEpoch++;
324
+ }
325
+ };
326
+ }
327
+ //#endregion
328
+ //#region session-core.ts
329
+ /**
330
+ * Framework-agnostic voice session core.
331
+ *
332
+ * Manages WebSocket communication, audio capture/playback, and agent state
333
+ * transitions using a subscribe/getSnapshot pattern compatible with React's
334
+ * `useSyncExternalStore` and other external store consumers.
335
+ *
336
+ * Server→client message interpretation lives in `session-core-messages.ts`;
337
+ * the public/internal type declarations live in `session-core-types.ts`.
338
+ *
339
+ * No dependency on React, Preact, or any UI framework.
340
+ */
341
+ /**
342
+ * Initialize audio capture and playback after the server sends a ready config.
343
+ *
344
+ * Lifecycle: dynamically import audio modules -> request microphone access ->
345
+ * register AudioWorklet processors -> create a `VoiceIO` instance -> send
346
+ * `audio_ready` to the server -> transition state to `"listening"`.
347
+ *
348
+ * Uses the connection `generation` counter to detect if `connect()` was called
349
+ * while awaiting async operations; if so, the stale VoiceIO is closed immediately
350
+ * to prevent it from being assigned to a newer connection.
351
+ *
352
+ * On failure (e.g. microphone permission denied, WebSocket closed mid-setup),
353
+ * sets the error state and transitions to `"disconnected"`.
354
+ */
355
+ async function initAudioCapture(conn, msg, deps) {
356
+ if (conn.audioSetupInFlight) return;
357
+ conn.audioSetupInFlight = true;
358
+ const gen = conn.generation;
359
+ try {
360
+ const [{ createVoiceIO }, captureWorklet, playbackWorklet] = await Promise.all([
361
+ import("./audio.js"),
362
+ import("./worklets/capture-processor.js").then((m) => m.default),
363
+ import("./worklets/playback-processor.js").then((m) => m.default)
364
+ ]);
365
+ const io = await createVoiceIO({
366
+ sttSampleRate: msg.sampleRate,
367
+ ttsSampleRate: msg.ttsSampleRate,
368
+ captureWorkletSrc: captureWorklet,
369
+ playbackWorkletSrc: playbackWorklet,
370
+ onMicData: (pcm16) => {
371
+ try {
372
+ deps.sendAudio(new Uint8Array(pcm16));
373
+ } catch {
374
+ console.debug("[aai-ui] sendAudio dropped: connection closed");
375
+ }
376
+ }
377
+ });
378
+ if (conn.generation !== gen || !conn.ws || conn.ws.readyState !== WS_OPEN) {
379
+ io.close();
380
+ return;
381
+ }
382
+ conn.voiceIO = io;
383
+ if (conn.preInitAudio.length > 0) {
384
+ for (const chunk of conn.preInitAudio) io.enqueue(chunk.buffer);
385
+ conn.preInitAudio = [];
386
+ }
387
+ deps.sendJson({ type: "audio_ready" });
388
+ deps.updateState({ recording: true });
389
+ if (conn.preInitDone) {
390
+ conn.preInitDone = false;
391
+ io.done().then(() => {
392
+ if (conn.generation !== gen) return;
393
+ deps.updateState({ state: "listening" });
394
+ }).catch(() => deps.updateState({ state: "listening" }));
395
+ } else deps.updateState({ state: "listening" });
396
+ } catch (err) {
397
+ if (conn.generation !== gen || !conn.ws || conn.ws.readyState !== WS_OPEN) return;
398
+ deps.updateState({
399
+ state: "error",
400
+ error: {
401
+ code: "audio",
402
+ message: `Microphone access failed: ${errorMessage(err)}`
403
+ },
404
+ running: false
405
+ });
406
+ } finally {
407
+ conn.audioSetupInFlight = false;
408
+ }
409
+ }
410
+ function buildWsUrl(platformUrl, resume, sessionId) {
411
+ const wsUrl = new URL("websocket", platformUrl.endsWith("/") ? platformUrl : `${platformUrl}/`);
412
+ wsUrl.protocol = wsUrl.protocol === "https:" ? "wss:" : "ws:";
413
+ if (sessionId) wsUrl.searchParams.set("sessionId", sessionId);
414
+ else if (resume) wsUrl.searchParams.set("resume", "1");
415
+ return wsUrl;
416
+ }
417
+ /**
418
+ * Create a framework-agnostic voice session core that connects to an AAI
419
+ * server via WebSocket.
420
+ *
421
+ * Uses a subscribe/getSnapshot pattern for state management, compatible with
422
+ * React's `useSyncExternalStore` and other external store integrations.
423
+ *
424
+ * @param options - Session configuration including the platform server URL.
425
+ * @returns A {@link SessionCore} handle for controlling the session.
426
+ *
427
+ * @public
428
+ */
429
+ function createSessionCore(options) {
430
+ const WS = options.WebSocket ?? WebSocket;
431
+ let currentSnapshot = {
432
+ ...CLEARED_SESSION_STATE,
433
+ state: "disconnected",
434
+ contentVersion: 0,
435
+ started: false,
436
+ running: false,
437
+ audioOut: true,
438
+ recording: false,
439
+ apiUrl: buildWsUrl(options.platformUrl, false).toString()
440
+ };
441
+ const subscribers = /* @__PURE__ */ new Set();
442
+ function notify() {
443
+ for (const sub of subscribers) sub();
444
+ }
445
+ /** Snapshot fields whose changes bump `contentVersion` (rendered conversation content). */
446
+ const contentKeys = [
447
+ "messages",
448
+ "toolCalls",
449
+ "userTranscript",
450
+ "agentTranscript"
451
+ ];
452
+ function updateState(partial) {
453
+ currentSnapshot = contentKeys.some((key) => key in partial && partial[key] !== currentSnapshot[key]) ? {
454
+ ...currentSnapshot,
455
+ ...partial,
456
+ contentVersion: currentSnapshot.contentVersion + 1
457
+ } : {
458
+ ...currentSnapshot,
459
+ ...partial
460
+ };
461
+ notify();
462
+ }
463
+ function getSnapshot() {
464
+ return currentSnapshot;
465
+ }
466
+ function subscribe(callback) {
467
+ subscribers.add(callback);
468
+ return () => {
469
+ subscribers.delete(callback);
470
+ };
471
+ }
472
+ const conn = {
473
+ ws: null,
474
+ voiceIO: null,
475
+ audioSetupInFlight: false,
476
+ generation: 0,
477
+ preInitAudio: [],
478
+ preInitDone: false,
479
+ readyConfig: null
480
+ };
481
+ let connectionController = null;
482
+ let hasConnected = false;
483
+ function cleanupAudio() {
484
+ upload.discard();
485
+ conn.audioSetupInFlight = false;
486
+ conn.voiceIO?.close();
487
+ conn.voiceIO = null;
488
+ conn.preInitAudio = [];
489
+ conn.preInitDone = false;
490
+ }
491
+ function resetState() {
492
+ updateState(CLEARED_SESSION_STATE);
493
+ }
494
+ function sendJson(msg) {
495
+ if (conn.ws && conn.ws.readyState === WS_OPEN) conn.ws.send(JSON.stringify(msg));
496
+ }
497
+ function sendAudio(bytes) {
498
+ if (!conn.ws || conn.ws.readyState !== WS_OPEN) return;
499
+ if (conn.ws.bufferedAmount > 65536) return;
500
+ conn.ws.send(bytes);
501
+ }
502
+ const audioDeps = {
503
+ sendJson,
504
+ sendAudio,
505
+ updateState
506
+ };
507
+ const upload = createUploadSender({
508
+ conn,
509
+ getSnapshot,
510
+ sendJson
511
+ });
512
+ const { handleMessage } = createMessageHandlers({
513
+ getSnapshot,
514
+ updateState,
515
+ conn
516
+ });
517
+ /** Abort the in-flight connection and release audio + WebSocket resources. */
518
+ function teardownConnection() {
519
+ connectionController?.abort();
520
+ connectionController = null;
521
+ cleanupAudio();
522
+ conn.ws?.close();
523
+ conn.ws = null;
524
+ }
525
+ /** React to the server's `config` message: record it, set up the audio
526
+ * path for the session's mode, and replay history on reconnect. */
527
+ function onServerConfig(config) {
528
+ if (config.sid) options.onSessionId?.(config.sid);
529
+ const isReconnect = hasConnected;
530
+ hasConnected = true;
531
+ conn.readyConfig = {
532
+ sampleRate: config.sampleRate,
533
+ ttsSampleRate: config.ttsSampleRate
534
+ };
535
+ const audioOut = config.audioOut !== false;
536
+ updateState({ audioOut });
537
+ if (audioOut) initAudioCapture(conn, config, audioDeps);
538
+ else {
539
+ sendJson({ type: "audio_ready" });
540
+ updateState({ state: "listening" });
541
+ }
542
+ if (isReconnect && currentSnapshot.messages.length > 0) sendJson({
543
+ type: "history",
544
+ messages: currentSnapshot.messages.map((m) => ({
545
+ role: m.role,
546
+ content: m.content
547
+ }))
548
+ });
549
+ }
550
+ function connect(opts) {
551
+ updateState({
552
+ state: "connecting",
553
+ error: null
554
+ });
555
+ teardownConnection();
556
+ conn.generation++;
557
+ const controller = new AbortController();
558
+ connectionController = controller;
559
+ const { signal: sig } = controller;
560
+ if (opts?.signal) opts.signal.addEventListener("abort", () => disconnect(), { signal: sig });
561
+ const resumeId = !hasConnected ? options.resumeSessionId : void 0;
562
+ const wsUrl = buildWsUrl(options.platformUrl, hasConnected, resumeId);
563
+ const socket = new WS(wsUrl.toString());
564
+ socket.binaryType = "arraybuffer";
565
+ conn.ws = socket;
566
+ socket.addEventListener("open", () => {
567
+ updateState({ state: "ready" });
568
+ }, { signal: sig });
569
+ socket.addEventListener("message", (event) => {
570
+ const config = handleMessage(event.data);
571
+ if (config) onServerConfig(config);
572
+ }, { signal: sig });
573
+ socket.addEventListener("close", () => {
574
+ if (sig.aborted) return;
575
+ controller.abort();
576
+ cleanupAudio();
577
+ updateState({
578
+ state: "disconnected",
579
+ running: false,
580
+ recording: false
581
+ });
582
+ }, { signal: sig });
583
+ }
584
+ function cancel() {
585
+ conn.voiceIO?.flush();
586
+ updateState({ state: "listening" });
587
+ sendJson({ type: "cancel" });
588
+ }
589
+ function reset() {
590
+ upload.discard();
591
+ conn.voiceIO?.flush();
592
+ if (conn.ws && conn.ws.readyState === WS_OPEN) {
593
+ sendJson({ type: "reset" });
594
+ return;
595
+ }
596
+ resetState();
597
+ disconnect();
598
+ connect();
599
+ }
600
+ function disconnect() {
601
+ teardownConnection();
602
+ updateState({
603
+ state: "disconnected",
604
+ running: false,
605
+ recording: false
606
+ });
607
+ }
608
+ function startRecording() {
609
+ if (currentSnapshot.audioOut || currentSnapshot.recording || conn.audioSetupInFlight || upload.inFlight()) return;
610
+ const cfg = conn.readyConfig;
611
+ if (!(cfg && conn.ws) || conn.ws.readyState !== WS_OPEN) return;
612
+ initAudioCapture(conn, cfg, audioDeps);
613
+ }
614
+ function stopRecording() {
615
+ if (currentSnapshot.audioOut || !currentSnapshot.recording) return;
616
+ cleanupAudio();
617
+ updateState({ recording: false });
618
+ }
619
+ function start() {
620
+ updateState({
621
+ started: true,
622
+ running: true
623
+ });
624
+ connect();
625
+ }
626
+ function toggle() {
627
+ if (currentSnapshot.running) disconnect();
628
+ else {
629
+ updateState({ running: true });
630
+ connect();
631
+ }
632
+ }
633
+ return {
634
+ getSnapshot,
635
+ subscribe,
636
+ connect,
637
+ cancel,
638
+ resetState,
639
+ reset,
640
+ disconnect,
641
+ start,
642
+ toggle,
643
+ startRecording,
644
+ stopRecording,
645
+ sendAudioFile: upload.sendAudioFile,
646
+ [Symbol.dispose]() {
647
+ disconnect();
648
+ }
649
+ };
650
+ }
651
+ //#endregion
652
+ export { createSessionCore as t };
@@ -0,0 +1,50 @@
1
+ import type { ConnState, SessionSnapshot } from "./session-core-types.ts";
2
+ /**
3
+ * Snapshot fields cleared when a session's conversation state is wiped —
4
+ * shared by the initial snapshot, `resetState()`, and the server `reset` event.
5
+ * The empty arrays are safe to share: snapshot collections are never mutated
6
+ * in place, only replaced.
7
+ */
8
+ export declare const CLEARED_SESSION_STATE: {
9
+ messages: never[];
10
+ toolCalls: never[];
11
+ customEvents: never[];
12
+ userTranscript: null;
13
+ agentTranscript: null;
14
+ error: null;
15
+ };
16
+ /** Config payload extracted from a `config` server message. */
17
+ export type SessionConfigMessage = {
18
+ sampleRate: number;
19
+ ttsSampleRate: number;
20
+ /** False for text-only agents (`tts: none()`) — see ReadyConfigSchema. */
21
+ audioOut?: boolean | undefined;
22
+ sid?: string | undefined;
23
+ };
24
+ /** Dependencies the message handlers need from the owning session core. */
25
+ type MessageHandlerDeps = {
26
+ getSnapshot: () => SessionSnapshot;
27
+ updateState: (partial: Partial<SessionSnapshot>) => void;
28
+ conn: ConnState;
29
+ };
30
+ type MessageHandlers = {
31
+ /**
32
+ * Dispatch an incoming WebSocket message.
33
+ *
34
+ * Binary frames carry raw PCM16 audio chunks. Text frames are JSON-encoded
35
+ * {@link ServerMessage} values validated via Zod.
36
+ *
37
+ * Returns the parsed config if the message is a `config` message,
38
+ * otherwise `undefined`.
39
+ */
40
+ handleMessage(data: unknown): SessionConfigMessage | undefined;
41
+ };
42
+ /**
43
+ * Create the server→client message handlers for one session core.
44
+ *
45
+ * Encapsulates the two turn-boundary counters (`handlerGeneration` for
46
+ * discarding stale async audio completions, `customEventSeq` for event
47
+ * dedup) that previously lived as closure locals in `createSessionCore`.
48
+ */
49
+ export declare function createMessageHandlers(deps: MessageHandlerDeps): MessageHandlers;
50
+ export {};