@alexkroman1/aai-ui 1.8.2 → 1.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/dist/_colors-DJUordGv.js +19 -0
  2. package/dist/_utils-5cs73OrA.js +16 -0
  3. package/dist/_utils.d.ts +4 -0
  4. package/dist/aai-logo-B8lDmsut.js +84 -0
  5. package/dist/audio.d.ts +9 -2
  6. package/dist/audio.js +41 -24
  7. package/dist/chat-view-C1XbqDk0.js +186 -0
  8. package/dist/components/_colors.d.ts +26 -0
  9. package/dist/components/aai-logo.d.ts +7 -6
  10. package/dist/components/button.d.ts +11 -13
  11. package/dist/components/button.js +7 -16
  12. package/dist/components/chat-view.d.ts +8 -19
  13. package/dist/components/chat-view.js +2 -98
  14. package/dist/components/controls.d.ts +1 -1
  15. package/dist/components/controls.js +2 -39
  16. package/dist/components/eyebrow.d.ts +12 -0
  17. package/dist/components/message-list.d.ts +1 -1
  18. package/dist/components/message-list.js +126 -56
  19. package/dist/components/sidebar-layout.d.ts +1 -1
  20. package/dist/components/start-screen.d.ts +4 -3
  21. package/dist/components/start-screen.js +20 -14
  22. package/dist/components/text-controls.d.ts +14 -0
  23. package/dist/components/tool-call-block.d.ts +5 -3
  24. package/dist/components/tool-call-block.js +1 -1
  25. package/dist/components/url-chips.d.ts +28 -0
  26. package/dist/context.d.ts +30 -4
  27. package/dist/context.js +86 -10
  28. package/dist/controls-BngrPbOC.js +135 -0
  29. package/dist/default-client/assets/audio-Cs-6t_Wd.js +1 -0
  30. package/dist/default-client/assets/capture-processor-C19oBn4L.js +105 -0
  31. package/dist/default-client/assets/index-Bf4ZTNcx.js +73 -0
  32. package/dist/default-client/assets/index-Bzlh9i7w.css +2 -0
  33. package/dist/default-client/assets/playback-processor-BtlzAH78.js +149 -0
  34. package/dist/default-client/index.html +3 -3
  35. package/dist/define-client.d.ts +2 -2
  36. package/dist/define-client.js +18 -24
  37. package/dist/eyebrow-C6ZFuiz6.js +27 -0
  38. package/dist/hooks.js +71 -42
  39. package/dist/index.d.ts +3 -1
  40. package/dist/index.js +6 -5
  41. package/dist/session-core-D3NDaySY.js +652 -0
  42. package/dist/session-core-messages.d.ts +50 -0
  43. package/dist/session-core-types.d.ts +147 -0
  44. package/dist/session-core-upload.d.ts +16 -0
  45. package/dist/session-core.d.ts +2 -68
  46. package/dist/session-core.js +1 -455
  47. package/dist/{tool-call-block-wby_jyoY.js → tool-call-block-7f1GTG-P.js} +33 -31
  48. package/dist/types.d.ts +29 -4
  49. package/dist/types.js +10 -1
  50. package/dist/worklets/capture-processor.d.ts +2 -0
  51. package/dist/worklets/capture-processor.js +80 -25
  52. package/dist/worklets/playback-processor.d.ts +2 -0
  53. package/dist/worklets/playback-processor.js +80 -29
  54. package/package.json +18 -18
  55. package/styles.css +30 -0
  56. package/dist/_react-test-utils.d.ts +0 -86
  57. package/dist/aai-logo-BqFv6JU6.js +0 -21
  58. package/dist/default-client/assets/audio-UowwhmYo.js +0 -1
  59. package/dist/default-client/assets/capture-processor-C26lSiVr.js +0 -53
  60. package/dist/default-client/assets/index-BZqVGR2o.css +0 -2
  61. package/dist/default-client/assets/index-YW9WUhbL.js +0 -49
  62. package/dist/default-client/assets/playback-processor-dSA8Im99.js +0 -101
@@ -0,0 +1,147 @@
1
+ /**
2
+ * Type declarations for the framework-agnostic voice session core.
3
+ *
4
+ * Split out of `session-core.ts` to keep that module focused on behaviour.
5
+ * The public types here are re-exported from `session-core.ts` for
6
+ * backwards compatibility.
7
+ */
8
+ import type { VoiceIO } from "./audio.ts";
9
+ import type { AgentState, ChatMessage, SessionError, ToolCallInfo, VoiceSessionOptions, WebSocketConstructor } from "./types.ts";
10
+ /**
11
+ * A custom event emitted by the agent via `ctx.send`.
12
+ *
13
+ * @public
14
+ */
15
+ export type CustomEvent = {
16
+ readonly id: number;
17
+ readonly event: string;
18
+ readonly data: unknown;
19
+ };
20
+ /**
21
+ * Immutable snapshot of the session state.
22
+ *
23
+ * Consumers (e.g. React hooks via `useSyncExternalStore`) read this to render.
24
+ * A new object reference is created on every state change.
25
+ *
26
+ * @public
27
+ */
28
+ export type SessionSnapshot = {
29
+ readonly state: AgentState;
30
+ /**
31
+ * False when the server declared the session text-only (`tts: none()`):
32
+ * no audio frames will arrive and replies render as text. True until the
33
+ * server's `config` message says otherwise (voice is the default).
34
+ */
35
+ readonly audioOut: boolean;
36
+ /**
37
+ * True while the microphone is live and streaming to the server. Voice
38
+ * sessions record for their whole lifetime; text-only sessions toggle this
39
+ * via `startRecording()` / `stopRecording()` (the record button).
40
+ */
41
+ readonly recording: boolean;
42
+ /**
43
+ * The WebSocket URL a program can connect to directly (the same endpoint
44
+ * this session uses), e.g. `wss://host/my-agent/websocket`. Derived from
45
+ * `platformUrl` at construction — available before connecting.
46
+ */
47
+ readonly apiUrl: string;
48
+ /**
49
+ * Monotonically increasing counter bumped whenever rendered conversation
50
+ * content changes (`messages`, `toolCalls`, or either live transcript).
51
+ * Cheap dependency for scroll-to-bottom effects — unlike summed lengths it
52
+ * never collides when the capped arrays slide.
53
+ */
54
+ readonly contentVersion: number;
55
+ readonly messages: ChatMessage[];
56
+ readonly toolCalls: ToolCallInfo[];
57
+ readonly customEvents: CustomEvent[];
58
+ readonly userTranscript: string | null;
59
+ readonly agentTranscript: string | null;
60
+ readonly error: SessionError | null;
61
+ readonly started: boolean;
62
+ readonly running: boolean;
63
+ };
64
+ /**
65
+ * A framework-agnostic voice session that manages WebSocket communication,
66
+ * audio capture/playback, and agent state transitions.
67
+ *
68
+ * Uses a subscribe/getSnapshot pattern (compatible with React's
69
+ * `useSyncExternalStore`). Implements `Disposable` for resource cleanup.
70
+ *
71
+ * @public
72
+ */
73
+ export type SessionCore = {
74
+ /** Return the current immutable state snapshot. */
75
+ getSnapshot(): SessionSnapshot;
76
+ /** Subscribe to state changes. Returns an unsubscribe function. */
77
+ subscribe(callback: () => void): () => void;
78
+ /**
79
+ * Open a WebSocket connection to the server and begin audio capture.
80
+ * @param options - Optional. `signal` is an AbortSignal that, when aborted, disconnects the session.
81
+ */
82
+ connect(options?: {
83
+ signal?: AbortSignal;
84
+ }): void;
85
+ /** Cancel the current agent turn and discard in-flight TTS audio. */
86
+ cancel(): void;
87
+ /** Clear messages, transcript, and error state without disconnecting. */
88
+ resetState(): void;
89
+ /** Reset the session: clear state and reconnect. */
90
+ reset(): void;
91
+ /** Close the WebSocket and release all audio resources. */
92
+ disconnect(): void;
93
+ /** Start the session for the first time (sets `started` and `running`). */
94
+ start(): void;
95
+ /** Toggle between connected and disconnected states. */
96
+ toggle(): void;
97
+ /**
98
+ * Start streaming microphone audio (text-only sessions). Requests mic
99
+ * access on first use. No-op in voice sessions, where the mic is always on.
100
+ */
101
+ startRecording(): void;
102
+ /** Stop streaming microphone audio (text-only sessions). No-op in voice sessions. */
103
+ stopRecording(): void;
104
+ /**
105
+ * Decode an audio file (any format the browser can decode), resample it to
106
+ * the session's STT rate, and stream it to the server for transcription.
107
+ * Text-only sessions (`tts: none()`) only — voice sessions stream the
108
+ * microphone instead and reject. Also rejects when no session is connected,
109
+ * the mic is recording, another upload is in flight, or the file cannot be
110
+ * decoded. Resolves once the audio has been handed to the socket.
111
+ */
112
+ sendAudioFile(file: Blob): Promise<void>;
113
+ /** Alias for `disconnect` for use with `using`. */
114
+ [Symbol.dispose](): void;
115
+ };
116
+ export type SessionCoreOptions = VoiceSessionOptions;
117
+ /**
118
+ * Shared mutable connection state for audio initialization.
119
+ *
120
+ * Tracks the active WebSocket, VoiceIO instance, and a generation counter
121
+ * that prevents stale async operations (e.g. a slow `initAudioCapture`) from
122
+ * assigning their results to a newer connection after a reconnect.
123
+ */
124
+ export type ConnState = {
125
+ ws: InstanceType<WebSocketConstructor> | null;
126
+ voiceIO: VoiceIO | null;
127
+ audioSetupInFlight: boolean;
128
+ /** Monotonically increasing counter bumped on each connect(). Prevents a stale
129
+ * initAudioCapture from assigning its voiceIO to a newer connection. */
130
+ generation: number;
131
+ /** Audio chunks that arrived before `voiceIO` was initialized — drained into
132
+ * the playback worklet once init completes. Closes the race between the
133
+ * server starting greeting audio (immediately on S2S connect) and the
134
+ * client awaiting mic permission + worklet registration. */
135
+ preInitAudio: Uint8Array[];
136
+ /** True if `audio_done` arrived before `voiceIO` was initialized. The done
137
+ * signal must be replayed after draining preInitAudio, or a short greeting
138
+ * buffered during mic-permission never finishes playing. */
139
+ preInitDone: boolean;
140
+ /** The server's `config` payload for the current connection — kept so
141
+ * text-only sessions can init the mic lazily (record button) and file
142
+ * uploads know the STT sample rate to resample to. */
143
+ readyConfig: {
144
+ sampleRate: number;
145
+ ttsSampleRate: number;
146
+ } | null;
147
+ };
@@ -0,0 +1,16 @@
1
+ import type { ClientMessage } from "@alexkroman1/aai/protocol";
2
+ import type { ConnState, SessionSnapshot } from "./session-core-types.ts";
3
+ /** Handle to the upload pipeline, owned by one session core. */
4
+ export type UploadSender = {
5
+ /** See {@link SessionCore.sendAudioFile}. */
6
+ sendAudioFile(file: Blob): Promise<void>;
7
+ /** True while an upload is decoding or streaming — the mic must stay off. */
8
+ inFlight(): boolean;
9
+ /** Invalidate any in-flight upload (reset / close / reconnect). */
10
+ discard(): void;
11
+ };
12
+ export declare function createUploadSender(deps: {
13
+ conn: ConnState;
14
+ getSnapshot: () => SessionSnapshot;
15
+ sendJson: (msg: ClientMessage) => void;
16
+ }): UploadSender;
@@ -1,71 +1,5 @@
1
- import type { AgentState, ChatMessage, SessionError, ToolCallInfo, VoiceSessionOptions } from "./types.ts";
2
- export type { AgentState, ChatMessage, SessionError, SessionErrorCode, ToolCallInfo, VoiceSessionOptions, WebSocketConstructor, } from "./types.ts";
3
- /**
4
- * A custom event emitted by the agent via `ctx.send`.
5
- *
6
- * @public
7
- */
8
- export type CustomEvent = {
9
- readonly id: number;
10
- readonly event: string;
11
- readonly data: unknown;
12
- };
13
- /**
14
- * Immutable snapshot of the session state.
15
- *
16
- * Consumers (e.g. React hooks via `useSyncExternalStore`) read this to render.
17
- * A new object reference is created on every state change.
18
- *
19
- * @public
20
- */
21
- export type SessionSnapshot = {
22
- readonly state: AgentState;
23
- readonly messages: ChatMessage[];
24
- readonly toolCalls: ToolCallInfo[];
25
- readonly customEvents: CustomEvent[];
26
- readonly userTranscript: string | null;
27
- readonly agentTranscript: string | null;
28
- readonly error: SessionError | null;
29
- readonly started: boolean;
30
- readonly running: boolean;
31
- };
32
- /**
33
- * A framework-agnostic voice session that manages WebSocket communication,
34
- * audio capture/playback, and agent state transitions.
35
- *
36
- * Uses a subscribe/getSnapshot pattern (compatible with React's
37
- * `useSyncExternalStore`). Implements `Disposable` for resource cleanup.
38
- *
39
- * @public
40
- */
41
- export type SessionCore = {
42
- /** Return the current immutable state snapshot. */
43
- getSnapshot(): SessionSnapshot;
44
- /** Subscribe to state changes. Returns an unsubscribe function. */
45
- subscribe(callback: () => void): () => void;
46
- /**
47
- * Open a WebSocket connection to the server and begin audio capture.
48
- * @param options - Optional. `signal` is an AbortSignal that, when aborted, disconnects the session.
49
- */
50
- connect(options?: {
51
- signal?: AbortSignal;
52
- }): void;
53
- /** Cancel the current agent turn and discard in-flight TTS audio. */
54
- cancel(): void;
55
- /** Clear messages, transcript, and error state without disconnecting. */
56
- resetState(): void;
57
- /** Reset the session: clear state and reconnect. */
58
- reset(): void;
59
- /** Close the WebSocket and release all audio resources. */
60
- disconnect(): void;
61
- /** Start the session for the first time (sets `started` and `running`). */
62
- start(): void;
63
- /** Toggle between connected and disconnected states. */
64
- toggle(): void;
65
- /** Alias for `disconnect` for use with `using`. */
66
- [Symbol.dispose](): void;
67
- };
68
- export type SessionCoreOptions = VoiceSessionOptions;
1
+ import type { SessionCore, SessionCoreOptions } from "./session-core-types.ts";
2
+ export type { CustomEvent, SessionCore, SessionCoreOptions, SessionSnapshot, } from "./session-core-types.ts";
69
3
  /**
70
4
  * Create a framework-agnostic voice session core that connects to an AAI
71
5
  * server via WebSocket.
@@ -1,456 +1,2 @@
1
- import { WS_OPEN, errorMessage } from "@alexkroman1/aai";
2
- import { ServerMessageSchema, lenientParse } from "@alexkroman1/aai/protocol";
3
- //#region session-core.ts
4
- /**
5
- * Framework-agnostic voice session core.
6
- *
7
- * Manages WebSocket communication, audio capture/playback, and agent state
8
- * transitions using a subscribe/getSnapshot pattern compatible with React's
9
- * `useSyncExternalStore` and other external store consumers.
10
- *
11
- * No dependency on React, Preact, or any UI framework.
12
- */
13
- /** Cap on `customEvents` retained in the session snapshot to avoid unbounded growth. */
14
- const MAX_CUSTOM_EVENTS = 200;
15
- /** Cap on `messages` retained in the session snapshot; mirrors host-side DEFAULT_MAX_HISTORY. */
16
- const MAX_MESSAGES = 200;
17
- function appendCapped(list, item, cap) {
18
- const next = [...list, item];
19
- return next.length > cap ? next.slice(-cap) : next;
20
- }
21
- /**
22
- * Initialize audio capture and playback after the server sends a ready config.
23
- *
24
- * Lifecycle: dynamically import audio modules -> request microphone access ->
25
- * register AudioWorklet processors -> create a `VoiceIO` instance -> send
26
- * `audio_ready` to the server -> transition state to `"listening"`.
27
- *
28
- * Uses the connection `generation` counter to detect if `connect()` was called
29
- * while awaiting async operations; if so, the stale VoiceIO is closed immediately
30
- * to prevent it from being assigned to a newer connection.
31
- *
32
- * On failure (e.g. microphone permission denied, WebSocket closed mid-setup),
33
- * sets the error state and transitions to `"disconnected"`.
34
- */
35
- async function initAudioCapture(conn, msg, deps) {
36
- if (conn.audioSetupInFlight) return;
37
- conn.audioSetupInFlight = true;
38
- const gen = conn.generation;
39
- try {
40
- const [{ createVoiceIO }, captureWorklet, playbackWorklet] = await Promise.all([
41
- import("./audio.js"),
42
- import("./worklets/capture-processor.js").then((m) => m.default),
43
- import("./worklets/playback-processor.js").then((m) => m.default)
44
- ]);
45
- const io = await createVoiceIO({
46
- sttSampleRate: msg.sampleRate,
47
- ttsSampleRate: msg.ttsSampleRate,
48
- captureWorkletSrc: captureWorklet,
49
- playbackWorkletSrc: playbackWorklet,
50
- onMicData: (pcm16) => {
51
- try {
52
- deps.sendAudio(new Uint8Array(pcm16));
53
- } catch {
54
- console.debug("[aai-ui] sendAudio dropped: connection closed");
55
- }
56
- }
57
- });
58
- if (conn.generation !== gen || !conn.ws || conn.ws.readyState !== WS_OPEN) {
59
- io.close();
60
- return;
61
- }
62
- conn.voiceIO = io;
63
- deps.sendJson({ type: "audio_ready" });
64
- deps.updateState({ state: "listening" });
65
- } catch (err) {
66
- if (conn.generation !== gen || !conn.ws || conn.ws.readyState !== WS_OPEN) return;
67
- deps.updateState({
68
- state: "error",
69
- error: {
70
- code: "audio",
71
- message: `Microphone access failed: ${errorMessage(err)}`
72
- },
73
- running: false
74
- });
75
- } finally {
76
- conn.audioSetupInFlight = false;
77
- }
78
- }
79
- function buildWsUrl(platformUrl, resume, sessionId) {
80
- const wsUrl = new URL("websocket", platformUrl.endsWith("/") ? platformUrl : `${platformUrl}/`);
81
- wsUrl.protocol = wsUrl.protocol === "https:" ? "wss:" : "ws:";
82
- if (sessionId) wsUrl.searchParams.set("sessionId", sessionId);
83
- else if (resume) wsUrl.searchParams.set("resume", "1");
84
- return wsUrl;
85
- }
86
- /**
87
- * Create a framework-agnostic voice session core that connects to an AAI
88
- * server via WebSocket.
89
- *
90
- * Uses a subscribe/getSnapshot pattern for state management, compatible with
91
- * React's `useSyncExternalStore` and other external store integrations.
92
- *
93
- * @param options - Session configuration including the platform server URL.
94
- * @returns A {@link SessionCore} handle for controlling the session.
95
- *
96
- * @public
97
- */
98
- function createSessionCore(options) {
99
- const WS = options.WebSocket ?? WebSocket;
100
- let currentSnapshot = {
101
- state: "disconnected",
102
- messages: [],
103
- toolCalls: [],
104
- customEvents: [],
105
- userTranscript: null,
106
- agentTranscript: null,
107
- error: null,
108
- started: false,
109
- running: false
110
- };
111
- const subscribers = /* @__PURE__ */ new Set();
112
- function notify() {
113
- for (const sub of subscribers) sub();
114
- }
115
- function updateState(partial) {
116
- currentSnapshot = {
117
- ...currentSnapshot,
118
- ...partial
119
- };
120
- notify();
121
- }
122
- function getSnapshot() {
123
- return currentSnapshot;
124
- }
125
- function subscribe(callback) {
126
- subscribers.add(callback);
127
- return () => {
128
- subscribers.delete(callback);
129
- };
130
- }
131
- const conn = {
132
- ws: null,
133
- voiceIO: null,
134
- audioSetupInFlight: false,
135
- generation: 0
136
- };
137
- let connectionController = null;
138
- let hasConnected = false;
139
- function cleanupAudio() {
140
- conn.audioSetupInFlight = false;
141
- conn.voiceIO?.close();
142
- conn.voiceIO = null;
143
- }
144
- function resetState() {
145
- updateState({
146
- messages: [],
147
- toolCalls: [],
148
- customEvents: [],
149
- userTranscript: null,
150
- agentTranscript: null,
151
- error: null
152
- });
153
- }
154
- function sendJson(msg) {
155
- if (conn.ws && conn.ws.readyState === WS_OPEN) conn.ws.send(JSON.stringify(msg));
156
- }
157
- function sendAudio(bytes) {
158
- if (conn.ws && conn.ws.readyState === WS_OPEN) conn.ws.send(bytes);
159
- }
160
- const audioDeps = {
161
- sendJson,
162
- sendAudio,
163
- updateState
164
- };
165
- /** Incremented on each turn boundary -- stale async callbacks compare against this. */
166
- let handlerGeneration = 0;
167
- /** Monotonically increasing counter for custom events -- used by useEvent to deduplicate. */
168
- let customEventSeq = 0;
169
- function appendCustomEvent(name, data) {
170
- updateState({ customEvents: appendCapped(currentSnapshot.customEvents, {
171
- id: ++customEventSeq,
172
- event: name,
173
- data
174
- }, MAX_CUSTOM_EVENTS) });
175
- }
176
- function handleUserTranscriptEvent(text) {
177
- handlerGeneration++;
178
- updateState({
179
- userTranscript: null,
180
- messages: appendCapped(currentSnapshot.messages, {
181
- role: "user",
182
- content: text
183
- }, MAX_MESSAGES),
184
- state: "thinking"
185
- });
186
- }
187
- function handleAgentTranscriptEvent(text) {
188
- updateState({
189
- agentTranscript: null,
190
- messages: appendCapped(currentSnapshot.messages, {
191
- role: "assistant",
192
- content: text
193
- }, MAX_MESSAGES)
194
- });
195
- }
196
- /** Single entry point for all server->client session events. */
197
- function handleEvent(e) {
198
- if (currentSnapshot.state === "error" && e.type !== "error") updateState({
199
- state: "disconnected",
200
- error: null
201
- });
202
- switch (e.type) {
203
- case "speech_started":
204
- updateState({ userTranscript: "" });
205
- break;
206
- case "speech_stopped": break;
207
- case "user_transcript":
208
- handleUserTranscriptEvent(e.text);
209
- break;
210
- case "agent_transcript":
211
- handleAgentTranscriptEvent(e.text);
212
- break;
213
- case "tool_call":
214
- updateState({ toolCalls: [...currentSnapshot.toolCalls, {
215
- callId: e.toolCallId,
216
- name: e.toolName,
217
- args: e.args ?? {},
218
- status: "pending",
219
- afterMessageIndex: currentSnapshot.messages.length - 1
220
- }] });
221
- break;
222
- case "tool_call_done": {
223
- const tcs = currentSnapshot.toolCalls;
224
- const idx = tcs.findIndex((tc) => tc.callId === e.toolCallId);
225
- if (idx !== -1) {
226
- const updated = [...tcs];
227
- const existing = updated[idx];
228
- if (existing) updated[idx] = {
229
- ...existing,
230
- status: "done",
231
- result: e.result
232
- };
233
- updateState({ toolCalls: updated });
234
- }
235
- break;
236
- }
237
- case "reply_done":
238
- updateState({ state: "listening" });
239
- break;
240
- case "cancelled":
241
- handlerGeneration++;
242
- conn.voiceIO?.flush();
243
- updateState({
244
- userTranscript: null,
245
- agentTranscript: null,
246
- state: "listening"
247
- });
248
- break;
249
- case "reset":
250
- handlerGeneration++;
251
- conn.voiceIO?.flush();
252
- updateState({
253
- messages: [],
254
- toolCalls: [],
255
- customEvents: [],
256
- userTranscript: null,
257
- agentTranscript: null,
258
- error: null,
259
- state: "listening"
260
- });
261
- break;
262
- case "custom_event":
263
- appendCustomEvent(e.event, e.data);
264
- break;
265
- case "error":
266
- console.error("Agent error:", e.message);
267
- updateState({
268
- state: "error",
269
- error: {
270
- code: e.code,
271
- message: e.message
272
- },
273
- running: false
274
- });
275
- break;
276
- case "idle_timeout": break;
277
- default: break;
278
- }
279
- }
280
- /** Enqueue a PCM16 audio chunk for playback. Transitions state to `"speaking"` on the first chunk. */
281
- function playAudioChunk(chunk) {
282
- if (currentSnapshot.state === "disconnected" && currentSnapshot.error !== null) return;
283
- if (currentSnapshot.state !== "speaking") updateState({ state: "speaking" });
284
- conn.voiceIO?.enqueue(chunk.buffer);
285
- }
286
- /**
287
- * Signal that the server has finished sending audio for this turn.
288
- * Waits for the audio queue to drain, then transitions state to `"listening"`.
289
- * Uses the `handlerGeneration` counter to discard stale completions from interrupted turns.
290
- */
291
- function playAudioDone() {
292
- const gen = handlerGeneration;
293
- const io = conn.voiceIO;
294
- if (io) io.done().then(() => {
295
- if (handlerGeneration !== gen) return;
296
- updateState({ state: "listening" });
297
- }).catch((err) => {
298
- console.warn("Audio playback done failed:", err);
299
- });
300
- else updateState({ state: "listening" });
301
- }
302
- /**
303
- * Dispatch an incoming WebSocket message.
304
- *
305
- * Binary frames carry raw PCM16 audio chunks. Text frames are JSON-encoded
306
- * {@link ServerMessage} values validated via Zod.
307
- *
308
- * Returns the parsed config if the message is a `config` message,
309
- * otherwise `undefined`.
310
- */
311
- function handleMessage(data) {
312
- if (data instanceof ArrayBuffer) {
313
- playAudioChunk(new Uint8Array(data));
314
- return;
315
- }
316
- if (typeof data !== "string") {
317
- console.warn("session-core: non-string, non-binary frame received; dropping");
318
- return;
319
- }
320
- let raw;
321
- try {
322
- raw = JSON.parse(data);
323
- } catch {
324
- console.warn("session-core: invalid JSON; dropping");
325
- return;
326
- }
327
- const parsed = lenientParse(ServerMessageSchema, raw);
328
- if (!parsed.ok) {
329
- if (parsed.malformed) console.warn("session-core: malformed server message", parsed.error);
330
- return;
331
- }
332
- const msg = parsed.data;
333
- if (msg.type === "config") return {
334
- sampleRate: msg.sampleRate,
335
- ttsSampleRate: msg.ttsSampleRate,
336
- sid: msg.sessionId
337
- };
338
- if (msg.type === "audio_done") {
339
- playAudioDone();
340
- return;
341
- }
342
- handleEvent(msg);
343
- }
344
- function connect(opts) {
345
- updateState({
346
- state: "connecting",
347
- error: null
348
- });
349
- connectionController?.abort();
350
- cleanupAudio();
351
- conn.ws?.close();
352
- conn.ws = null;
353
- conn.generation++;
354
- const controller = new AbortController();
355
- connectionController = controller;
356
- const { signal: sig } = controller;
357
- if (opts?.signal) opts.signal.addEventListener("abort", () => disconnect(), { signal: sig });
358
- const resumeId = !hasConnected ? options.resumeSessionId : void 0;
359
- const socket = new WS(buildWsUrl(options.platformUrl, hasConnected, resumeId).toString());
360
- socket.binaryType = "arraybuffer";
361
- conn.ws = socket;
362
- socket.addEventListener("open", () => {
363
- updateState({ state: "ready" });
364
- }, { signal: sig });
365
- socket.addEventListener("message", (event) => {
366
- const config = handleMessage(event.data);
367
- if (config) {
368
- if (config.sid) options.onSessionId?.(config.sid);
369
- const isReconnect = hasConnected;
370
- hasConnected = true;
371
- initAudioCapture(conn, config, audioDeps).catch((err) => {
372
- audioDeps.updateState({
373
- state: "error",
374
- error: {
375
- code: "audio",
376
- message: `Audio capture failed: ${errorMessage(err)}`
377
- },
378
- running: false
379
- });
380
- });
381
- if (isReconnect && currentSnapshot.messages.length > 0) sendJson({
382
- type: "history",
383
- messages: currentSnapshot.messages.map((m) => ({
384
- role: m.role,
385
- content: m.content
386
- }))
387
- });
388
- }
389
- }, { signal: sig });
390
- socket.addEventListener("close", () => {
391
- if (sig.aborted) return;
392
- controller.abort();
393
- cleanupAudio();
394
- updateState({
395
- state: "disconnected",
396
- running: false
397
- });
398
- }, { signal: sig });
399
- }
400
- function cancel() {
401
- conn.voiceIO?.flush();
402
- updateState({ state: "listening" });
403
- sendJson({ type: "cancel" });
404
- }
405
- function reset() {
406
- conn.voiceIO?.flush();
407
- if (conn.ws && conn.ws.readyState === WS_OPEN) {
408
- sendJson({ type: "reset" });
409
- return;
410
- }
411
- resetState();
412
- disconnect();
413
- connect();
414
- }
415
- function disconnect() {
416
- connectionController?.abort();
417
- connectionController = null;
418
- cleanupAudio();
419
- conn.ws?.close();
420
- conn.ws = null;
421
- updateState({
422
- state: "disconnected",
423
- running: false
424
- });
425
- }
426
- function start() {
427
- updateState({
428
- started: true,
429
- running: true
430
- });
431
- connect();
432
- }
433
- function toggle() {
434
- if (currentSnapshot.running) disconnect();
435
- else {
436
- updateState({ running: true });
437
- connect();
438
- }
439
- }
440
- return {
441
- getSnapshot,
442
- subscribe,
443
- connect,
444
- cancel,
445
- resetState,
446
- reset,
447
- disconnect,
448
- start,
449
- toggle,
450
- [Symbol.dispose]() {
451
- disconnect();
452
- }
453
- };
454
- }
455
- //#endregion
1
+ import { t as createSessionCore } from "./session-core-D3NDaySY.js";
456
2
  export { createSessionCore };