@alexkroman1/aai-ui 1.8.2 → 1.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/_colors-DJUordGv.js +19 -0
- package/dist/_utils-5cs73OrA.js +16 -0
- package/dist/_utils.d.ts +4 -0
- package/dist/aai-logo-B8lDmsut.js +84 -0
- package/dist/audio.d.ts +9 -2
- package/dist/audio.js +41 -24
- package/dist/chat-view-C1XbqDk0.js +186 -0
- package/dist/components/_colors.d.ts +26 -0
- package/dist/components/aai-logo.d.ts +7 -6
- package/dist/components/button.d.ts +11 -13
- package/dist/components/button.js +7 -16
- package/dist/components/chat-view.d.ts +8 -19
- package/dist/components/chat-view.js +2 -98
- package/dist/components/controls.d.ts +1 -1
- package/dist/components/controls.js +2 -39
- package/dist/components/eyebrow.d.ts +12 -0
- package/dist/components/message-list.d.ts +1 -1
- package/dist/components/message-list.js +126 -56
- package/dist/components/sidebar-layout.d.ts +1 -1
- package/dist/components/start-screen.d.ts +4 -3
- package/dist/components/start-screen.js +20 -14
- package/dist/components/text-controls.d.ts +14 -0
- package/dist/components/tool-call-block.d.ts +5 -3
- package/dist/components/tool-call-block.js +1 -1
- package/dist/components/url-chips.d.ts +28 -0
- package/dist/context.d.ts +30 -4
- package/dist/context.js +86 -10
- package/dist/controls-BngrPbOC.js +135 -0
- package/dist/default-client/assets/audio-Cs-6t_Wd.js +1 -0
- package/dist/default-client/assets/capture-processor-C19oBn4L.js +105 -0
- package/dist/default-client/assets/index-Bf4ZTNcx.js +73 -0
- package/dist/default-client/assets/index-Bzlh9i7w.css +2 -0
- package/dist/default-client/assets/playback-processor-BtlzAH78.js +149 -0
- package/dist/default-client/index.html +3 -3
- package/dist/define-client.d.ts +2 -2
- package/dist/define-client.js +18 -24
- package/dist/eyebrow-C6ZFuiz6.js +27 -0
- package/dist/hooks.js +71 -42
- package/dist/index.d.ts +3 -1
- package/dist/index.js +6 -5
- package/dist/session-core-D3NDaySY.js +652 -0
- package/dist/session-core-messages.d.ts +50 -0
- package/dist/session-core-types.d.ts +147 -0
- package/dist/session-core-upload.d.ts +16 -0
- package/dist/session-core.d.ts +2 -68
- package/dist/session-core.js +1 -455
- package/dist/{tool-call-block-wby_jyoY.js → tool-call-block-7f1GTG-P.js} +33 -31
- package/dist/types.d.ts +29 -4
- package/dist/types.js +10 -1
- package/dist/worklets/capture-processor.d.ts +2 -0
- package/dist/worklets/capture-processor.js +80 -25
- package/dist/worklets/playback-processor.d.ts +2 -0
- package/dist/worklets/playback-processor.js +80 -29
- package/package.json +18 -18
- package/styles.css +30 -0
- package/dist/_react-test-utils.d.ts +0 -86
- package/dist/aai-logo-BqFv6JU6.js +0 -21
- package/dist/default-client/assets/audio-UowwhmYo.js +0 -1
- package/dist/default-client/assets/capture-processor-C26lSiVr.js +0 -53
- package/dist/default-client/assets/index-BZqVGR2o.css +0 -2
- package/dist/default-client/assets/index-YW9WUhbL.js +0 -49
- package/dist/default-client/assets/playback-processor-dSA8Im99.js +0 -101
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Type declarations for the framework-agnostic voice session core.
|
|
3
|
+
*
|
|
4
|
+
* Split out of `session-core.ts` to keep that module focused on behaviour.
|
|
5
|
+
* The public types here are re-exported from `session-core.ts` for
|
|
6
|
+
* backwards compatibility.
|
|
7
|
+
*/
|
|
8
|
+
import type { VoiceIO } from "./audio.ts";
|
|
9
|
+
import type { AgentState, ChatMessage, SessionError, ToolCallInfo, VoiceSessionOptions, WebSocketConstructor } from "./types.ts";
|
|
10
|
+
/**
|
|
11
|
+
* A custom event emitted by the agent via `ctx.send`.
|
|
12
|
+
*
|
|
13
|
+
* @public
|
|
14
|
+
*/
|
|
15
|
+
export type CustomEvent = {
|
|
16
|
+
readonly id: number;
|
|
17
|
+
readonly event: string;
|
|
18
|
+
readonly data: unknown;
|
|
19
|
+
};
|
|
20
|
+
/**
|
|
21
|
+
* Immutable snapshot of the session state.
|
|
22
|
+
*
|
|
23
|
+
* Consumers (e.g. React hooks via `useSyncExternalStore`) read this to render.
|
|
24
|
+
* A new object reference is created on every state change.
|
|
25
|
+
*
|
|
26
|
+
* @public
|
|
27
|
+
*/
|
|
28
|
+
export type SessionSnapshot = {
|
|
29
|
+
readonly state: AgentState;
|
|
30
|
+
/**
|
|
31
|
+
* False when the server declared the session text-only (`tts: none()`):
|
|
32
|
+
* no audio frames will arrive and replies render as text. True until the
|
|
33
|
+
* server's `config` message says otherwise (voice is the default).
|
|
34
|
+
*/
|
|
35
|
+
readonly audioOut: boolean;
|
|
36
|
+
/**
|
|
37
|
+
* True while the microphone is live and streaming to the server. Voice
|
|
38
|
+
* sessions record for their whole lifetime; text-only sessions toggle this
|
|
39
|
+
* via `startRecording()` / `stopRecording()` (the record button).
|
|
40
|
+
*/
|
|
41
|
+
readonly recording: boolean;
|
|
42
|
+
/**
|
|
43
|
+
* The WebSocket URL a program can connect to directly (the same endpoint
|
|
44
|
+
* this session uses), e.g. `wss://host/my-agent/websocket`. Derived from
|
|
45
|
+
* `platformUrl` at construction — available before connecting.
|
|
46
|
+
*/
|
|
47
|
+
readonly apiUrl: string;
|
|
48
|
+
/**
|
|
49
|
+
* Monotonically increasing counter bumped whenever rendered conversation
|
|
50
|
+
* content changes (`messages`, `toolCalls`, or either live transcript).
|
|
51
|
+
* Cheap dependency for scroll-to-bottom effects — unlike summed lengths it
|
|
52
|
+
* never collides when the capped arrays slide.
|
|
53
|
+
*/
|
|
54
|
+
readonly contentVersion: number;
|
|
55
|
+
readonly messages: ChatMessage[];
|
|
56
|
+
readonly toolCalls: ToolCallInfo[];
|
|
57
|
+
readonly customEvents: CustomEvent[];
|
|
58
|
+
readonly userTranscript: string | null;
|
|
59
|
+
readonly agentTranscript: string | null;
|
|
60
|
+
readonly error: SessionError | null;
|
|
61
|
+
readonly started: boolean;
|
|
62
|
+
readonly running: boolean;
|
|
63
|
+
};
|
|
64
|
+
/**
|
|
65
|
+
* A framework-agnostic voice session that manages WebSocket communication,
|
|
66
|
+
* audio capture/playback, and agent state transitions.
|
|
67
|
+
*
|
|
68
|
+
* Uses a subscribe/getSnapshot pattern (compatible with React's
|
|
69
|
+
* `useSyncExternalStore`). Implements `Disposable` for resource cleanup.
|
|
70
|
+
*
|
|
71
|
+
* @public
|
|
72
|
+
*/
|
|
73
|
+
export type SessionCore = {
|
|
74
|
+
/** Return the current immutable state snapshot. */
|
|
75
|
+
getSnapshot(): SessionSnapshot;
|
|
76
|
+
/** Subscribe to state changes. Returns an unsubscribe function. */
|
|
77
|
+
subscribe(callback: () => void): () => void;
|
|
78
|
+
/**
|
|
79
|
+
* Open a WebSocket connection to the server and begin audio capture.
|
|
80
|
+
* @param options - Optional. `signal` is an AbortSignal that, when aborted, disconnects the session.
|
|
81
|
+
*/
|
|
82
|
+
connect(options?: {
|
|
83
|
+
signal?: AbortSignal;
|
|
84
|
+
}): void;
|
|
85
|
+
/** Cancel the current agent turn and discard in-flight TTS audio. */
|
|
86
|
+
cancel(): void;
|
|
87
|
+
/** Clear messages, transcript, and error state without disconnecting. */
|
|
88
|
+
resetState(): void;
|
|
89
|
+
/** Reset the session: clear state and reconnect. */
|
|
90
|
+
reset(): void;
|
|
91
|
+
/** Close the WebSocket and release all audio resources. */
|
|
92
|
+
disconnect(): void;
|
|
93
|
+
/** Start the session for the first time (sets `started` and `running`). */
|
|
94
|
+
start(): void;
|
|
95
|
+
/** Toggle between connected and disconnected states. */
|
|
96
|
+
toggle(): void;
|
|
97
|
+
/**
|
|
98
|
+
* Start streaming microphone audio (text-only sessions). Requests mic
|
|
99
|
+
* access on first use. No-op in voice sessions, where the mic is always on.
|
|
100
|
+
*/
|
|
101
|
+
startRecording(): void;
|
|
102
|
+
/** Stop streaming microphone audio (text-only sessions). No-op in voice sessions. */
|
|
103
|
+
stopRecording(): void;
|
|
104
|
+
/**
|
|
105
|
+
* Decode an audio file (any format the browser can decode), resample it to
|
|
106
|
+
* the session's STT rate, and stream it to the server for transcription.
|
|
107
|
+
* Text-only sessions (`tts: none()`) only — voice sessions stream the
|
|
108
|
+
* microphone instead and reject. Also rejects when no session is connected,
|
|
109
|
+
* the mic is recording, another upload is in flight, or the file cannot be
|
|
110
|
+
* decoded. Resolves once the audio has been handed to the socket.
|
|
111
|
+
*/
|
|
112
|
+
sendAudioFile(file: Blob): Promise<void>;
|
|
113
|
+
/** Alias for `disconnect` for use with `using`. */
|
|
114
|
+
[Symbol.dispose](): void;
|
|
115
|
+
};
|
|
116
|
+
export type SessionCoreOptions = VoiceSessionOptions;
|
|
117
|
+
/**
|
|
118
|
+
* Shared mutable connection state for audio initialization.
|
|
119
|
+
*
|
|
120
|
+
* Tracks the active WebSocket, VoiceIO instance, and a generation counter
|
|
121
|
+
* that prevents stale async operations (e.g. a slow `initAudioCapture`) from
|
|
122
|
+
* assigning their results to a newer connection after a reconnect.
|
|
123
|
+
*/
|
|
124
|
+
export type ConnState = {
|
|
125
|
+
ws: InstanceType<WebSocketConstructor> | null;
|
|
126
|
+
voiceIO: VoiceIO | null;
|
|
127
|
+
audioSetupInFlight: boolean;
|
|
128
|
+
/** Monotonically increasing counter bumped on each connect(). Prevents a stale
|
|
129
|
+
* initAudioCapture from assigning its voiceIO to a newer connection. */
|
|
130
|
+
generation: number;
|
|
131
|
+
/** Audio chunks that arrived before `voiceIO` was initialized — drained into
|
|
132
|
+
* the playback worklet once init completes. Closes the race between the
|
|
133
|
+
* server starting greeting audio (immediately on S2S connect) and the
|
|
134
|
+
* client awaiting mic permission + worklet registration. */
|
|
135
|
+
preInitAudio: Uint8Array[];
|
|
136
|
+
/** True if `audio_done` arrived before `voiceIO` was initialized. The done
|
|
137
|
+
* signal must be replayed after draining preInitAudio, or a short greeting
|
|
138
|
+
* buffered during mic-permission never finishes playing. */
|
|
139
|
+
preInitDone: boolean;
|
|
140
|
+
/** The server's `config` payload for the current connection — kept so
|
|
141
|
+
* text-only sessions can init the mic lazily (record button) and file
|
|
142
|
+
* uploads know the STT sample rate to resample to. */
|
|
143
|
+
readyConfig: {
|
|
144
|
+
sampleRate: number;
|
|
145
|
+
ttsSampleRate: number;
|
|
146
|
+
} | null;
|
|
147
|
+
};
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import type { ClientMessage } from "@alexkroman1/aai/protocol";
|
|
2
|
+
import type { ConnState, SessionSnapshot } from "./session-core-types.ts";
|
|
3
|
+
/** Handle to the upload pipeline, owned by one session core. */
|
|
4
|
+
export type UploadSender = {
|
|
5
|
+
/** See {@link SessionCore.sendAudioFile}. */
|
|
6
|
+
sendAudioFile(file: Blob): Promise<void>;
|
|
7
|
+
/** True while an upload is decoding or streaming — the mic must stay off. */
|
|
8
|
+
inFlight(): boolean;
|
|
9
|
+
/** Invalidate any in-flight upload (reset / close / reconnect). */
|
|
10
|
+
discard(): void;
|
|
11
|
+
};
|
|
12
|
+
export declare function createUploadSender(deps: {
|
|
13
|
+
conn: ConnState;
|
|
14
|
+
getSnapshot: () => SessionSnapshot;
|
|
15
|
+
sendJson: (msg: ClientMessage) => void;
|
|
16
|
+
}): UploadSender;
|
package/dist/session-core.d.ts
CHANGED
|
@@ -1,71 +1,5 @@
|
|
|
1
|
-
import type {
|
|
2
|
-
export type {
|
|
3
|
-
/**
|
|
4
|
-
* A custom event emitted by the agent via `ctx.send`.
|
|
5
|
-
*
|
|
6
|
-
* @public
|
|
7
|
-
*/
|
|
8
|
-
export type CustomEvent = {
|
|
9
|
-
readonly id: number;
|
|
10
|
-
readonly event: string;
|
|
11
|
-
readonly data: unknown;
|
|
12
|
-
};
|
|
13
|
-
/**
|
|
14
|
-
* Immutable snapshot of the session state.
|
|
15
|
-
*
|
|
16
|
-
* Consumers (e.g. React hooks via `useSyncExternalStore`) read this to render.
|
|
17
|
-
* A new object reference is created on every state change.
|
|
18
|
-
*
|
|
19
|
-
* @public
|
|
20
|
-
*/
|
|
21
|
-
export type SessionSnapshot = {
|
|
22
|
-
readonly state: AgentState;
|
|
23
|
-
readonly messages: ChatMessage[];
|
|
24
|
-
readonly toolCalls: ToolCallInfo[];
|
|
25
|
-
readonly customEvents: CustomEvent[];
|
|
26
|
-
readonly userTranscript: string | null;
|
|
27
|
-
readonly agentTranscript: string | null;
|
|
28
|
-
readonly error: SessionError | null;
|
|
29
|
-
readonly started: boolean;
|
|
30
|
-
readonly running: boolean;
|
|
31
|
-
};
|
|
32
|
-
/**
|
|
33
|
-
* A framework-agnostic voice session that manages WebSocket communication,
|
|
34
|
-
* audio capture/playback, and agent state transitions.
|
|
35
|
-
*
|
|
36
|
-
* Uses a subscribe/getSnapshot pattern (compatible with React's
|
|
37
|
-
* `useSyncExternalStore`). Implements `Disposable` for resource cleanup.
|
|
38
|
-
*
|
|
39
|
-
* @public
|
|
40
|
-
*/
|
|
41
|
-
export type SessionCore = {
|
|
42
|
-
/** Return the current immutable state snapshot. */
|
|
43
|
-
getSnapshot(): SessionSnapshot;
|
|
44
|
-
/** Subscribe to state changes. Returns an unsubscribe function. */
|
|
45
|
-
subscribe(callback: () => void): () => void;
|
|
46
|
-
/**
|
|
47
|
-
* Open a WebSocket connection to the server and begin audio capture.
|
|
48
|
-
* @param options - Optional. `signal` is an AbortSignal that, when aborted, disconnects the session.
|
|
49
|
-
*/
|
|
50
|
-
connect(options?: {
|
|
51
|
-
signal?: AbortSignal;
|
|
52
|
-
}): void;
|
|
53
|
-
/** Cancel the current agent turn and discard in-flight TTS audio. */
|
|
54
|
-
cancel(): void;
|
|
55
|
-
/** Clear messages, transcript, and error state without disconnecting. */
|
|
56
|
-
resetState(): void;
|
|
57
|
-
/** Reset the session: clear state and reconnect. */
|
|
58
|
-
reset(): void;
|
|
59
|
-
/** Close the WebSocket and release all audio resources. */
|
|
60
|
-
disconnect(): void;
|
|
61
|
-
/** Start the session for the first time (sets `started` and `running`). */
|
|
62
|
-
start(): void;
|
|
63
|
-
/** Toggle between connected and disconnected states. */
|
|
64
|
-
toggle(): void;
|
|
65
|
-
/** Alias for `disconnect` for use with `using`. */
|
|
66
|
-
[Symbol.dispose](): void;
|
|
67
|
-
};
|
|
68
|
-
export type SessionCoreOptions = VoiceSessionOptions;
|
|
1
|
+
import type { SessionCore, SessionCoreOptions } from "./session-core-types.ts";
|
|
2
|
+
export type { CustomEvent, SessionCore, SessionCoreOptions, SessionSnapshot, } from "./session-core-types.ts";
|
|
69
3
|
/**
|
|
70
4
|
* Create a framework-agnostic voice session core that connects to an AAI
|
|
71
5
|
* server via WebSocket.
|
package/dist/session-core.js
CHANGED
|
@@ -1,456 +1,2 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import { ServerMessageSchema, lenientParse } from "@alexkroman1/aai/protocol";
|
|
3
|
-
//#region session-core.ts
|
|
4
|
-
/**
|
|
5
|
-
* Framework-agnostic voice session core.
|
|
6
|
-
*
|
|
7
|
-
* Manages WebSocket communication, audio capture/playback, and agent state
|
|
8
|
-
* transitions using a subscribe/getSnapshot pattern compatible with React's
|
|
9
|
-
* `useSyncExternalStore` and other external store consumers.
|
|
10
|
-
*
|
|
11
|
-
* No dependency on React, Preact, or any UI framework.
|
|
12
|
-
*/
|
|
13
|
-
/** Cap on `customEvents` retained in the session snapshot to avoid unbounded growth. */
|
|
14
|
-
const MAX_CUSTOM_EVENTS = 200;
|
|
15
|
-
/** Cap on `messages` retained in the session snapshot; mirrors host-side DEFAULT_MAX_HISTORY. */
|
|
16
|
-
const MAX_MESSAGES = 200;
|
|
17
|
-
function appendCapped(list, item, cap) {
|
|
18
|
-
const next = [...list, item];
|
|
19
|
-
return next.length > cap ? next.slice(-cap) : next;
|
|
20
|
-
}
|
|
21
|
-
/**
|
|
22
|
-
* Initialize audio capture and playback after the server sends a ready config.
|
|
23
|
-
*
|
|
24
|
-
* Lifecycle: dynamically import audio modules -> request microphone access ->
|
|
25
|
-
* register AudioWorklet processors -> create a `VoiceIO` instance -> send
|
|
26
|
-
* `audio_ready` to the server -> transition state to `"listening"`.
|
|
27
|
-
*
|
|
28
|
-
* Uses the connection `generation` counter to detect if `connect()` was called
|
|
29
|
-
* while awaiting async operations; if so, the stale VoiceIO is closed immediately
|
|
30
|
-
* to prevent it from being assigned to a newer connection.
|
|
31
|
-
*
|
|
32
|
-
* On failure (e.g. microphone permission denied, WebSocket closed mid-setup),
|
|
33
|
-
* sets the error state and transitions to `"disconnected"`.
|
|
34
|
-
*/
|
|
35
|
-
async function initAudioCapture(conn, msg, deps) {
|
|
36
|
-
if (conn.audioSetupInFlight) return;
|
|
37
|
-
conn.audioSetupInFlight = true;
|
|
38
|
-
const gen = conn.generation;
|
|
39
|
-
try {
|
|
40
|
-
const [{ createVoiceIO }, captureWorklet, playbackWorklet] = await Promise.all([
|
|
41
|
-
import("./audio.js"),
|
|
42
|
-
import("./worklets/capture-processor.js").then((m) => m.default),
|
|
43
|
-
import("./worklets/playback-processor.js").then((m) => m.default)
|
|
44
|
-
]);
|
|
45
|
-
const io = await createVoiceIO({
|
|
46
|
-
sttSampleRate: msg.sampleRate,
|
|
47
|
-
ttsSampleRate: msg.ttsSampleRate,
|
|
48
|
-
captureWorkletSrc: captureWorklet,
|
|
49
|
-
playbackWorkletSrc: playbackWorklet,
|
|
50
|
-
onMicData: (pcm16) => {
|
|
51
|
-
try {
|
|
52
|
-
deps.sendAudio(new Uint8Array(pcm16));
|
|
53
|
-
} catch {
|
|
54
|
-
console.debug("[aai-ui] sendAudio dropped: connection closed");
|
|
55
|
-
}
|
|
56
|
-
}
|
|
57
|
-
});
|
|
58
|
-
if (conn.generation !== gen || !conn.ws || conn.ws.readyState !== WS_OPEN) {
|
|
59
|
-
io.close();
|
|
60
|
-
return;
|
|
61
|
-
}
|
|
62
|
-
conn.voiceIO = io;
|
|
63
|
-
deps.sendJson({ type: "audio_ready" });
|
|
64
|
-
deps.updateState({ state: "listening" });
|
|
65
|
-
} catch (err) {
|
|
66
|
-
if (conn.generation !== gen || !conn.ws || conn.ws.readyState !== WS_OPEN) return;
|
|
67
|
-
deps.updateState({
|
|
68
|
-
state: "error",
|
|
69
|
-
error: {
|
|
70
|
-
code: "audio",
|
|
71
|
-
message: `Microphone access failed: ${errorMessage(err)}`
|
|
72
|
-
},
|
|
73
|
-
running: false
|
|
74
|
-
});
|
|
75
|
-
} finally {
|
|
76
|
-
conn.audioSetupInFlight = false;
|
|
77
|
-
}
|
|
78
|
-
}
|
|
79
|
-
function buildWsUrl(platformUrl, resume, sessionId) {
|
|
80
|
-
const wsUrl = new URL("websocket", platformUrl.endsWith("/") ? platformUrl : `${platformUrl}/`);
|
|
81
|
-
wsUrl.protocol = wsUrl.protocol === "https:" ? "wss:" : "ws:";
|
|
82
|
-
if (sessionId) wsUrl.searchParams.set("sessionId", sessionId);
|
|
83
|
-
else if (resume) wsUrl.searchParams.set("resume", "1");
|
|
84
|
-
return wsUrl;
|
|
85
|
-
}
|
|
86
|
-
/**
|
|
87
|
-
* Create a framework-agnostic voice session core that connects to an AAI
|
|
88
|
-
* server via WebSocket.
|
|
89
|
-
*
|
|
90
|
-
* Uses a subscribe/getSnapshot pattern for state management, compatible with
|
|
91
|
-
* React's `useSyncExternalStore` and other external store integrations.
|
|
92
|
-
*
|
|
93
|
-
* @param options - Session configuration including the platform server URL.
|
|
94
|
-
* @returns A {@link SessionCore} handle for controlling the session.
|
|
95
|
-
*
|
|
96
|
-
* @public
|
|
97
|
-
*/
|
|
98
|
-
function createSessionCore(options) {
|
|
99
|
-
const WS = options.WebSocket ?? WebSocket;
|
|
100
|
-
let currentSnapshot = {
|
|
101
|
-
state: "disconnected",
|
|
102
|
-
messages: [],
|
|
103
|
-
toolCalls: [],
|
|
104
|
-
customEvents: [],
|
|
105
|
-
userTranscript: null,
|
|
106
|
-
agentTranscript: null,
|
|
107
|
-
error: null,
|
|
108
|
-
started: false,
|
|
109
|
-
running: false
|
|
110
|
-
};
|
|
111
|
-
const subscribers = /* @__PURE__ */ new Set();
|
|
112
|
-
function notify() {
|
|
113
|
-
for (const sub of subscribers) sub();
|
|
114
|
-
}
|
|
115
|
-
function updateState(partial) {
|
|
116
|
-
currentSnapshot = {
|
|
117
|
-
...currentSnapshot,
|
|
118
|
-
...partial
|
|
119
|
-
};
|
|
120
|
-
notify();
|
|
121
|
-
}
|
|
122
|
-
function getSnapshot() {
|
|
123
|
-
return currentSnapshot;
|
|
124
|
-
}
|
|
125
|
-
function subscribe(callback) {
|
|
126
|
-
subscribers.add(callback);
|
|
127
|
-
return () => {
|
|
128
|
-
subscribers.delete(callback);
|
|
129
|
-
};
|
|
130
|
-
}
|
|
131
|
-
const conn = {
|
|
132
|
-
ws: null,
|
|
133
|
-
voiceIO: null,
|
|
134
|
-
audioSetupInFlight: false,
|
|
135
|
-
generation: 0
|
|
136
|
-
};
|
|
137
|
-
let connectionController = null;
|
|
138
|
-
let hasConnected = false;
|
|
139
|
-
function cleanupAudio() {
|
|
140
|
-
conn.audioSetupInFlight = false;
|
|
141
|
-
conn.voiceIO?.close();
|
|
142
|
-
conn.voiceIO = null;
|
|
143
|
-
}
|
|
144
|
-
function resetState() {
|
|
145
|
-
updateState({
|
|
146
|
-
messages: [],
|
|
147
|
-
toolCalls: [],
|
|
148
|
-
customEvents: [],
|
|
149
|
-
userTranscript: null,
|
|
150
|
-
agentTranscript: null,
|
|
151
|
-
error: null
|
|
152
|
-
});
|
|
153
|
-
}
|
|
154
|
-
function sendJson(msg) {
|
|
155
|
-
if (conn.ws && conn.ws.readyState === WS_OPEN) conn.ws.send(JSON.stringify(msg));
|
|
156
|
-
}
|
|
157
|
-
function sendAudio(bytes) {
|
|
158
|
-
if (conn.ws && conn.ws.readyState === WS_OPEN) conn.ws.send(bytes);
|
|
159
|
-
}
|
|
160
|
-
const audioDeps = {
|
|
161
|
-
sendJson,
|
|
162
|
-
sendAudio,
|
|
163
|
-
updateState
|
|
164
|
-
};
|
|
165
|
-
/** Incremented on each turn boundary -- stale async callbacks compare against this. */
|
|
166
|
-
let handlerGeneration = 0;
|
|
167
|
-
/** Monotonically increasing counter for custom events -- used by useEvent to deduplicate. */
|
|
168
|
-
let customEventSeq = 0;
|
|
169
|
-
function appendCustomEvent(name, data) {
|
|
170
|
-
updateState({ customEvents: appendCapped(currentSnapshot.customEvents, {
|
|
171
|
-
id: ++customEventSeq,
|
|
172
|
-
event: name,
|
|
173
|
-
data
|
|
174
|
-
}, MAX_CUSTOM_EVENTS) });
|
|
175
|
-
}
|
|
176
|
-
function handleUserTranscriptEvent(text) {
|
|
177
|
-
handlerGeneration++;
|
|
178
|
-
updateState({
|
|
179
|
-
userTranscript: null,
|
|
180
|
-
messages: appendCapped(currentSnapshot.messages, {
|
|
181
|
-
role: "user",
|
|
182
|
-
content: text
|
|
183
|
-
}, MAX_MESSAGES),
|
|
184
|
-
state: "thinking"
|
|
185
|
-
});
|
|
186
|
-
}
|
|
187
|
-
function handleAgentTranscriptEvent(text) {
|
|
188
|
-
updateState({
|
|
189
|
-
agentTranscript: null,
|
|
190
|
-
messages: appendCapped(currentSnapshot.messages, {
|
|
191
|
-
role: "assistant",
|
|
192
|
-
content: text
|
|
193
|
-
}, MAX_MESSAGES)
|
|
194
|
-
});
|
|
195
|
-
}
|
|
196
|
-
/** Single entry point for all server->client session events. */
|
|
197
|
-
function handleEvent(e) {
|
|
198
|
-
if (currentSnapshot.state === "error" && e.type !== "error") updateState({
|
|
199
|
-
state: "disconnected",
|
|
200
|
-
error: null
|
|
201
|
-
});
|
|
202
|
-
switch (e.type) {
|
|
203
|
-
case "speech_started":
|
|
204
|
-
updateState({ userTranscript: "" });
|
|
205
|
-
break;
|
|
206
|
-
case "speech_stopped": break;
|
|
207
|
-
case "user_transcript":
|
|
208
|
-
handleUserTranscriptEvent(e.text);
|
|
209
|
-
break;
|
|
210
|
-
case "agent_transcript":
|
|
211
|
-
handleAgentTranscriptEvent(e.text);
|
|
212
|
-
break;
|
|
213
|
-
case "tool_call":
|
|
214
|
-
updateState({ toolCalls: [...currentSnapshot.toolCalls, {
|
|
215
|
-
callId: e.toolCallId,
|
|
216
|
-
name: e.toolName,
|
|
217
|
-
args: e.args ?? {},
|
|
218
|
-
status: "pending",
|
|
219
|
-
afterMessageIndex: currentSnapshot.messages.length - 1
|
|
220
|
-
}] });
|
|
221
|
-
break;
|
|
222
|
-
case "tool_call_done": {
|
|
223
|
-
const tcs = currentSnapshot.toolCalls;
|
|
224
|
-
const idx = tcs.findIndex((tc) => tc.callId === e.toolCallId);
|
|
225
|
-
if (idx !== -1) {
|
|
226
|
-
const updated = [...tcs];
|
|
227
|
-
const existing = updated[idx];
|
|
228
|
-
if (existing) updated[idx] = {
|
|
229
|
-
...existing,
|
|
230
|
-
status: "done",
|
|
231
|
-
result: e.result
|
|
232
|
-
};
|
|
233
|
-
updateState({ toolCalls: updated });
|
|
234
|
-
}
|
|
235
|
-
break;
|
|
236
|
-
}
|
|
237
|
-
case "reply_done":
|
|
238
|
-
updateState({ state: "listening" });
|
|
239
|
-
break;
|
|
240
|
-
case "cancelled":
|
|
241
|
-
handlerGeneration++;
|
|
242
|
-
conn.voiceIO?.flush();
|
|
243
|
-
updateState({
|
|
244
|
-
userTranscript: null,
|
|
245
|
-
agentTranscript: null,
|
|
246
|
-
state: "listening"
|
|
247
|
-
});
|
|
248
|
-
break;
|
|
249
|
-
case "reset":
|
|
250
|
-
handlerGeneration++;
|
|
251
|
-
conn.voiceIO?.flush();
|
|
252
|
-
updateState({
|
|
253
|
-
messages: [],
|
|
254
|
-
toolCalls: [],
|
|
255
|
-
customEvents: [],
|
|
256
|
-
userTranscript: null,
|
|
257
|
-
agentTranscript: null,
|
|
258
|
-
error: null,
|
|
259
|
-
state: "listening"
|
|
260
|
-
});
|
|
261
|
-
break;
|
|
262
|
-
case "custom_event":
|
|
263
|
-
appendCustomEvent(e.event, e.data);
|
|
264
|
-
break;
|
|
265
|
-
case "error":
|
|
266
|
-
console.error("Agent error:", e.message);
|
|
267
|
-
updateState({
|
|
268
|
-
state: "error",
|
|
269
|
-
error: {
|
|
270
|
-
code: e.code,
|
|
271
|
-
message: e.message
|
|
272
|
-
},
|
|
273
|
-
running: false
|
|
274
|
-
});
|
|
275
|
-
break;
|
|
276
|
-
case "idle_timeout": break;
|
|
277
|
-
default: break;
|
|
278
|
-
}
|
|
279
|
-
}
|
|
280
|
-
/** Enqueue a PCM16 audio chunk for playback. Transitions state to `"speaking"` on the first chunk. */
|
|
281
|
-
function playAudioChunk(chunk) {
|
|
282
|
-
if (currentSnapshot.state === "disconnected" && currentSnapshot.error !== null) return;
|
|
283
|
-
if (currentSnapshot.state !== "speaking") updateState({ state: "speaking" });
|
|
284
|
-
conn.voiceIO?.enqueue(chunk.buffer);
|
|
285
|
-
}
|
|
286
|
-
/**
|
|
287
|
-
* Signal that the server has finished sending audio for this turn.
|
|
288
|
-
* Waits for the audio queue to drain, then transitions state to `"listening"`.
|
|
289
|
-
* Uses the `handlerGeneration` counter to discard stale completions from interrupted turns.
|
|
290
|
-
*/
|
|
291
|
-
function playAudioDone() {
|
|
292
|
-
const gen = handlerGeneration;
|
|
293
|
-
const io = conn.voiceIO;
|
|
294
|
-
if (io) io.done().then(() => {
|
|
295
|
-
if (handlerGeneration !== gen) return;
|
|
296
|
-
updateState({ state: "listening" });
|
|
297
|
-
}).catch((err) => {
|
|
298
|
-
console.warn("Audio playback done failed:", err);
|
|
299
|
-
});
|
|
300
|
-
else updateState({ state: "listening" });
|
|
301
|
-
}
|
|
302
|
-
/**
|
|
303
|
-
* Dispatch an incoming WebSocket message.
|
|
304
|
-
*
|
|
305
|
-
* Binary frames carry raw PCM16 audio chunks. Text frames are JSON-encoded
|
|
306
|
-
* {@link ServerMessage} values validated via Zod.
|
|
307
|
-
*
|
|
308
|
-
* Returns the parsed config if the message is a `config` message,
|
|
309
|
-
* otherwise `undefined`.
|
|
310
|
-
*/
|
|
311
|
-
function handleMessage(data) {
|
|
312
|
-
if (data instanceof ArrayBuffer) {
|
|
313
|
-
playAudioChunk(new Uint8Array(data));
|
|
314
|
-
return;
|
|
315
|
-
}
|
|
316
|
-
if (typeof data !== "string") {
|
|
317
|
-
console.warn("session-core: non-string, non-binary frame received; dropping");
|
|
318
|
-
return;
|
|
319
|
-
}
|
|
320
|
-
let raw;
|
|
321
|
-
try {
|
|
322
|
-
raw = JSON.parse(data);
|
|
323
|
-
} catch {
|
|
324
|
-
console.warn("session-core: invalid JSON; dropping");
|
|
325
|
-
return;
|
|
326
|
-
}
|
|
327
|
-
const parsed = lenientParse(ServerMessageSchema, raw);
|
|
328
|
-
if (!parsed.ok) {
|
|
329
|
-
if (parsed.malformed) console.warn("session-core: malformed server message", parsed.error);
|
|
330
|
-
return;
|
|
331
|
-
}
|
|
332
|
-
const msg = parsed.data;
|
|
333
|
-
if (msg.type === "config") return {
|
|
334
|
-
sampleRate: msg.sampleRate,
|
|
335
|
-
ttsSampleRate: msg.ttsSampleRate,
|
|
336
|
-
sid: msg.sessionId
|
|
337
|
-
};
|
|
338
|
-
if (msg.type === "audio_done") {
|
|
339
|
-
playAudioDone();
|
|
340
|
-
return;
|
|
341
|
-
}
|
|
342
|
-
handleEvent(msg);
|
|
343
|
-
}
|
|
344
|
-
function connect(opts) {
|
|
345
|
-
updateState({
|
|
346
|
-
state: "connecting",
|
|
347
|
-
error: null
|
|
348
|
-
});
|
|
349
|
-
connectionController?.abort();
|
|
350
|
-
cleanupAudio();
|
|
351
|
-
conn.ws?.close();
|
|
352
|
-
conn.ws = null;
|
|
353
|
-
conn.generation++;
|
|
354
|
-
const controller = new AbortController();
|
|
355
|
-
connectionController = controller;
|
|
356
|
-
const { signal: sig } = controller;
|
|
357
|
-
if (opts?.signal) opts.signal.addEventListener("abort", () => disconnect(), { signal: sig });
|
|
358
|
-
const resumeId = !hasConnected ? options.resumeSessionId : void 0;
|
|
359
|
-
const socket = new WS(buildWsUrl(options.platformUrl, hasConnected, resumeId).toString());
|
|
360
|
-
socket.binaryType = "arraybuffer";
|
|
361
|
-
conn.ws = socket;
|
|
362
|
-
socket.addEventListener("open", () => {
|
|
363
|
-
updateState({ state: "ready" });
|
|
364
|
-
}, { signal: sig });
|
|
365
|
-
socket.addEventListener("message", (event) => {
|
|
366
|
-
const config = handleMessage(event.data);
|
|
367
|
-
if (config) {
|
|
368
|
-
if (config.sid) options.onSessionId?.(config.sid);
|
|
369
|
-
const isReconnect = hasConnected;
|
|
370
|
-
hasConnected = true;
|
|
371
|
-
initAudioCapture(conn, config, audioDeps).catch((err) => {
|
|
372
|
-
audioDeps.updateState({
|
|
373
|
-
state: "error",
|
|
374
|
-
error: {
|
|
375
|
-
code: "audio",
|
|
376
|
-
message: `Audio capture failed: ${errorMessage(err)}`
|
|
377
|
-
},
|
|
378
|
-
running: false
|
|
379
|
-
});
|
|
380
|
-
});
|
|
381
|
-
if (isReconnect && currentSnapshot.messages.length > 0) sendJson({
|
|
382
|
-
type: "history",
|
|
383
|
-
messages: currentSnapshot.messages.map((m) => ({
|
|
384
|
-
role: m.role,
|
|
385
|
-
content: m.content
|
|
386
|
-
}))
|
|
387
|
-
});
|
|
388
|
-
}
|
|
389
|
-
}, { signal: sig });
|
|
390
|
-
socket.addEventListener("close", () => {
|
|
391
|
-
if (sig.aborted) return;
|
|
392
|
-
controller.abort();
|
|
393
|
-
cleanupAudio();
|
|
394
|
-
updateState({
|
|
395
|
-
state: "disconnected",
|
|
396
|
-
running: false
|
|
397
|
-
});
|
|
398
|
-
}, { signal: sig });
|
|
399
|
-
}
|
|
400
|
-
function cancel() {
|
|
401
|
-
conn.voiceIO?.flush();
|
|
402
|
-
updateState({ state: "listening" });
|
|
403
|
-
sendJson({ type: "cancel" });
|
|
404
|
-
}
|
|
405
|
-
function reset() {
|
|
406
|
-
conn.voiceIO?.flush();
|
|
407
|
-
if (conn.ws && conn.ws.readyState === WS_OPEN) {
|
|
408
|
-
sendJson({ type: "reset" });
|
|
409
|
-
return;
|
|
410
|
-
}
|
|
411
|
-
resetState();
|
|
412
|
-
disconnect();
|
|
413
|
-
connect();
|
|
414
|
-
}
|
|
415
|
-
function disconnect() {
|
|
416
|
-
connectionController?.abort();
|
|
417
|
-
connectionController = null;
|
|
418
|
-
cleanupAudio();
|
|
419
|
-
conn.ws?.close();
|
|
420
|
-
conn.ws = null;
|
|
421
|
-
updateState({
|
|
422
|
-
state: "disconnected",
|
|
423
|
-
running: false
|
|
424
|
-
});
|
|
425
|
-
}
|
|
426
|
-
function start() {
|
|
427
|
-
updateState({
|
|
428
|
-
started: true,
|
|
429
|
-
running: true
|
|
430
|
-
});
|
|
431
|
-
connect();
|
|
432
|
-
}
|
|
433
|
-
function toggle() {
|
|
434
|
-
if (currentSnapshot.running) disconnect();
|
|
435
|
-
else {
|
|
436
|
-
updateState({ running: true });
|
|
437
|
-
connect();
|
|
438
|
-
}
|
|
439
|
-
}
|
|
440
|
-
return {
|
|
441
|
-
getSnapshot,
|
|
442
|
-
subscribe,
|
|
443
|
-
connect,
|
|
444
|
-
cancel,
|
|
445
|
-
resetState,
|
|
446
|
-
reset,
|
|
447
|
-
disconnect,
|
|
448
|
-
start,
|
|
449
|
-
toggle,
|
|
450
|
-
[Symbol.dispose]() {
|
|
451
|
-
disconnect();
|
|
452
|
-
}
|
|
453
|
-
};
|
|
454
|
-
}
|
|
455
|
-
//#endregion
|
|
1
|
+
import { t as createSessionCore } from "./session-core-D3NDaySY.js";
|
|
456
2
|
export { createSessionCore };
|