@alexkroman1/aai-ui 1.16.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/dist/audio.d.ts +43 -5
  2. package/dist/audio.js +39 -20
  3. package/dist/chat-view-DKFhxMAT.js +129 -0
  4. package/dist/client-config.d.ts +7 -6
  5. package/dist/components/chat-view.d.ts +1 -8
  6. package/dist/components/chat-view.js +2 -2
  7. package/dist/components/console-shell.d.ts +37 -0
  8. package/dist/components/controls.js +1 -1
  9. package/dist/components/url-chips.d.ts +1 -1
  10. package/dist/components/workflow-view.d.ts +13 -0
  11. package/dist/{controls-BbZcmnJf.js → controls-4OJoekj6.js} +1 -1
  12. package/dist/default-client/assets/audio-Cgviqo9t.js +1 -0
  13. package/dist/default-client/assets/capture-processor-DLHxAIfT.js +90 -0
  14. package/dist/default-client/assets/index-BgbIWnfG.css +2 -0
  15. package/dist/default-client/assets/index-DbKKR3UE.js +62 -0
  16. package/dist/default-client/assets/playback-processor-bMTdFp-8.js +271 -0
  17. package/dist/default-client/assets/rolldown-runtime-BpQH8Ho1.js +1 -0
  18. package/dist/default-client/assets/types-Bpg3ZIZK.js +64 -0
  19. package/dist/default-client/index.html +5 -2
  20. package/dist/define-client-yoGybEYR.js +710 -0
  21. package/dist/define-client.d.ts +0 -9
  22. package/dist/define-client.js +1 -1
  23. package/dist/index.d.ts +3 -4
  24. package/dist/index.js +6 -5
  25. package/dist/{session-core-B64kau_v.js → session-core-BLhiQ18c.js} +6 -0
  26. package/dist/session-core.js +1 -1
  27. package/dist/sync-mic.d.ts +11 -43
  28. package/dist/types.d.ts +24 -1
  29. package/dist/types.js +32 -2
  30. package/dist/worklets/capture-processor.d.ts +1 -1
  31. package/dist/worklets/capture-processor.js +32 -49
  32. package/dist/worklets/playback-processor.d.ts +1 -1
  33. package/dist/worklets/playback-processor.js +125 -9
  34. package/package.json +2 -2
  35. package/dist/chat-view-gi6FccZq.js +0 -193
  36. package/dist/components/sync-chat-view.d.ts +0 -18
  37. package/dist/components/text-controls.d.ts +0 -14
  38. package/dist/default-client/assets/audio-DNDZgEZp.js +0 -1
  39. package/dist/default-client/assets/capture-processor-UlKEKyIW.js +0 -108
  40. package/dist/default-client/assets/index-BogmeUln.css +0 -2
  41. package/dist/default-client/assets/index-BunwIXSP.js +0 -124
  42. package/dist/default-client/assets/playback-processor-C5HVRVbu.js +0 -156
  43. package/dist/define-client-DdpijqAu.js +0 -906
  44. package/dist/sync-vad.d.ts +0 -54
@@ -1,906 +0,0 @@
1
- import { SessionProvider, ThemeProvider, useTheme } from "./context.js";
2
- import { Button } from "./components/button.js";
3
- import { i as TEXT_MUTED, r as TEXT_FAINT, t as ERROR_COLOR } from "./_colors-DYX7XRTr.js";
4
- import { t as AaiLogo } from "./aai-logo-B8lDmsut.js";
5
- import { a as UrlChip } from "./controls-BbZcmnJf.js";
6
- import { t as Eyebrow } from "./eyebrow-C6ZFuiz6.js";
7
- import { n as ToolConfigContext } from "./tool-call-block-DIxpG8GM.js";
8
- import { ThinkingDots } from "./components/message-list.js";
9
- import { n as stateColor, t as ChatView } from "./chat-view-gi6FccZq.js";
10
- import { SidebarLayout } from "./components/sidebar-layout.js";
11
- import { StartScreen } from "./components/start-screen.js";
12
- import { t as createSessionCore } from "./session-core-B64kau_v.js";
13
- import { CLIENT_CONFIG_PATH, ClientConfigResponseSchema, SyncTurnResponseSchema } from "@alexkroman1/aai/protocol";
14
- import clsx from "clsx";
15
- import { createElement, useEffect, useRef, useState } from "react";
16
- import { jsx, jsxs } from "react/jsx-runtime";
17
- import { DEFAULT_MAX_HISTORY, errorMessage, safeJsonParse } from "@alexkroman1/aai";
18
- import { flushSync } from "react-dom";
19
- import { createRoot } from "react-dom/client";
20
- //#region client-config.ts
21
- /**
22
- * Pre-connection client-config lookup.
23
- *
24
- * `GET client-config` (relative to the agent's base URL — see
25
- * `sdk/client-config.ts` in `@alexkroman1/aai`) tells the default client how
26
- * to talk to the agent before any connection exists, most importantly which
27
- * transport `agent({ transport })` declared. Every failure path — network
28
- * error, 404 from an older server, malformed body — degrades to the
29
- * WebSocket default, so this lookup can never break an existing agent.
30
- */
31
- /** Resolve a relative endpoint path against the agent's base URL. */
32
- function buildAgentUrl(platformUrl, endpointPath) {
33
- return new URL(endpointPath, platformUrl.endsWith("/") ? platformUrl : `${platformUrl}/`);
34
- }
35
- const WEBSOCKET_DEFAULT = { transport: "websocket" };
36
- /** Fetch the agent's client config; any failure yields the WebSocket default. */
37
- async function fetchClientConfig(platformUrl, fetchFn) {
38
- const doFetch = fetchFn ?? ((input, init) => globalThis.fetch(input, init));
39
- try {
40
- const resp = await doFetch(buildAgentUrl(platformUrl, CLIENT_CONFIG_PATH).href);
41
- if (!resp.ok) return WEBSOCKET_DEFAULT;
42
- const parsed = ClientConfigResponseSchema.safeParse(await resp.json());
43
- return parsed.success ? parsed.data : WEBSOCKET_DEFAULT;
44
- } catch {
45
- return WEBSOCKET_DEFAULT;
46
- }
47
- }
48
- //#endregion
49
- //#region sync-vad.ts
50
- const DEFAULT_SPEECH_RMS = .015;
51
- const DEFAULT_MAX_UTTERANCE_MS = 3e4;
52
- function rmsOf(frame) {
53
- if (frame.length === 0) return 0;
54
- let sum = 0;
55
- for (const s of frame) {
56
- const f = s / 32768;
57
- sum += f * f;
58
- }
59
- return Math.sqrt(sum / frame.length);
60
- }
61
- function concat(frames) {
62
- let total = 0;
63
- for (const f of frames) total += f.length;
64
- const out = new Int16Array(total);
65
- let offset = 0;
66
- for (const f of frames) {
67
- out.set(f, offset);
68
- offset += f.length;
69
- }
70
- return out;
71
- }
72
- /** Create an energy-based {@link UtteranceDetector}. */
73
- function createUtteranceDetector(opts) {
74
- const { sampleRate, speechRms = DEFAULT_SPEECH_RMS, minSpeechMs = 150, hangoverMs = 700, prerollMs = 300, maxUtteranceMs = DEFAULT_MAX_UTTERANCE_MS } = opts;
75
- const msOf = (frame) => frame.length / sampleRate * 1e3;
76
- let preroll = [];
77
- let prerollTotalMs = 0;
78
- let utterance = [];
79
- let utteranceMs = 0;
80
- let speaking = false;
81
- /** Consecutive voiced ms while still a candidate (below minSpeechMs). */
82
- let candidateVoicedMs = 0;
83
- /** Consecutive silent ms while speaking (toward hangoverMs). */
84
- let silenceMs = 0;
85
- function toIdle() {
86
- utterance = [];
87
- utteranceMs = 0;
88
- speaking = false;
89
- candidateVoicedMs = 0;
90
- silenceMs = 0;
91
- }
92
- function pushPreroll(frame) {
93
- preroll.push(frame);
94
- prerollTotalMs += msOf(frame);
95
- while (preroll.length > 0 && prerollTotalMs - msOf(preroll[0]) >= prerollMs) prerollTotalMs -= msOf(preroll.shift());
96
- }
97
- function beginCandidate(frame) {
98
- utterance = [...preroll, frame];
99
- utteranceMs = prerollTotalMs + msOf(frame);
100
- preroll = [];
101
- prerollTotalMs = 0;
102
- candidateVoicedMs = msOf(frame);
103
- }
104
- function finalize() {
105
- const out = concat(utterance);
106
- toIdle();
107
- return out;
108
- }
109
- function pushIdle(frame, voiced) {
110
- if (!voiced) {
111
- pushPreroll(frame);
112
- return;
113
- }
114
- beginCandidate(frame);
115
- }
116
- function pushCandidate(frame, voiced) {
117
- if (!voiced) {
118
- for (const f of utterance) pushPreroll(f);
119
- toIdle();
120
- return;
121
- }
122
- utterance.push(frame);
123
- utteranceMs += msOf(frame);
124
- candidateVoicedMs += msOf(frame);
125
- }
126
- function pushSpeaking(frame, voiced) {
127
- utterance.push(frame);
128
- utteranceMs += msOf(frame);
129
- silenceMs = voiced ? 0 : silenceMs + msOf(frame);
130
- return silenceMs >= hangoverMs ? finalize() : null;
131
- }
132
- return {
133
- get speaking() {
134
- return speaking;
135
- },
136
- push(frame) {
137
- if (frame.length === 0) return null;
138
- const voiced = rmsOf(frame) >= speechRms;
139
- let closed = null;
140
- if (speaking) closed = pushSpeaking(frame, voiced);
141
- else if (utterance.length > 0) pushCandidate(frame, voiced);
142
- else pushIdle(frame, voiced);
143
- if (closed) return closed;
144
- if (!speaking && candidateVoicedMs >= minSpeechMs) speaking = true;
145
- if (speaking && utteranceMs >= maxUtteranceMs) return finalize();
146
- return null;
147
- },
148
- flush() {
149
- const out = speaking ? finalize() : null;
150
- this.reset();
151
- return out;
152
- },
153
- reset() {
154
- preroll = [];
155
- prerollTotalMs = 0;
156
- toIdle();
157
- }
158
- };
159
- }
160
- //#endregion
161
- //#region sync-mic.ts
162
- /**
163
- * WebRTC microphone capture for sync mode.
164
- *
165
- * Captures voice through `getUserMedia` with the WebRTC voice-processing
166
- * constraints (echo cancellation, noise suppression, auto gain — the
167
- * processing that makes the energy VAD in `sync-vad.ts` reliable), runs an
168
- * AudioWorklet that batches raw frames to the main thread, feeds them
169
- * through the utterance detector, and hands each completed utterance to
170
- * the sync session as one HTTP turn. No WebSocket anywhere on the path.
171
- *
172
- * The worklet module ships inline as a blob URL (same pattern as the
173
- * WebSocket path's worklets), so sync mode needs no separately-served
174
- * processor file. A blob URL rather than a data URI because the agent
175
- * page's CSP allows `script-src blob:` but not `data:` — a data-URI
176
- * module fails `addModule` with "Unable to load a worklet's module".
177
- */
178
- /** Default capture rate — what the STT providers expect. */
179
- const DEFAULT_SYNC_MIC_SAMPLE_RATE = 16e3;
180
- /** ~128 ms at 16 kHz: few messages per second, fine-enough VAD granularity. */
181
- const CAPTURE_BATCH_SAMPLES = 2048;
182
- /**
183
- * The capture processor: coalesces 128-sample render quanta into
184
- * {@link CAPTURE_BATCH_SAMPLES} batches and posts them (transferred, so no
185
- * per-batch copy). Inlined as source because it must be stringified into a
186
- * blob URL.
187
- *
188
- * `batch` is held as a field rather than re-read from the posted view:
189
- * `postMessage` with a transfer list detaches the buffer, so `out.length` is
190
- * 0 by the time the next buffer is allocated. Allocating a zero-length `buf`
191
- * from it made `n` 0 forever, so `read` stopped advancing and the render
192
- * thread spun inside `process()` posting empty chunks — the mic went
193
- * permanently deaf on its first flush.
194
- *
195
- * Exported for the worklet unit tests (`sync-mic-worklet.test.ts`), which
196
- * evaluate this source directly; it is not part of the package surface.
197
- */
198
- const CAPTURE_PROCESSOR_SRC = `
199
- registerProcessor("aai-sync-capture", class extends AudioWorkletProcessor {
200
- constructor(options) {
201
- super();
202
- const batch = (options && options.processorOptions && options.processorOptions.batchSamples) || ${CAPTURE_BATCH_SAMPLES};
203
- this.batch = batch;
204
- this.buf = new Float32Array(batch);
205
- this.len = 0;
206
- }
207
- process(inputs) {
208
- const ch = inputs[0] && inputs[0][0];
209
- if (!ch) return true;
210
- let read = 0;
211
- while (read < ch.length) {
212
- const n = Math.min(ch.length - read, this.buf.length - this.len);
213
- this.buf.set(ch.subarray(read, read + n), this.len);
214
- this.len += n;
215
- read += n;
216
- if (this.len === this.batch) {
217
- const out = this.buf;
218
- // Size the next buffer from this.batch, never from \`out\`: the
219
- // transfer below detaches out.buffer, so out.length reads 0 here.
220
- this.buf = new Float32Array(this.batch);
221
- this.len = 0;
222
- this.port.postMessage({ event: "chunk", samples: out }, [out.buffer]);
223
- }
224
- }
225
- return true;
226
- }
227
- });
228
- `;
229
- /**
230
- * Blob-URL module for the capture processor (no served asset). Satisfies the
231
- * agent page's `script-src blob:` CSP, which rejects data-URI modules.
232
- */
233
- const CAPTURE_WORKLET_MODULE_URL = URL.createObjectURL(new Blob([CAPTURE_PROCESSOR_SRC], { type: "application/javascript" }));
234
- /** Clamp-and-convert one Float32 capture batch to PCM16. */
235
- function floatToPcm16(samples) {
236
- const pcm = new Int16Array(samples.length);
237
- let i = 0;
238
- for (const sample of samples) {
239
- const s = Math.max(-1, Math.min(1, sample));
240
- pcm[i++] = s < 0 ? s * 32768 : s * 32767;
241
- }
242
- return pcm;
243
- }
244
- /**
245
- * Push-to-talk recorder on the same WebRTC capture pipeline as
246
- * {@link startSyncMicrophone} — `getUserMedia` voice processing feeding the
247
- * capture worklet — minus the VAD: the caller's button is the endpointing.
248
- * Recording runs exactly between `start()` and `stop()`; the mic stays open
249
- * across presses until `close()`.
250
- *
251
- * @public
252
- */
253
- function createPttRecorder(sampleRate = DEFAULT_SYNC_MIC_SAMPLE_RATE) {
254
- let ctx = null;
255
- let stream = null;
256
- let node = null;
257
- let chunks = [];
258
- let recording = false;
259
- async function ensureOpen() {
260
- if (ctx) return;
261
- const streamPromise = navigator.mediaDevices.getUserMedia({ audio: {
262
- echoCancellation: true,
263
- noiseSuppression: true,
264
- autoGainControl: true
265
- } });
266
- const audioCtx = new AudioContext({
267
- sampleRate,
268
- latencyHint: "interactive"
269
- });
270
- try {
271
- const [media] = await Promise.all([
272
- streamPromise,
273
- audioCtx.resume(),
274
- audioCtx.audioWorklet.addModule(CAPTURE_WORKLET_MODULE_URL)
275
- ]);
276
- stream = media;
277
- } catch (err) {
278
- streamPromise.then((s) => {
279
- for (const t of s.getTracks()) t.stop();
280
- }).catch(() => {});
281
- await audioCtx.close().catch(() => {});
282
- throw err;
283
- }
284
- const workletNode = new AudioWorkletNode(audioCtx, "aai-sync-capture", {
285
- channelCount: 1,
286
- channelCountMode: "explicit",
287
- processorOptions: { batchSamples: CAPTURE_BATCH_SAMPLES }
288
- });
289
- workletNode.port.onmessage = (e) => {
290
- const data = e.data;
291
- if (recording && data.event === "chunk" && data.samples) chunks.push(data.samples);
292
- };
293
- audioCtx.createMediaStreamSource(stream).connect(workletNode);
294
- ctx = audioCtx;
295
- node = workletNode;
296
- }
297
- return {
298
- async start() {
299
- await ensureOpen();
300
- chunks = [];
301
- recording = true;
302
- },
303
- async stop() {
304
- await new Promise((r) => setTimeout(r, 150));
305
- recording = false;
306
- const total = chunks.reduce((n, c) => n + c.length, 0);
307
- const all = new Float32Array(total);
308
- let offset = 0;
309
- for (const c of chunks) {
310
- all.set(c, offset);
311
- offset += c.length;
312
- }
313
- chunks = [];
314
- return floatToPcm16(all);
315
- },
316
- async close() {
317
- recording = false;
318
- node?.disconnect();
319
- if (stream) for (const t of stream.getTracks()) t.stop();
320
- await ctx?.close().catch(() => {});
321
- ctx = null;
322
- node = null;
323
- stream = null;
324
- }
325
- };
326
- }
327
- /**
328
- * Open the microphone and stream endpointed utterances into a sync session.
329
- *
330
- * @throws If microphone access is denied or worklet registration fails.
331
- */
332
- async function startSyncMicrophone(opts) {
333
- const sampleRate = opts.sampleRate ?? 16e3;
334
- const detector = createUtteranceDetector({
335
- sampleRate,
336
- ...opts.vad
337
- });
338
- const fail = (err) => {
339
- opts.onError?.(err instanceof Error ? err : new Error(errorMessage(err)));
340
- };
341
- const streamPromise = navigator.mediaDevices.getUserMedia({ audio: {
342
- echoCancellation: true,
343
- noiseSuppression: true,
344
- autoGainControl: true
345
- } });
346
- const ctx = new AudioContext({
347
- sampleRate,
348
- latencyHint: "interactive"
349
- });
350
- let stream;
351
- try {
352
- [stream] = await Promise.all([
353
- streamPromise,
354
- ctx.resume(),
355
- ctx.audioWorklet.addModule(CAPTURE_WORKLET_MODULE_URL)
356
- ]);
357
- } catch (err) {
358
- streamPromise.then((s) => {
359
- for (const t of s.getTracks()) t.stop();
360
- }).catch(() => {});
361
- await ctx.close().catch(() => {});
362
- throw err;
363
- }
364
- const mic = ctx.createMediaStreamSource(stream);
365
- const node = new AudioWorkletNode(ctx, "aai-sync-capture", {
366
- channelCount: 1,
367
- channelCountMode: "explicit",
368
- processorOptions: { batchSamples: CAPTURE_BATCH_SAMPLES }
369
- });
370
- mic.connect(node);
371
- let stopped = false;
372
- let wasSpeaking = false;
373
- function dispatch(utterance) {
374
- if (!utterance) return;
375
- opts.onSpeechEnd?.();
376
- opts.session.sendPcm16(utterance, sampleRate).catch(fail);
377
- }
378
- function handleFrame(samples) {
379
- if (stopped) return;
380
- dispatch(detector.push(floatToPcm16(samples)));
381
- if (detector.speaking && !wasSpeaking) opts.onSpeechStart?.();
382
- wasSpeaking = detector.speaking;
383
- }
384
- node.port.onmessage = (e) => {
385
- const data = e.data;
386
- if (data.event === "chunk" && data.samples) handleFrame(data.samples);
387
- };
388
- node.onprocessorerror = () => fail(/* @__PURE__ */ new Error("Sync capture worklet crashed"));
389
- return {
390
- get speaking() {
391
- return detector.speaking;
392
- },
393
- async stop() {
394
- if (stopped) return;
395
- stopped = true;
396
- dispatch(detector.flush());
397
- mic.disconnect();
398
- node.disconnect();
399
- for (const t of stream.getTracks()) t.stop();
400
- await ctx.close().catch(() => {});
401
- }
402
- };
403
- }
404
- //#endregion
405
- //#region sync-session.ts
406
- /**
407
- * Sync-mode browser session — HTTP turns, no WebSocket.
408
- *
409
- * The client half of the server's `POST /sync` endpoint (see
410
- * `host/sync-turn.ts` in `@alexkroman1/aai`): each turn is one request
411
- * carrying committed text or one endpointed utterance of PCM16 audio plus
412
- * the conversation history, answered with the transcript, the reply text,
413
- * and (when the agent's TTS provider supports one-shot synthesis) the
414
- * spoken reply. The server holds no session state — this object owns the
415
- * history and replays it every turn.
416
- *
417
- * Microphone capture and utterance endpointing live in `sync-mic.ts` /
418
- * `sync-vad.ts`; this module is transport only, so it also runs in
419
- * non-browser clients that bring their own audio.
420
- */
421
- /** Base64-encode PCM16 samples (chunked — `btoa` takes a binary string). */
422
- function pcm16ToBase64(pcm) {
423
- const bytes = new Uint8Array(pcm.buffer, pcm.byteOffset, pcm.byteLength);
424
- let binary = "";
425
- const CHUNK = 32768;
426
- for (let i = 0; i < bytes.length; i += CHUNK) binary += String.fromCharCode(...bytes.subarray(i, i + CHUNK));
427
- return btoa(binary);
428
- }
429
- /** Decode base64 PCM16LE (the sync response's `audio` field) into samples. */
430
- function base64ToPcm16(base64) {
431
- const binary = atob(base64);
432
- const samples = binary.length >> 1;
433
- const bytes = new Uint8Array(samples * 2);
434
- for (let i = 0; i < bytes.length; i++) bytes[i] = binary.charCodeAt(i);
435
- return new Int16Array(bytes.buffer, 0, samples);
436
- }
437
- /** Create a {@link SyncSession} against a sync-mode agent server. */
438
- function createSyncSession(opts) {
439
- const fetchFn = opts.fetch ?? ((input, init) => globalThis.fetch(input, init));
440
- const history = [];
441
- let queue = Promise.resolve();
442
- async function parseTurn(resp) {
443
- if (!resp.ok) {
444
- const detail = safeJsonParse(await resp.text().catch(() => "")) ?? {};
445
- throw new Error(`Sync turn failed: HTTP ${resp.status}${detail.error ? ` (${detail.error})` : ""}`);
446
- }
447
- const parsed = SyncTurnResponseSchema.safeParse(await resp.json());
448
- if (!parsed.success) throw new Error("Sync turn failed: malformed server response");
449
- return parsed.data;
450
- }
451
- async function runTurn(body) {
452
- try {
453
- const turn = await parseTurn(await fetchFn(opts.url, {
454
- method: "POST",
455
- headers: { "Content-Type": "application/json" },
456
- body: JSON.stringify({
457
- ...body,
458
- history: [...history]
459
- })
460
- }));
461
- history.push({
462
- role: "user",
463
- content: turn.transcript
464
- });
465
- if (turn.reply.length > 0) history.push({
466
- role: "assistant",
467
- content: turn.reply
468
- });
469
- if (history.length > DEFAULT_MAX_HISTORY) history.splice(0, history.length - DEFAULT_MAX_HISTORY);
470
- const result = {
471
- ...turn,
472
- pcm: turn.audio !== void 0 ? base64ToPcm16(turn.audio) : null
473
- };
474
- opts.onTurn?.(result);
475
- return result;
476
- } catch (err) {
477
- const wrapped = err instanceof Error ? err : new Error(errorMessage(err));
478
- opts.onError?.(wrapped);
479
- throw wrapped;
480
- }
481
- }
482
- function enqueue(body) {
483
- const next = queue.then(() => runTurn(body), () => runTurn(body));
484
- queue = next.catch(() => void 0);
485
- return next;
486
- }
487
- return {
488
- get history() {
489
- return history;
490
- },
491
- sendText(text) {
492
- return enqueue({ text });
493
- },
494
- sendPcm16(pcm, sampleRate) {
495
- return enqueue({
496
- audio: pcm16ToBase64(pcm),
497
- sampleRate
498
- });
499
- },
500
- reset() {
501
- history.length = 0;
502
- }
503
- };
504
- }
505
- //#endregion
506
- //#region components/sync-chat-view.tsx
507
- /** @jsxImportSource react */
508
- /**
509
- * Default shell for sync-transport agents (`agent({ transport: "sync" })`).
510
- *
511
- * A hands-free voice agent: one toggle starts the conversation, and from
512
- * then on the mic stays open — `startSyncMicrophone` runs the WebRTC
513
- * voice-processing capture through the energy VAD (`sync-vad.ts`), which
514
- * endpoints each utterance automatically and sends it as one `POST /sync`
515
- * request through `createSyncSession`. No button per turn: speak, pause,
516
- * and the reply comes back and plays. The view shows what was heard, the
517
- * agent's reply, and — via the endpoint chip next to the toggle — exactly
518
- * where each utterance is being sent.
519
- *
520
- * Visually it is the same "voice agent console" as the WebSocket
521
- * {@link ChatView}: header with logo + live-status eyebrow, the output on a
522
- * raised card, controls beneath — built from the same shared pieces
523
- * ({@link ThinkingDots}, {@link Eyebrow}, {@link Button}, {@link UrlChip})
524
- * so the two transports are indistinguishable at a glance.
525
- *
526
- * Rendered by `client()` when the agent's `GET /client-config` declares
527
- * `transport: "sync"`; also exported for custom clients that want the stock
528
- * sync UI with their own chrome around it.
529
- */
530
- /**
531
- * Play one reply's PCM16 through a shared AudioContext. `onPlaying`
532
- * tracks playback so the eyebrow can show "speaking" while the reply is
533
- * audible.
534
- */
535
- function playReply(ctxRef, turn, onPlaying) {
536
- if (!(turn.pcm && turn.sampleRate)) return;
537
- ctxRef.current ??= new AudioContext();
538
- const ctx = ctxRef.current;
539
- const buffer = ctx.createBuffer(1, turn.pcm.length, turn.sampleRate);
540
- const channel = buffer.getChannelData(0);
541
- turn.pcm.forEach((sample, i) => {
542
- channel[i] = sample / 32768;
543
- });
544
- const source = ctx.createBufferSource();
545
- source.buffer = buffer;
546
- source.connect(ctx.destination);
547
- source.onended = () => onPlaying(false);
548
- onPlaying(true);
549
- source.start();
550
- }
551
- /**
552
- * Map the hands-free conversation lifecycle onto the same states the
553
- * WebSocket eyebrow shows, so the status chip reads identically across
554
- * transports: an endpointed utterance in flight is "thinking", an audible
555
- * reply is "speaking", a live mic is "listening".
556
- */
557
- function syncState(opts) {
558
- if (opts.pending > 0) return "thinking";
559
- if (opts.agentSpeaking) return "speaking";
560
- if (opts.live) return "listening";
561
- if (opts.error) return "error";
562
- return "ready";
563
- }
564
- /**
565
- * Sync-transport view: a hands-free VAD-endpointed conversation, transcript
566
- * + reply output, and the endpoint each utterance is POSTed to — one HTTP
567
- * request per turn.
568
- *
569
- * @public
570
- */
571
- function SyncChatView({ syncUrl, title, greeting }) {
572
- const theme = useTheme();
573
- const [exchanges, setExchanges] = useState([]);
574
- const [live, setLive] = useState(false);
575
- const [userSpeaking, setUserSpeaking] = useState(false);
576
- const [pending, setPending] = useState(0);
577
- const [agentSpeaking, setAgentSpeaking] = useState(false);
578
- const [error, setError] = useState(null);
579
- const playbackCtx = useRef(null);
580
- const mic = useRef(null);
581
- const toggling = useRef(false);
582
- const disposed = useRef(false);
583
- const anchorRef = useRef(null);
584
- const sessionRef = useRef(createSyncSession({
585
- url: syncUrl,
586
- onTurn: (turn) => {
587
- if (disposed.current) return;
588
- setExchanges((prev) => [...prev, {
589
- id: prev.length,
590
- heard: turn.transcript,
591
- reply: turn.reply
592
- }]);
593
- setPending((n) => Math.max(0, n - 1));
594
- setError(turn.ttsError ? `TTS unavailable: ${turn.ttsError}` : null);
595
- playReply(playbackCtx, turn, setAgentSpeaking);
596
- },
597
- onError: (err) => {
598
- if (disposed.current) return;
599
- setPending((n) => Math.max(0, n - 1));
600
- setError(err.message);
601
- }
602
- }));
603
- useEffect(() => () => {
604
- disposed.current = true;
605
- mic.current?.stop();
606
- mic.current = null;
607
- playbackCtx.current?.close();
608
- playbackCtx.current = null;
609
- }, []);
610
- const contentCount = exchanges.length + (pending > 0 ? 1 : 0);
611
- useEffect(() => {
612
- if (contentCount === 0) return;
613
- anchorRef.current?.scrollIntoView({
614
- behavior: "smooth",
615
- block: "end"
616
- });
617
- }, [contentCount]);
618
- async function toggleConversation() {
619
- if (toggling.current) return;
620
- toggling.current = true;
621
- try {
622
- if (mic.current) {
623
- const handle = mic.current;
624
- mic.current = null;
625
- setLive(false);
626
- setUserSpeaking(false);
627
- await handle.stop();
628
- return;
629
- }
630
- const handle = await startSyncMicrophone({
631
- session: sessionRef.current,
632
- onSpeechStart: () => setUserSpeaking(true),
633
- onSpeechEnd: () => {
634
- setUserSpeaking(false);
635
- setPending((n) => n + 1);
636
- },
637
- onError: (err) => setError(err.message)
638
- });
639
- if (disposed.current) {
640
- await handle.stop();
641
- return;
642
- }
643
- mic.current = handle;
644
- setLive(true);
645
- setError(null);
646
- } catch (err) {
647
- setError(err instanceof Error ? err.message : String(err));
648
- } finally {
649
- toggling.current = false;
650
- }
651
- }
652
- const state = syncState({
653
- error,
654
- live,
655
- pending,
656
- agentSpeaking
657
- });
658
- const buttonLabel = live ? "End conversation" : "Start conversation";
659
- const pulsing = userSpeaking || agentSpeaking;
660
- return /* @__PURE__ */ jsxs("div", {
661
- className: "flex flex-col h-screen w-full max-w-190 mx-auto box-border px-6 py-8 gap-5 font-aai text-sm",
662
- style: {
663
- background: theme.bg,
664
- color: theme.text
665
- },
666
- children: [
667
- /* @__PURE__ */ jsxs("div", {
668
- className: "flex items-center justify-between shrink-0",
669
- children: [/* @__PURE__ */ jsxs("div", {
670
- className: "flex items-center gap-3 min-w-0",
671
- children: [/* @__PURE__ */ jsx(AaiLogo, { size: 22 }), /* @__PURE__ */ jsx("span", {
672
- className: "font-aai-serif text-[22px] leading-[1.2] font-normal truncate",
673
- style: { color: theme.text },
674
- children: title ?? "Voice Agent"
675
- })]
676
- }), /* @__PURE__ */ jsxs(Eyebrow, {
677
- className: "shrink-0",
678
- "data-state": state,
679
- children: [/* @__PURE__ */ jsx("span", {
680
- className: "w-[7px] h-[7px] rounded-full",
681
- style: {
682
- background: stateColor(state, theme.primary),
683
- animation: pulsing ? "aai-pulse 1.6s ease-in-out infinite" : "none"
684
- }
685
- }), state]
686
- })]
687
- }),
688
- error && /* @__PURE__ */ jsx("div", {
689
- className: "px-3.5 py-2.5 rounded-aai border text-[13px] leading-[130%] shrink-0",
690
- style: {
691
- borderColor: "rgba(179,38,30,0.35)",
692
- background: "rgba(179,38,30,0.06)",
693
- color: "#B3261E"
694
- },
695
- children: error
696
- }),
697
- /* @__PURE__ */ jsx("div", {
698
- className: "flex flex-col flex-1 min-h-0 border rounded-lg overflow-hidden",
699
- style: {
700
- background: theme.surface,
701
- borderColor: theme.border,
702
- boxShadow: "0 1px 3px 0 rgb(20 18 12 / 0.06)"
703
- },
704
- children: /* @__PURE__ */ jsx("div", {
705
- role: "log",
706
- className: "flex-1 overflow-y-auto [scrollbar-width:none]",
707
- style: { background: theme.surface },
708
- children: /* @__PURE__ */ jsxs("div", {
709
- className: "flex flex-col gap-5 p-7",
710
- children: [
711
- greeting !== void 0 && greeting.length > 0 && /* @__PURE__ */ jsx("p", {
712
- className: "text-[15px] leading-[23px]",
713
- style: { color: theme.text },
714
- children: greeting
715
- }),
716
- exchanges.length === 0 && /* @__PURE__ */ jsx("p", {
717
- className: "text-sm",
718
- style: { color: "#57534B" },
719
- children: "Start the conversation and just talk — each pause endpoints an utterance, which goes out as one HTTP request to the endpoint below."
720
- }),
721
- exchanges.map((ex) => /* @__PURE__ */ jsxs("div", {
722
- className: "flex flex-col gap-1.5",
723
- children: [
724
- /* @__PURE__ */ jsx("span", {
725
- className: "text-[10px] font-medium tracking-[1.2px] uppercase leading-none",
726
- style: { color: TEXT_FAINT },
727
- children: "Heard"
728
- }),
729
- /* @__PURE__ */ jsx("p", {
730
- className: "text-[15px] leading-[22px]",
731
- style: { color: TEXT_MUTED },
732
- children: ex.heard
733
- }),
734
- /* @__PURE__ */ jsx("span", {
735
- className: "text-[10px] font-medium tracking-[1.2px] uppercase leading-none mt-1.5",
736
- style: { color: TEXT_FAINT },
737
- children: "Agent"
738
- }),
739
- /* @__PURE__ */ jsx("p", {
740
- className: "whitespace-pre-wrap wrap-break-word text-[15px] font-normal leading-[23px]",
741
- style: { color: theme.text },
742
- children: ex.reply
743
- })
744
- ]
745
- }, ex.id)),
746
- pending > 0 && /* @__PURE__ */ jsx("div", {
747
- "data-testid": "thinking",
748
- children: /* @__PURE__ */ jsx(ThinkingDots, {})
749
- }),
750
- /* @__PURE__ */ jsx("div", { ref: anchorRef })
751
- ]
752
- })
753
- })
754
- }),
755
- /* @__PURE__ */ jsxs("div", {
756
- className: "flex items-center gap-2 shrink-0",
757
- children: [/* @__PURE__ */ jsxs(Button, {
758
- size: "lg",
759
- variant: live ? "default" : "secondary",
760
- className: "select-none",
761
- style: live ? {
762
- background: ERROR_COLOR,
763
- borderColor: "transparent"
764
- } : void 0,
765
- onClick: () => void toggleConversation(),
766
- "aria-pressed": live,
767
- title: "Start or end the conversation",
768
- children: [/* @__PURE__ */ jsx("span", {
769
- className: clsx("w-2 h-2 rounded-full mr-2", pulsing && "animate-pulse"),
770
- style: { background: live ? "#fff" : ERROR_COLOR }
771
- }), buttonLabel]
772
- }), /* @__PURE__ */ jsx(UrlChip, {
773
- label: "Sync",
774
- url: syncUrl,
775
- hint: "Each utterance is one POST to this endpoint",
776
- testId: "sync-url-chip",
777
- className: "ml-auto min-w-0 max-w-[55%]"
778
- })]
779
- })
780
- ]
781
- });
782
- }
783
- //#endregion
784
- //#region define-client.tsx
785
- /** @jsxImportSource react */
786
- function resolveContainer(target = "#app") {
787
- if (typeof target !== "string") return target;
788
- const el = document.querySelector(target);
789
- if (!el) throw new Error(`Element not found: ${target}`);
790
- return el;
791
- }
792
- /**
793
- * Default shell rendered in config tier.
794
- * Wraps StartScreen → (SidebarLayout →) ChatView.
795
- */
796
- function DefaultShell({ name, Sidebar, sidebarWidth }) {
797
- const chat = /* @__PURE__ */ jsx(ChatView, { title: name });
798
- return /* @__PURE__ */ jsx(StartScreen, {
799
- title: name,
800
- children: Sidebar ? /* @__PURE__ */ jsx(SidebarLayout, {
801
- sidebar: /* @__PURE__ */ jsx(Sidebar, {}),
802
- sidebarWidth,
803
- children: chat
804
- }) : chat
805
- });
806
- }
807
- /**
808
- * Config-tier root: resolves the transport (explicit option, else the
809
- * server's `GET client-config`) and renders the WebSocket shell or the
810
- * sync shell.
811
- *
812
- * The WebSocket shell renders immediately — optimistically — while the
813
- * lookup is in flight, then swaps if the agent declared `"sync"`. That
814
- * keeps mounting synchronous, keeps agents on servers without the endpoint
815
- * (every lookup failure resolves to `"websocket"`) exactly as before, and
816
- * the worst-case race — Start clicked on a sync agent before the lookup
817
- * lands — still yields a working session: sync agents answer WebSocket
818
- * sessions too.
819
- */
820
- function DefaultRoot({ platformUrl, transport, name, Sidebar, sidebarWidth }) {
821
- const [resolved, setResolved] = useState(transport !== void 0 ? { transport } : null);
822
- useEffect(() => {
823
- if (transport !== void 0) return;
824
- let cancelled = false;
825
- fetchClientConfig(platformUrl).then((cfg) => {
826
- if (!cancelled) setResolved(cfg);
827
- });
828
- return () => {
829
- cancelled = true;
830
- };
831
- }, [platformUrl, transport]);
832
- if (resolved?.transport === "sync") return /* @__PURE__ */ jsx(SyncChatView, {
833
- syncUrl: buildAgentUrl(platformUrl, "sync").href,
834
- title: name ?? resolved.name,
835
- greeting: resolved.greeting
836
- });
837
- return /* @__PURE__ */ jsx(DefaultShell, {
838
- name,
839
- Sidebar,
840
- sidebarWidth
841
- });
842
- }
843
- /**
844
- * Define and mount a client UI for a voice agent.
845
- *
846
- * **Tier 1 (config-only):** Pass options without `component` to get the
847
- * default shell (StartScreen + ChatView, optional sidebar).
848
- *
849
- * **Tier 2 (custom component):** Pass `component` to render a fully custom
850
- * root component inside the providers.
851
- *
852
- * @example Tier 1
853
- * ```tsx
854
- * client({
855
- * name: "Pizza Ordering",
856
- * theme: { bg: "#1a1a1a", primary: "#e55" },
857
- * sidebar: OrderPanel,
858
- * tools: { add_pizza: { icon: "🍕", label: "Adding pizza" } },
859
- * });
860
- * ```
861
- *
862
- * @example Tier 2
863
- * ```tsx
864
- * client({ component: MyCustomApp });
865
- * ```
866
- *
867
- * @returns A {@link ClientHandle} for cleanup.
868
- * @throws If the target element is not found in the DOM.
869
- *
870
- * @public
871
- */
872
- function client(config) {
873
- const container = resolveContainer(config.target);
874
- const platformUrl = config.platformUrl ?? globalThis.location.origin + globalThis.location.pathname;
875
- const session = createSessionCore({
876
- platformUrl,
877
- onSessionId: config.onSessionId,
878
- resumeSessionId: config.resumeSessionId,
879
- WebSocket: config.WebSocket
880
- });
881
- const rootNode = config.component ? createElement(config.component) : createElement(DefaultRoot, {
882
- platformUrl,
883
- transport: config.transport,
884
- name: config.name,
885
- Sidebar: config.sidebar,
886
- sidebarWidth: config.sidebarWidth
887
- });
888
- const toolConfig = config.tools ?? {};
889
- const root = createRoot(container);
890
- flushSync(() => {
891
- root.render(createElement(ToolConfigContext.Provider, { value: toolConfig }, createElement(ThemeProvider, { value: config.theme }, createElement(SessionProvider, { value: session }, rootNode))));
892
- });
893
- const handle = {
894
- session,
895
- dispose() {
896
- root.unmount();
897
- session[Symbol.dispose]();
898
- },
899
- [Symbol.dispose]() {
900
- handle.dispose();
901
- }
902
- };
903
- return handle;
904
- }
905
- //#endregion
906
- export { pcm16ToBase64 as a, createPttRecorder as c, createUtteranceDetector as d, buildAgentUrl as f, createSyncSession as i, floatToPcm16 as l, SyncChatView as n, CAPTURE_WORKLET_MODULE_URL as o, fetchClientConfig as p, base64ToPcm16 as r, DEFAULT_SYNC_MIC_SAMPLE_RATE as s, client as t, startSyncMicrophone as u };