@alexkroman1/aai-ui 1.16.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/_module-url-C_4gRVL0.js +15 -0
  2. package/dist/audio.d.ts +85 -11
  3. package/dist/audio.js +112 -83
  4. package/dist/chat-view-C1oJxsWz.js +129 -0
  5. package/dist/client-config.d.ts +6 -6
  6. package/dist/components/chat-view.d.ts +1 -8
  7. package/dist/components/chat-view.js +2 -2
  8. package/dist/components/console-shell.d.ts +37 -0
  9. package/dist/components/controls.js +1 -1
  10. package/dist/components/url-chips.d.ts +1 -4
  11. package/dist/context.d.ts +1 -1
  12. package/dist/context.js +1 -4
  13. package/dist/{controls-BbZcmnJf.js → controls-DV368uhb.js} +2 -5
  14. package/dist/default-client/assets/_module-url-BX0RuRU2.js +1 -0
  15. package/dist/default-client/assets/audio-BGHiiDY_.js +1 -0
  16. package/dist/default-client/assets/capture-processor-BLPsqnGl.js +90 -0
  17. package/dist/default-client/assets/{index-BunwIXSP.js → index-DXf7-M0Q.js} +156 -36
  18. package/dist/default-client/assets/index-Dv-Q5VRL.css +2 -0
  19. package/dist/default-client/assets/playback-processor-B_T_1KQP.js +278 -0
  20. package/dist/default-client/index.html +2 -2
  21. package/dist/define-client-Cyf1jqAl.js +150 -0
  22. package/dist/define-client.d.ts +0 -9
  23. package/dist/define-client.js +1 -1
  24. package/dist/index.d.ts +2 -5
  25. package/dist/index.js +7 -5
  26. package/dist/{session-core-B64kau_v.js → session-core-BM3WHeoY.js} +36 -128
  27. package/dist/session-core-messages.d.ts +0 -4
  28. package/dist/session-core-types.d.ts +1 -27
  29. package/dist/session-core.js +1 -1
  30. package/dist/types.d.ts +24 -1
  31. package/dist/types.js +32 -2
  32. package/dist/worklets/_module-url.d.ts +10 -0
  33. package/dist/worklets/capture-processor.d.ts +3 -3
  34. package/dist/worklets/capture-processor.js +35 -52
  35. package/dist/worklets/playback-processor.d.ts +3 -3
  36. package/dist/worklets/playback-processor.js +149 -26
  37. package/package.json +2 -2
  38. package/dist/chat-view-gi6FccZq.js +0 -193
  39. package/dist/components/sync-chat-view.d.ts +0 -18
  40. package/dist/components/text-controls.d.ts +0 -14
  41. package/dist/default-client/assets/audio-DNDZgEZp.js +0 -1
  42. package/dist/default-client/assets/capture-processor-UlKEKyIW.js +0 -108
  43. package/dist/default-client/assets/index-BogmeUln.css +0 -2
  44. package/dist/default-client/assets/playback-processor-C5HVRVbu.js +0 -156
  45. package/dist/define-client-DdpijqAu.js +0 -906
  46. package/dist/session-core-upload.d.ts +0 -16
  47. package/dist/sync-mic.d.ts +0 -93
  48. package/dist/sync-session.d.ts +0 -49
  49. package/dist/sync-vad.d.ts +0 -54
@@ -1,16 +0,0 @@
1
- import type { ClientMessage } from "@alexkroman1/aai/protocol";
2
- import type { ConnState, SessionSnapshot } from "./session-core-types.ts";
3
- /** Handle to the upload pipeline, owned by one session core. */
4
- export type UploadSender = {
5
- /** See {@link SessionCore.sendAudioFile}. */
6
- sendAudioFile(file: Blob): Promise<void>;
7
- /** True while an upload is decoding or streaming — the mic must stay off. */
8
- inFlight(): boolean;
9
- /** Invalidate any in-flight upload (reset / close / reconnect). */
10
- discard(): void;
11
- };
12
- export declare function createUploadSender(deps: {
13
- conn: ConnState;
14
- getSnapshot: () => SessionSnapshot;
15
- sendJson: (msg: ClientMessage) => void;
16
- }): UploadSender;
@@ -1,93 +0,0 @@
1
- /**
2
- * WebRTC microphone capture for sync mode.
3
- *
4
- * Captures voice through `getUserMedia` with the WebRTC voice-processing
5
- * constraints (echo cancellation, noise suppression, auto gain — the
6
- * processing that makes the energy VAD in `sync-vad.ts` reliable), runs an
7
- * AudioWorklet that batches raw frames to the main thread, feeds them
8
- * through the utterance detector, and hands each completed utterance to
9
- * the sync session as one HTTP turn. No WebSocket anywhere on the path.
10
- *
11
- * The worklet module ships inline as a blob URL (same pattern as the
12
- * WebSocket path's worklets), so sync mode needs no separately-served
13
- * processor file. A blob URL rather than a data URI because the agent
14
- * page's CSP allows `script-src blob:` but not `data:` — a data-URI
15
- * module fails `addModule` with "Unable to load a worklet's module".
16
- */
17
- import type { SyncSession } from "./sync-session.ts";
18
- import { type UtteranceDetectorOptions } from "./sync-vad.ts";
19
- /** Default capture rate — what the STT providers expect. */
20
- export declare const DEFAULT_SYNC_MIC_SAMPLE_RATE = 16000;
21
- /** ~128 ms at 16 kHz: few messages per second, fine-enough VAD granularity. */
22
- export declare const CAPTURE_BATCH_SAMPLES = 2048;
23
- /**
24
- * The capture processor: coalesces 128-sample render quanta into
25
- * {@link CAPTURE_BATCH_SAMPLES} batches and posts them (transferred, so no
26
- * per-batch copy). Inlined as source because it must be stringified into a
27
- * blob URL.
28
- *
29
- * `batch` is held as a field rather than re-read from the posted view:
30
- * `postMessage` with a transfer list detaches the buffer, so `out.length` is
31
- * 0 by the time the next buffer is allocated. Allocating a zero-length `buf`
32
- * from it made `n` 0 forever, so `read` stopped advancing and the render
33
- * thread spun inside `process()` posting empty chunks — the mic went
34
- * permanently deaf on its first flush.
35
- *
36
- * Exported for the worklet unit tests (`sync-mic-worklet.test.ts`), which
37
- * evaluate this source directly; it is not part of the package surface.
38
- */
39
- export declare const CAPTURE_PROCESSOR_SRC = "\nregisterProcessor(\"aai-sync-capture\", class extends AudioWorkletProcessor {\n constructor(options) {\n super();\n const batch = (options && options.processorOptions && options.processorOptions.batchSamples) || 2048;\n this.batch = batch;\n this.buf = new Float32Array(batch);\n this.len = 0;\n }\n process(inputs) {\n const ch = inputs[0] && inputs[0][0];\n if (!ch) return true;\n let read = 0;\n while (read < ch.length) {\n const n = Math.min(ch.length - read, this.buf.length - this.len);\n this.buf.set(ch.subarray(read, read + n), this.len);\n this.len += n;\n read += n;\n if (this.len === this.batch) {\n const out = this.buf;\n // Size the next buffer from this.batch, never from `out`: the\n // transfer below detaches out.buffer, so out.length reads 0 here.\n this.buf = new Float32Array(this.batch);\n this.len = 0;\n this.port.postMessage({ event: \"chunk\", samples: out }, [out.buffer]);\n }\n }\n return true;\n }\n});\n";
40
- /**
41
- * Blob-URL module for the capture processor (no served asset). Satisfies the
42
- * agent page's `script-src blob:` CSP, which rejects data-URI modules.
43
- */
44
- export declare const CAPTURE_WORKLET_MODULE_URL: string;
45
- /** Clamp-and-convert one Float32 capture batch to PCM16. */
46
- export declare function floatToPcm16(samples: Float32Array): Int16Array;
47
- /** Configuration for {@link startSyncMicrophone}. */
48
- export type SyncMicrophoneOptions = {
49
- /** The session each completed utterance is sent through. */
50
- session: Pick<SyncSession, "sendPcm16">;
51
- /** Capture/STT sample rate. Defaults to {@link DEFAULT_SYNC_MIC_SAMPLE_RATE}. */
52
- sampleRate?: number | undefined;
53
- /** VAD tuning overrides (see {@link UtteranceDetectorOptions}). */
54
- vad?: Omit<UtteranceDetectorOptions, "sampleRate"> | undefined;
55
- /** Speech onset confirmed — a turn will follow once the user pauses. */
56
- onSpeechStart?: (() => void) | undefined;
57
- /** An utterance was endpointed and its turn dispatched. */
58
- onSpeechEnd?: (() => void) | undefined;
59
- /** Capture or turn failure (the mic keeps running unless stopped). */
60
- onError?: ((err: Error) => void) | undefined;
61
- };
62
- /** Live microphone handle returned by {@link startSyncMicrophone}. */
63
- export type SyncMicrophone = {
64
- /** True while the detector is inside an utterance. */
65
- readonly speaking: boolean;
66
- /** Release the mic, the AudioContext, and flush a trailing utterance. */
67
- stop(): Promise<void>;
68
- };
69
- /** Hold-to-record handle returned by {@link createPttRecorder}. */
70
- export type PttRecorder = {
71
- /** Open the mic (first call) and start collecting frames. */
72
- start(): Promise<void>;
73
- /** Stop collecting and return everything recorded since `start()` as PCM16. */
74
- stop(): Promise<Int16Array>;
75
- /** Release the mic and the AudioContext. */
76
- close(): Promise<void>;
77
- };
78
- /**
79
- * Push-to-talk recorder on the same WebRTC capture pipeline as
80
- * {@link startSyncMicrophone} — `getUserMedia` voice processing feeding the
81
- * capture worklet — minus the VAD: the caller's button is the endpointing.
82
- * Recording runs exactly between `start()` and `stop()`; the mic stays open
83
- * across presses until `close()`.
84
- *
85
- * @public
86
- */
87
- export declare function createPttRecorder(sampleRate?: number): PttRecorder;
88
- /**
89
- * Open the microphone and stream endpointed utterances into a sync session.
90
- *
91
- * @throws If microphone access is denied or worklet registration fails.
92
- */
93
- export declare function startSyncMicrophone(opts: SyncMicrophoneOptions): Promise<SyncMicrophone>;
@@ -1,49 +0,0 @@
1
- /**
2
- * Sync-mode browser session — HTTP turns, no WebSocket.
3
- *
4
- * The client half of the server's `POST /sync` endpoint (see
5
- * `host/sync-turn.ts` in `@alexkroman1/aai`): each turn is one request
6
- * carrying committed text or one endpointed utterance of PCM16 audio plus
7
- * the conversation history, answered with the transcript, the reply text,
8
- * and (when the agent's TTS provider supports one-shot synthesis) the
9
- * spoken reply. The server holds no session state — this object owns the
10
- * history and replays it every turn.
11
- *
12
- * Microphone capture and utterance endpointing live in `sync-mic.ts` /
13
- * `sync-vad.ts`; this module is transport only, so it also runs in
14
- * non-browser clients that bring their own audio.
15
- */
16
- import { type SyncHistoryMessage, type SyncTurnResponse } from "@alexkroman1/aai/protocol";
17
- /** Base64-encode PCM16 samples (chunked — `btoa` takes a binary string). */
18
- export declare function pcm16ToBase64(pcm: Int16Array): string;
19
- /** Decode base64 PCM16LE (the sync response's `audio` field) into samples. */
20
- export declare function base64ToPcm16(base64: string): Int16Array;
21
- /** One completed sync turn: the wire response plus decoded reply audio. */
22
- export type SyncTurnResult = SyncTurnResponse & {
23
- /** Decoded reply audio (PCM16 at `sampleRate`), when the server spoke. */
24
- pcm: Int16Array | null;
25
- };
26
- /** Configuration for {@link createSyncSession}. */
27
- export type SyncSessionOptions = {
28
- /** The agent server's sync endpoint, e.g. `http://localhost:3000/sync`. */
29
- url: string;
30
- /** Fetch override (tests / custom transports). */
31
- fetch?: typeof globalThis.fetch | undefined;
32
- /** Called after every completed turn (text and voice alike). */
33
- onTurn?: ((turn: SyncTurnResult) => void) | undefined;
34
- /** Called when a turn fails; the rejection still propagates to the caller. */
35
- onError?: ((err: Error) => void) | undefined;
36
- };
37
- /** Sync-mode session handle. */
38
- export type SyncSession = {
39
- /** Conversation so far, oldest first — replayed to the server each turn. */
40
- readonly history: readonly SyncHistoryMessage[];
41
- /** Run one turn from committed text. */
42
- sendText(text: string): Promise<SyncTurnResult>;
43
- /** Run one turn from one utterance of mono PCM16 audio. */
44
- sendPcm16(pcm: Int16Array, sampleRate: number): Promise<SyncTurnResult>;
45
- /** Forget the conversation. */
46
- reset(): void;
47
- };
48
- /** Create a {@link SyncSession} against a sync-mode agent server. */
49
- export declare function createSyncSession(opts: SyncSessionOptions): SyncSession;
@@ -1,54 +0,0 @@
1
- /**
2
- * Client-side utterance detection (VAD) for sync mode.
3
- *
4
- * Sync mode has no streaming STT to endpoint speech server-side, so the
5
- * browser decides where an utterance ends: this module is a pure
6
- * energy-based voice-activity state machine fed PCM16 frames (from the
7
- * WebRTC capture pipeline in `sync-mic.ts` — `getUserMedia`'s voice
8
- * processing handles echo cancellation and noise suppression upstream,
9
- * which is what makes a simple RMS gate workable). No browser APIs are
10
- * touched here, so the state machine is fully unit-testable.
11
- *
12
- * Lifecycle per utterance: idle (keeping a pre-roll ring of recent audio)
13
- * → candidate speech (voiced frames accumulating toward `minSpeechMs`)
14
- * → speaking → `hangoverMs` of silence closes the utterance, which is
15
- * returned as one PCM16 buffer including the pre-roll and the hangover
16
- * tail (leading context and trailing silence both help one-shot STT).
17
- */
18
- /** Tuning for {@link createUtteranceDetector}. */
19
- export type UtteranceDetectorOptions = {
20
- /** Sample rate of pushed PCM16 frames, in Hz. */
21
- sampleRate: number;
22
- /** RMS level (0..1 of full scale) a frame must exceed to count as voiced. */
23
- speechRms?: number;
24
- /** Sustained voiced audio required before an utterance starts. Filters clicks. */
25
- minSpeechMs?: number;
26
- /** Silence that ends an utterance once one has started. */
27
- hangoverMs?: number;
28
- /** Audio kept from before speech onset, so a soft first syllable survives. */
29
- prerollMs?: number;
30
- /** Hard cap: a monologue this long is emitted as an utterance mid-speech. */
31
- maxUtteranceMs?: number;
32
- };
33
- export declare const DEFAULT_SPEECH_RMS = 0.015;
34
- export declare const DEFAULT_MIN_SPEECH_MS = 150;
35
- export declare const DEFAULT_HANGOVER_MS = 700;
36
- export declare const DEFAULT_PREROLL_MS = 300;
37
- export declare const DEFAULT_MAX_UTTERANCE_MS = 30000;
38
- /** Utterance detector — feed PCM16 frames, get complete utterances back. */
39
- export type UtteranceDetector = {
40
- /**
41
- * Push one capture frame (any length). Returns a complete utterance when
42
- * this frame closed one, else null.
43
- */
44
- push(frame: Int16Array): Int16Array | null;
45
- /** End of stream: returns any in-progress utterance (candidate audio that
46
- * never reached `minSpeechMs` is discarded as noise). */
47
- flush(): Int16Array | null;
48
- /** True while an utterance is being captured. */
49
- readonly speaking: boolean;
50
- /** Discard all buffered audio and return to idle. */
51
- reset(): void;
52
- };
53
- /** Create an energy-based {@link UtteranceDetector}. */
54
- export declare function createUtteranceDetector(opts: UtteranceDetectorOptions): UtteranceDetector;