@alexkroman1/aai-ui 5.14.0 → 6.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/dist/_repeat-until.d.ts +30 -0
- package/dist/_sse.d.ts +56 -0
- package/dist/_workflow-api-ref.d.ts +37 -0
- package/dist/audio.js +26 -26
- package/dist/{chat-view-CgFytvGy.js → chat-view-CK61bWWx.js} +2 -1
- package/dist/components/_form-values.d.ts +19 -0
- package/dist/components/chat-view.js +1 -1
- package/dist/components/form-types.d.ts +67 -0
- package/dist/components/form.d.ts +138 -0
- package/dist/components/message-list.js +1 -1
- package/dist/components/workflow-fields.d.ts +57 -0
- package/dist/components/workflow-progress.d.ts +55 -0
- package/dist/default-client/assets/audio-fO7SVU64.js +1 -0
- package/dist/default-client/assets/{capture-processor-B_5Ive8e.js → capture-processor-Dmc-KEpb.js} +4 -4
- package/dist/default-client/assets/client-audio-constants-Ck0IJO4c.js +1 -0
- package/dist/default-client/assets/index-CDugAuLK.css +2 -0
- package/dist/default-client/assets/index-DCI51Xz_.js +293 -0
- package/dist/default-client/assets/{playback-processor-6L8SIQ_l.js → playback-processor-DwQ9tE7X.js} +16 -13
- package/dist/default-client/index.html +3 -2
- package/dist/define-client.d.ts +40 -1
- package/dist/define-client.js +59 -17
- package/dist/index.d.ts +10 -0
- package/dist/index.js +1595 -5
- package/dist/{message-list-CcjgWRVZ.js → message-list-BwA3rdPi.js} +15 -1
- package/dist/page.d.ts +88 -0
- package/dist/{session-core-BA8H3qtF.js → session-core-ClKdVgRU.js} +245 -112
- package/dist/session-core-dial.d.ts +38 -0
- package/dist/session-core-handshake.d.ts +16 -1
- package/dist/session-core-messages.d.ts +2 -2
- package/dist/session-core-reconnect.d.ts +2 -7
- package/dist/session-core.js +1 -1
- package/dist/session-resume-store.d.ts +43 -0
- package/dist/types.d.ts +1 -1
- package/dist/types.js +2 -2
- package/dist/use-user-transcript.d.ts +70 -0
- package/dist/use-workflow-form.d.ts +136 -0
- package/dist/use-workflow-progress.d.ts +100 -0
- package/dist/use-workflow-run.d.ts +56 -0
- package/dist/use-workflow-runs.d.ts +71 -0
- package/dist/workflow-client.d.ts +97 -0
- package/dist/workflow-events.d.ts +39 -0
- package/dist/worklets/_playback-bench-harness.d.ts +181 -0
- package/dist/worklets/_playback-bench-host.d.ts +63 -0
- package/dist/worklets/_playback-bench-page.d.ts +65 -0
- package/dist/worklets/_tts-trace-harness.d.ts +142 -0
- package/dist/worklets/_worklet-test-utils.d.ts +27 -0
- package/dist/worklets/playback-processor.d.ts +1 -1
- package/dist/worklets/playback-processor.js +15 -12
- package/package.json +9 -8
- package/dist/default-client/assets/audio-CsQVQn3f.js +0 -1
- package/dist/default-client/assets/index-D35_z2WM.js +0 -293
- package/dist/default-client/assets/index-DCjB3qtb.css +0 -2
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Browser client for the workflow HTTP API (`aai/host/workflow-api.ts`).
|
|
3
|
+
*
|
|
4
|
+
* This is the whole client half of a WORKFLOW APP: an agent whose front door is
|
|
5
|
+
* a form rather than a microphone (`workflowApp()`) starts runs
|
|
6
|
+
* here and watches them for the answer. It deliberately does NOT go through
|
|
7
|
+
* `SessionCore` — there is no socket, no audio graph, and no session to resume.
|
|
8
|
+
*
|
|
9
|
+
* **The requests themselves are the SDK's now**
|
|
10
|
+
* (`createWorkflowApiClient`, `@alexkroman1/aai/workflow-api`), and what is left
|
|
11
|
+
* here is the one thing that is genuinely a BROWSER's: the default base URL.
|
|
12
|
+
* Every route, every query, the 404-is-an-answer rule and the `wait` clamp were
|
|
13
|
+
* written three times over — here, in the studio's Workflows card, and in
|
|
14
|
+
* `aai workflow` — and the parts the copies disagreed on were exactly the ones a
|
|
15
|
+
* reader cannot check by eye. The SDK module's doc carries that argument; this
|
|
16
|
+
* file must not grow a second implementation of any of it.
|
|
17
|
+
*
|
|
18
|
+
* The one thing worth knowing before using it: **a run outlives the page.**
|
|
19
|
+
* Starting one resolves as soon as the run is created, so `runId` is the only
|
|
20
|
+
* handle that matters and it stays valid across a reload, a different device, or
|
|
21
|
+
* `curl` — which is what makes `useWorkflowRun` a watch rather than a
|
|
22
|
+
* subscription to something the page owns.
|
|
23
|
+
*
|
|
24
|
+
* The loop that keeps asking lives in `use-workflow-run.ts`, and the streaming
|
|
25
|
+
* fast path under it in `workflow-events.ts`.
|
|
26
|
+
*/
|
|
27
|
+
import type { WorkflowRunSnapshot } from "@alexkroman1/aai";
|
|
28
|
+
import { type WorkflowApi } from "@alexkroman1/aai/workflow-api";
|
|
29
|
+
/**
|
|
30
|
+
* A run's observable state.
|
|
31
|
+
*
|
|
32
|
+
* Aliased from the SDK rather than restated. `import type` is erased entirely,
|
|
33
|
+
* so a second definition of the fields and the five-member status union would
|
|
34
|
+
* buy nothing and cost the one thing that matters — nothing would assert the two
|
|
35
|
+
* agree, so a status added to the SDK would never reach the browser type.
|
|
36
|
+
*
|
|
37
|
+
* `WorkflowRun` keeps the shorter name because it is what a page's own code
|
|
38
|
+
* writes; nothing in a browser needs the word "snapshot" to know a read returns
|
|
39
|
+
* one.
|
|
40
|
+
*
|
|
41
|
+
* It is GENERIC on the run's output, and a page supplies it — see
|
|
42
|
+
* {@link useWorkflowRun}. It does NOT have to restate that type: a page can name
|
|
43
|
+
* its own workflow and derive the rest with `WorkflowOutputOf`, pulling no
|
|
44
|
+
* server graph into the bundle.
|
|
45
|
+
*
|
|
46
|
+
* @public
|
|
47
|
+
*/
|
|
48
|
+
export type WorkflowRun<R = unknown> = WorkflowRunSnapshot<R>;
|
|
49
|
+
/**
|
|
50
|
+
* A workflow's own output type, and the shape `GET /workflows` lists — both
|
|
51
|
+
* re-exported so a page needs ONE import to type its runs and render its form.
|
|
52
|
+
*/
|
|
53
|
+
export type { WorkflowOutputOf, WorkflowSummary } from "@alexkroman1/aai";
|
|
54
|
+
/**
|
|
55
|
+
* A run status nothing will change again.
|
|
56
|
+
*
|
|
57
|
+
* Re-exported from the SDK rather than defined here. A second implementation
|
|
58
|
+
* listing two of the three terminal statuses would leave a cancelled run polled
|
|
59
|
+
* forever by a page while the agent considered it finished — the kind of drift a
|
|
60
|
+
* status predicate beside the status union cannot have.
|
|
61
|
+
*/
|
|
62
|
+
export { isTerminal } from "@alexkroman1/aai";
|
|
63
|
+
/**
|
|
64
|
+
* The call set {@link createWorkflowApi} returns.
|
|
65
|
+
*
|
|
66
|
+
* Re-exported from the SDK rather than declared here: it IS the SDK's client,
|
|
67
|
+
* and a structural restatement would be a second thing to keep in step with the
|
|
68
|
+
* routes for no gain.
|
|
69
|
+
*/
|
|
70
|
+
export type { WorkflowApi } from "@alexkroman1/aai/workflow-api";
|
|
71
|
+
export type WorkflowApiOptions = {
|
|
72
|
+
/**
|
|
73
|
+
* Base URL of the agent. Defaults to the page's own origin + path, which is
|
|
74
|
+
* right for a page the agent itself serves — the only case that exists today,
|
|
75
|
+
* and the reason this wrapper exists at all: the SDK client requires a base
|
|
76
|
+
* URL, because `location` does not exist in that half of the SDK.
|
|
77
|
+
*/
|
|
78
|
+
baseUrl?: string;
|
|
79
|
+
/**
|
|
80
|
+
* Bearer for an agent whose operator set `AAI_WORKFLOW_API_TOKEN`. A page
|
|
81
|
+
* served to the public has nothing to put here (and should not — it would be
|
|
82
|
+
* readable in the bundle); this exists for a programmatic caller written
|
|
83
|
+
* against the same client.
|
|
84
|
+
*/
|
|
85
|
+
token?: string;
|
|
86
|
+
};
|
|
87
|
+
/**
|
|
88
|
+
* Create a workflow API client aimed at the agent serving this page.
|
|
89
|
+
*
|
|
90
|
+
* Hoist it out of the component that uses it. `useWorkflowRun` holds the client
|
|
91
|
+
* in a ref precisely so a fresh object per render does not restart its watch,
|
|
92
|
+
* but a client built in render is still a new `fetch` closure every time and
|
|
93
|
+
* reads as though it were free.
|
|
94
|
+
*
|
|
95
|
+
* @public
|
|
96
|
+
*/
|
|
97
|
+
export declare function createWorkflowApi(opts?: WorkflowApiOptions): WorkflowApi;
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Watching a run over server-sent events — the PUSH half of `useWorkflowRun`.
|
|
3
|
+
*
|
|
4
|
+
* Its own module because the seam is clean: everything in `workflow-client.ts`
|
|
5
|
+
* is request/response shaping plus the poll, and this is one long-lived stream
|
|
6
|
+
* and the SSE parser it needs.
|
|
7
|
+
*
|
|
8
|
+
* @internal
|
|
9
|
+
*/
|
|
10
|
+
import type { WorkflowApi, WorkflowRun } from "./workflow-client.ts";
|
|
11
|
+
/**
|
|
12
|
+
* The slice of the client this needs: one method.
|
|
13
|
+
*
|
|
14
|
+
* Narrowed rather than taking the whole `WorkflowApi`, and the narrowing is the
|
|
15
|
+
* honest statement — nothing here reads a run, starts one, or cancels one, it
|
|
16
|
+
* opens ONE stream. It also makes a test double a plain object rather than a
|
|
17
|
+
* six-method stub cast into shape. A real client satisfies it structurally.
|
|
18
|
+
*/
|
|
19
|
+
export type RunWatcher = Pick<WorkflowApi, "watch">;
|
|
20
|
+
/**
|
|
21
|
+
* Watch a run over SSE, falling back to the caller's poll on any failure.
|
|
22
|
+
*
|
|
23
|
+
* The poll stays the fallback rather than being replaced, and that is the whole
|
|
24
|
+
* shape of this: a stream is an optimisation over a mechanism that already
|
|
25
|
+
* works, so every way it can fail — an older agent with no `/events` route, a
|
|
26
|
+
* proxy that buffers, a network that drops it — has to degrade to the thing that
|
|
27
|
+
* does. What it buys is real, though: on the platform every polled read BROKERS,
|
|
28
|
+
* so N open tabs at `DEFAULT_WORKFLOW_POLL_MS` is N/2 brokered requests a
|
|
29
|
+
* second, each able to boot a sandbox. One stream per tab replaces all of it.
|
|
30
|
+
*
|
|
31
|
+
* `EventSource` is not used, for two reasons that both matter here: it cannot
|
|
32
|
+
* send an `Authorization` header (an agent with `AAI_WORKFLOW_API_TOKEN` set
|
|
33
|
+
* would be unreachable), and it reconnects on its own schedule, which would
|
|
34
|
+
* fight the caller's. A `fetch` stream gives both back.
|
|
35
|
+
*
|
|
36
|
+
* Returns a stop function. `onFallback` is called at most once, when this stream
|
|
37
|
+
* cannot be relied on and the poll should take over.
|
|
38
|
+
*/
|
|
39
|
+
export declare function watchRunEvents<R>(getClient: () => RunWatcher, runId: string, onRun: (run: WorkflowRun<R>) => void, onSettled: () => void, onFallback: () => void): () => void;
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A bench for the playback jitter buffer: real TTS, the real worklet, a
|
|
3
|
+
* deterministic clock, and one setting changed at a time.
|
|
4
|
+
*
|
|
5
|
+
* Why this exists. `PLAYBACK_JITTER_MS` and `PLAYBACK_REFILL_MS` are the two
|
|
6
|
+
* numbers that decide whether a reply is heard as speech or as stutter, and
|
|
7
|
+
* every test that touches them supplies its own arrival pattern:
|
|
8
|
+
* `playback-processor.test.ts` hand-feeds quanta, and `audio-stress.test.ts`
|
|
9
|
+
* records in its own header that its chunk sizes outrun the render loop by an
|
|
10
|
+
* order of magnitude so the buffer "effectively never starves". Neither can
|
|
11
|
+
* answer "is 400 the right number", because neither has ever seen how audio
|
|
12
|
+
* actually arrives.
|
|
13
|
+
*
|
|
14
|
+
* So the bench replays a RECORDED reply (`_tts-trace-harness.ts`) through the
|
|
15
|
+
* chain the audio really crosses:
|
|
16
|
+
*
|
|
17
|
+
* provider frames (recorded arrival times)
|
|
18
|
+
* -> the server's bounded-lead pacer <- MODELLED, see below
|
|
19
|
+
* -> a network profile (latency, jitter, stalls)
|
|
20
|
+
* -> the real playback worklet <- `playbackProcessorSource`
|
|
21
|
+
* -> rendered PCM + the worklet's own stats
|
|
22
|
+
*
|
|
23
|
+
* Everything is virtual-time: the clock advances one 128-sample render quantum
|
|
24
|
+
* per iteration, so a sweep of a hundred settings over a 30-second reply costs
|
|
25
|
+
* seconds and gives byte-identical results every run. {@link renderToWav} is
|
|
26
|
+
* the other half — the rendered output written where a human can listen to it,
|
|
27
|
+
* because a concealment counter does not tell you whether a reply sounds
|
|
28
|
+
* broken.
|
|
29
|
+
*
|
|
30
|
+
* **The pacer is a MODEL, and that is this bench's one fidelity gap.**
|
|
31
|
+
* `createAudioPacer` (`aai/host/audio-pacer.ts`) is not on any published
|
|
32
|
+
* subpath and this package may not import a sibling's internals, so
|
|
33
|
+
* {@link pacedSends} is a transcription of its algorithm rather than the
|
|
34
|
+
* algorithm. It is ~20 lines and reproduces the two properties the pacer's own
|
|
35
|
+
* doc states — free flow until the lead ceiling, then one release per
|
|
36
|
+
* `burstMs` — but a change to the real pacer will NOT fail this file. If the
|
|
37
|
+
* pacer moves, re-read it against `pacedSends`. Exporting the real one and
|
|
38
|
+
* deleting the model is the fix; it needs a non-test change.
|
|
39
|
+
*/
|
|
40
|
+
import { type TtsTrace } from "./_tts-trace-harness.ts";
|
|
41
|
+
/** The render quantum every AudioWorkletProcessor is called with. */
|
|
42
|
+
export declare const QUANTUM = 128;
|
|
43
|
+
/** Sampling interval of {@link RenderResult.earMs}. */
|
|
44
|
+
export declare const EAR_SAMPLE_MS = 20;
|
|
45
|
+
/** One frame arriving at the client: when it lands, and what it carries. */
|
|
46
|
+
export type Delivery = {
|
|
47
|
+
atMs: number;
|
|
48
|
+
bytes: Uint8Array;
|
|
49
|
+
};
|
|
50
|
+
/**
|
|
51
|
+
* How the server releases audio. Production values are
|
|
52
|
+
* `CLIENT_AUDIO_LEAD_MS` (1000) and `PACER_BURST_MS` (200); they are
|
|
53
|
+
* parameters here because they are half of what the bench is sweeping — the
|
|
54
|
+
* client's cushion is the server's lead, so tuning the jitter buffer without
|
|
55
|
+
* them is tuning one end of one number.
|
|
56
|
+
*/
|
|
57
|
+
export type PacerProfile = {
|
|
58
|
+
leadMs: number;
|
|
59
|
+
burstMs: number;
|
|
60
|
+
};
|
|
61
|
+
/** What the link does to a frame between the server and the ear. */
|
|
62
|
+
export type NetworkProfile = {
|
|
63
|
+
name: string;
|
|
64
|
+
/** One-way latency floor, ms. */
|
|
65
|
+
latencyMs: number;
|
|
66
|
+
/** Peak extra delay added on top, ms (0 = a perfectly even link). */
|
|
67
|
+
jitterMs: number;
|
|
68
|
+
/**
|
|
69
|
+
* Freezes: `{ atMs, forMs }` pairs during which nothing is delivered, every
|
|
70
|
+
* held frame landing at the end. This is what a jitter buffer is FOR, and
|
|
71
|
+
* the only part of a profile that has to be scripted rather than sampled —
|
|
72
|
+
* a stall is an event, not a distribution.
|
|
73
|
+
*/
|
|
74
|
+
stalls?: {
|
|
75
|
+
atMs: number;
|
|
76
|
+
forMs: number;
|
|
77
|
+
}[];
|
|
78
|
+
/**
|
|
79
|
+
* Throughput ceiling in bits/s, or `Infinity`. Both legs are uncompressed
|
|
80
|
+
* PCM16 (384 kbps down at 24 kHz), so a link that cannot carry the bitrate
|
|
81
|
+
* cannot be fixed by any buffer — that claim is in the aai-ui guide and this
|
|
82
|
+
* is what makes it measurable.
|
|
83
|
+
*/
|
|
84
|
+
bitsPerSecond?: number;
|
|
85
|
+
};
|
|
86
|
+
/**
|
|
87
|
+
* The worklet's one knob. It was two — a startup target and a refill target —
|
|
88
|
+
* until this bench showed the startup one was redundant by construction; see
|
|
89
|
+
* `PLAYBACK_FILL_MS`.
|
|
90
|
+
*/
|
|
91
|
+
export type PlaybackSettings = {
|
|
92
|
+
fillMs: number;
|
|
93
|
+
};
|
|
94
|
+
/**
|
|
95
|
+
* Model the server's bounded-lead pacer over a trace's frames.
|
|
96
|
+
*
|
|
97
|
+
* A transcription of `createAudioPacer`, whose loop is: a frame is sent
|
|
98
|
+
* immediately while the lead (everything sent so far, minus now) is under the
|
|
99
|
+
* ceiling; otherwise it waits until the lead has drained `burstMs` below the
|
|
100
|
+
* ceiling and then goes out with everything else the drain releases.
|
|
101
|
+
*/
|
|
102
|
+
export declare function pacedSends(trace: TtsTrace, pacer: PacerProfile): Delivery[];
|
|
103
|
+
/** Apply a network profile to the server's send schedule. */
|
|
104
|
+
export declare function overNetwork(sends: Delivery[], net: NetworkProfile): Delivery[];
|
|
105
|
+
/** Everything one render of one setting produced. */
|
|
106
|
+
export type RenderResult = {
|
|
107
|
+
/** The audio the ear actually received, at the context sample rate. */
|
|
108
|
+
rendered: Float32Array;
|
|
109
|
+
sampleRate: number;
|
|
110
|
+
/** Ms from the first frame LEAVING the provider to the first audible sample. */
|
|
111
|
+
timeToFirstAudioMs: number;
|
|
112
|
+
/** The worklet's own report, as `onPlaybackStats` would receive it. */
|
|
113
|
+
stats: {
|
|
114
|
+
concealedSamples: number;
|
|
115
|
+
silentConcealedSamples: number;
|
|
116
|
+
concealmentEvents: number;
|
|
117
|
+
silentConcealmentEvents: number;
|
|
118
|
+
};
|
|
119
|
+
/** Every concealment episode's length in ms, in order. */
|
|
120
|
+
gapsMs: number[];
|
|
121
|
+
/** `bufferedMs` values the worklet reported, as `playback_progress` would. */
|
|
122
|
+
progressMs: number[];
|
|
123
|
+
/**
|
|
124
|
+
* Ground truth for the heard cursor: cumulative ms of the REPLY's own audio
|
|
125
|
+
* the ear had received, sampled every {@link EAR_SAMPLE_MS}.
|
|
126
|
+
*
|
|
127
|
+
* Concealed samples are excluded — they are fabricated, not reply audio — so
|
|
128
|
+
* this is exactly the quantity `heardMs()` in
|
|
129
|
+
* `aai/host/transports/pipeline-heard.ts` estimates, which is what makes the
|
|
130
|
+
* two comparable.
|
|
131
|
+
*/
|
|
132
|
+
earMs: number[];
|
|
133
|
+
/** How long the reply took to play out, first sample to last. */
|
|
134
|
+
playedMs: number;
|
|
135
|
+
};
|
|
136
|
+
/**
|
|
137
|
+
* Render one delivery schedule through the real playback worklet.
|
|
138
|
+
*
|
|
139
|
+
* The clock is the render loop itself: quantum `q` happens at
|
|
140
|
+
* `q * QUANTUM / sampleRate` seconds, every delivery at or before that instant
|
|
141
|
+
* is written first, and `done` is posted once the last frame has landed. That
|
|
142
|
+
* is the ordering the audio thread really sees (`onmessage` and `process()`
|
|
143
|
+
* never interleave), and it makes the whole render a pure function of the
|
|
144
|
+
* schedule.
|
|
145
|
+
*/
|
|
146
|
+
export declare function renderSchedule(deliveries: Delivery[], opts: {
|
|
147
|
+
sampleRate: number;
|
|
148
|
+
settings: PlaybackSettings;
|
|
149
|
+
maxSeconds?: number;
|
|
150
|
+
}): RenderResult;
|
|
151
|
+
/** One end-to-end run: trace + pacer + network + settings. */
|
|
152
|
+
export declare function runBench(opts: {
|
|
153
|
+
trace: TtsTrace;
|
|
154
|
+
pacer: PacerProfile;
|
|
155
|
+
net: NetworkProfile;
|
|
156
|
+
settings: PlaybackSettings;
|
|
157
|
+
}): RenderResult;
|
|
158
|
+
/**
|
|
159
|
+
* Score one render. Lower is better, and the WEIGHTS are the opinion in this
|
|
160
|
+
* file — everything above is measurement.
|
|
161
|
+
*
|
|
162
|
+
* The three terms are not interchangeable:
|
|
163
|
+
*
|
|
164
|
+
* - **Startup latency** is paid on every single turn, so it is the term that
|
|
165
|
+
* compounds over a conversation.
|
|
166
|
+
* - **Silent concealment** is an audible hole. Concealment that stays under
|
|
167
|
+
* the fade is a smear the ear largely forgives, which is why the worklet
|
|
168
|
+
* reports the two separately, so silence is weighted an order of magnitude
|
|
169
|
+
* higher than concealment in general.
|
|
170
|
+
* - **Episode COUNT** matters independently of total length: one 300 ms pause
|
|
171
|
+
* reads as a network hiccup, where six 50 ms ones read as a broken codec.
|
|
172
|
+
* That is the whole finding behind the refill re-arm, so a score that
|
|
173
|
+
* summed milliseconds alone would rank the failure it was built to prevent
|
|
174
|
+
* as equal to the fix.
|
|
175
|
+
*/
|
|
176
|
+
export declare function scoreRender(r: RenderResult): {
|
|
177
|
+
score: number;
|
|
178
|
+
parts: Record<string, number>;
|
|
179
|
+
};
|
|
180
|
+
/** Minimal 16-bit PCM WAV, so a render can be listened to. */
|
|
181
|
+
export declare function toWav(samples: Float32Array, sampleRate: number): Buffer;
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The bench's HOST-side half: what the server believes about the caller's
|
|
3
|
+
* playback, against what the caller actually heard.
|
|
4
|
+
*
|
|
5
|
+
* Split from `_playback-bench-harness.ts` at the seam it already had — that file
|
|
6
|
+
* models the wire and drives the worklet, and this one models
|
|
7
|
+
* `aai/host/transports/pipeline-heard.ts`'s arithmetic over the result. It went
|
|
8
|
+
* out when the file hit the 500-line cap, which is the right time rather than a
|
|
9
|
+
* line later.
|
|
10
|
+
*
|
|
11
|
+
* Everything here is a TRANSCRIPTION of host code this package may not import
|
|
12
|
+
* (`aai`'s `host/` is Node-only and internal), so it carries the same warning the
|
|
13
|
+
* pacer model does: a change to `createPlaybackClock` will not fail this file.
|
|
14
|
+
* Re-read it against these functions if that clock moves.
|
|
15
|
+
*/
|
|
16
|
+
import type { Delivery, RenderResult } from "./_playback-bench-harness.ts";
|
|
17
|
+
/**
|
|
18
|
+
* What the HOST believes about playback, and what actually happened.
|
|
19
|
+
*
|
|
20
|
+
* The barge-in floor is `pending()` in `aai/host/transports/pipeline-heard.ts`:
|
|
21
|
+
* `now() < endsAtMs + PIPELINE_PLAYBACK_GRACE_MS`, where `endsAtMs` accumulates
|
|
22
|
+
* each forwarded chunk's duration from `max(endsAtMs, now())` — an OPEN-LOOP
|
|
23
|
+
* model that assumes playback starts the instant a chunk is forwarded and runs at
|
|
24
|
+
* exactly 1.0x. No real client beats that, so the model is a lower bound and the
|
|
25
|
+
* grace is what covers the difference.
|
|
26
|
+
*
|
|
27
|
+
* This measures the difference. `requiredGraceMs` is how long after the host's
|
|
28
|
+
* estimate the caller was still hearing audio — i.e. the smallest grace that
|
|
29
|
+
* keeps barge-in working to the end of a reply. It is a REQUIREMENT, so a grace
|
|
30
|
+
* at or above it is correct and one below it means a barge-in in the reply's tail
|
|
31
|
+
* is not recognised as arriving during playback.
|
|
32
|
+
*
|
|
33
|
+
* Two clients are modelled because the host serves both: one that wires
|
|
34
|
+
* `onPlaybackProgress` (the browser does) and one that does not (a telephony
|
|
35
|
+
* bridge, a harness). The reports clamp `endsAtMs` UPWARD only, so a reporting
|
|
36
|
+
* client shrinks the requirement and a silent one leaves the host on the
|
|
37
|
+
* open-loop estimate — which is the case the constant has to be safe for.
|
|
38
|
+
*/
|
|
39
|
+
export type PlayoutVsHost = {
|
|
40
|
+
/** When the host's open-loop model thinks forwarded audio finishes playing. */
|
|
41
|
+
openLoopEndMs: number;
|
|
42
|
+
/** The same, with the client's `playback_progress` reports clamped in. */
|
|
43
|
+
reportedEndMs: number;
|
|
44
|
+
/** When the caller actually stopped hearing audio. */
|
|
45
|
+
realEndMs: number;
|
|
46
|
+
/** Smallest grace that keeps `pending()` true to `realEndMs`, unreported. */
|
|
47
|
+
requiredGraceMs: number;
|
|
48
|
+
/** The same for a client that DOES report its backlog. */
|
|
49
|
+
requiredGraceReportingMs: number;
|
|
50
|
+
};
|
|
51
|
+
/**
|
|
52
|
+
* Replay a render against the host's own playback-clock arithmetic.
|
|
53
|
+
*
|
|
54
|
+
* `deliveries` is what the host FORWARDED (the pacer's output, before the link),
|
|
55
|
+
* because that is what `onChunk` sees — the host has no view of the wire.
|
|
56
|
+
*/
|
|
57
|
+
export declare function playoutVsHost(opts: {
|
|
58
|
+
forwarded: Delivery[];
|
|
59
|
+
render: RenderResult;
|
|
60
|
+
sampleRate: number;
|
|
61
|
+
/** Interval of the client's backlog reports; production is 500 ms. */
|
|
62
|
+
reportIntervalMs: number;
|
|
63
|
+
}): PlayoutVsHost;
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The browser half of the playback bench: a page that plays a recorded reply
|
|
3
|
+
* through the REAL playback worklet in a REAL `AudioContext`, so the tuning can
|
|
4
|
+
* be listened to rather than only counted.
|
|
5
|
+
*
|
|
6
|
+
* This is the part the offline renderer cannot be: `renderSchedule` drives
|
|
7
|
+
* `process()` on a synthetic sample clock with `postMessage` delivered exactly
|
|
8
|
+
* between quanta, which is the ordering the audio thread guarantees but says
|
|
9
|
+
* nothing about the browser's own render cadence, its output latency, or what
|
|
10
|
+
* a `port.postMessage` burst does to a live audio callback. If the two agree,
|
|
11
|
+
* the offline sweep is trustworthy; where they disagree, the browser is right.
|
|
12
|
+
*
|
|
13
|
+
* Two ways in, same page:
|
|
14
|
+
*
|
|
15
|
+
* - **A human opens it** and gets a slider for `fillMs`, the
|
|
16
|
+
* network profile, and the pacer lead, with the worklet's own concealment
|
|
17
|
+
* counters live. Audio comes out of the speakers.
|
|
18
|
+
* - **Playwright drives it** through {@link BENCH_API} on `window`, and pulls
|
|
19
|
+
* back the samples the tap captured — the audio that actually reached the
|
|
20
|
+
* destination — so a test can diff it against the offline render.
|
|
21
|
+
*
|
|
22
|
+
* The page is a STRING rather than a file under a bundler because it must stay
|
|
23
|
+
* test-only: aai-ui's build (`tsdown` + `build-default-client.ts`) ships
|
|
24
|
+
* `dist/default-client/`, and a second HTML entry there is a product artifact
|
|
25
|
+
* with a coverage floor and a size budget. Generated into `reports/playback/`,
|
|
26
|
+
* it is neither.
|
|
27
|
+
*/
|
|
28
|
+
/** The name of the object the page exposes for a driver to call. */
|
|
29
|
+
export declare const BENCH_API = "__aaiPlaybackBench";
|
|
30
|
+
/** Options a driver or a human can set on one run. */
|
|
31
|
+
export type BenchRunOptions = {
|
|
32
|
+
fillMs: number;
|
|
33
|
+
/** Deliveries: ms since run start, and the slice of the PCM to write. */
|
|
34
|
+
schedule: {
|
|
35
|
+
atMs: number;
|
|
36
|
+
offset: number;
|
|
37
|
+
length: number;
|
|
38
|
+
}[];
|
|
39
|
+
/** Whether the reply is audible. A sweep run in a headless browser is not. */
|
|
40
|
+
muted?: boolean;
|
|
41
|
+
};
|
|
42
|
+
/**
|
|
43
|
+
* Build the bench page.
|
|
44
|
+
*
|
|
45
|
+
* `pcmUrl` is fetched once and sliced per delivery, so the page replays the
|
|
46
|
+
* same bytes the offline renderer does. `sampleRate` must be the trace's: the
|
|
47
|
+
* context is created at it and the page REFUSES to run if the browser grants
|
|
48
|
+
* another, exactly as `audio.ts` does — PCM written into a context at the wrong
|
|
49
|
+
* rate plays at the wrong speed, which would look like a tuning result.
|
|
50
|
+
*/
|
|
51
|
+
export declare function benchPageHtml(opts: {
|
|
52
|
+
pcmUrl: string;
|
|
53
|
+
sampleRate: number;
|
|
54
|
+
/** Shown in the header, so a saved page names what it is playing. */
|
|
55
|
+
title: string;
|
|
56
|
+
/** Default schedules a human can pick between, by profile name. */
|
|
57
|
+
profiles: Record<string, {
|
|
58
|
+
atMs: number;
|
|
59
|
+
offset: number;
|
|
60
|
+
length: number;
|
|
61
|
+
}[]>;
|
|
62
|
+
defaults?: {
|
|
63
|
+
fillMs: number;
|
|
64
|
+
};
|
|
65
|
+
}): string;
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Capture and replay of REAL TTS arrival timing.
|
|
3
|
+
*
|
|
4
|
+
* The playback worklet's tuning question — how deep should
|
|
5
|
+
* `PLAYBACK_JITTER_MS` be — is entirely a question about how unevenly audio
|
|
6
|
+
* arrives, and every existing test answers it with a generated arrival pattern.
|
|
7
|
+
* `audio-stress.test.ts` says so itself: its chunk-size arbitrary averages
|
|
8
|
+
* ~750 samples against 128 consumed per render, so writes outrun renders by an
|
|
9
|
+
* order of magnitude and the buffer effectively never starves. A jitter buffer
|
|
10
|
+
* tuned against that is tuned against nothing.
|
|
11
|
+
*
|
|
12
|
+
* So a trace is a recording of one real reply as the provider actually emitted
|
|
13
|
+
* it: PCM16 bytes plus the millisecond each frame ARRIVED, relative to the
|
|
14
|
+
* first. Replayed through the real pacer and the real worklet, that makes the
|
|
15
|
+
* tuning question measurable and repeatable — the same reply, the same
|
|
16
|
+
* arrival pattern, one setting changed.
|
|
17
|
+
*
|
|
18
|
+
* Two halves, deliberately separate:
|
|
19
|
+
*
|
|
20
|
+
* - {@link captureTtsTrace} needs a live provider and an API key. It runs once,
|
|
21
|
+
* by hand (`AAI_CAPTURE_TTS_TRACE=1`), and commits its result.
|
|
22
|
+
* - {@link readTtsTrace} needs neither, so every test that CONSUMES a trace is
|
|
23
|
+
* keyless and offline.
|
|
24
|
+
*
|
|
25
|
+
* The bytes live beside the JSON rather than inside it: base64 in a fixture
|
|
26
|
+
* inflates ~1 MB of PCM by a third and makes the file unreadable in a diff,
|
|
27
|
+
* where a `.pcm` sidecar is `ffplay`-able and reviewed by its length alone.
|
|
28
|
+
*/
|
|
29
|
+
/** One provider audio frame, with the moment it arrived. */
|
|
30
|
+
export type TraceFrame = {
|
|
31
|
+
/** Arrival time in ms, relative to the first frame of the reply. */
|
|
32
|
+
tMs: number;
|
|
33
|
+
/** Byte offset of this frame's PCM16 in the sidecar. */
|
|
34
|
+
offset: number;
|
|
35
|
+
/** Byte length of this frame's PCM16. */
|
|
36
|
+
length: number;
|
|
37
|
+
};
|
|
38
|
+
/** A recorded reply: what the provider sent, and when. */
|
|
39
|
+
export type TtsTrace = {
|
|
40
|
+
/** Sample rate of the PCM16 in the sidecar. */
|
|
41
|
+
sampleRate: number;
|
|
42
|
+
/** Provider kind and voice, so a trace names the thing it recorded. */
|
|
43
|
+
provider: string;
|
|
44
|
+
voice: string;
|
|
45
|
+
/** The reply text that was synthesized. */
|
|
46
|
+
text: string;
|
|
47
|
+
/** Ms from the first `sendText` to the first audio frame. */
|
|
48
|
+
firstAudioMs: number;
|
|
49
|
+
/** Ms from the first `sendText` to `done`. */
|
|
50
|
+
doneMs: number;
|
|
51
|
+
frames: TraceFrame[];
|
|
52
|
+
/** All frames' PCM16, concatenated in arrival order. */
|
|
53
|
+
pcm: Int16Array;
|
|
54
|
+
};
|
|
55
|
+
/**
|
|
56
|
+
* Total audio duration in ms. Not the same as {@link TtsTrace.doneMs}: a
|
|
57
|
+
* provider that synthesizes faster than real time produces more audio than the
|
|
58
|
+
* wall clock it took to produce it, and the RATIO of the two is the whole
|
|
59
|
+
* reason a jitter buffer can ever fill.
|
|
60
|
+
*/
|
|
61
|
+
export declare function traceAudioMs(trace: TtsTrace): number;
|
|
62
|
+
/**
|
|
63
|
+
* The slice of the SDK's `TtsSession` a capture drives. Declared structurally
|
|
64
|
+
* so the harness needs no host-only type import: a real session satisfies it.
|
|
65
|
+
*/
|
|
66
|
+
export type CapturableTtsSession = {
|
|
67
|
+
sendText(text: string): void;
|
|
68
|
+
flush(): void;
|
|
69
|
+
on(event: "audio", fn: (pcm: Int16Array) => void): unknown;
|
|
70
|
+
on(event: "done", fn: () => void): unknown;
|
|
71
|
+
on(event: "error", fn: (err: {
|
|
72
|
+
message?: string;
|
|
73
|
+
}) => void): unknown;
|
|
74
|
+
close(): Promise<void>;
|
|
75
|
+
};
|
|
76
|
+
/**
|
|
77
|
+
* Open a real TTS session, synthesize `text`, and record every frame's arrival.
|
|
78
|
+
*
|
|
79
|
+
* The text is sent the way the PIPELINE sends it — as deltas, with a final
|
|
80
|
+
* `flush()` — because the AssemblyAI adapter segments on what it receives and
|
|
81
|
+
* `Generate` only buffers until a `Flush`. Handing it the whole reply in one
|
|
82
|
+
* call would record a different arrival pattern from the one a real turn
|
|
83
|
+
* produces (see `providers/tts/assemblyai-segment.ts`).
|
|
84
|
+
*/
|
|
85
|
+
export declare function captureTtsTrace(opts: {
|
|
86
|
+
text: string;
|
|
87
|
+
/**
|
|
88
|
+
* Opens the real provider session. INJECTED rather than resolved here: the
|
|
89
|
+
* resolver (`aai`'s `host/providers/resolve.ts`) is not on any published
|
|
90
|
+
* subpath, and this package may not import a sibling's internals. The caller
|
|
91
|
+
* that has one is the capture runner, which is not part of the package.
|
|
92
|
+
*/
|
|
93
|
+
open: (o: {
|
|
94
|
+
sampleRate: number;
|
|
95
|
+
signal: AbortSignal;
|
|
96
|
+
}) => Promise<CapturableTtsSession>;
|
|
97
|
+
/** LLM-shaped deltas. Defaults to splitting `text` on word boundaries. */
|
|
98
|
+
deltas?: string[];
|
|
99
|
+
/** Recorded into the trace so a fixture names the voice it captured. */
|
|
100
|
+
voice?: string;
|
|
101
|
+
provider?: string;
|
|
102
|
+
sampleRate?: number;
|
|
103
|
+
/**
|
|
104
|
+
* Gap between deltas, in ms. **Set this to represent a real turn.**
|
|
105
|
+
*
|
|
106
|
+
* Sending every delta in one burst is what a capture does by default, and for
|
|
107
|
+
* the PLAYBACK question that is harmless — the pacer reshapes arrival anyway.
|
|
108
|
+
* For anything about SEGMENTATION it invalidates the measurement outright: the
|
|
109
|
+
* segmenter is handed the whole reply before it makes its first cut, so
|
|
110
|
+
* time-to-first-audio collapses to the service's own latency (~40 ms measured)
|
|
111
|
+
* and every segmentation rule scores the same. An LLM streams at ~30 ms a
|
|
112
|
+
* delta, which is what the segmenter really sees.
|
|
113
|
+
*/
|
|
114
|
+
deltaIntervalMs?: number;
|
|
115
|
+
/** Hard cap, so a provider that never sends `done` cannot hang the capture. */
|
|
116
|
+
timeoutMs?: number;
|
|
117
|
+
}): Promise<TtsTrace>;
|
|
118
|
+
/**
|
|
119
|
+
* Split a reply into LLM-shaped deltas: a few words at a time, which is what
|
|
120
|
+
* `streamText` emits and therefore what the adapter's segmenter sees.
|
|
121
|
+
*/
|
|
122
|
+
export declare function splitIntoDeltas(text: string, wordsPer?: number): string[];
|
|
123
|
+
/** Where a trace's two files live, given its directory and name. */
|
|
124
|
+
export declare function tracePaths(dir: string, name: string): {
|
|
125
|
+
index: string;
|
|
126
|
+
pcm: string;
|
|
127
|
+
};
|
|
128
|
+
export declare function writeTtsTrace(dir: string, name: string, trace: TtsTrace): Promise<void>;
|
|
129
|
+
/** Whether both halves of a trace are present, for a `skipIf` that announces. */
|
|
130
|
+
export declare function hasTtsTrace(dir: string, name: string): boolean;
|
|
131
|
+
export declare function readTtsTrace(dir: string, name: string): Promise<TtsTrace>;
|
|
132
|
+
/**
|
|
133
|
+
* The same read, synchronously.
|
|
134
|
+
*
|
|
135
|
+
* A `describe` body may not `await` — vitest collects it synchronously — so a
|
|
136
|
+
* suite whose every case shares one trace has no other way to load it once.
|
|
137
|
+
* Reading it per `test` instead would decode 375 KiB of PCM ten times over for
|
|
138
|
+
* a value that is immutable.
|
|
139
|
+
*/
|
|
140
|
+
export declare function readTtsTraceSync(dir: string, name: string): TtsTrace;
|
|
141
|
+
/** The PCM16 bytes of one frame, as the wire would carry them. */
|
|
142
|
+
export declare function frameBytes(trace: TtsTrace, frame: TraceFrame): Uint8Array;
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Test harness for AudioWorklet processor sources.
|
|
3
|
+
*
|
|
4
|
+
* The worklets ship as source strings (compiled to Blob URLs for the real
|
|
5
|
+
* AudioWorklet). This harness evaluates a source string with stubbed
|
|
6
|
+
* AudioWorkletGlobalScope globals so the processor's runtime behavior
|
|
7
|
+
* (batching, resampling, ring buffer) can be exercised directly in unit tests.
|
|
8
|
+
*/
|
|
9
|
+
export type WorkletPort = {
|
|
10
|
+
onmessage: ((e: {
|
|
11
|
+
data: unknown;
|
|
12
|
+
}) => void) | null;
|
|
13
|
+
postMessage(data: unknown, transfer?: unknown[]): void;
|
|
14
|
+
};
|
|
15
|
+
export type WorkletInstance = {
|
|
16
|
+
port: WorkletPort;
|
|
17
|
+
process(inputs: Float32Array[][], outputs: Float32Array[][]): boolean;
|
|
18
|
+
};
|
|
19
|
+
export type WorkletHarness = {
|
|
20
|
+
instance: WorkletInstance;
|
|
21
|
+
/** Messages the processor posted to the main thread, in order. */
|
|
22
|
+
posted: unknown[];
|
|
23
|
+
/** Deliver a message from the main thread to the processor. */
|
|
24
|
+
sendMessage(data: unknown): void;
|
|
25
|
+
};
|
|
26
|
+
/** Evaluate a worklet source string and instantiate its registered processor. */
|
|
27
|
+
export declare function instantiateWorklet(source: string, processorOptions?: Record<string, unknown>, contextSampleRate?: number): WorkletHarness;
|