@alexkroman1/aai-ui 6.1.0 → 6.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -33,6 +33,7 @@
33
33
  * followed from the same id either way.
34
34
  */
35
35
  import { type WorkflowSummary } from "@alexkroman1/aai";
36
+ import type { UploadProgress } from "@alexkroman1/aai/workflow-api";
36
37
  import type { WorkflowApi, WorkflowRun } from "./workflow-client.ts";
37
38
  /** Options for {@link useWorkflows}. */
38
39
  export type UseWorkflowsOptions = {
@@ -71,6 +72,23 @@ export type UseWorkflowsResult = {
71
72
  * @public
72
73
  */
73
74
  export declare function useWorkflows(opts?: UseWorkflowsOptions): UseWorkflowsResult;
75
+ /**
76
+ * What {@link WorkflowSubmission.upload} reports while the bytes are going.
77
+ *
78
+ * The SDK's per-request {@link UploadProgress} plus WHICH file it describes,
79
+ * because a form is allowed more than one and a bar over "the upload" would
80
+ * restart at zero partway through with nothing to say why.
81
+ *
82
+ * @public
83
+ */
84
+ export type UploadStatus = UploadProgress & {
85
+ /** The file being sent, by the name the picker gave it. */
86
+ name: string;
87
+ /** Which file of the submission this is, counting from 1. */
88
+ index: number;
89
+ /** How many files this submission sends in total. */
90
+ count: number;
91
+ };
74
92
  /** What {@link useWorkflowSubmit} returns. */
75
93
  export type WorkflowSubmission<R = unknown> = {
76
94
  /**
@@ -91,6 +109,19 @@ export type WorkflowSubmission<R = unknown> = {
91
109
  * already in flight.
92
110
  */
93
111
  pending: boolean;
112
+ /**
113
+ * How far the submission's files have got, while any are still going.
114
+ *
115
+ * Undefined before the first byte and again from the moment the last one
116
+ * lands, so a page can render `{upload && <UploadProgressBar upload={upload} />}`
117
+ * and the bar exists exactly for as long as there is an upload to describe. A
118
+ * form with no files never sets it at all.
119
+ *
120
+ * The wait it covers is the one `run` cannot: a run does not EXIST until its
121
+ * input is stored, so `pending` is true and there is nothing to poll — which
122
+ * for a 200 MB recording is minutes of a page that looks stuck.
123
+ */
124
+ upload: UploadStatus | undefined;
94
125
  /** The submit's own failure (a rejected input), or the watch's. */
95
126
  error: string | undefined;
96
127
  };
@@ -0,0 +1,181 @@
1
+ /**
2
+ * A bench for the playback jitter buffer: real TTS, the real worklet, a
3
+ * deterministic clock, and one setting changed at a time.
4
+ *
5
+ * Why this exists. `PLAYBACK_JITTER_MS` and `PLAYBACK_REFILL_MS` are the two
6
+ * numbers that decide whether a reply is heard as speech or as stutter, and
7
+ * every test that touches them supplies its own arrival pattern:
8
+ * `playback-processor.test.ts` hand-feeds quanta, and `audio-stress.test.ts`
9
+ * records in its own header that its chunk sizes outrun the render loop by an
10
+ * order of magnitude so the buffer "effectively never starves". Neither can
11
+ * answer "is 400 the right number", because neither has ever seen how audio
12
+ * actually arrives.
13
+ *
14
+ * So the bench replays a RECORDED reply (`_tts-trace-harness.ts`) through the
15
+ * chain the audio really crosses:
16
+ *
17
+ * provider frames (recorded arrival times)
18
+ * -> the server's bounded-lead pacer <- MODELLED, see below
19
+ * -> a network profile (latency, jitter, stalls)
20
+ * -> the real playback worklet <- `playbackProcessorSource`
21
+ * -> rendered PCM + the worklet's own stats
22
+ *
23
+ * Everything is virtual-time: the clock advances one 128-sample render quantum
24
+ * per iteration, so a sweep of a hundred settings over a 30-second reply costs
25
+ * seconds and gives byte-identical results every run. {@link renderToWav} is
26
+ * the other half — the rendered output written where a human can listen to it,
27
+ * because a concealment counter does not tell you whether a reply sounds
28
+ * broken.
29
+ *
30
+ * **The pacer is a MODEL, and that is this bench's one fidelity gap.**
31
+ * `createAudioPacer` (`aai/host/audio-pacer.ts`) is not on any published
32
+ * subpath and this package may not import a sibling's internals, so
33
+ * {@link pacedSends} is a transcription of its algorithm rather than the
34
+ * algorithm. It is ~20 lines and reproduces the two properties the pacer's own
35
+ * doc states — free flow until the lead ceiling, then one release per
36
+ * `burstMs` — but a change to the real pacer will NOT fail this file. If the
37
+ * pacer moves, re-read it against `pacedSends`. Exporting the real one and
38
+ * deleting the model is the fix; it needs a non-test change.
39
+ */
40
+ import { type TtsTrace } from "./_tts-trace-harness.ts";
41
+ /** The render quantum every AudioWorkletProcessor is called with. */
42
+ export declare const QUANTUM = 128;
43
+ /** Sampling interval of {@link RenderResult.earMs}. */
44
+ export declare const EAR_SAMPLE_MS = 20;
45
+ /** One frame arriving at the client: when it lands, and what it carries. */
46
+ export type Delivery = {
47
+ atMs: number;
48
+ bytes: Uint8Array;
49
+ };
50
+ /**
51
+ * How the server releases audio. Production values are
52
+ * `CLIENT_AUDIO_LEAD_MS` (1000) and `PACER_BURST_MS` (200); they are
53
+ * parameters here because they are half of what the bench is sweeping — the
54
+ * client's cushion is the server's lead, so tuning the jitter buffer without
55
+ * them is tuning one end of one number.
56
+ */
57
+ export type PacerProfile = {
58
+ leadMs: number;
59
+ burstMs: number;
60
+ };
61
+ /** What the link does to a frame between the server and the ear. */
62
+ export type NetworkProfile = {
63
+ name: string;
64
+ /** One-way latency floor, ms. */
65
+ latencyMs: number;
66
+ /** Peak extra delay added on top, ms (0 = a perfectly even link). */
67
+ jitterMs: number;
68
+ /**
69
+ * Freezes: `{ atMs, forMs }` pairs during which nothing is delivered, every
70
+ * held frame landing at the end. This is what a jitter buffer is FOR, and
71
+ * the only part of a profile that has to be scripted rather than sampled —
72
+ * a stall is an event, not a distribution.
73
+ */
74
+ stalls?: {
75
+ atMs: number;
76
+ forMs: number;
77
+ }[];
78
+ /**
79
+ * Throughput ceiling in bits/s, or `Infinity`. Both legs are uncompressed
80
+ * PCM16 (384 kbps down at 24 kHz), so a link that cannot carry the bitrate
81
+ * cannot be fixed by any buffer — that claim is in the aai-ui guide and this
82
+ * is what makes it measurable.
83
+ */
84
+ bitsPerSecond?: number;
85
+ };
86
+ /**
87
+ * The worklet's one knob. It was two — a startup target and a refill target —
88
+ * until this bench showed the startup one was redundant by construction; see
89
+ * `PLAYBACK_FILL_MS`.
90
+ */
91
+ export type PlaybackSettings = {
92
+ fillMs: number;
93
+ };
94
+ /**
95
+ * Model the server's bounded-lead pacer over a trace's frames.
96
+ *
97
+ * A transcription of `createAudioPacer`, whose loop is: a frame is sent
98
+ * immediately while the lead (everything sent so far, minus now) is under the
99
+ * ceiling; otherwise it waits until the lead has drained `burstMs` below the
100
+ * ceiling and then goes out with everything else the drain releases.
101
+ */
102
+ export declare function pacedSends(trace: TtsTrace, pacer: PacerProfile): Delivery[];
103
+ /** Apply a network profile to the server's send schedule. */
104
+ export declare function overNetwork(sends: Delivery[], net: NetworkProfile): Delivery[];
105
+ /** Everything one render of one setting produced. */
106
+ export type RenderResult = {
107
+ /** The audio the ear actually received, at the context sample rate. */
108
+ rendered: Float32Array;
109
+ sampleRate: number;
110
+ /** Ms from the first frame LEAVING the provider to the first audible sample. */
111
+ timeToFirstAudioMs: number;
112
+ /** The worklet's own report, as `onPlaybackStats` would receive it. */
113
+ stats: {
114
+ concealedSamples: number;
115
+ silentConcealedSamples: number;
116
+ concealmentEvents: number;
117
+ silentConcealmentEvents: number;
118
+ };
119
+ /** Every concealment episode's length in ms, in order. */
120
+ gapsMs: number[];
121
+ /** `bufferedMs` values the worklet reported, as `playback_progress` would. */
122
+ progressMs: number[];
123
+ /**
124
+ * Ground truth for the heard cursor: cumulative ms of the REPLY's own audio
125
+ * the ear had received, sampled every {@link EAR_SAMPLE_MS}.
126
+ *
127
+ * Concealed samples are excluded — they are fabricated, not reply audio — so
128
+ * this is exactly the quantity `heardMs()` in
129
+ * `aai/host/transports/pipeline-heard.ts` estimates, which is what makes the
130
+ * two comparable.
131
+ */
132
+ earMs: number[];
133
+ /** How long the reply took to play out, first sample to last. */
134
+ playedMs: number;
135
+ };
136
+ /**
137
+ * Render one delivery schedule through the real playback worklet.
138
+ *
139
+ * The clock is the render loop itself: quantum `q` happens at
140
+ * `q * QUANTUM / sampleRate` seconds, every delivery at or before that instant
141
+ * is written first, and `done` is posted once the last frame has landed. That
142
+ * is the ordering the audio thread really sees (`onmessage` and `process()`
143
+ * never interleave), and it makes the whole render a pure function of the
144
+ * schedule.
145
+ */
146
+ export declare function renderSchedule(deliveries: Delivery[], opts: {
147
+ sampleRate: number;
148
+ settings: PlaybackSettings;
149
+ maxSeconds?: number;
150
+ }): RenderResult;
151
+ /** One end-to-end run: trace + pacer + network + settings. */
152
+ export declare function runBench(opts: {
153
+ trace: TtsTrace;
154
+ pacer: PacerProfile;
155
+ net: NetworkProfile;
156
+ settings: PlaybackSettings;
157
+ }): RenderResult;
158
+ /**
159
+ * Score one render. Lower is better, and the WEIGHTS are the opinion in this
160
+ * file — everything above is measurement.
161
+ *
162
+ * The three terms are not interchangeable:
163
+ *
164
+ * - **Startup latency** is paid on every single turn, so it is the term that
165
+ * compounds over a conversation.
166
+ * - **Silent concealment** is an audible hole. Concealment that stays under
167
+ * the fade is a smear the ear largely forgives, which is why the worklet
168
+ * reports the two separately, so silence is weighted an order of magnitude
169
+ * higher than concealment in general.
170
+ * - **Episode COUNT** matters independently of total length: one 300 ms pause
171
+ * reads as a network hiccup, where six 50 ms ones read as a broken codec.
172
+ * That is the whole finding behind the refill re-arm, so a score that
173
+ * summed milliseconds alone would rank the failure it was built to prevent
174
+ * as equal to the fix.
175
+ */
176
+ export declare function scoreRender(r: RenderResult): {
177
+ score: number;
178
+ parts: Record<string, number>;
179
+ };
180
+ /** Minimal 16-bit PCM WAV, so a render can be listened to. */
181
+ export declare function toWav(samples: Float32Array, sampleRate: number): Buffer;
@@ -0,0 +1,63 @@
1
+ /**
2
+ * The bench's HOST-side half: what the server believes about the caller's
3
+ * playback, against what the caller actually heard.
4
+ *
5
+ * Split from `_playback-bench-harness.ts` at the seam it already had — that file
6
+ * models the wire and drives the worklet, and this one models
7
+ * `aai/host/transports/pipeline-heard.ts`'s arithmetic over the result. It went
8
+ * out when the file hit the 500-line cap, which is the right time rather than a
9
+ * line later.
10
+ *
11
+ * Everything here is a TRANSCRIPTION of host code this package may not import
12
+ * (`aai`'s `host/` is Node-only and internal), so it carries the same warning the
13
+ * pacer model does: a change to `createPlaybackClock` will not fail this file.
14
+ * Re-read it against these functions if that clock moves.
15
+ */
16
+ import type { Delivery, RenderResult } from "./_playback-bench-harness.ts";
17
+ /**
18
+ * What the HOST believes about playback, and what actually happened.
19
+ *
20
+ * The barge-in floor is `pending()` in `aai/host/transports/pipeline-heard.ts`:
21
+ * `now() < endsAtMs + PIPELINE_PLAYBACK_GRACE_MS`, where `endsAtMs` accumulates
22
+ * each forwarded chunk's duration from `max(endsAtMs, now())` — an OPEN-LOOP
23
+ * model that assumes playback starts the instant a chunk is forwarded and runs at
24
+ * exactly 1.0x. No real client beats that, so the model is a lower bound and the
25
+ * grace is what covers the difference.
26
+ *
27
+ * This measures the difference. `requiredGraceMs` is how long after the host's
28
+ * estimate the caller was still hearing audio — i.e. the smallest grace that
29
+ * keeps barge-in working to the end of a reply. It is a REQUIREMENT, so a grace
30
+ * at or above it is correct and one below it means a barge-in in the reply's tail
31
+ * is not recognised as arriving during playback.
32
+ *
33
+ * Two clients are modelled because the host serves both: one that wires
34
+ * `onPlaybackProgress` (the browser does) and one that does not (a telephony
35
+ * bridge, a harness). The reports clamp `endsAtMs` UPWARD only, so a reporting
36
+ * client shrinks the requirement and a silent one leaves the host on the
37
+ * open-loop estimate — which is the case the constant has to be safe for.
38
+ */
39
+ export type PlayoutVsHost = {
40
+ /** When the host's open-loop model thinks forwarded audio finishes playing. */
41
+ openLoopEndMs: number;
42
+ /** The same, with the client's `playback_progress` reports clamped in. */
43
+ reportedEndMs: number;
44
+ /** When the caller actually stopped hearing audio. */
45
+ realEndMs: number;
46
+ /** Smallest grace that keeps `pending()` true to `realEndMs`, unreported. */
47
+ requiredGraceMs: number;
48
+ /** The same for a client that DOES report its backlog. */
49
+ requiredGraceReportingMs: number;
50
+ };
51
+ /**
52
+ * Replay a render against the host's own playback-clock arithmetic.
53
+ *
54
+ * `deliveries` is what the host FORWARDED (the pacer's output, before the link),
55
+ * because that is what `onChunk` sees — the host has no view of the wire.
56
+ */
57
+ export declare function playoutVsHost(opts: {
58
+ forwarded: Delivery[];
59
+ render: RenderResult;
60
+ sampleRate: number;
61
+ /** Interval of the client's backlog reports; production is 500 ms. */
62
+ reportIntervalMs: number;
63
+ }): PlayoutVsHost;
@@ -0,0 +1,65 @@
1
+ /**
2
+ * The browser half of the playback bench: a page that plays a recorded reply
3
+ * through the REAL playback worklet in a REAL `AudioContext`, so the tuning can
4
+ * be listened to rather than only counted.
5
+ *
6
+ * This is the part the offline renderer cannot be: `renderSchedule` drives
7
+ * `process()` on a synthetic sample clock with `postMessage` delivered exactly
8
+ * between quanta, which is the ordering the audio thread guarantees but says
9
+ * nothing about the browser's own render cadence, its output latency, or what
10
+ * a `port.postMessage` burst does to a live audio callback. If the two agree,
11
+ * the offline sweep is trustworthy; where they disagree, the browser is right.
12
+ *
13
+ * Two ways in, same page:
14
+ *
15
+ * - **A human opens it** and gets a slider for `fillMs`, the
16
+ * network profile, and the pacer lead, with the worklet's own concealment
17
+ * counters live. Audio comes out of the speakers.
18
+ * - **Playwright drives it** through {@link BENCH_API} on `window`, and pulls
19
+ * back the samples the tap captured — the audio that actually reached the
20
+ * destination — so a test can diff it against the offline render.
21
+ *
22
+ * The page is a STRING rather than a file under a bundler because it must stay
23
+ * test-only: aai-ui's build (`tsdown` + `build-default-client.ts`) ships
24
+ * `dist/default-client/`, and a second HTML entry there is a product artifact
25
+ * with a coverage floor and a size budget. Generated into `reports/playback/`,
26
+ * it is neither.
27
+ */
28
+ /** The name of the object the page exposes for a driver to call. */
29
+ export declare const BENCH_API = "__aaiPlaybackBench";
30
+ /** Options a driver or a human can set on one run. */
31
+ export type BenchRunOptions = {
32
+ fillMs: number;
33
+ /** Deliveries: ms since run start, and the slice of the PCM to write. */
34
+ schedule: {
35
+ atMs: number;
36
+ offset: number;
37
+ length: number;
38
+ }[];
39
+ /** Whether the reply is audible. A sweep run in a headless browser is not. */
40
+ muted?: boolean;
41
+ };
42
+ /**
43
+ * Build the bench page.
44
+ *
45
+ * `pcmUrl` is fetched once and sliced per delivery, so the page replays the
46
+ * same bytes the offline renderer does. `sampleRate` must be the trace's: the
47
+ * context is created at it and the page REFUSES to run if the browser grants
48
+ * another, exactly as `audio.ts` does — PCM written into a context at the wrong
49
+ * rate plays at the wrong speed, which would look like a tuning result.
50
+ */
51
+ export declare function benchPageHtml(opts: {
52
+ pcmUrl: string;
53
+ sampleRate: number;
54
+ /** Shown in the header, so a saved page names what it is playing. */
55
+ title: string;
56
+ /** Default schedules a human can pick between, by profile name. */
57
+ profiles: Record<string, {
58
+ atMs: number;
59
+ offset: number;
60
+ length: number;
61
+ }[]>;
62
+ defaults?: {
63
+ fillMs: number;
64
+ };
65
+ }): string;
@@ -0,0 +1,142 @@
1
+ /**
2
+ * Capture and replay of REAL TTS arrival timing.
3
+ *
4
+ * The playback worklet's tuning question — how deep should
5
+ * `PLAYBACK_JITTER_MS` be — is entirely a question about how unevenly audio
6
+ * arrives, and every existing test answers it with a generated arrival pattern.
7
+ * `audio-stress.test.ts` says so itself: its chunk-size arbitrary averages
8
+ * ~750 samples against 128 consumed per render, so writes outrun renders by an
9
+ * order of magnitude and the buffer effectively never starves. A jitter buffer
10
+ * tuned against that is tuned against nothing.
11
+ *
12
+ * So a trace is a recording of one real reply as the provider actually emitted
13
+ * it: PCM16 bytes plus the millisecond each frame ARRIVED, relative to the
14
+ * first. Replayed through the real pacer and the real worklet, that makes the
15
+ * tuning question measurable and repeatable — the same reply, the same
16
+ * arrival pattern, one setting changed.
17
+ *
18
+ * Two halves, deliberately separate:
19
+ *
20
+ * - {@link captureTtsTrace} needs a live provider and an API key. It runs once,
21
+ * by hand (`AAI_CAPTURE_TTS_TRACE=1`), and commits its result.
22
+ * - {@link readTtsTrace} needs neither, so every test that CONSUMES a trace is
23
+ * keyless and offline.
24
+ *
25
+ * The bytes live beside the JSON rather than inside it: base64 in a fixture
26
+ * inflates ~1 MB of PCM by a third and makes the file unreadable in a diff,
27
+ * where a `.pcm` sidecar is `ffplay`-able and reviewed by its length alone.
28
+ */
29
+ /** One provider audio frame, with the moment it arrived. */
30
+ export type TraceFrame = {
31
+ /** Arrival time in ms, relative to the first frame of the reply. */
32
+ tMs: number;
33
+ /** Byte offset of this frame's PCM16 in the sidecar. */
34
+ offset: number;
35
+ /** Byte length of this frame's PCM16. */
36
+ length: number;
37
+ };
38
+ /** A recorded reply: what the provider sent, and when. */
39
+ export type TtsTrace = {
40
+ /** Sample rate of the PCM16 in the sidecar. */
41
+ sampleRate: number;
42
+ /** Provider kind and voice, so a trace names the thing it recorded. */
43
+ provider: string;
44
+ voice: string;
45
+ /** The reply text that was synthesized. */
46
+ text: string;
47
+ /** Ms from the first `sendText` to the first audio frame. */
48
+ firstAudioMs: number;
49
+ /** Ms from the first `sendText` to `done`. */
50
+ doneMs: number;
51
+ frames: TraceFrame[];
52
+ /** All frames' PCM16, concatenated in arrival order. */
53
+ pcm: Int16Array;
54
+ };
55
+ /**
56
+ * Total audio duration in ms. Not the same as {@link TtsTrace.doneMs}: a
57
+ * provider that synthesizes faster than real time produces more audio than the
58
+ * wall clock it took to produce it, and the RATIO of the two is the whole
59
+ * reason a jitter buffer can ever fill.
60
+ */
61
+ export declare function traceAudioMs(trace: TtsTrace): number;
62
+ /**
63
+ * The slice of the SDK's `TtsSession` a capture drives. Declared structurally
64
+ * so the harness needs no host-only type import: a real session satisfies it.
65
+ */
66
+ export type CapturableTtsSession = {
67
+ sendText(text: string): void;
68
+ flush(): void;
69
+ on(event: "audio", fn: (pcm: Int16Array) => void): unknown;
70
+ on(event: "done", fn: () => void): unknown;
71
+ on(event: "error", fn: (err: {
72
+ message?: string;
73
+ }) => void): unknown;
74
+ close(): Promise<void>;
75
+ };
76
+ /**
77
+ * Open a real TTS session, synthesize `text`, and record every frame's arrival.
78
+ *
79
+ * The text is sent the way the PIPELINE sends it — as deltas, with a final
80
+ * `flush()` — because the AssemblyAI adapter segments on what it receives and
81
+ * `Generate` only buffers until a `Flush`. Handing it the whole reply in one
82
+ * call would record a different arrival pattern from the one a real turn
83
+ * produces (see `providers/tts/assemblyai-segment.ts`).
84
+ */
85
+ export declare function captureTtsTrace(opts: {
86
+ text: string;
87
+ /**
88
+ * Opens the real provider session. INJECTED rather than resolved here: the
89
+ * resolver (`aai`'s `host/providers/resolve.ts`) is not on any published
90
+ * subpath, and this package may not import a sibling's internals. The caller
91
+ * that has one is the capture runner, which is not part of the package.
92
+ */
93
+ open: (o: {
94
+ sampleRate: number;
95
+ signal: AbortSignal;
96
+ }) => Promise<CapturableTtsSession>;
97
+ /** LLM-shaped deltas. Defaults to splitting `text` on word boundaries. */
98
+ deltas?: string[];
99
+ /** Recorded into the trace so a fixture names the voice it captured. */
100
+ voice?: string;
101
+ provider?: string;
102
+ sampleRate?: number;
103
+ /**
104
+ * Gap between deltas, in ms. **Set this to represent a real turn.**
105
+ *
106
+ * Sending every delta in one burst is what a capture does by default, and for
107
+ * the PLAYBACK question that is harmless — the pacer reshapes arrival anyway.
108
+ * For anything about SEGMENTATION it invalidates the measurement outright: the
109
+ * segmenter is handed the whole reply before it makes its first cut, so
110
+ * time-to-first-audio collapses to the service's own latency (~40 ms measured)
111
+ * and every segmentation rule scores the same. An LLM streams at ~30 ms a
112
+ * delta, which is what the segmenter really sees.
113
+ */
114
+ deltaIntervalMs?: number;
115
+ /** Hard cap, so a provider that never sends `done` cannot hang the capture. */
116
+ timeoutMs?: number;
117
+ }): Promise<TtsTrace>;
118
+ /**
119
+ * Split a reply into LLM-shaped deltas: a few words at a time, which is what
120
+ * `streamText` emits and therefore what the adapter's segmenter sees.
121
+ */
122
+ export declare function splitIntoDeltas(text: string, wordsPer?: number): string[];
123
+ /** Where a trace's two files live, given its directory and name. */
124
+ export declare function tracePaths(dir: string, name: string): {
125
+ index: string;
126
+ pcm: string;
127
+ };
128
+ export declare function writeTtsTrace(dir: string, name: string, trace: TtsTrace): Promise<void>;
129
+ /** Whether both halves of a trace are present, for a `skipIf` that announces. */
130
+ export declare function hasTtsTrace(dir: string, name: string): boolean;
131
+ export declare function readTtsTrace(dir: string, name: string): Promise<TtsTrace>;
132
+ /**
133
+ * The same read, synchronously.
134
+ *
135
+ * A `describe` body may not `await` — vitest collects it synchronously — so a
136
+ * suite whose every case shares one trace has no other way to load it once.
137
+ * Reading it per `test` instead would decode 375 KiB of PCM ten times over for
138
+ * a value that is immutable.
139
+ */
140
+ export declare function readTtsTraceSync(dir: string, name: string): TtsTrace;
141
+ /** The PCM16 bytes of one frame, as the wire would carry them. */
142
+ export declare function frameBytes(trace: TtsTrace, frame: TraceFrame): Uint8Array;
@@ -0,0 +1,27 @@
1
+ /**
2
+ * Test harness for AudioWorklet processor sources.
3
+ *
4
+ * The worklets ship as source strings (compiled to Blob URLs for the real
5
+ * AudioWorklet). This harness evaluates a source string with stubbed
6
+ * AudioWorkletGlobalScope globals so the processor's runtime behavior
7
+ * (batching, resampling, ring buffer) can be exercised directly in unit tests.
8
+ */
9
+ export type WorkletPort = {
10
+ onmessage: ((e: {
11
+ data: unknown;
12
+ }) => void) | null;
13
+ postMessage(data: unknown, transfer?: unknown[]): void;
14
+ };
15
+ export type WorkletInstance = {
16
+ port: WorkletPort;
17
+ process(inputs: Float32Array[][], outputs: Float32Array[][]): boolean;
18
+ };
19
+ export type WorkletHarness = {
20
+ instance: WorkletInstance;
21
+ /** Messages the processor posted to the main thread, in order. */
22
+ posted: unknown[];
23
+ /** Deliver a message from the main thread to the processor. */
24
+ sendMessage(data: unknown): void;
25
+ };
26
+ /** Evaluate a worklet source string and instantiate its registered processor. */
27
+ export declare function instantiateWorklet(source: string, processorOptions?: Record<string, unknown>, contextSampleRate?: number): WorkletHarness;
@@ -1,4 +1,4 @@
1
1
  /** Raw worklet source — exported so tests can evaluate the processor directly. */
2
- export declare const playbackProcessorSource = "\nclass PlaybackProcessor extends AudioWorkletProcessor {\n constructor(options) {\n super();\n const opts = options.processorOptions || {};\n // The node's context IS the playback context (audio.ts asserts its rate),\n // so the worklet-global sampleRate is authoritative; the option exists\n // for the node-less test harness.\n const rate = opts.sampleRate ?? sampleRate;\n // Kept, because every derived quantity below reads it and `bufferedMs` in\n // process() is the one that used to read the worklet global instead. Those\n // agree in production (the option is unset, so `rate` IS `sampleRate`) and\n // diverge only in the node-less harness \u2014 which is the one place a spec\n // could ever assert on the reported backlog, so the divergence made the\n // number untestable rather than wrong.\n this.rate = rate;\n // Fill target for the start of a turn. If 'done' arrives first (short\n // utterance), start immediately instead of waiting for audio that is\n // never coming.\n this.jitterSamples = Math.floor((rate * (opts.jitterMs ?? 400)) / 1000);\n // Fill target after an underrun \u2014 see PLAYBACK_REFILL_MS.\n this.refillSamples = Math.floor((rate * (opts.refillMs ?? 200)) / 1000);\n // Concealment source: a ring of the most recently played samples, looped\n // under a decaying gain to cover a gap. Sized to the fade window, with a\n // per-sample decay that reaches the floor exactly at its end.\n this.concealCapacity = Math.max(1, Math.floor((rate * 40) / 1000));\n this.concealBuf = new Float32Array(this.concealCapacity);\n this.concealDecay = Math.exp(Math.log(0.001) / this.concealCapacity);\n // Float32 ring buffer \u2014 PLAYBACK_BUFFER_SECONDS at the context sample\n // rate. Allocated once for the node's lifetime; per-turn state resets via\n // resetTurn(). writePos and readPos are absolute (monotonic) sample\n // counts; the buffer is indexed modulo capacity so a longer reply keeps\n // playing instead of writing past the end and going silent.\n this.capacity = rate * 60;\n this.samples = new Float32Array(this.capacity);\n // Cadence for the 'progress' report in process(), counted in samples so\n // the hot path never reads a clock. Survives resetTurn(): the host clamps\n // upward only, so a stale count costs at most one extra report.\n this.reportIntervalSamples = Math.floor((rate * 500) / 1000);\n this.sinceReportSamples = 0;\n // Platform endianness probe: the wire format is PCM16 little-endian, so\n // the Int16Array fast path in ingestBytes is only valid on LE hosts\n // (every shipping browser target; the DataView path is the fallback).\n this.littleEndian = new Uint8Array(new Uint16Array([1]).buffer)[0] === 1;\n this.resetTurn();\n\n this.port.onmessage = (e) => {\n const d = e.data;\n if (d.event === 'write') {\n this.ingestBytes(d.buffer);\n } else if (d.event === 'interrupt') {\n // Applied eagerly, not deferred to the next process(): onmessage and\n // process() never interleave (one audio thread), and a deferred\n // interrupt would let 'write'/'done' frames for the NEXT turn\n // coalesce in ahead of it and be wiped by the reset along with the\n // cancelled turn's audio.\n this.stopTurn('interrupt');\n } else if (d.event === 'done') {\n this.isDone = true;\n // Echoed back on this turn's 'stop' so the host can tell WHICH turn\n // drained: a stop already in flight when a barge-in lands would\n // otherwise settle the next turn's done() the moment it arrives.\n this.doneTurn = d.turn ?? null;\n }\n };\n }\n\n // Reset per-turn state so the node is reusable across replies without\n // reallocating the sample buffer or re-instantiating the worklet.\n resetTurn() {\n this.isDone = false;\n // Host turn id carried by this turn's 'done' (null until one arrives).\n this.doneTurn = null;\n this.playing = false;\n // Whether any real audio has been rendered this turn. Separates a turn's\n // pre-roll (nothing to extrapolate from, and not a defect) from a\n // mid-turn underrun.\n this.hasPlayed = false;\n this.fillTarget = this.jitterSamples;\n // Carry-over byte for split samples across chunks\n this.carry = null;\n this.writePos = 0;\n this.readPos = 0;\n // Concealment ring state and the current fade position.\n this.concealLen = 0;\n this.concealWrite = 0;\n this.concealPos = 0;\n this.concealGain = 1;\n // Episode flags, so a multi-quantum gap counts as one event.\n this.concealing = false;\n this.concealedSilence = false;\n // Reported to the host on 'stop'. A fresh object per turn: the one just\n // posted must not be mutated by the next turn.\n this.stats = {\n concealedSamples: 0,\n silentConcealedSamples: 0,\n concealmentEvents: 0,\n silentConcealmentEvents: 0,\n };\n }\n\n // End the current turn: notify the host and rearm for the next reply.\n // Must NOT return false from process() \u2014 a processor that stops is dead\n // for good, forcing a new node (and buffer) per reply.\n // `reason` ('interrupt' | 'done') tells the host which turn boundary this\n // stop belongs to \u2014 interrupt-stops are dropped host-side, flush() having\n // already settled that turn \u2014 and `turn` names WHICH turn drained (the id\n // this turn's 'done' carried), so a drain-stop still in flight when a\n // barge-in lands cannot settle the next turn's done() early.\n stopTurn(reason) {\n this.port.postMessage({ event: 'stop', reason, turn: this.doneTurn, stats: this.stats });\n this.resetTurn();\n }\n\n // Cover a quantum (from `start`) where real audio should have been.\n //\n // Before the turn's first samples there is nothing to extrapolate from, so\n // the gap is plain silence and counted as nothing \u2014 WebRTC likewise only\n // counts concealment once playout has begun. After that, loop the retained\n // tail under a decaying gain: a hard zero-fill is a discontinuity mid-word,\n // which is the click that makes a brief stall sound like breakage.\n coverGap(out, start) {\n if (!this.hasPlayed) {\n out.fill(0, start);\n return;\n }\n if (!this.concealing) {\n this.concealing = true;\n this.concealedSilence = false;\n this.stats.concealmentEvents++;\n }\n const total = out.length - start;\n const len = this.concealLen;\n let silent = 0;\n // Second condition: once the fade has decayed to the floor, every sample\n // the loop would emit is 0 anyway \u2014 bulk-fill instead of running 128\n // branchy iterations per quantum for the whole tail of a long stall.\n if (len === 0 || this.concealGain < 0.001) {\n out.fill(0, start);\n silent = total;\n } else {\n let g = this.concealGain;\n for (let i = start; i < out.length; i++) {\n if (g < 0.001) {\n // The fade has run out: keep counting the gap, but stop looping a\n // fragment that is now inaudible anyway.\n out[i] = 0;\n silent++;\n continue;\n }\n out[i] = this.concealBuf[this.concealPos] * g;\n this.concealPos = this.concealPos + 1 === len ? 0 : this.concealPos + 1;\n g *= this.concealDecay;\n }\n this.concealGain = g;\n }\n this.stats.concealedSamples += total;\n if (silent > 0) {\n this.stats.silentConcealedSamples += silent;\n if (!this.concealedSilence) {\n this.concealedSilence = true;\n this.stats.silentConcealmentEvents++;\n }\n }\n }\n\n // Retain the tail of a rendered quantum as the next gap's concealment\n // source, and close any episode the real audio just ended.\n rememberTail(out, n) {\n const cap = this.concealCapacity;\n const take = Math.min(n, cap);\n // Bulk copies (this runs on every cleanly rendered quantum): the tail is\n // one contiguous source run, landing in at most two ring runs.\n const tail = out.subarray(n - take, n);\n const first = Math.min(take, cap - this.concealWrite);\n this.concealBuf.set(tail.subarray(0, first), this.concealWrite);\n if (take > first) this.concealBuf.set(tail.subarray(first), 0);\n this.concealWrite = (this.concealWrite + take) % cap;\n this.concealLen = Math.min(cap, this.concealLen + take);\n // Read the loop oldest-first; once the ring is full the write cursor is\n // the oldest retained sample.\n this.concealPos = this.concealLen === cap ? this.concealWrite : 0;\n this.concealing = false;\n this.concealGain = 1;\n }\n\n ingestBytes(uint8) {\n let bytes = uint8;\n\n if (this.carry !== null) {\n const merged = new Uint8Array(1 + bytes.length);\n merged[0] = this.carry;\n merged.set(bytes, 1);\n bytes = merged;\n this.carry = null;\n }\n\n if (bytes.length % 2 !== 0) {\n this.carry = bytes[bytes.length - 1];\n bytes = bytes.subarray(0, bytes.length - 1);\n }\n\n if (bytes.length === 0) return;\n const numSamples = bytes.length / 2;\n const cap = this.capacity;\n const samples = this.samples;\n if (this.littleEndian && (bytes.byteOffset & 1) === 0) {\n // Fast path: 2-byte-aligned LE bytes wrap directly as an Int16Array;\n // copy wrap-aware in at most two runs with no per-sample DataView call\n // or modulo. This runs on the realtime audio thread.\n const int16 = new Int16Array(bytes.buffer, bytes.byteOffset, numSamples);\n let src = 0;\n let dst = this.writePos % cap;\n while (src < numSamples) {\n const run = Math.min(numSamples - src, cap - dst);\n for (let j = 0; j < run; j++) {\n samples[dst + j] = int16[src + j] / 0x8000;\n }\n src += run;\n dst = 0;\n }\n } else {\n // Odd byte offset (or big-endian host): fall back to per-sample reads.\n const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.length);\n let dst = this.writePos % cap;\n for (let i = 0; i < numSamples; i++) {\n samples[dst] = view.getInt16(i * 2, true) / 0x8000;\n dst++;\n if (dst === cap) dst = 0;\n }\n }\n this.writePos += numSamples;\n // If the producer outran the consumer by more than the buffer holds, drop\n // the oldest unplayed audio rather than reading samples we've overwritten.\n if (this.writePos - this.readPos > this.capacity) {\n this.readPos = this.writePos - this.capacity;\n }\n }\n\n process(inputs, outputs) {\n // No output wired up yet \u2014 nothing to render this quantum. Throwing here\n // would permanently kill the processor (the node is persistent per\n // session), so guard like the capture processor does.\n if (!outputs[0] || !outputs[0][0]) return true;\n const out = outputs[0][0];\n const avail = this.writePos - this.readPos;\n\n // Report the unplayed backlog to the host (`playback_progress`). This IS\n // the closed-loop signal: the host otherwise models playback open-loop \u2014\n // every forwarded chunk assumed to start playing on arrival at exactly\n // 1.0x \u2014 and cannot see a buffer that has run ahead of the wall clock.\n // Measured against a client draining at 0.6x, the host declared the line\n // silent while it still held seconds of the reply, and then opened the\n // speaking-edge gate over speech the caller had not heard.\n //\n // Sent from process() rather than a timer because this is the only place\n // that sees the buffer on the audio thread; the counter is in QUANTA, so\n // it costs an integer compare per render and no clock read. Silence is not\n // reported: an empty buffer is what the host already assumes.\n this.sinceReportSamples += out.length;\n if (this.sinceReportSamples >= this.reportIntervalSamples) {\n this.sinceReportSamples = 0;\n if (avail > 0) {\n this.port.postMessage({ event: 'progress', bufferedMs: (avail / this.rate) * 1000 });\n }\n }\n\n // Filling: wait for the target. 'done' short-circuits it \u2014 what is\n // buffered is all there will be, so there is nothing left to wait for.\n if (!this.playing) {\n if (avail >= this.fillTarget || this.isDone) {\n this.playing = true;\n } else {\n this.coverGap(out, 0);\n return true;\n }\n }\n\n // Underrun: this quantum cannot be filled and more audio is still coming.\n // Go back to filling (at the refill target) and cover the gap, leaving\n // readPos untouched \u2014 the fragment stays buffered and plays intact once\n // the buffer recovers, instead of being dribbled out a few samples at a\n // time for the rest of the turn.\n if (avail < out.length && !this.isDone) {\n this.playing = false;\n this.fillTarget = this.refillSamples;\n this.coverGap(out, 0);\n return true;\n }\n\n if (avail > 0) {\n const n = Math.min(avail, out.length);\n // Copy from the ring buffer, splitting across the wrap boundary.\n const start = this.readPos % this.capacity;\n const first = Math.min(n, this.capacity - start);\n out.set(this.samples.subarray(start, start + first), 0);\n if (n > first) out.set(this.samples.subarray(0, n - first), first);\n this.readPos += n;\n // Only reachable with n < out.length on the turn's final partial\n // quantum (the underrun branch above catches every other case).\n out.fill(0, n);\n this.hasPlayed = true;\n this.rememberTail(out, n);\n return true;\n }\n\n // Drained and done: end the turn. Not reachable mid-turn \u2014 an empty\n // buffer with audio still coming is the underrun branch above.\n out.fill(0);\n if (this.isDone) {\n this.stopTurn('done');\n }\n return true;\n }\n}\n\nregisterProcessor('playback-processor', PlaybackProcessor);\n";
2
+ export declare const playbackProcessorSource = "\nclass PlaybackProcessor extends AudioWorkletProcessor {\n constructor(options) {\n super();\n const opts = options.processorOptions || {};\n // The node's context IS the playback context (audio.ts asserts its rate),\n // so the worklet-global sampleRate is authoritative; the option exists\n // for the node-less test harness.\n const rate = opts.sampleRate ?? sampleRate;\n // Kept, because every derived quantity below reads it and `bufferedMs` in\n // process() is the one that used to read the worklet global instead. Those\n // agree in production (the option is unset, so `rate` IS `sampleRate`) and\n // diverge only in the node-less harness \u2014 which is the one place a spec\n // could ever assert on the reported backlog, so the divergence made the\n // number untestable rather than wrong.\n this.rate = rate;\n // The one fill target \u2014 for the start of a turn and for recovery after an\n // underrun alike. If 'done' arrives first (short utterance), start\n // immediately instead of waiting for audio that is never coming.\n this.fillSamples = Math.floor((rate * (opts.fillMs ?? 200)) / 1000);\n // Concealment source: a ring of the most recently played samples, looped\n // under a decaying gain to cover a gap. Sized to the fade window, with a\n // per-sample decay that reaches the floor exactly at its end.\n this.concealCapacity = Math.max(1, Math.floor((rate * 40) / 1000));\n this.concealBuf = new Float32Array(this.concealCapacity);\n this.concealDecay = Math.exp(Math.log(0.001) / this.concealCapacity);\n // Float32 ring buffer \u2014 PLAYBACK_BUFFER_SECONDS at the context sample\n // rate. Allocated once for the node's lifetime; per-turn state resets via\n // resetTurn(). writePos and readPos are absolute (monotonic) sample\n // counts; the buffer is indexed modulo capacity so a longer reply keeps\n // playing instead of writing past the end and going silent.\n this.capacity = rate * 60;\n this.samples = new Float32Array(this.capacity);\n // Cadence for the 'progress' report in process(), counted in samples so\n // the hot path never reads a clock. Survives resetTurn(): the host clamps\n // upward only, so a stale count costs at most one extra report.\n this.reportIntervalSamples = Math.floor((rate * 500) / 1000);\n this.sinceReportSamples = 0;\n // Platform endianness probe: the wire format is PCM16 little-endian, so\n // the Int16Array fast path in ingestBytes is only valid on LE hosts\n // (every shipping browser target; the DataView path is the fallback).\n this.littleEndian = new Uint8Array(new Uint16Array([1]).buffer)[0] === 1;\n this.resetTurn();\n\n this.port.onmessage = (e) => {\n const d = e.data;\n if (d.event === 'write') {\n this.ingestBytes(d.buffer);\n } else if (d.event === 'interrupt') {\n // Applied eagerly, not deferred to the next process(): onmessage and\n // process() never interleave (one audio thread), and a deferred\n // interrupt would let 'write'/'done' frames for the NEXT turn\n // coalesce in ahead of it and be wiped by the reset along with the\n // cancelled turn's audio.\n this.stopTurn('interrupt');\n } else if (d.event === 'done') {\n this.isDone = true;\n // Echoed back on this turn's 'stop' so the host can tell WHICH turn\n // drained: a stop already in flight when a barge-in lands would\n // otherwise settle the next turn's done() the moment it arrives.\n this.doneTurn = d.turn ?? null;\n }\n };\n }\n\n // Reset per-turn state so the node is reusable across replies without\n // reallocating the sample buffer or re-instantiating the worklet.\n resetTurn() {\n this.isDone = false;\n // Host turn id carried by this turn's 'done' (null until one arrives).\n this.doneTurn = null;\n this.playing = false;\n // Whether any real audio has been rendered this turn. Separates a turn's\n // pre-roll (nothing to extrapolate from, and not a defect) from a\n // mid-turn underrun.\n this.hasPlayed = false;\n // Carry-over byte for split samples across chunks\n this.carry = null;\n this.writePos = 0;\n this.readPos = 0;\n // Concealment ring state and the current fade position.\n this.concealLen = 0;\n this.concealWrite = 0;\n this.concealPos = 0;\n this.concealGain = 1;\n // Episode flags, so a multi-quantum gap counts as one event.\n this.concealing = false;\n this.concealedSilence = false;\n // Reported to the host on 'stop'. A fresh object per turn: the one just\n // posted must not be mutated by the next turn.\n this.stats = {\n concealedSamples: 0,\n silentConcealedSamples: 0,\n concealmentEvents: 0,\n silentConcealmentEvents: 0,\n };\n }\n\n // End the current turn: notify the host and rearm for the next reply.\n // Must NOT return false from process() \u2014 a processor that stops is dead\n // for good, forcing a new node (and buffer) per reply.\n // `reason` ('interrupt' | 'done') tells the host which turn boundary this\n // stop belongs to \u2014 interrupt-stops are dropped host-side, flush() having\n // already settled that turn \u2014 and `turn` names WHICH turn drained (the id\n // this turn's 'done' carried), so a drain-stop still in flight when a\n // barge-in lands cannot settle the next turn's done() early.\n stopTurn(reason) {\n this.port.postMessage({ event: 'stop', reason, turn: this.doneTurn, stats: this.stats });\n this.resetTurn();\n }\n\n // Cover a quantum (from `start`) where real audio should have been.\n //\n // Before the turn's first samples there is nothing to extrapolate from, so\n // the gap is plain silence and counted as nothing \u2014 WebRTC likewise only\n // counts concealment once playout has begun. After that, loop the retained\n // tail under a decaying gain: a hard zero-fill is a discontinuity mid-word,\n // which is the click that makes a brief stall sound like breakage.\n coverGap(out, start) {\n if (!this.hasPlayed) {\n out.fill(0, start);\n return;\n }\n if (!this.concealing) {\n this.concealing = true;\n this.concealedSilence = false;\n this.stats.concealmentEvents++;\n }\n const total = out.length - start;\n const len = this.concealLen;\n let silent = 0;\n // Second condition: once the fade has decayed to the floor, every sample\n // the loop would emit is 0 anyway \u2014 bulk-fill instead of running 128\n // branchy iterations per quantum for the whole tail of a long stall.\n if (len === 0 || this.concealGain < 0.001) {\n out.fill(0, start);\n silent = total;\n } else {\n let g = this.concealGain;\n for (let i = start; i < out.length; i++) {\n if (g < 0.001) {\n // The fade has run out: keep counting the gap, but stop looping a\n // fragment that is now inaudible anyway.\n out[i] = 0;\n silent++;\n continue;\n }\n out[i] = this.concealBuf[this.concealPos] * g;\n this.concealPos = this.concealPos + 1 === len ? 0 : this.concealPos + 1;\n g *= this.concealDecay;\n }\n this.concealGain = g;\n }\n this.stats.concealedSamples += total;\n if (silent > 0) {\n this.stats.silentConcealedSamples += silent;\n if (!this.concealedSilence) {\n this.concealedSilence = true;\n this.stats.silentConcealmentEvents++;\n }\n }\n }\n\n // Retain the tail of a rendered quantum as the next gap's concealment\n // source, and close any episode the real audio just ended.\n rememberTail(out, n) {\n const cap = this.concealCapacity;\n const take = Math.min(n, cap);\n // Bulk copies (this runs on every cleanly rendered quantum): the tail is\n // one contiguous source run, landing in at most two ring runs.\n const tail = out.subarray(n - take, n);\n const first = Math.min(take, cap - this.concealWrite);\n this.concealBuf.set(tail.subarray(0, first), this.concealWrite);\n if (take > first) this.concealBuf.set(tail.subarray(first), 0);\n this.concealWrite = (this.concealWrite + take) % cap;\n this.concealLen = Math.min(cap, this.concealLen + take);\n // Read the loop oldest-first; once the ring is full the write cursor is\n // the oldest retained sample.\n this.concealPos = this.concealLen === cap ? this.concealWrite : 0;\n this.concealing = false;\n this.concealGain = 1;\n }\n\n ingestBytes(uint8) {\n let bytes = uint8;\n\n if (this.carry !== null) {\n const merged = new Uint8Array(1 + bytes.length);\n merged[0] = this.carry;\n merged.set(bytes, 1);\n bytes = merged;\n this.carry = null;\n }\n\n if (bytes.length % 2 !== 0) {\n this.carry = bytes[bytes.length - 1];\n bytes = bytes.subarray(0, bytes.length - 1);\n }\n\n if (bytes.length === 0) return;\n const numSamples = bytes.length / 2;\n const cap = this.capacity;\n const samples = this.samples;\n if (this.littleEndian && (bytes.byteOffset & 1) === 0) {\n // Fast path: 2-byte-aligned LE bytes wrap directly as an Int16Array;\n // copy wrap-aware in at most two runs with no per-sample DataView call\n // or modulo. This runs on the realtime audio thread.\n const int16 = new Int16Array(bytes.buffer, bytes.byteOffset, numSamples);\n let src = 0;\n let dst = this.writePos % cap;\n while (src < numSamples) {\n const run = Math.min(numSamples - src, cap - dst);\n for (let j = 0; j < run; j++) {\n samples[dst + j] = int16[src + j] / 0x8000;\n }\n src += run;\n dst = 0;\n }\n } else {\n // Odd byte offset (or big-endian host): fall back to per-sample reads.\n const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.length);\n let dst = this.writePos % cap;\n for (let i = 0; i < numSamples; i++) {\n samples[dst] = view.getInt16(i * 2, true) / 0x8000;\n dst++;\n if (dst === cap) dst = 0;\n }\n }\n this.writePos += numSamples;\n // If the producer outran the consumer by more than the buffer holds, drop\n // the oldest unplayed audio rather than reading samples we've overwritten.\n if (this.writePos - this.readPos > this.capacity) {\n this.readPos = this.writePos - this.capacity;\n }\n }\n\n process(inputs, outputs) {\n // No output wired up yet \u2014 nothing to render this quantum. Throwing here\n // would permanently kill the processor (the node is persistent per\n // session), so guard like the capture processor does.\n if (!outputs[0] || !outputs[0][0]) return true;\n const out = outputs[0][0];\n const avail = this.writePos - this.readPos;\n\n // Report the unplayed backlog to the host (`playback_progress`). This IS\n // the closed-loop signal: the host otherwise models playback open-loop \u2014\n // every forwarded chunk assumed to start playing on arrival at exactly\n // 1.0x \u2014 and cannot see a buffer that has run ahead of the wall clock.\n // Measured against a client draining at 0.6x, the host declared the line\n // silent while it still held seconds of the reply, and then opened the\n // speaking-edge gate over speech the caller had not heard.\n //\n // Sent from process() rather than a timer because this is the only place\n // that sees the buffer on the audio thread; the counter is in QUANTA, so\n // it costs an integer compare per render and no clock read. Silence is not\n // reported: an empty buffer is what the host already assumes.\n this.sinceReportSamples += out.length;\n if (this.sinceReportSamples >= this.reportIntervalSamples) {\n this.sinceReportSamples = 0;\n if (avail > 0) {\n this.port.postMessage({ event: 'progress', bufferedMs: (avail / this.rate) * 1000 });\n }\n }\n\n // Filling: wait for the target. 'done' short-circuits it \u2014 what is\n // buffered is all there will be, so there is nothing left to wait for.\n if (!this.playing) {\n if (avail >= this.fillSamples || this.isDone) {\n this.playing = true;\n } else {\n this.coverGap(out, 0);\n return true;\n }\n }\n\n // Underrun: this quantum cannot be filled and more audio is still coming.\n // Go back to filling and cover the gap, leaving\n // readPos untouched \u2014 the fragment stays buffered and plays intact once\n // the buffer recovers, instead of being dribbled out a few samples at a\n // time for the rest of the turn.\n if (avail < out.length && !this.isDone) {\n this.playing = false;\n this.coverGap(out, 0);\n return true;\n }\n\n if (avail > 0) {\n const n = Math.min(avail, out.length);\n // Copy from the ring buffer, splitting across the wrap boundary.\n const start = this.readPos % this.capacity;\n const first = Math.min(n, this.capacity - start);\n out.set(this.samples.subarray(start, start + first), 0);\n if (n > first) out.set(this.samples.subarray(0, n - first), first);\n this.readPos += n;\n // Only reachable with n < out.length on the turn's final partial\n // quantum (the underrun branch above catches every other case).\n out.fill(0, n);\n this.hasPlayed = true;\n this.rememberTail(out, n);\n return true;\n }\n\n // Drained and done: end the turn. Not reachable mid-turn \u2014 an empty\n // buffer with audio still coming is the underrun branch above.\n out.fill(0);\n if (this.isDone) {\n this.stopTurn('done');\n }\n return true;\n }\n}\n\nregisterProcessor('playback-processor', PlaybackProcessor);\n";
3
3
  declare const _default: string;
4
4
  export default _default;
@@ -1,4 +1,4 @@
1
- import { PLAYBACK_BUFFER_SECONDS, PLAYBACK_CONCEAL_FADE_MS, PLAYBACK_CONCEAL_FLOOR, PLAYBACK_JITTER_MS, PLAYBACK_PROGRESS_INTERVAL_MS, PLAYBACK_REFILL_MS } from "../types.js";
1
+ import { PLAYBACK_BUFFER_SECONDS, PLAYBACK_CONCEAL_FADE_MS, PLAYBACK_CONCEAL_FLOOR, PLAYBACK_FILL_MS, PLAYBACK_PROGRESS_INTERVAL_MS } from "../types.js";
2
2
  import { t as workletModuleUrl } from "../_module-url-C_4gRVL0.js";
3
3
  //#region worklets/playback-processor.ts
4
4
  const PlaybackProcessorWorklet = `
@@ -17,12 +17,10 @@ class PlaybackProcessor extends AudioWorkletProcessor {
17
17
  // could ever assert on the reported backlog, so the divergence made the
18
18
  // number untestable rather than wrong.
19
19
  this.rate = rate;
20
- // Fill target for the start of a turn. If 'done' arrives first (short
21
- // utterance), start immediately instead of waiting for audio that is
22
- // never coming.
23
- this.jitterSamples = Math.floor((rate * (opts.jitterMs ?? ${PLAYBACK_JITTER_MS})) / 1000);
24
- // Fill target after an underrun — see PLAYBACK_REFILL_MS.
25
- this.refillSamples = Math.floor((rate * (opts.refillMs ?? ${PLAYBACK_REFILL_MS})) / 1000);
20
+ // The one fill target — for the start of a turn and for recovery after an
21
+ // underrun alike. If 'done' arrives first (short utterance), start
22
+ // immediately instead of waiting for audio that is never coming.
23
+ this.fillSamples = Math.floor((rate * (opts.fillMs ?? ${PLAYBACK_FILL_MS})) / 1000);
26
24
  // Concealment source: a ring of the most recently played samples, looped
27
25
  // under a decaying gain to cover a gap. Sized to the fade window, with a
28
26
  // per-sample decay that reaches the floor exactly at its end.
@@ -79,7 +77,6 @@ class PlaybackProcessor extends AudioWorkletProcessor {
79
77
  // pre-roll (nothing to extrapolate from, and not a defect) from a
80
78
  // mid-turn underrun.
81
79
  this.hasPlayed = false;
82
- this.fillTarget = this.jitterSamples;
83
80
  // Carry-over byte for split samples across chunks
84
81
  this.carry = null;
85
82
  this.writePos = 0;
@@ -271,7 +268,7 @@ class PlaybackProcessor extends AudioWorkletProcessor {
271
268
  // Filling: wait for the target. 'done' short-circuits it — what is
272
269
  // buffered is all there will be, so there is nothing left to wait for.
273
270
  if (!this.playing) {
274
- if (avail >= this.fillTarget || this.isDone) {
271
+ if (avail >= this.fillSamples || this.isDone) {
275
272
  this.playing = true;
276
273
  } else {
277
274
  this.coverGap(out, 0);
@@ -280,13 +277,12 @@ class PlaybackProcessor extends AudioWorkletProcessor {
280
277
  }
281
278
 
282
279
  // Underrun: this quantum cannot be filled and more audio is still coming.
283
- // Go back to filling (at the refill target) and cover the gap, leaving
280
+ // Go back to filling and cover the gap, leaving
284
281
  // readPos untouched — the fragment stays buffered and plays intact once
285
282
  // the buffer recovers, instead of being dribbled out a few samples at a
286
283
  // time for the rest of the turn.
287
284
  if (avail < out.length && !this.isDone) {
288
285
  this.playing = false;
289
- this.fillTarget = this.refillSamples;
290
286
  this.coverGap(out, 0);
291
287
  return true;
292
288
  }