@bojackduy/opencode-voice 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1322 @@
1
+ // Local streaming dictation controller (STAGE 1).
2
+ //
3
+ // Rolling-window live transcription over a persistent local whisper-server.
4
+ // No cloud/LLM calls: snapshots of bounded rolling audio are transcribed as
5
+ // plain text hypotheses and folded into stable/tentative transcript state
6
+ // (see lib/streaming-transcript.js). Editor/command wiring is stage 2 - this
7
+ // module exposes a small documented controller/callback contract for it.
8
+ //
9
+ // Architecture (SoX capture is independent of recognition):
10
+ //
11
+ // capture lane - SoX streams PCM continuously into a bounded ring
12
+ // (windowMs + margin, oldest bytes discarded). Started
13
+ // BEFORE server readiness so initial speech is never
14
+ // dropped during cold start.
15
+ // recognition lane - a chained-setTimeout scheduler snapshots the newest
16
+ // rolling window on a fixed cadence and transcribes it.
17
+ // Exactly one request is ever in flight and at most one
18
+ // timer is ever pending: the next tick is armed at
19
+ // launch (due one cadence later, even while decoding),
20
+ // a tick firing while busy only sets a coalesce flag,
21
+ // and completion fires one immediate catch-up tick on
22
+ // the newest audio instead of queueing stale snapshots.
23
+ //
24
+ // Window geometry: windowMs=10000, cadenceMs=1000 means consecutive snapshots
25
+ // overlap by ~9s; only the newest ~1s of audio is new each tick. The
26
+ // "~1s overlap" from the plan is the per-tick advance, not the snapshot
27
+ // overlap - consecutive snapshots deliberately share most of their audio so
28
+ // no word is cut at a boundary, and the stability tracker dedupes the rest.
29
+ //
30
+ // Controller contract (for stage 2):
31
+ // start() -> true when a dictation session begins
32
+ // stop() -> Promise<{text}> final transcript (captures the SoX
33
+ // tail and transcribes it once more before commit)
34
+ // cancel() -> promptly drops the session; late in-flight
35
+ // responses are invalidated by generation and ignored
36
+ // dispose() -> cancel + teardown (server released, capture freed)
37
+ // setModel({modelPath, language}) -> teardown + recreate transcriber; only
38
+ // when idle/errored, never mid-dictation
39
+ // getState() -> { status, stableText, tentativeText, inFlight,
40
+ // lastError, hasFallbackAudio, fallbackCoverage,
41
+ // coverage, model } (coverage = live capture
42
+ // retention; null for scripted captures)
43
+ // getTranscript() -> { stableText, tentativeText, text }
44
+ // getFallbackAudio() -> WAV Buffer | null (for a later batch fallback;
45
+ // populated on stop/error paths, cleared on start)
46
+ // getFallbackCoverage() -> { source, fromMs, toMs, totalMs, complete,
47
+ // droppedRanges, message } | null: exactly which audio
48
+ // the fallback WAV covers. `complete: false` means the
49
+ // spool cap was hit and the middle is explicitly
50
+ // listed as dropped - never a "full recording" claim.
51
+ //
52
+ // Callbacks: onPartial({stableText, tentativeText}), onFinal({text}),
53
+ // onStateChange(status), onError({code, message, hasFallbackAudio}).
54
+ //
55
+ // Statuses: idle | starting | streaming | stopping | cancelled | error.
56
+ // The whisper-server model stays loaded across start/stop cycles (the
57
+ // transcriber is created once); teardown happens on dispose() or setModel().
58
+ // A missing/unreachable server is an explicit error (codes from
59
+ // lib/whisper-server.js: SERVER_BINARY_MISSING, PORT_IN_USE, START_TIMEOUT,
60
+ // ...) - the controller never silently falls back to whisper-cli; the
61
+ // preserved fallback audio lets a later stage run the batch path instead.
62
+ //
63
+ // Bounded memory: the PCM ring is capped (oldest bytes discarded); a disk
64
+ // spool preserves contiguous audio from t=0 up to spoolMaxMs (bounded disk
65
+ // policy: single capped file per session, consumed+unlinked on stopFinal,
66
+ // unlinked on cancel/dispose, best-effort sweep of stale files on start);
67
+ // overflow past the cap is counted and reported via getCoverage/onOverflow,
68
+ // never silently dropped. At most one in-flight request plus one coalesce
69
+ // flag (no snapshot queue); the transcript grows only with spoken words
70
+ // (the output itself). No spawnSync anywhere in the capture/recognition hot
71
+ // path (async spawn + HTTP only; spool appends are async chained writes).
72
+
73
+ import { spawn } from "node:child_process";
74
+ import fs from "node:fs";
75
+ import os from "node:os";
76
+ import path from "node:path";
77
+ import { buildRecordArgs, detectAudioBackend } from "./stt.js";
78
+ import { wrapPcmAsWav } from "./audio-chunker.js";
79
+ import { createStabilityTracker } from "./streaming-transcript.js";
80
+ import { acquireSharedWhisperServer } from "./whisper-server.js";
81
+
82
+ export const STREAMING_DEFAULTS = {
83
+ windowMs: 10000,
84
+ cadenceMs: 1000,
85
+ // Ring headroom beyond the window so a slow tick still snapshots windowMs.
86
+ ringMarginMs: 3000,
87
+ sampleRate: 16000,
88
+ bytesPerSecond: 32000, // 16kHz mono 16-bit
89
+ stopDrainTimeoutMs: 5000,
90
+ soxExitTimeoutMs: 2000,
91
+ // Grace for the SoX 'close' event after 'exit': 'exit' fires before stdio
92
+ // is flushed, so the tail PCM is only guaranteed delivered at 'close'.
93
+ closeGraceMs: 200,
94
+ // Wall-clock bounds for stop (real timers, never the injectable manual
95
+ // clock): how long stop waits for a stuck in-flight request before
96
+ // aborting it, how long the final tail transcription may take, and how
97
+ // long a cold-start stop waits for the startup sequence to settle.
98
+ stopFinalTimeoutMs: 30000,
99
+ stopReadyTimeoutMs: 20000,
100
+ // Disk spool: every captured byte is appended from t=0 so audio the RAM
101
+ // ring can no longer hold (cold start longer than the window, inference
102
+ // slower than the window) is never silently dropped. Bounded: spooling
103
+ // stops at spoolMaxMs and the overflow is counted + reported (see
104
+ // getCoverage), never silently discarded.
105
+ spoolMaxMs: 300000, // 5 minutes ~= 9.6 MB of 16kHz mono 16-bit PCM
106
+ spoolFilePrefix: "opencode-voice-stream-",
107
+ spoolStaleMs: 3600000, // best-effort sweep of abandoned spool files
108
+ };
109
+
110
+ // ---- Bounded rolling PCM ring (pure, shared by SoX capture and tests) ----
111
+
112
+ export function createRollingPcmBuffer({ capacityBytes }) {
113
+ let chunks = [];
114
+ let size = 0;
115
+
116
+ function push(buf) {
117
+ if (!buf || buf.length === 0) return;
118
+ chunks.push(Buffer.from(buf));
119
+ size += buf.length;
120
+ while (size > capacityBytes && chunks.length > 0) {
121
+ const head = chunks[0];
122
+ const over = size - capacityBytes;
123
+ if (head.length <= over) {
124
+ chunks.shift();
125
+ size -= head.length;
126
+ } else {
127
+ chunks[0] = head.subarray(over);
128
+ size -= over;
129
+ }
130
+ }
131
+ }
132
+
133
+ function snapshotLast(nBytes) {
134
+ const want = Math.min(nBytes, size);
135
+ if (want <= 0) return Buffer.alloc(0);
136
+ const out = Buffer.alloc(want);
137
+ let pos = want;
138
+ for (let i = chunks.length - 1; i >= 0 && pos > 0; i--) {
139
+ const take = Math.min(chunks[i].length, pos);
140
+ pos -= take;
141
+ chunks[i].copy(out, pos, chunks[i].length - take);
142
+ }
143
+ return out;
144
+ }
145
+
146
+ function drainAll() {
147
+ const out = Buffer.concat(chunks, size);
148
+ chunks = [];
149
+ size = 0;
150
+ return out;
151
+ }
152
+
153
+ return {
154
+ push,
155
+ snapshotLast,
156
+ drainAll,
157
+ size: () => size,
158
+ capacity: () => capacityBytes,
159
+ };
160
+ }
161
+
162
+ // ---- Default SoX rolling capture (async spawn only, no spawnSync) ----
163
+
164
+ export function createSoxRollingCapture({
165
+ mic = null,
166
+ backend = detectAudioBackend(),
167
+ sampleRate = STREAMING_DEFAULTS.sampleRate,
168
+ windowMs = STREAMING_DEFAULTS.windowMs,
169
+ ringMarginMs = STREAMING_DEFAULTS.ringMarginMs,
170
+ spoolMaxMs = STREAMING_DEFAULTS.spoolMaxMs,
171
+ spoolDir = os.tmpdir(),
172
+ logger = null,
173
+ spawnFn = spawn,
174
+ onOverflow = null,
175
+ // onError({code, message, ...}): async process failures - spawn 'error'
176
+ // events (e.g. ENOENT) and unexpected exits. Sync spawn throws still throw
177
+ // (the controller maps them to CAPTURE_FAILED). Expected exits (code 0 or
178
+ // our own SIGINT/SIGKILL/SIGTERM from stopFinal/cancel) never fire it.
179
+ onError = null,
180
+ } = {}) {
181
+ const capacityBytes = Math.ceil(
182
+ ((windowMs + ringMarginMs) / 1000) * STREAMING_DEFAULTS.bytesPerSecond,
183
+ );
184
+ const spoolMaxBytes = Math.ceil((spoolMaxMs / 1000) * STREAMING_DEFAULTS.bytesPerSecond);
185
+ let ring = createRollingPcmBuffer({ capacityBytes });
186
+ let proc = null;
187
+ let startedAtMs = 0;
188
+ // Absolute sample-offset accounting: total PCM bytes ever pushed, plus a
189
+ // per-snapshot sequence number. Snapshots report which absolute audio
190
+ // range they cover so callers (and fallback-audio coverage text) can tell
191
+ // retained audio from dropped audio instead of claiming "full recording".
192
+ let totalPushedBytes = 0;
193
+ let snapshotSeq = 0;
194
+ // Disk spool: contiguous audio from t=0, bounded by spoolMaxBytes.
195
+ // Overflow is counted (droppedBytes) and reported via onOverflow +
196
+ // getCoverage - audio is never silently discarded.
197
+ let spoolPath = null;
198
+ let spoolChain = Promise.resolve();
199
+ let spooledBytes = 0;
200
+ let droppedBytes = 0;
201
+ let spoolTruncated = false;
202
+ let overflowNotified = false;
203
+ let spoolConsumed = false;
204
+ let closeGraceTimer = null;
205
+ // Last process failure (async spawn error / unexpected exit). Never
206
+ // silent: delivered via onError AND retained here for getLastError().
207
+ let captureError = null;
208
+ let lastExit = null;
209
+
210
+ // Best-effort sweep of spool files abandoned by crashed sessions. Never
211
+ // blocks start and never throws: a dirty tmpdir must not break capture.
212
+ function sweepStaleSpools() {
213
+ const prefix = STREAMING_DEFAULTS.spoolFilePrefix;
214
+ fs.promises
215
+ .readdir(spoolDir)
216
+ .then((names) => {
217
+ const now = Date.now();
218
+ return Promise.all(
219
+ names
220
+ .filter((n) => n.startsWith(prefix))
221
+ .map(async (n) => {
222
+ const p = path.join(spoolDir, n);
223
+ if (p === spoolPath) return;
224
+ try {
225
+ const st = await fs.promises.stat(p);
226
+ if (now - st.mtimeMs > STREAMING_DEFAULTS.spoolStaleMs) {
227
+ await fs.promises.unlink(p);
228
+ }
229
+ } catch {}
230
+ }),
231
+ );
232
+ })
233
+ .catch(() => {});
234
+ }
235
+
236
+ function reportCaptureError(err) {
237
+ captureError = err;
238
+ try {
239
+ onError?.(err);
240
+ } catch {}
241
+ }
242
+
243
+ function unlinkSpool() {
244
+ if (!spoolPath) return Promise.resolve();
245
+ const p = spoolPath;
246
+ spoolPath = null;
247
+ return fs.promises.unlink(p).catch(() => {});
248
+ }
249
+
250
+ function spoolAppend(buf) {
251
+ if (spoolConsumed) return;
252
+ if (!spoolPath) {
253
+ const name = `${STREAMING_DEFAULTS.spoolFilePrefix}${process.pid}-${Date.now()}-${Math.floor(Math.random() * 1e6)}.pcm`;
254
+ spoolPath = path.join(spoolDir, name);
255
+ }
256
+ const target = spoolPath;
257
+ if (spoolTruncated) {
258
+ droppedBytes += buf.length;
259
+ return;
260
+ }
261
+ if (spooledBytes + buf.length > spoolMaxBytes) {
262
+ spoolTruncated = true;
263
+ droppedBytes += buf.length;
264
+ if (!overflowNotified) {
265
+ overflowNotified = true;
266
+ const droppedMs = (droppedBytes / STREAMING_DEFAULTS.bytesPerSecond) * 1000;
267
+ try {
268
+ onOverflow?.({
269
+ code: "AUDIO_SPOOL_OVERFLOW",
270
+ message:
271
+ `Audio spool reached its ${spoolMaxMs}ms cap; ` +
272
+ `retaining the first ${((spooledBytes / STREAMING_DEFAULTS.bytesPerSecond) * 1000).toFixed(0)}ms plus the rolling window, ` +
273
+ `dropping the middle (${droppedMs.toFixed(0)}ms so far)`,
274
+ recoverable: true,
275
+ spooledMs: (spooledBytes / STREAMING_DEFAULTS.bytesPerSecond) * 1000,
276
+ droppedMs,
277
+ });
278
+ } catch {}
279
+ }
280
+ return;
281
+ }
282
+ spooledBytes += buf.length;
283
+ const chunk = Buffer.from(buf);
284
+ spoolChain = spoolChain.then(() => fs.promises.appendFile(target, chunk)).catch(() => {});
285
+ }
286
+
287
+ function flushSpool() {
288
+ return spoolChain;
289
+ }
290
+
291
+ // Honest coverage of retained audio. The spool holds a contiguous prefix
292
+ // from t=0; the RAM ring holds the trailing window. With no overflow the
293
+ // session is fully retained; with overflow the middle is explicitly listed
294
+ // as dropped - never presented as a complete recording.
295
+ function getCoverage() {
296
+ const totalMs = (totalPushedBytes / STREAMING_DEFAULTS.bytesPerSecond) * 1000;
297
+ const spooledMs = (spooledBytes / STREAMING_DEFAULTS.bytesPerSecond) * 1000;
298
+ const ringMs = (ring.size() / STREAMING_DEFAULTS.bytesPerSecond) * 1000;
299
+ const ringStartMs = Math.max(0, totalMs - ringMs);
300
+ if (!spoolTruncated) {
301
+ return {
302
+ source: "spool",
303
+ fromMs: 0,
304
+ toMs: totalMs,
305
+ totalMs,
306
+ complete: true,
307
+ droppedRanges: [],
308
+ message: `complete recording 0.0-${totalMs.toFixed(1)}s`,
309
+ };
310
+ }
311
+ const droppedRanges = ringStartMs > spooledMs ? [{ fromMs: spooledMs, toMs: ringStartMs }] : [];
312
+ const retainedRanges = [{ fromMs: 0, toMs: spooledMs }];
313
+ if (ring.size() > 0) retainedRanges.push({ fromMs: ringStartMs, toMs: totalMs });
314
+ return {
315
+ source: "spool-head-plus-ring-tail",
316
+ fromMs: 0,
317
+ toMs: totalMs,
318
+ totalMs,
319
+ complete: false,
320
+ retainedRanges,
321
+ droppedRanges,
322
+ message:
323
+ `spool cap ${spoolMaxMs}ms hit: fallback holds 0.0-${spooledMs.toFixed(1)}s, ` +
324
+ `rolling window holds ${ringStartMs.toFixed(1)}-${totalMs.toFixed(1)}s` +
325
+ (droppedRanges.length > 0
326
+ ? `; missing ${spooledMs.toFixed(1)}-${ringStartMs.toFixed(1)}s`
327
+ : `; no single buffer holds the full ${totalMs.toFixed(1)}s`) +
328
+ `; NOT a complete recording`,
329
+ };
330
+ }
331
+
332
+ function start() {
333
+ if (proc) return;
334
+ // A previous session's spool was never collected (restart without
335
+ // stopFinal/cancel): drop it promptly - a fresh start is a new t=0, and
336
+ // abandoned files must not accumulate beyond the stale sweep. Chained
337
+ // AFTER in-flight appends so a racing write cannot resurrect the file.
338
+ if (spoolPath) {
339
+ const abandoned = spoolPath;
340
+ spoolPath = null;
341
+ spoolChain = spoolChain
342
+ .then(() => fs.promises.unlink(abandoned).catch(() => {}))
343
+ .catch(() => {});
344
+ }
345
+ ring = createRollingPcmBuffer({ capacityBytes });
346
+ totalPushedBytes = 0;
347
+ snapshotSeq = 0;
348
+ spoolPath = null;
349
+ spoolChain = Promise.resolve();
350
+ spooledBytes = 0;
351
+ droppedBytes = 0;
352
+ spoolTruncated = false;
353
+ overflowNotified = false;
354
+ spoolConsumed = false;
355
+ captureError = null;
356
+ lastExit = null;
357
+ if (closeGraceTimer) {
358
+ clearTimeout(closeGraceTimer);
359
+ closeGraceTimer = null;
360
+ }
361
+ sweepStaleSpools();
362
+ const inputArgs = buildRecordArgs(backend, mic);
363
+ logger?.log("STT", `Streaming capture starting backend=${backend}`, "debug");
364
+ try {
365
+ proc = spawnFn(
366
+ "sox",
367
+ [...inputArgs, "-r", String(sampleRate), "-c", "1", "-b", "16", "-t", "raw", "-"],
368
+ { stdio: ["ignore", "pipe", "pipe"] },
369
+ );
370
+ } catch (err) {
371
+ // Sync spawn failure (e.g. ENOENT thrown synchronously): record for
372
+ // getLastError() AND throw so the controller maps it to CAPTURE_FAILED.
373
+ captureError = {
374
+ code: err?.code === "ENOENT" ? "CAPTURE_SPAWN_ENOENT" : "CAPTURE_SPAWN_FAILED",
375
+ message: `Failed to spawn sox: ${err?.message || String(err)}`,
376
+ recoverable: true,
377
+ };
378
+ proc = null;
379
+ throw err;
380
+ }
381
+ startedAtMs = Date.now();
382
+ proc.stdout?.on("data", (buf) => {
383
+ totalPushedBytes += buf.length;
384
+ ring.push(buf);
385
+ spoolAppend(buf);
386
+ });
387
+ proc.stderr?.on("data", () => {});
388
+ // Owned-process guards: handlers from a previous start() must never
389
+ // null a restarted session's proc. 'close' (stdio flushed) releases the
390
+ // handle; 'exit' alone only arms a grace fallback - the stop tail must
391
+ // wait for drained stdout, not merely exit.
392
+ const owned = proc;
393
+ proc.on("error", (err) => {
394
+ if (proc !== owned) return;
395
+ if (closeGraceTimer) {
396
+ clearTimeout(closeGraceTimer);
397
+ closeGraceTimer = null;
398
+ }
399
+ proc = null;
400
+ // Async spawn failure (e.g. ENOENT delivered as an event): explicit
401
+ // and recoverable, with whatever audio was captured so far retained.
402
+ reportCaptureError({
403
+ code: err?.code === "ENOENT" ? "CAPTURE_SPAWN_ENOENT" : "CAPTURE_SPAWN_ERROR",
404
+ message: `SoX process error: ${err?.message || String(err)}`,
405
+ recoverable: true,
406
+ });
407
+ });
408
+ proc.on("exit", (code, signal) => {
409
+ if (proc !== owned) return;
410
+ lastExit = { code: code ?? null, signal: signal ?? null };
411
+ const expected = code === 0 || (signal && ["SIGINT", "SIGKILL", "SIGTERM"].includes(signal));
412
+ if (!expected) {
413
+ // Unexpected exit (crash, external kill): explicit error. Audio
414
+ // captured so far stays in ring+spool; stdout arriving between exit
415
+ // and close is still drained (data handler is independent of proc).
416
+ reportCaptureError({
417
+ code: "CAPTURE_EXITED",
418
+ message: `SoX exited unexpectedly (code=${lastExit.code} signal=${lastExit.signal}); audio retained for fallback`,
419
+ recoverable: true,
420
+ exit: { ...lastExit },
421
+ });
422
+ }
423
+ if (closeGraceTimer) clearTimeout(closeGraceTimer);
424
+ closeGraceTimer = setTimeout(() => {
425
+ closeGraceTimer = null;
426
+ if (proc === owned) proc = null;
427
+ }, STREAMING_DEFAULTS.closeGraceMs);
428
+ closeGraceTimer.unref?.();
429
+ });
430
+ proc.on("close", () => {
431
+ if (closeGraceTimer) {
432
+ clearTimeout(closeGraceTimer);
433
+ closeGraceTimer = null;
434
+ }
435
+ if (proc === owned) proc = null;
436
+ });
437
+ }
438
+
439
+ function snapshot() {
440
+ const pcm = ring.snapshotLast(Math.ceil((windowMs / 1000) * STREAMING_DEFAULTS.bytesPerSecond));
441
+ if (pcm.length === 0) return null;
442
+ const absoluteEndMs = (totalPushedBytes / STREAMING_DEFAULTS.bytesPerSecond) * 1000;
443
+ const absoluteStartMs = absoluteEndMs - (pcm.length / STREAMING_DEFAULTS.bytesPerSecond) * 1000;
444
+ snapshotSeq += 1;
445
+ return {
446
+ wav: wrapPcmAsWav(pcm, { sampleRate }),
447
+ durationMs: Date.now() - startedAtMs,
448
+ absoluteStartMs,
449
+ absoluteEndMs,
450
+ seq: snapshotSeq,
451
+ };
452
+ }
453
+
454
+ function waitForExit(timeoutMs) {
455
+ return new Promise((resolve) => {
456
+ const start = Date.now();
457
+ const check = () => {
458
+ if (!proc || Date.now() - start >= timeoutMs) {
459
+ if (proc) {
460
+ try {
461
+ proc.kill("SIGKILL");
462
+ } catch {}
463
+ proc = null;
464
+ }
465
+ resolve();
466
+ return;
467
+ }
468
+ setTimeout(check, 50);
469
+ };
470
+ check();
471
+ });
472
+ }
473
+
474
+ // Stop capture and return retained audio for the stop-tail transcription
475
+ // and batch fallback. The fallback prefers the disk spool: contiguous
476
+ // audio from t=0 (covers cold-start speech the RAM ring aged out), with a
477
+ // coverage descriptor that says exactly what is retained. Bounded: SIGINT,
478
+ // then SIGKILL after soxExitTimeoutMs. Consumes the spool file (cleanup).
479
+ async function stopFinal({ soxExitTimeoutMs = STREAMING_DEFAULTS.soxExitTimeoutMs } = {}) {
480
+ if (proc) {
481
+ try {
482
+ proc.kill("SIGINT");
483
+ } catch {}
484
+ await waitForExit(soxExitTimeoutMs);
485
+ }
486
+ await spoolChain;
487
+ const coverage = getCoverage();
488
+ let wav = null;
489
+ let wavFromMs = null;
490
+ let wavToMs = null;
491
+ if (spoolPath && !spoolConsumed && spooledBytes > 0) {
492
+ const target = spoolPath;
493
+ try {
494
+ const spoolPcm = await fs.promises.readFile(target);
495
+ if (spoolPcm.length > 0) {
496
+ wav = wrapPcmAsWav(spoolPcm, { sampleRate });
497
+ wavFromMs = 0;
498
+ wavToMs = (spoolPcm.length / STREAMING_DEFAULTS.bytesPerSecond) * 1000;
499
+ }
500
+ } catch {}
501
+ spoolConsumed = true;
502
+ await unlinkSpool();
503
+ }
504
+ if (!wav) {
505
+ const snap = snapshot();
506
+ wav = snap?.wav || null;
507
+ wavFromMs = snap?.absoluteStartMs ?? null;
508
+ wavToMs = snap?.absoluteEndMs ?? null;
509
+ coverage.source = "ring-tail";
510
+ }
511
+ return {
512
+ wav,
513
+ durationMs: Date.now() - startedAtMs,
514
+ absoluteStartMs: wavFromMs,
515
+ absoluteEndMs: wavToMs,
516
+ seq: snapshotSeq,
517
+ coverage,
518
+ };
519
+ }
520
+
521
+ async function cancel() {
522
+ if (proc) {
523
+ const owned = proc;
524
+ proc = null;
525
+ try {
526
+ owned.kill("SIGKILL");
527
+ } catch {}
528
+ }
529
+ await spoolChain;
530
+ spoolConsumed = true;
531
+ await unlinkSpool();
532
+ ring = createRollingPcmBuffer({ capacityBytes });
533
+ }
534
+
535
+ function dispose() {
536
+ return cancel();
537
+ }
538
+
539
+ function isAlive() {
540
+ return proc !== null;
541
+ }
542
+
543
+ function getLastError() {
544
+ return captureError;
545
+ }
546
+
547
+ function getLastExit() {
548
+ return lastExit;
549
+ }
550
+
551
+ return {
552
+ start,
553
+ snapshot,
554
+ stopFinal,
555
+ cancel,
556
+ dispose,
557
+ getCoverage,
558
+ flushSpool,
559
+ isAlive,
560
+ getLastError,
561
+ getLastExit,
562
+ };
563
+ }
564
+
565
+ // ---- Default server transcriber (persistent whisper-server, no fallback) ----
566
+
567
+ export function createServerTranscriber({
568
+ modelPath,
569
+ language,
570
+ host,
571
+ port,
572
+ logger = null,
573
+ serverFactory = acquireSharedWhisperServer,
574
+ // Default tick format is plain "json" (~0.9s per 10s window): the
575
+ // stability tracker is text-anchored and never reads segments, so the
576
+ // verbose payload only adds latency. Pass "verbose_json" explicitly to
577
+ // opt into segments; old builds that reject it still downgrade to "json".
578
+ responseFormat: preferredFormat = "json",
579
+ } = {}) {
580
+ let held = null;
581
+ let responseFormat = preferredFormat === "verbose_json" ? "verbose_json" : "json";
582
+ // "json" needs no probing (single attempt). "verbose_json" probes once:
583
+ // on rejection it retries as plain json and stays there.
584
+ let formatProbed = responseFormat === "json";
585
+
586
+ function handle() {
587
+ if (!held) held = serverFactory({ modelPath, language, host, port, logger });
588
+ return held.client;
589
+ }
590
+
591
+ async function ensureReady() {
592
+ const client = handle();
593
+ const ready = await client.start();
594
+ if (ready) return { ready: true };
595
+ const err = client.getLastError() || { code: "NOT_READY", message: "whisper-server not ready" };
596
+ return { ready: false, error: err };
597
+ }
598
+
599
+ async function transcribe(wavBuffer, opts = {}) {
600
+ const client = handle();
601
+ // An explicit per-call format wins for this call only (tests + callers
602
+ // that need segments); otherwise the session default applies.
603
+ const callFormat = opts.responseFormat === "verbose_json" ? "verbose_json" : null;
604
+ const effective = callFormat || responseFormat;
605
+ let result = await client.transcribeBuffer(wavBuffer, {
606
+ responseFormat: effective,
607
+ signal: opts.signal,
608
+ });
609
+ // Old builds may reject verbose_json: retry once as plain json and stay
610
+ // there. The stability logic is text-anchored either way (segments are
611
+ // snapshot-relative hints, never cross-window truth).
612
+ if (result.error && !formatProbed && effective === "verbose_json") {
613
+ formatProbed = true;
614
+ responseFormat = "json";
615
+ result = await client.transcribeBuffer(wavBuffer, { responseFormat, signal: opts.signal });
616
+ } else {
617
+ formatProbed = true;
618
+ }
619
+ return result;
620
+ }
621
+
622
+ function dispose() {
623
+ held?.release();
624
+ held = null;
625
+ }
626
+
627
+ return {
628
+ ensureReady,
629
+ transcribe,
630
+ dispose,
631
+ getResponseFormat: () => responseFormat,
632
+ // boundPort is the auto-scan claim (8090, 8091, ...) once the server is
633
+ // up; port is the requested port (undefined = default scan path).
634
+ describe: () => ({
635
+ modelPath,
636
+ language,
637
+ port,
638
+ boundPort: held?.client?.getPort?.() ?? null,
639
+ responseFormat,
640
+ }),
641
+ };
642
+ }
643
+
644
+ // ---- Streaming controller ----
645
+
646
+ export function createStreamingController({
647
+ captureFactory = (opts = {}) => createSoxRollingCapture({ windowMs, ...opts }),
648
+ transcriberFactory = (spec) => createServerTranscriber({ ...spec }),
649
+ trackerFactory = () => createStabilityTracker(),
650
+ clock = { setTimeout, clearTimeout },
651
+ windowMs = STREAMING_DEFAULTS.windowMs,
652
+ cadenceMs = STREAMING_DEFAULTS.cadenceMs,
653
+ stopDrainTimeoutMs = STREAMING_DEFAULTS.stopDrainTimeoutMs,
654
+ stopFinalTimeoutMs = STREAMING_DEFAULTS.stopFinalTimeoutMs,
655
+ stopReadyTimeoutMs = STREAMING_DEFAULTS.stopReadyTimeoutMs,
656
+ model = null,
657
+ logger = null,
658
+ onPartial = null,
659
+ onFinal = null,
660
+ onStateChange = null,
661
+ onError = null,
662
+ onOverflow = null,
663
+ } = {}) {
664
+ let capture = null;
665
+ let transcriber = null;
666
+ let tracker = trackerFactory();
667
+ let status = "idle";
668
+ let timer = null;
669
+ let inFlight = false;
670
+ let inFlightPromise = null;
671
+ // Ownership fencing: every async continuation re-checks `generation`
672
+ // after each await, and every inference request carries the AbortSignal of
673
+ // its own generation. A stale continuation must NEVER touch shared state
674
+ // (inFlight flags, transcript, timers, statuses, callbacks) - it returns
675
+ // without effect so a new session's request cannot be clobbered.
676
+ let activeReqController = null;
677
+ let startupPromise = null;
678
+ let stopPromise = null;
679
+ let coalesced = false;
680
+ let finalizing = false;
681
+ let generation = 0;
682
+ let lastError = null;
683
+ let fallbackAudio = null;
684
+ let fallbackCoverage = null;
685
+ let pendingModel = model ? { ...model } : null;
686
+
687
+ function setStatus(next) {
688
+ status = next;
689
+ try {
690
+ onStateChange?.(next);
691
+ } catch {}
692
+ }
693
+
694
+ function emitPartial() {
695
+ const s = tracker.getState();
696
+ try {
697
+ onPartial?.({ stableText: s.stableText, tentativeText: s.tentativeText });
698
+ } catch {}
699
+ }
700
+
701
+ function currentTranscript() {
702
+ const s = tracker.getState();
703
+ return {
704
+ stableText: s.stableText,
705
+ tentativeText: s.tentativeText,
706
+ text: `${s.stableText} ${s.tentativeText}`.trim(),
707
+ };
708
+ }
709
+
710
+ // Coverage for a fallback WAV when the capture reported none (scripted or
711
+ // legacy captures): approximate from the tail offsets, marked explicitly
712
+ // unverified - a fallback MUST never look complete without evidence.
713
+ function coverageForTail(fin) {
714
+ if (!fin?.wav) return null;
715
+ const toMs = fin.absoluteEndMs ?? fin.durationMs ?? 0;
716
+ const fromMs = fin.absoluteStartMs ?? 0;
717
+ return {
718
+ source: "capture-tail-unverified",
719
+ fromMs,
720
+ toMs,
721
+ totalMs: toMs,
722
+ complete: false,
723
+ retainedRanges: [{ fromMs, toMs }],
724
+ droppedRanges: [],
725
+ message:
726
+ `capture reported no coverage; fallback holds ~${fromMs.toFixed(1)}-${toMs.toFixed(1)}ms; ` +
727
+ `completeness NOT verified`,
728
+ };
729
+ }
730
+
731
+ function storeFallback(fin) {
732
+ fallbackAudio = fin?.wav || null;
733
+ fallbackCoverage = fin?.coverage || coverageForTail(fin);
734
+ }
735
+
736
+ function schedule(delayMs) {
737
+ clearTimer();
738
+ timer = clock.setTimeout(() => {
739
+ timer = null;
740
+ void tick();
741
+ }, delayMs);
742
+ }
743
+
744
+ function clearTimer() {
745
+ if (timer !== null && timer !== undefined) {
746
+ try {
747
+ clock.clearTimeout(timer);
748
+ } catch {}
749
+ timer = null;
750
+ }
751
+ }
752
+
753
+ function describeTranscriber() {
754
+ try {
755
+ return transcriber?.describe?.() || null;
756
+ } catch {
757
+ return null;
758
+ }
759
+ }
760
+
761
+ // Actual whisper-server port for the warm-lease identity (stt.js keys the
762
+ // parked lease per model/language/port). Null until the server is up.
763
+ function getServerPort() {
764
+ try {
765
+ const d = describeTranscriber();
766
+ return d?.boundPort ?? d?.port ?? null;
767
+ } catch {
768
+ return null;
769
+ }
770
+ }
771
+
772
+ function abortActiveRequest() {
773
+ const controller = activeReqController;
774
+ activeReqController = null;
775
+ if (controller) {
776
+ try {
777
+ controller.abort();
778
+ } catch {}
779
+ }
780
+ }
781
+
782
+ // Wall-clock timeout that never holds the event loop open: callers must
783
+ // cancel() it once their race settles so `--test` and short-lived CLIs
784
+ // don't wait out the full bound.
785
+ function wallTimeout(ms, value) {
786
+ let timer = null;
787
+ const promise = new Promise((resolve) => {
788
+ timer = setTimeout(() => {
789
+ timer = null;
790
+ resolve(value);
791
+ }, ms);
792
+ timer.unref?.();
793
+ });
794
+ return {
795
+ promise,
796
+ cancel() {
797
+ if (timer) {
798
+ clearTimeout(timer);
799
+ timer = null;
800
+ }
801
+ },
802
+ };
803
+ }
804
+
805
+ // Settle the in-flight request within a wall-clock bound: await it, and on
806
+ // timeout abort its signal and proceed WITHOUT waiting further. A
807
+ // transcriber that ignores the abort settles whenever it settles; the
808
+ // generation/promise-identity guards in tick() drop its result, so the
809
+ // final transcription still never runs concurrently at the controller
810
+ // level. Never throws, never hangs.
811
+ async function settleInFlight(timeoutMs) {
812
+ const pending = inFlightPromise;
813
+ if (!pending) return;
814
+ const timeout = wallTimeout(timeoutMs, false);
815
+ const settled = await Promise.race([
816
+ pending.then(
817
+ () => true,
818
+ () => true,
819
+ ),
820
+ timeout.promise,
821
+ ]);
822
+ timeout.cancel();
823
+ if (!settled) abortActiveRequest();
824
+ }
825
+
826
+ async function tick() {
827
+ if (status !== "streaming" || finalizing) return;
828
+ if (inFlight) {
829
+ // Exactly one in-flight request: coalesce to the newest audio instead
830
+ // of queueing a stale snapshot behind the slow one.
831
+ coalesced = true;
832
+ return;
833
+ }
834
+ const snap = capture?.snapshot?.();
835
+ if (!snap || !snap.wav || snap.wav.length === 0) {
836
+ schedule(cadenceMs); // no audio yet (cold start): keep polling
837
+ return;
838
+ }
839
+ const myGen = generation;
840
+ const reqController = new AbortController();
841
+ activeReqController = reqController;
842
+ inFlight = true;
843
+ const myPromise = (async () => {
844
+ try {
845
+ return await transcriber.transcribe(snap.wav, { signal: reqController.signal });
846
+ } catch (err) {
847
+ return { error: err?.message || String(err), code: "TRANSCRIBE_THROW" };
848
+ }
849
+ })();
850
+ inFlightPromise = myPromise;
851
+ // Cadence continues DURING inference: the next tick is due one cadence
852
+ // after this launch regardless of decode duration. A tick that fires
853
+ // while busy only sets the coalesce flag (no snapshot queue, no
854
+ // backlog); completion then fires one immediate catch-up tick on the
855
+ // newest audio. At most one timer is ever pending.
856
+ schedule(cadenceMs);
857
+ const result = await myPromise;
858
+ // Stale (cancelled/disposed/restarted while awaiting): touch NOTHING.
859
+ // Clearing inFlight here would clobber the new session's request and
860
+ // permit two overlapping physical inferences.
861
+ if (myGen !== generation || inFlightPromise !== myPromise) return;
862
+ inFlight = false;
863
+ inFlightPromise = null;
864
+ if (activeReqController === reqController) activeReqController = null;
865
+ if (finalizing || status !== "streaming") return; // stopped while awaiting
866
+ if (result.error) {
867
+ await handleTranscribeError(result, myGen);
868
+ return;
869
+ }
870
+ const text = (result.text || "").trim();
871
+ if (text) {
872
+ tracker.update(text, {
873
+ absoluteStartMs: snap.absoluteStartMs ?? null,
874
+ absoluteEndMs: snap.absoluteEndMs ?? null,
875
+ seq: snap.seq ?? null,
876
+ });
877
+ emitPartial();
878
+ }
879
+ if (coalesced) {
880
+ coalesced = false;
881
+ schedule(0); // ticks arrived during decode: one catch-up on newest audio
882
+ }
883
+ // Otherwise the cadence timer armed at launch is still pending: the next
884
+ // tick fires exactly one cadence after the last launch, not one cadence
885
+ // after this decode.
886
+ }
887
+
888
+ async function handleTranscribeError(result, myGen) {
889
+ // Explicit failure: stop scheduling, preserve what capture has for a
890
+ // later batch fallback. Never silently switch to whisper-cli here.
891
+ // Generation-guarded after the await: a cancel/restart that landed
892
+ // during stopFinal must win over this stale error path.
893
+ clearTimer();
894
+ lastError = { code: result.code || "TRANSCRIBE_FAILED", message: result.error };
895
+ try {
896
+ const fin = await capture?.stopFinal?.();
897
+ if (myGen !== generation) return;
898
+ storeFallback(fin);
899
+ } catch {
900
+ if (myGen !== generation) return;
901
+ fallbackAudio = null;
902
+ fallbackCoverage = null;
903
+ }
904
+ if (myGen !== generation) return;
905
+ setStatus("error");
906
+ try {
907
+ onError?.({
908
+ ...lastError,
909
+ hasFallbackAudio: fallbackAudio !== null,
910
+ fallbackCoverage,
911
+ });
912
+ } catch {}
913
+ }
914
+
915
+ // Capture-process failure (async spawn error / unexpected exit): stop the
916
+ // scheduler with an explicit error and retained transcript/audio - never
917
+ // stream forever on a dead mic. During "starting" only records: the
918
+ // startup continuation converts a dead capture into its failure path.
919
+ // During stop/cancel/dispose the owner of that transition wins instead.
920
+ function handleCaptureError(err) {
921
+ const myGen = generation;
922
+ if (status !== "streaming" && status !== "starting") return;
923
+ if (status === "starting") {
924
+ lastError = {
925
+ code: err?.code || "CAPTURE_FAILED",
926
+ message: err?.message || String(err),
927
+ recoverable: true,
928
+ };
929
+ return;
930
+ }
931
+ void (async () => {
932
+ clearTimer();
933
+ lastError = {
934
+ code: err?.code || "CAPTURE_FAILED",
935
+ message: err?.message || String(err),
936
+ recoverable: true,
937
+ };
938
+ try {
939
+ const fin = await capture?.stopFinal?.();
940
+ if (myGen !== generation) return;
941
+ storeFallback(fin);
942
+ } catch {
943
+ if (myGen !== generation) return;
944
+ fallbackAudio = null;
945
+ fallbackCoverage = null;
946
+ }
947
+ if (myGen !== generation) return;
948
+ if (status !== "streaming") return;
949
+ setStatus("error");
950
+ try {
951
+ onError?.({
952
+ ...lastError,
953
+ hasFallbackAudio: fallbackAudio !== null,
954
+ fallbackCoverage,
955
+ });
956
+ } catch {}
957
+ })();
958
+ }
959
+
960
+ function start() {
961
+ if (status === "streaming" || status === "starting" || status === "stopping") return false;
962
+ generation += 1; // invalidate any ancient late responses
963
+ abortActiveRequest(); // defensive: no lingering signal from a dead session
964
+ stopPromise = null;
965
+ tracker = trackerFactory();
966
+ lastError = null;
967
+ fallbackAudio = null;
968
+ fallbackCoverage = null;
969
+ coalesced = false;
970
+ finalizing = false;
971
+ if (!capture) {
972
+ capture = captureFactory({
973
+ onError: (err) => handleCaptureError(err),
974
+ onOverflow: (info) => {
975
+ try {
976
+ onOverflow?.(info);
977
+ } catch {}
978
+ },
979
+ });
980
+ }
981
+ if (!transcriber) {
982
+ transcriber = transcriberFactory(pendingModel ? { ...pendingModel } : {});
983
+ model = describeTranscriber() || pendingModel;
984
+ }
985
+ setStatus("starting");
986
+ // Capture FIRST so nothing spoken during model cold start is lost; the
987
+ // recognition ticks only begin after ensureReady resolves.
988
+ try {
989
+ capture.start();
990
+ } catch (err) {
991
+ lastError = { code: "CAPTURE_FAILED", message: err?.message || String(err) };
992
+ setStatus("error");
993
+ try {
994
+ onError?.({ ...lastError, hasFallbackAudio: false });
995
+ } catch {}
996
+ return true;
997
+ }
998
+ void (startupPromise = (async () => {
999
+ const myGen = generation;
1000
+ let ready;
1001
+ try {
1002
+ ready = await transcriber.ensureReady();
1003
+ } catch (err) {
1004
+ ready = { ready: false, error: { code: "ENSURE_READY_THROW", message: err?.message } };
1005
+ }
1006
+ if (myGen !== generation) return; // cancelled/stopped/restarted during startup
1007
+ // The mic died while the model loaded (capture error already
1008
+ // recorded): never enter streaming on a dead capture.
1009
+ if (capture?.isAlive?.() === false) {
1010
+ const captureErr = capture?.getLastError?.() || lastError;
1011
+ ready = {
1012
+ ready: false,
1013
+ error: captureErr || { code: "CAPTURE_FAILED", message: "capture died during startup" },
1014
+ };
1015
+ }
1016
+ if (!ready?.ready) {
1017
+ // Model never loaded, but capture has the audio since t=0: keep it
1018
+ // for a later batch fallback.
1019
+ try {
1020
+ const fin = await capture?.stopFinal?.();
1021
+ if (myGen !== generation) return; // lost the race during teardown
1022
+ storeFallback(fin);
1023
+ } catch {
1024
+ if (myGen !== generation) return;
1025
+ fallbackAudio = null;
1026
+ fallbackCoverage = null;
1027
+ }
1028
+ if (myGen !== generation) return;
1029
+ lastError = ready?.error || { code: "NOT_READY", message: "whisper-server not ready" };
1030
+ setStatus("error");
1031
+ try {
1032
+ onError?.({
1033
+ ...lastError,
1034
+ hasFallbackAudio: fallbackAudio !== null,
1035
+ fallbackCoverage,
1036
+ });
1037
+ } catch {}
1038
+ return;
1039
+ }
1040
+ if (myGen !== generation || status !== "starting") return;
1041
+ setStatus("streaming");
1042
+ schedule(0); // first snapshot covers audio buffered since t=0
1043
+ })());
1044
+ return true;
1045
+ }
1046
+
1047
+ // Stop: freeze the transcript and report it. Ordering guarantees:
1048
+ // 1. The mic stops FIRST (no extra speech captured after stop) while a
1049
+ // stuck in-flight request is settled within stopDrainTimeoutMs, then
1050
+ // aborted - the final transcription never runs concurrently with a
1051
+ // stale request (one-in-flight holds through finalization).
1052
+ // 2. The final transcription is bounded by stopFinalTimeoutMs.
1053
+ // 3. A cold-start stop (status "starting") awaits the startup sequence
1054
+ // within stopReadyTimeoutMs, else returns an explicit recoverable
1055
+ // failure - never a silent empty success.
1056
+ // 4. Concurrent stop calls share one finalization promise.
1057
+ // Every await is followed by a generation/status guard: a cancel/dispose
1058
+ // that lands mid-stop owns the outcome, and stop resolves quietly without
1059
+ // onPartial/onFinal/state changes.
1060
+ async function stop() {
1061
+ if (status === "idle" || status === "cancelled") return currentTranscript();
1062
+ if (stopPromise) return stopPromise;
1063
+ const stoppingFrom = status;
1064
+ const myGen = generation;
1065
+ setStatus("stopping");
1066
+ clearTimer();
1067
+ finalizing = true;
1068
+ stopPromise = (async () => {
1069
+ try {
1070
+ if (stoppingFrom === "starting") return await stopFromStarting(myGen);
1071
+ return await stopFromActive(myGen, stoppingFrom);
1072
+ } finally {
1073
+ finalizing = false;
1074
+ }
1075
+ })();
1076
+ const result = await stopPromise;
1077
+ stopPromise = null;
1078
+ return result;
1079
+ }
1080
+
1081
+ async function stopFromActive(myGen, stoppingFrom, cachedFin = undefined) {
1082
+ // Mic first, drain concurrently: stopFinal kills SoX promptly while the
1083
+ // old request settles (bounded, then aborted). A cold-start stop passes
1084
+ // its already-obtained tail so the spool fallback is never replaced by
1085
+ // a second stopFinal's ring tail.
1086
+ const finPromise =
1087
+ cachedFin !== undefined
1088
+ ? Promise.resolve(cachedFin)
1089
+ : (async () => {
1090
+ try {
1091
+ return await capture?.stopFinal?.();
1092
+ } catch {
1093
+ return null;
1094
+ }
1095
+ })();
1096
+ await settleInFlight(stopDrainTimeoutMs);
1097
+ if (myGen !== generation || status !== "stopping") return currentTranscript();
1098
+ const fin = await finPromise;
1099
+ if (myGen !== generation || status !== "stopping") return currentTranscript();
1100
+ storeFallback(fin);
1101
+ if (stoppingFrom !== "error" && fin?.wav) {
1102
+ const finalController = new AbortController();
1103
+ activeReqController = finalController;
1104
+ let result;
1105
+ const finalTimeout = wallTimeout(stopFinalTimeoutMs, {
1106
+ error: `final transcription timed out after ${stopFinalTimeoutMs}ms`,
1107
+ code: "STOP_FINAL_TIMEOUT",
1108
+ });
1109
+ try {
1110
+ result = await Promise.race([
1111
+ transcriber?.transcribe(fin.wav, { signal: finalController.signal }),
1112
+ finalTimeout.promise,
1113
+ ]);
1114
+ } catch (err) {
1115
+ result = { error: err?.message || String(err), code: "TRANSCRIBE_THROW" };
1116
+ }
1117
+ finalTimeout.cancel();
1118
+ if (activeReqController === finalController) activeReqController = null;
1119
+ if (result?.code === "STOP_FINAL_TIMEOUT") {
1120
+ try {
1121
+ finalController.abort();
1122
+ } catch {}
1123
+ }
1124
+ if (myGen !== generation || status !== "stopping") return currentTranscript();
1125
+ if (result && !result.error && (result.text || "").trim()) {
1126
+ tracker.update(result.text.trim(), {
1127
+ absoluteStartMs: fin?.absoluteStartMs ?? null,
1128
+ absoluteEndMs: fin?.absoluteEndMs ?? null,
1129
+ seq: fin?.seq ?? null,
1130
+ final: true,
1131
+ });
1132
+ emitPartial();
1133
+ }
1134
+ }
1135
+ tracker.commitTail();
1136
+ if (myGen !== generation || status !== "stopping") return currentTranscript();
1137
+ const done = currentTranscript();
1138
+ try {
1139
+ onFinal?.({ text: done.text });
1140
+ } catch {}
1141
+ inFlight = false;
1142
+ inFlightPromise = null;
1143
+ coalesced = false;
1144
+ setStatus("idle"); // server stays loaded for the next dictation
1145
+ return done;
1146
+ }
1147
+
1148
+ async function stopFromStarting(myGen) {
1149
+ // Cold-start stop: capture holds speech since t=0 but the model may
1150
+ // never have become ready. Stop the mic promptly, then await the
1151
+ // startup sequence within a bound.
1152
+ let fin = null;
1153
+ try {
1154
+ fin = await capture?.stopFinal?.();
1155
+ } catch {
1156
+ fin = null;
1157
+ }
1158
+ if (myGen !== generation || status !== "stopping") return currentTranscript();
1159
+ storeFallback(fin);
1160
+ let startupSettled = false;
1161
+ if (startupPromise) {
1162
+ const readyTimeout = wallTimeout(stopReadyTimeoutMs, false);
1163
+ await Promise.race([
1164
+ startupPromise.then(
1165
+ () => {
1166
+ startupSettled = true;
1167
+ },
1168
+ () => {
1169
+ startupSettled = true;
1170
+ },
1171
+ ),
1172
+ readyTimeout.promise,
1173
+ ]);
1174
+ readyTimeout.cancel();
1175
+ } else {
1176
+ startupSettled = true;
1177
+ }
1178
+ if (myGen !== generation || status !== "stopping") return currentTranscript();
1179
+ if (!startupSettled) {
1180
+ // Model never became ready: explicit recoverable failure with the
1181
+ // t=0 audio preserved - never a silent empty success.
1182
+ lastError = {
1183
+ code: "STOP_NOT_READY",
1184
+ message:
1185
+ `stop during cold start: model not ready within ${stopReadyTimeoutMs}ms; ` +
1186
+ (fallbackAudio ? "audio preserved for batch fallback" : "no audio captured"),
1187
+ recoverable: true,
1188
+ };
1189
+ setStatus("error");
1190
+ try {
1191
+ onError?.({
1192
+ ...lastError,
1193
+ hasFallbackAudio: fallbackAudio !== null,
1194
+ fallbackCoverage,
1195
+ });
1196
+ } catch {}
1197
+ return { ...currentTranscript(), error: lastError };
1198
+ }
1199
+ // Startup settled while we were stopping: its continuation stood down
1200
+ // (status check), and readiness is confirmed - finalize with the
1201
+ // already-captured tail (never a second stopFinal).
1202
+ return await stopFromActive(myGen, "streaming", fin);
1203
+ }
1204
+
1205
+ // Cancel: prompt, bounded, invalidates late responses via generation.
1206
+ // Aborts the physical in-flight request (no overlapping inference on
1207
+ // restart), stops capture, and wins over any concurrent stop/error path
1208
+ // via the generation bump. Never waits for inference.
1209
+ async function cancel() {
1210
+ if (status === "idle") return false;
1211
+ generation += 1;
1212
+ abortActiveRequest();
1213
+ setStatus("cancelled");
1214
+ clearTimer();
1215
+ finalizing = false;
1216
+ inFlight = false;
1217
+ inFlightPromise = null;
1218
+ coalesced = false;
1219
+ try {
1220
+ await capture?.cancel?.();
1221
+ } catch {}
1222
+ if (status === "cancelled") setStatus("idle");
1223
+ return true;
1224
+ }
1225
+
1226
+ async function dispose() {
1227
+ generation += 1;
1228
+ abortActiveRequest();
1229
+ stopPromise = null;
1230
+ startupPromise = null;
1231
+ clearTimer();
1232
+ finalizing = false;
1233
+ inFlight = false;
1234
+ inFlightPromise = null;
1235
+ coalesced = false;
1236
+ try {
1237
+ await capture?.cancel?.();
1238
+ } catch {}
1239
+ try {
1240
+ transcriber?.dispose?.();
1241
+ } catch {}
1242
+ try {
1243
+ await capture?.dispose?.();
1244
+ } catch {}
1245
+ capture = null;
1246
+ transcriber = null;
1247
+ // pendingModel/model are configuration, not runtime: a later start()
1248
+ // recreates the transcriber from the same spec (model stays "loaded"
1249
+ // conceptually until setModel/dispose semantics change it in stage 2).
1250
+ setStatus("idle");
1251
+ }
1252
+
1253
+ // Teardown + recreate on model/language change. Refused mid-dictation so a
1254
+ // loaded model is never pulled out from under a running session. The next
1255
+ // start() builds the transcriber from the new spec, so the model stays
1256
+ // loaded across dictations until explicitly changed or disposed.
1257
+ // Rebind per-session callbacks when a warm controller is reused across
1258
+ // dictations: the transcriber lease (and capture shell) stay, but partials
1259
+ // and errors must reach the NEW session's editor/toast, never a stale one.
1260
+ // Only honored while idle/errored (never mid-dictation); returns false
1261
+ // otherwise so a caller cannot rewire a live session.
1262
+ function updateCallbacks(next = {}) {
1263
+ if (status === "streaming" || status === "starting" || status === "stopping") return false;
1264
+ if (next && typeof next === "object") {
1265
+ if ("onPartial" in next) onPartial = next.onPartial ?? null;
1266
+ if ("onFinal" in next) onFinal = next.onFinal ?? null;
1267
+ if ("onStateChange" in next) onStateChange = next.onStateChange ?? null;
1268
+ if ("onError" in next) onError = next.onError ?? null;
1269
+ if ("onOverflow" in next) onOverflow = next.onOverflow ?? null;
1270
+ }
1271
+ return true;
1272
+ }
1273
+ function setModel(next) {
1274
+ if (status === "streaming" || status === "starting" || status === "stopping") return false;
1275
+ try {
1276
+ transcriber?.dispose?.();
1277
+ } catch {}
1278
+ transcriber = null;
1279
+ pendingModel = next ? { ...next } : null;
1280
+ model = pendingModel;
1281
+ fallbackAudio = null;
1282
+ fallbackCoverage = null;
1283
+ lastError = null;
1284
+ tracker = trackerFactory();
1285
+ return true;
1286
+ }
1287
+
1288
+ function getState() {
1289
+ const t = currentTranscript();
1290
+ let coverage = null;
1291
+ try {
1292
+ coverage = capture?.getCoverage?.() || null;
1293
+ } catch {
1294
+ coverage = null;
1295
+ }
1296
+ return {
1297
+ status,
1298
+ stableText: t.stableText,
1299
+ tentativeText: t.tentativeText,
1300
+ inFlight,
1301
+ lastError,
1302
+ hasFallbackAudio: fallbackAudio !== null,
1303
+ fallbackCoverage,
1304
+ coverage,
1305
+ model,
1306
+ };
1307
+ }
1308
+
1309
+ return {
1310
+ start,
1311
+ stop,
1312
+ cancel,
1313
+ dispose,
1314
+ setModel,
1315
+ updateCallbacks,
1316
+ getState,
1317
+ getServerPort,
1318
+ getTranscript: currentTranscript,
1319
+ getFallbackAudio: () => fallbackAudio,
1320
+ getFallbackCoverage: () => fallbackCoverage,
1321
+ };
1322
+ }