omnirush 0.8.6 → 0.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/assets/CHANGELOG.md +120 -0
  2. package/assets/extensions/omnirush/agents-lib.ts +501 -63
  3. package/assets/extensions/omnirush/agents.ts +140 -115
  4. package/assets/extensions/omnirush/bgshell-lib.ts +432 -0
  5. package/assets/extensions/omnirush/bgshell.ts +392 -0
  6. package/assets/extensions/omnirush/capture/workspace-collector.ts +86 -22
  7. package/assets/extensions/omnirush/collector.ts +275 -43
  8. package/assets/extensions/omnirush/commands.ts +2 -0
  9. package/assets/extensions/omnirush/deliveries.ts +145 -0
  10. package/assets/extensions/omnirush/guard/UPSTREAM +2 -0
  11. package/assets/extensions/omnirush/guard/git-command-policy.ts +885 -0
  12. package/assets/extensions/omnirush/guard-lib.ts +230 -0
  13. package/assets/extensions/omnirush/guard.ts +340 -0
  14. package/assets/extensions/omnirush/index.ts +32 -6
  15. package/assets/extensions/omnirush/mcp.ts +103 -8
  16. package/assets/extensions/omnirush/memory-lib.ts +516 -0
  17. package/assets/extensions/omnirush/pi-engine.ts +115 -3
  18. package/assets/extensions/omnirush/plan-lib.ts +38 -9
  19. package/assets/extensions/omnirush/plan.ts +85 -30
  20. package/assets/extensions/omnirush/sota.ts +52 -4
  21. package/assets/extensions/omnirush/status-lib.ts +3 -0
  22. package/assets/extensions/omnirush/subagent-marker.ts +21 -0
  23. package/assets/extensions/omnirush/subagents-lib.ts +623 -0
  24. package/assets/extensions/omnirush/subagents.ts +305 -0
  25. package/assets/extensions/omnirush/swarm-lib.ts +142 -0
  26. package/assets/extensions/omnirush/swarm.ts +95 -0
  27. package/assets/extensions/omnirush/voice/capture.ts +502 -0
  28. package/assets/extensions/omnirush/voice/core/UPSTREAM +16 -0
  29. package/assets/extensions/omnirush/voice/core/file-source.ts +70 -0
  30. package/assets/extensions/omnirush/voice/core/index.ts +21 -0
  31. package/assets/extensions/omnirush/voice/core/keyterms.ts +117 -0
  32. package/assets/extensions/omnirush/voice/core/resample.ts +63 -0
  33. package/assets/extensions/omnirush/voice/core/segmenter.ts +231 -0
  34. package/assets/extensions/omnirush/voice/core/session.ts +403 -0
  35. package/assets/extensions/omnirush/voice/core/text.ts +81 -0
  36. package/assets/extensions/omnirush/voice/core/transcriber.ts +135 -0
  37. package/assets/extensions/omnirush/voice/core/types.ts +102 -0
  38. package/assets/extensions/omnirush/voice/core/wav.ts +95 -0
  39. package/assets/extensions/omnirush/voice/keys.ts +435 -0
  40. package/assets/extensions/omnirush/voice/kitty.ts +64 -0
  41. package/assets/extensions/omnirush/voice/pvrecorder-worker.cjs +43 -0
  42. package/assets/extensions/omnirush/voice/settings.ts +67 -0
  43. package/assets/extensions/omnirush/voice.ts +838 -0
  44. package/assets/extensions/omnirush/yolo-lib.ts +80 -0
  45. package/assets/extensions/omnirush/yolo.ts +85 -0
  46. package/package.json +7 -3
  47. package/scripts/brand-engine.js +526 -0
  48. package/scripts/build-all-packages.py +29 -1
  49. package/scripts/smoke-packages.py +33 -1
  50. package/src/bin.js +232 -33
  51. package/src/compat.js +272 -0
  52. package/src/lib.js +64 -0
  53. package/src/sessions.js +222 -0
  54. package/scripts/patch-pi-branding.js +0 -251
@@ -0,0 +1,403 @@
1
+ import { applyKeyterms } from "./keyterms";
2
+ import { LevelMeter, Segmenter, type Segment, type SegmenterOptions } from "./segmenter";
3
+ import { countWords, isLikelyHallucination, stitch } from "./text";
4
+ import {
5
+ VOICE_SAMPLE_RATE,
6
+ VoiceError,
7
+ isVoiceError,
8
+ type AudioSource,
9
+ type Transcriber,
10
+ type VoicePhase,
11
+ type VoiceResult,
12
+ type VoiceSnapshot,
13
+ } from "./types";
14
+ import { encodeWav } from "./wav";
15
+
16
+ /** Pauses voice after `threshold` failures within `windowMs`, for `cooldownMs`. Shared by every recording. */
17
+ export class CircuitBreaker {
18
+ private failures: number[] = [];
19
+ private openUntil = 0;
20
+
21
+ constructor(
22
+ private readonly threshold = 3,
23
+ private readonly windowMs = 10_000,
24
+ private readonly cooldownMs = 30_000,
25
+ private readonly now: () => number = Date.now,
26
+ ) {}
27
+
28
+ recordFailure(): void {
29
+ const at = this.now();
30
+ this.failures = [...this.failures.filter((time) => at - time < this.windowMs), at];
31
+ if (this.failures.length >= this.threshold) {
32
+ this.openUntil = at + this.cooldownMs;
33
+ this.failures = [];
34
+ }
35
+ }
36
+
37
+ recordSuccess(): void {
38
+ this.failures = [];
39
+ }
40
+
41
+ isOpen(): boolean {
42
+ return this.now() < this.openUntil;
43
+ }
44
+ }
45
+
46
+ export type VoiceSessionOptions = {
47
+ source: AudioSource;
48
+ transcriber: Transcriber;
49
+ language?: string | null;
50
+ keyterms?: readonly string[];
51
+ /** Uploads in flight at once. */
52
+ concurrency?: number;
53
+ /** Pauses before each retry of a failed upload; its length is the retry count. */
54
+ retryDelaysMs?: readonly number[];
55
+ /** Hard cap on one recording. */
56
+ maxRecordingMs?: number;
57
+ /** Stop by itself after this much silence (tap mode); null never. */
58
+ silenceAutoStopMs?: number | null;
59
+ /** How long stop() waits for the last uploads. */
60
+ finalizeTimeoutMs?: number;
61
+ /** Below this peak RMS the microphone is reported as delivering no signal. */
62
+ noSignalRms?: number;
63
+ segmenter?: SegmenterOptions;
64
+ breaker?: CircuitBreaker;
65
+ recordingId?: string;
66
+ /** Every state change, and level updates at most every `snapshotIntervalMs`. */
67
+ onChange?: (snapshot: VoiceSnapshot) => void;
68
+ /** The recording hit its cap or the silence limit and is finalizing on its own. */
69
+ onAutoStop?: (reason: "max_duration" | "silence") => void;
70
+ snapshotIntervalMs?: number;
71
+ now?: () => number;
72
+ };
73
+
74
+ type SlotState =
75
+ | { status: "pending" }
76
+ | { status: "done"; text: string }
77
+ | { status: "failed"; error: VoiceError };
78
+
79
+ type Job = { segment: Segment; wav: Uint8Array | null };
80
+
81
+ function randomId(): string {
82
+ const bytes = new Uint8Array(8);
83
+ globalThis.crypto.getRandomValues(bytes);
84
+ return Array.from(bytes, (byte) => byte.toString(16).padStart(2, "0")).join("");
85
+ }
86
+
87
+ function wait(ms: number, signal: AbortSignal): Promise<void> {
88
+ return new Promise((resolve) => {
89
+ if (signal.aborted || ms <= 0) return resolve();
90
+ const timer = setTimeout(done, ms);
91
+ function done() {
92
+ clearTimeout(timer);
93
+ signal.removeEventListener("abort", done);
94
+ resolve();
95
+ }
96
+ signal.addEventListener("abort", done, { once: true });
97
+ });
98
+ }
99
+
100
+ /**
101
+ * One dictation: capture → VAD segments → one upload per segment while the
102
+ * user keeps talking (at most `concurrency` at once, retried with backoff) →
103
+ * texts stitched back in recording order.
104
+ *
105
+ * Audio is disposable: each segment's PCM is encoded, uploaded and dropped
106
+ * as soon as its transcript (or final failure) is known; silence is dropped
107
+ * as it arrives; cancel drops everything. Nothing is written anywhere.
108
+ */
109
+ export class VoiceSession {
110
+ readonly recordingId: string;
111
+ private readonly options: VoiceSessionOptions;
112
+ private readonly segmenter: Segmenter;
113
+ private readonly meter = new LevelMeter();
114
+ private readonly abort = new AbortController();
115
+ private readonly now: () => number;
116
+ private phaseValue: VoicePhase = "idle";
117
+ private slots: SlotState[] = [];
118
+ private queue: Job[] = [];
119
+ private active = 0;
120
+ private startedAt = 0;
121
+ private lastSpeechAt = 0;
122
+ private lastSnapshotAt = 0;
123
+ private error: VoiceError | null = null;
124
+ private idleWaiters: Array<() => void> = [];
125
+ private capTimer: ReturnType<typeof setInterval> | null = null;
126
+ private result: VoiceResult | null = null;
127
+ private stopping: Promise<VoiceResult> | null = null;
128
+
129
+ constructor(options: VoiceSessionOptions) {
130
+ this.options = options;
131
+ this.recordingId = options.recordingId ?? randomId();
132
+ this.segmenter = new Segmenter({ sampleRate: VOICE_SAMPLE_RATE, ...options.segmenter });
133
+ this.now = options.now ?? Date.now;
134
+ }
135
+
136
+ get phase(): VoicePhase {
137
+ return this.phaseValue;
138
+ }
139
+
140
+ snapshot(): VoiceSnapshot {
141
+ return {
142
+ phase: this.phaseValue,
143
+ levels: this.meter.levels(),
144
+ elapsedMs: this.startedAt ? this.now() - this.startedAt : 0,
145
+ text: this.orderedText(true),
146
+ pending: this.slots.filter((slot) => slot.status === "pending").length,
147
+ failed: this.slots.filter((slot) => slot.status === "failed").length,
148
+ error: this.error,
149
+ };
150
+ }
151
+
152
+ async start(): Promise<void> {
153
+ if (this.phaseValue !== "idle") throw new Error("A voice session starts once");
154
+ if (this.options.breaker?.isOpen()) {
155
+ this.fail(new VoiceError("breaker_open", "Voice paused after repeated errors. Try again in a moment."));
156
+ throw this.error;
157
+ }
158
+ this.setPhase("arming");
159
+ try {
160
+ await this.options.source.start(
161
+ (frames) => this.onFrames(frames),
162
+ (error) => {
163
+ if (error) {
164
+ this.abortAll();
165
+ this.fail(error);
166
+ } else if (this.phaseValue === "recording") {
167
+ // The source ran dry (a file input): finalize like a release.
168
+ void this.stop();
169
+ }
170
+ },
171
+ );
172
+ } catch (error) {
173
+ this.options.source.stop();
174
+ this.fail(isVoiceError(error) ? error : new VoiceError("capture_failed", "The microphone could not be opened."));
175
+ throw this.error;
176
+ }
177
+ if (this.phase !== "arming") return;
178
+ this.startedAt = this.now();
179
+ this.lastSpeechAt = this.startedAt;
180
+ this.setPhase("recording");
181
+ this.capTimer = setInterval(() => this.checkCaps(), 250);
182
+ }
183
+
184
+ /** Ends capture and waits for the last segment's text. Idempotent. */
185
+ stop(): Promise<VoiceResult> {
186
+ if (this.result) return Promise.resolve(this.result);
187
+ this.stopping ??= this.finalize();
188
+ return this.stopping;
189
+ }
190
+
191
+ /** Drops the recording: capture stops, uploads are aborted, no text is returned. */
192
+ cancel(): void {
193
+ if (this.phaseValue === "done" || this.phaseValue === "cancelled") return;
194
+ this.abortAll();
195
+ this.slots = [];
196
+ this.result = { text: "", words: 0, segments: 0, error: null };
197
+ this.setPhase("cancelled");
198
+ this.releaseWaiters();
199
+ }
200
+
201
+ private abortAll(): void {
202
+ this.clearCapTimer();
203
+ this.options.source.stop();
204
+ this.abort.abort();
205
+ for (const job of this.queue) job.wav = null;
206
+ this.queue = [];
207
+ }
208
+
209
+ private async finalize(): Promise<VoiceResult> {
210
+ if (this.phaseValue === "idle" || this.phaseValue === "arming") this.cancel();
211
+ if (this.phaseValue === "cancelled" || this.phaseValue === "error") {
212
+ this.result ??= { text: "", words: 0, segments: 0, error: this.error };
213
+ return this.result;
214
+ }
215
+ this.clearCapTimer();
216
+ this.options.source.stop();
217
+ this.setPhase("finalizing");
218
+ const last = this.segmenter.flush();
219
+ if (last) this.enqueue(last);
220
+ const deadline = this.options.finalizeTimeoutMs ?? 45_000;
221
+ // CLI: the deadline timer is cleared once the uploads are done, so a
222
+ // finished recording leaves no 45 s timer holding the event loop.
223
+ const deadlineDone = new AbortController();
224
+ const timedOut = await Promise.race([
225
+ this.whenIdle().then(() => false),
226
+ wait(deadline, AbortSignal.any([this.abort.signal, deadlineDone.signal])).then(() => !deadlineDone.signal.aborted),
227
+ ]);
228
+ deadlineDone.abort();
229
+ if (this.phase === "cancelled") return this.result ?? { text: "", words: 0, segments: 0, error: null };
230
+ if (timedOut && this.active + this.queue.length > 0) {
231
+ this.abort.abort();
232
+ this.queue = [];
233
+ this.slots = this.slots.map((slot) => slot.status === "pending"
234
+ ? { status: "failed", error: new VoiceError("network", "The transcription service took too long.") }
235
+ : slot);
236
+ }
237
+ const text = this.orderedText(false);
238
+ const error = this.outcomeError(text);
239
+ this.result = { text, words: countWords(text), segments: this.slots.length, error };
240
+ this.error = error;
241
+ this.slots = [];
242
+ this.setPhase("done");
243
+ return this.result;
244
+ }
245
+
246
+ private outcomeError(text: string): VoiceError | null {
247
+ const failed = this.slots.find((slot): slot is Extract<SlotState, { status: "failed" }> => slot.status === "failed");
248
+ if (failed) return failed.error;
249
+ if (text) return null;
250
+ if (this.slots.length === 0 && this.segmenter.maxRms < (this.options.noSignalRms ?? 0.003)) {
251
+ return new VoiceError("no_signal", "No audio came from the microphone. Check the input device in Settings → Voice.");
252
+ }
253
+ return new VoiceError("no_speech", "No speech was detected.");
254
+ }
255
+
256
+ private onFrames(frames: Int16Array): void {
257
+ // A source may deliver its first frames before start() has resolved.
258
+ if (this.phaseValue !== "recording" && this.phaseValue !== "arming") return;
259
+ const threshold = this.segmenter.threshold;
260
+ const closed = this.segmenter.push(frames, (value) => {
261
+ this.meter.push(value);
262
+ if (value > threshold) this.lastSpeechAt = this.now();
263
+ });
264
+ for (const segment of closed) this.enqueue(segment);
265
+ const at = this.now();
266
+ if (closed.length > 0 || at - this.lastSnapshotAt >= (this.options.snapshotIntervalMs ?? 50)) this.emit();
267
+ }
268
+
269
+ private checkCaps(): void {
270
+ if (this.phaseValue !== "recording") return;
271
+ const at = this.now();
272
+ if (at - this.startedAt >= (this.options.maxRecordingMs ?? 120_000)) {
273
+ this.options.onAutoStop?.("max_duration");
274
+ void this.stop();
275
+ return;
276
+ }
277
+ const silence = this.options.silenceAutoStopMs;
278
+ if (silence != null && at - this.lastSpeechAt >= silence) {
279
+ this.options.onAutoStop?.("silence");
280
+ void this.stop();
281
+ }
282
+ }
283
+
284
+ private enqueue(segment: Segment): void {
285
+ this.slots[segment.index] = { status: "pending" };
286
+ this.queue.push({ segment, wav: null });
287
+ this.pump();
288
+ this.emit();
289
+ }
290
+
291
+ private pump(): void {
292
+ const limit = Math.max(1, this.options.concurrency ?? 2);
293
+ while (this.active < limit && this.queue.length > 0 && !this.abort.signal.aborted) {
294
+ const job = this.queue.shift()!;
295
+ this.active += 1;
296
+ void this.run(job).finally(() => {
297
+ this.active -= 1;
298
+ this.pump();
299
+ if (this.active === 0 && this.queue.length === 0) this.releaseWaiters();
300
+ });
301
+ }
302
+ }
303
+
304
+ private async run(job: Job): Promise<void> {
305
+ const { segment } = job;
306
+ job.wav = encodeWav(segment.samples);
307
+ // The PCM is not needed once it is encoded.
308
+ const segmentMeta = { peakRms: segment.peakRms, threshold: segment.threshold, speechMs: segment.speechMs };
309
+ const durationMs = segment.endMs - segment.startMs;
310
+ const delays = this.options.retryDelaysMs ?? [300, 1_200];
311
+ const keyterms = this.options.keyterms ?? [];
312
+ try {
313
+ for (let attempt = 0; ; attempt += 1) {
314
+ if (this.abort.signal.aborted || !job.wav) return;
315
+ try {
316
+ const { text } = await this.options.transcriber.transcribe({
317
+ wav: job.wav,
318
+ segmentIndex: segment.index,
319
+ recordingId: this.recordingId,
320
+ language: this.options.language ?? null,
321
+ keyterms,
322
+ durationMs,
323
+ }, this.abort.signal);
324
+ if (this.abort.signal.aborted) return;
325
+ this.options.breaker?.recordSuccess();
326
+ const cleaned = isLikelyHallucination(text, segmentMeta) ? "" : applyKeyterms(text.trim(), keyterms);
327
+ this.slots[segment.index] = { status: "done", text: cleaned };
328
+ return;
329
+ } catch (error) {
330
+ if (this.abort.signal.aborted) return;
331
+ const voiceError = isVoiceError(error)
332
+ ? error
333
+ : new VoiceError("network", "The transcription service could not be reached.", { retryable: true });
334
+ if (voiceError.retryable && attempt < delays.length) {
335
+ const pause = Math.min(5_000, Math.max(delays[attempt]!, voiceError.retryAfterMs ?? 0));
336
+ await wait(pause, this.abort.signal);
337
+ continue;
338
+ }
339
+ if (voiceError.kind === "network") this.options.breaker?.recordFailure();
340
+ this.slots[segment.index] = { status: "failed", error: voiceError };
341
+ return;
342
+ }
343
+ }
344
+ } finally {
345
+ // Disposable audio: the segment's bytes go as soon as its outcome is known.
346
+ job.wav = null;
347
+ segment.samples = new Int16Array(0);
348
+ this.emit();
349
+ }
350
+ }
351
+
352
+ /** Finished texts in order; with `stopAtPending`, up to the first segment still in flight. */
353
+ private orderedText(stopAtPending: boolean): string {
354
+ const parts: string[] = [];
355
+ for (const slot of this.slots) {
356
+ if (!slot) continue;
357
+ if (slot.status === "pending") {
358
+ if (stopAtPending) break;
359
+ continue;
360
+ }
361
+ if (slot.status === "done") parts.push(slot.text);
362
+ }
363
+ return stitch(parts);
364
+ }
365
+
366
+ private whenIdle(): Promise<void> {
367
+ if (this.active === 0 && this.queue.length === 0) return Promise.resolve();
368
+ return new Promise((resolve) => this.idleWaiters.push(resolve));
369
+ }
370
+
371
+ private releaseWaiters(): void {
372
+ const waiters = this.idleWaiters;
373
+ this.idleWaiters = [];
374
+ for (const resolve of waiters) resolve();
375
+ }
376
+
377
+ private clearCapTimer(): void {
378
+ if (this.capTimer) clearInterval(this.capTimer);
379
+ this.capTimer = null;
380
+ }
381
+
382
+ private fail(error: VoiceError | null): void {
383
+ this.error = error;
384
+ this.clearCapTimer();
385
+ this.slots = [];
386
+ this.setPhase("error");
387
+ this.releaseWaiters();
388
+ }
389
+
390
+ private setPhase(phase: VoicePhase): void {
391
+ this.phaseValue = phase;
392
+ this.emit();
393
+ }
394
+
395
+ private emit(): void {
396
+ this.lastSnapshotAt = this.now();
397
+ try {
398
+ this.options.onChange?.(this.snapshot());
399
+ } catch {
400
+ // A listener never breaks the recording.
401
+ }
402
+ }
403
+ }
@@ -0,0 +1,81 @@
1
+ /** Han, Hiragana, Katakana, Hangul, and CJK punctuation: written without spaces between words. */
2
+ const CJK = /[ -〿぀-ヿ㐀-䶿一-鿿가-힯豈-﫿＀-￯]/;
3
+ const CJK_GLOBAL = new RegExp(CJK.source, "gu");
4
+ /** Punctuation that attaches to the word before it. */
5
+ const CLOSING = /^[.,!?;:%)\]}'"’”…、。,!?:;)」』]/u;
6
+ const OPENING = /[([{'"‘“「『]$/u;
7
+
8
+ export function isCjk(char: string): boolean {
9
+ return CJK.test(char);
10
+ }
11
+
12
+ /** Words the way a person counts them: whitespace-separated, and every CJK character on its own. */
13
+ export function countWords(text: string): number {
14
+ const cjk = text.match(CJK_GLOBAL)?.length ?? 0;
15
+ const rest = text.replace(CJK_GLOBAL, " ").trim();
16
+ return cjk + (rest ? rest.split(/\s+/).length : 0);
17
+ }
18
+
19
+ /** Whether a space belongs between `left` and `right` (the text on either side of a join). */
20
+ function needsSpace(left: string, right: string): boolean {
21
+ if (!left || !right) return false;
22
+ const last = left.slice(-1);
23
+ const first = right.slice(0, 1);
24
+ if (/\s/.test(last) || /\s/.test(first)) return false;
25
+ if (CLOSING.test(right) || OPENING.test(left)) return false;
26
+ if (isCjk(last) && isCjk(first)) return false;
27
+ return true;
28
+ }
29
+
30
+ /** Segment texts in recording order, joined the way they were spoken. */
31
+ export function stitch(parts: readonly string[]): string {
32
+ let out = "";
33
+ for (const raw of parts) {
34
+ const part = raw.trim();
35
+ if (!part) continue;
36
+ out += needsSpace(out, part) ? ` ${part}` : part;
37
+ }
38
+ return out;
39
+ }
40
+
41
+ /**
42
+ * `text` as it should be inserted between `before` and `after` (the draft
43
+ * around the cursor): a separating space on each side where a word would
44
+ * otherwise run into its neighbour.
45
+ */
46
+ export function padInsertion(before: string, text: string, after: string): string {
47
+ const body = text.trim();
48
+ if (!body) return "";
49
+ return `${needsSpace(before, body) ? " " : ""}${body}${needsSpace(body, after) ? " " : ""}`;
50
+ }
51
+
52
+ /**
53
+ * Whisper-style models answer silence and breath with stock phrases. A
54
+ * segment whose whole text is one of these, from quiet or very short audio,
55
+ * is dropped.
56
+ */
57
+ const HALLUCINATIONS = new Set([
58
+ "thank you",
59
+ "thank you very much",
60
+ "thanks",
61
+ "thanks for watching",
62
+ "thank you for watching",
63
+ "please subscribe",
64
+ "you",
65
+ "bye",
66
+ "bye bye",
67
+ "okay",
68
+ "so",
69
+ "uh",
70
+ "um",
71
+ "hmm",
72
+ "music",
73
+ "applause",
74
+ ]);
75
+
76
+ export function isLikelyHallucination(text: string, audio: { peakRms: number; threshold: number; speechMs: number }): boolean {
77
+ const normalized = text.toLowerCase().replace(/[^\p{L}\p{N}\s]/gu, " ").replace(/\s+/g, " ").trim();
78
+ if (!normalized) return true;
79
+ if (!HALLUCINATIONS.has(normalized)) return false;
80
+ return audio.speechMs < 600 || audio.peakRms < audio.threshold * 2;
81
+ }
@@ -0,0 +1,135 @@
1
+ import { VoiceError, type TranscribeRequest, type TranscribeResult, type Transcriber } from "./types";
2
+
3
+ type FetchLike = (input: string, init: RequestInit) => Promise<Response>;
4
+
5
+ export type RemoteTranscriberOptions = {
6
+ /** POST target: the backend's /omnirush/v1/audio/transcriptions, or a local broker route in front of it. */
7
+ url: string;
8
+ /** Called per request, so a rotated bearer is picked up. */
9
+ headers?: () => Record<string, string> | Promise<Record<string, string>>;
10
+ fetch?: FetchLike;
11
+ /** Per-request deadline. */
12
+ timeoutMs?: number;
13
+ };
14
+
15
+ function retryAfterMs(response: Response): number | null {
16
+ const header = response.headers.get("retry-after");
17
+ if (!header) return null;
18
+ const seconds = Number(header);
19
+ if (Number.isFinite(seconds)) return Math.max(0, seconds * 1000);
20
+ const date = Date.parse(header);
21
+ return Number.isFinite(date) ? Math.max(0, date - Date.now()) : null;
22
+ }
23
+
24
+ /** The error code of a JSON error body: `{"detail":"x"}`, `{"error":"x"}`, `{"error":{"code":"x"}}` or `{"code":"x"}`. */
25
+ export function errorCode(body: unknown): string | null {
26
+ if (!body || typeof body !== "object") return null;
27
+ const record = body as Record<string, unknown>;
28
+ for (const value of [record.detail, record.error, record.code]) {
29
+ if (typeof value === "string" && value.trim()) return value.trim();
30
+ if (value && typeof value === "object") {
31
+ const code = (value as Record<string, unknown>).code;
32
+ if (typeof code === "string" && code.trim()) return code.trim();
33
+ }
34
+ }
35
+ return null;
36
+ }
37
+
38
+ /** Maps a transcription service answer to the error the user reads. */
39
+ export function voiceErrorFromResponse(status: number, code: string | null, retryAfter: number | null): VoiceError {
40
+ if (status === 401 || code === "omnirush_account_required") {
41
+ return new VoiceError("signed_out", "Sign in to omnirush.ai to use voice.", { status });
42
+ }
43
+ if (code === "voice_unavailable" || code === "voice_disabled" || status === 404) {
44
+ return new VoiceError("unavailable", "Voice input is not available on this account yet.", { status });
45
+ }
46
+ if (status === 422 || code === "audio_unreadable") {
47
+ return new VoiceError("unreadable", "The recording could not be read. Try again.", { status });
48
+ }
49
+ if (status === 400 || status === 415) {
50
+ return new VoiceError("unreadable", "The recording was not in a format the transcription service accepts.", { status });
51
+ }
52
+ if (status === 413 || code === "audio_too_long" || code === "audio_too_large") {
53
+ return new VoiceError("too_large", "That recording is too long to transcribe.", { status });
54
+ }
55
+ if (status === 429) {
56
+ const budget = code === "voice_daily_budget_exhausted" || code === "daily_grant_exhausted";
57
+ return new VoiceError(
58
+ "rate_limited",
59
+ code === "daily_grant_exhausted"
60
+ ? "You have used today's model allowance, which voice falls back on. It refills at 00:00 UTC."
61
+ : budget ? "You have used today's voice allowance. It refills at 00:00 UTC." : "Voice is busy. Try again in a moment.",
62
+ { status, retryable: !budget, retryAfterMs: retryAfter },
63
+ );
64
+ }
65
+ return new VoiceError("network", "The transcription service did not answer. Try again.", {
66
+ status,
67
+ retryable: status >= 500 || status === 408 || status === 0,
68
+ retryAfterMs: retryAfter,
69
+ });
70
+ }
71
+
72
+ /**
73
+ * multipart/form-data as plain bytes. Not FormData with a Blob: a browser
74
+ * may page Blob contents into its blob store on disk, and audio must only
75
+ * ever be in memory.
76
+ */
77
+ export function multipartBody(wav: Uint8Array, filename: string, fields: ReadonlyArray<[string, string]>): { body: Uint8Array<ArrayBuffer>; contentType: string } {
78
+ const random = new Uint8Array(12);
79
+ globalThis.crypto.getRandomValues(random);
80
+ const boundary = `----omnirush-voice-${Array.from(random, (byte) => byte.toString(16).padStart(2, "0")).join("")}`;
81
+ const encoder = new TextEncoder();
82
+ const head = encoder.encode(
83
+ fields.map(([name, value]) => `--${boundary}\r\nContent-Disposition: form-data; name="${name}"\r\n\r\n${value.replace(/[\r\n]+/g, " ")}\r\n`).join("")
84
+ + `--${boundary}\r\nContent-Disposition: form-data; name="file"; filename="${filename}"\r\nContent-Type: audio/wav\r\n\r\n`,
85
+ );
86
+ const tail = encoder.encode(`\r\n--${boundary}--\r\n`);
87
+ const body = new Uint8Array(head.length + wav.length + tail.length);
88
+ body.set(head, 0);
89
+ body.set(wav, head.length);
90
+ body.set(tail, head.length + wav.length);
91
+ return { body, contentType: `multipart/form-data; boundary=${boundary}` };
92
+ }
93
+
94
+ /**
95
+ * A Transcriber over HTTP: one multipart POST per segment (`file`,
96
+ * `language`, `prompt_terms`, `segment_index`, `recording_id`), answered
97
+ * with `{"text": "..."}`. The audio lives only in the request body.
98
+ */
99
+ export class RemoteTranscriber implements Transcriber {
100
+ constructor(private readonly options: RemoteTranscriberOptions) {}
101
+
102
+ async transcribe(request: TranscribeRequest, signal: AbortSignal): Promise<TranscribeResult> {
103
+ const fields: Array<[string, string]> = [["segment_index", String(request.segmentIndex)], ["recording_id", request.recordingId]];
104
+ if (request.language) fields.push(["language", request.language]);
105
+ if (request.keyterms.length > 0) fields.push(["prompt_terms", request.keyterms.join(", ")]);
106
+ const { body: upload, contentType } = multipartBody(request.wav, `segment-${request.segmentIndex}.wav`, fields);
107
+ const fetcher = this.options.fetch ?? ((input: string, init: RequestInit) => fetch(input, init));
108
+ const timeout = AbortSignal.timeout(this.options.timeoutMs ?? 30_000);
109
+ let response: Response;
110
+ try {
111
+ response = await fetcher(this.options.url, {
112
+ method: "POST",
113
+ headers: { Accept: "application/json", "Content-Type": contentType, ...(await this.options.headers?.()) },
114
+ body: upload,
115
+ signal: AbortSignal.any([signal, timeout]),
116
+ });
117
+ } catch (error) {
118
+ if (signal.aborted) throw error;
119
+ throw new VoiceError("network", "The transcription service could not be reached.", { retryable: true, status: 0 });
120
+ }
121
+ const text = await response.text().catch(() => "");
122
+ let body: unknown = null;
123
+ try {
124
+ body = text ? JSON.parse(text) : null;
125
+ } catch {
126
+ body = null;
127
+ }
128
+ if (!response.ok) throw voiceErrorFromResponse(response.status, errorCode(body), retryAfterMs(response));
129
+ const value = body && typeof body === "object" ? (body as Record<string, unknown>).text : null;
130
+ if (typeof value !== "string") {
131
+ throw new VoiceError("network", "The transcription service sent an unexpected answer.", { retryable: true, status: response.status });
132
+ }
133
+ return { text: value };
134
+ }
135
+ }