@bojackduy/opencode-voice 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/stt.js ADDED
@@ -0,0 +1,2339 @@
1
+ // Speech-to-text: sox recording, whisper-cpp or API transcription, LLM normalization.
2
+
3
+ import fs from "node:fs";
4
+ import path from "node:path";
5
+ import os from "node:os";
6
+ import { spawn, execSync } from "node:child_process";
7
+ import { getActiveSessionTitle, getRecentConversationContext } from "./session.js";
8
+
9
+ let sttApiEndpoint = null;
10
+ let sttApiModel = null;
11
+ let sttApiKeyEnv = null;
12
+
13
+ const WAV_FILENAME = "opencode-stt.wav";
14
+ let tmpDir = "/tmp";
15
+
16
+ const MODELS_DIRS = [
17
+ path.join(os.homedir(), ".local", "share", "whisper-cpp"),
18
+ "/opt/homebrew/share/whisper-cpp/models",
19
+ "/usr/local/share/whisper-cpp/models",
20
+ ];
21
+
22
+ const MODELS = {
23
+ "large-v3-turbo-q5_0": {
24
+ label: "Large v3 Turbo Q5 (recommended)",
25
+ file: "ggml-large-v3-turbo-q5_0.bin",
26
+ },
27
+ "large-v3-turbo-q8_0": { label: "Large v3 Turbo Q8", file: "ggml-large-v3-turbo-q8_0.bin" },
28
+ "large-v3-turbo": { label: "Large v3 Turbo (full)", file: "ggml-large-v3-turbo.bin" },
29
+ "medium-q5_0": { label: "Medium Q5 (multilingual, faster)", file: "ggml-medium-q5_0.bin" },
30
+ "small.en": { label: "Small English", file: "ggml-small.en.bin" },
31
+ small: { label: "Small Multilingual", file: "ggml-small.bin" },
32
+ "base.en": { label: "Base English", file: "ggml-base.en.bin" },
33
+ base: { label: "Base Multilingual", file: "ggml-base.bin" },
34
+ "tiny.en": { label: "Tiny English (fastest)", file: "ggml-tiny.en.bin" },
35
+ tiny: { label: "Tiny Multilingual (fastest)", file: "ggml-tiny.bin" },
36
+ };
37
+ const DEFAULT_MODEL = "large-v3-turbo-q5_0";
38
+
39
+ const DEFAULT_LANGUAGE = "auto";
40
+ // Curated subset for the /stt-language picker; options.sttLanguage accepts any
41
+ // whisper.cpp language code.
42
+ const LANGUAGES = {
43
+ auto: { label: "Auto-detect" },
44
+ en: { label: "English" },
45
+ zh: { label: "Chinese" },
46
+ yue: { label: "Cantonese" },
47
+ ja: { label: "Japanese" },
48
+ ko: { label: "Korean" },
49
+ de: { label: "German" },
50
+ fr: { label: "French" },
51
+ es: { label: "Spanish" },
52
+ pt: { label: "Portuguese" },
53
+ ru: { label: "Russian" },
54
+ it: { label: "Italian" },
55
+ };
56
+ let sttDefaultLanguage = DEFAULT_LANGUAGE;
57
+ // Bounds for the recent-turns knowledge block sent with normalization.
58
+ let sttContextMessages = 8;
59
+ let sttContextChars = 3000;
60
+ // interpretive (default): fix misheard words using context; strict: transcribe exactly.
61
+ let sttNormalizeMode = "interpretive";
62
+ // Worst-case budget for one normalize call; on timeout the raw transcript is
63
+ // used so a stalled LLM never blocks the turn.
64
+ let sttNormalizeTimeoutMs = 15000;
65
+
66
+ // Pronouns, demonstratives, and capitalized words signal that the transcript
67
+ // may reference the conversation (he/she/Cristina/...). Without them the
68
+ // knowledge block only adds input tokens, so it is skipped.
69
+
70
+ const CONTEXT_SIGNALS =
71
+ /\b(he|him|his|she|her|hers|they|them|their|it|its|this|that|these|those|anh|chị|em|cô|chú|bác|ông|bà|nó|họ|này|kia|đó|ấy)\b|[A-ZÀ-ỴĐ][a-zà-ỹ]{1,}/u;
72
+
73
+ export function needsContext(rawText) {
74
+ if (!rawText || rawText.length < 3) return false;
75
+ return CONTEXT_SIGNALS.test(rawText);
76
+ }
77
+
78
+ // Prefetched conversation context, kicked off while the user is still
79
+ // speaking so the fetch usually finishes before the turn ends.
80
+
81
+ let prefetchedContext = null; // { at, sessionID, promise }
82
+
83
+ function currentRouteSessionID(api) {
84
+ const route = api?.route?.current;
85
+ return route?.name === "session" ? route?.params?.sessionID || null : null;
86
+ }
87
+
88
+ async function fetchContextParts(client, api) {
89
+ const [sessionTitle, recent] = await Promise.all([
90
+ getActiveSessionTitle(client),
91
+ getRecentConversationContext(client, api, {
92
+ maxMessages: sttContextMessages,
93
+ maxChars: sttContextChars,
94
+ }),
95
+ ]);
96
+ return { sessionTitle, recent };
97
+ }
98
+
99
+ function prefetchContextParts(client, api) {
100
+ try {
101
+ prefetchedContext = {
102
+ at: Date.now(),
103
+ sessionID: currentRouteSessionID(api),
104
+ promise: fetchContextParts(client, api),
105
+ };
106
+ } catch {
107
+ prefetchedContext = null;
108
+ }
109
+ }
110
+
111
+ export function isOpenRouterEndpoint(endpoint) {
112
+ return /(^https?:\/\/)?([^/]+\.)?openrouter\.ai(\/|$)/i.test(endpoint || "");
113
+ }
114
+
115
+ function buildMultipartTranscriptionRequest(model, audioBuffer, apiKey) {
116
+ const blob = new Blob([audioBuffer], { type: "audio/wav" });
117
+ const form = new FormData();
118
+ form.append("file", blob, "audio.wav");
119
+ form.append("model", model);
120
+ form.append("response_format", "json");
121
+
122
+ const headers = {};
123
+ if (apiKey) headers["Authorization"] = "Bearer " + apiKey;
124
+
125
+ return {
126
+ headers,
127
+ body: form,
128
+ };
129
+ }
130
+
131
+ export function buildOpenRouterTranscriptionRequest(model, audioBuffer, apiKey) {
132
+ const headers = { "Content-Type": "application/json" };
133
+ if (apiKey) headers["Authorization"] = "Bearer " + apiKey;
134
+
135
+ const payload = {
136
+ model,
137
+ input_audio: {
138
+ data: audioBuffer.toString("base64"),
139
+ format: "wav",
140
+ },
141
+ };
142
+
143
+ return {
144
+ headers,
145
+ body: JSON.stringify(payload),
146
+ };
147
+ }
148
+
149
+ function getModelsDir() {
150
+ for (const dir of MODELS_DIRS) {
151
+ if (fs.existsSync(dir)) return dir;
152
+ }
153
+ return MODELS_DIRS[0];
154
+ }
155
+
156
+ // ---- Audio backend detection (coreaudio / pulseaudio / sox default) ----
157
+
158
+ export function detectAudioBackend() {
159
+ if (process.platform === "darwin") return "coreaudio";
160
+ try {
161
+ execSync("pactl --version", { stdio: "ignore", timeout: 3000 });
162
+ return "pulseaudio";
163
+ } catch {
164
+ return "default";
165
+ }
166
+ }
167
+
168
+ // ---- Audio server diagnostics ----
169
+
170
+ export function isWSL() {
171
+ return Boolean(process.env.WSL_DISTRO_NAME || process.env.WSL_INTEROP);
172
+ }
173
+
174
+ // Unlike `pactl --version`, `pactl info` actually connects to the server.
175
+ function pulseServerHealth() {
176
+ try {
177
+ execSync("pactl info", { stdio: "ignore", timeout: 3000 });
178
+ return { ok: true };
179
+ } catch {
180
+ return { ok: false };
181
+ }
182
+ }
183
+
184
+ // User-facing hint for missing devices / recording failures. On WSL the audio
185
+ // server is WSLg's PulseAudio, which can wedge and needs a `wsl --shutdown`.
186
+ export function buildAudioHint({ backend, serverOk, isWsl }) {
187
+ if (backend !== "pulseaudio") return "No input devices found";
188
+ if (serverOk) return "No input devices found - check your audio input source configuration";
189
+ if (isWsl) {
190
+ return 'Audio server unreachable. On WSL, WSLg\'s PulseAudio may be stuck - run "wsl --shutdown" on Windows, then reopen Ubuntu';
191
+ }
192
+ return "Audio server unreachable - check that PipeWire/PulseAudio is running";
193
+ }
194
+
195
+ // Appends a server-health hint to recording failure messages when the
196
+ // PulseAudio server is unreachable (e.g. wedged WSLg on WSL).
197
+ function audioFailureSuffix(backend) {
198
+ if (backend !== "pulseaudio") return "";
199
+ if (pulseServerHealth().ok) return "";
200
+ return `. ${buildAudioHint({ backend, serverOk: false, isWsl: isWSL() })}`;
201
+ }
202
+
203
+ // Input device descriptors: name is the value passed to sox, label is shown in the UI.
204
+ export function parsePactlSources(jsonText) {
205
+ const data = JSON.parse(jsonText);
206
+ return (Array.isArray(data) ? data : [])
207
+ .filter((s) => s?.name && !s.name.endsWith(".monitor"))
208
+ .map((s) => ({
209
+ name: s.name,
210
+ label: s.description ? `${s.description} (${s.name})` : s.name,
211
+ }));
212
+ }
213
+
214
+ export function parsePactlSourcesShort(text) {
215
+ return text
216
+ .split("\n")
217
+ .map((line) => line.trim().split(/\s+/)[1])
218
+ .filter((name) => name && !name.endsWith(".monitor"))
219
+ .map((name) => ({ name, label: name }));
220
+ }
221
+
222
+ function listInputDevices(backend) {
223
+ if (backend === "coreaudio") {
224
+ try {
225
+ const json = execSync("system_profiler SPAudioDataType -json 2>/dev/null", {
226
+ encoding: "utf-8",
227
+ timeout: 5000,
228
+ });
229
+ const data = JSON.parse(json);
230
+ return (data.SPAudioDataType?.[0]?._items || [])
231
+ .filter((d) => d.coreaudio_input_source != null)
232
+ .map((d) => {
233
+ const name = d.coreaudio_device_name || d._name;
234
+ return { name, label: name };
235
+ });
236
+ } catch {
237
+ return [];
238
+ }
239
+ }
240
+ if (backend === "pulseaudio") {
241
+ try {
242
+ const json = execSync("pactl -f json list sources 2>/dev/null", {
243
+ encoding: "utf-8",
244
+ timeout: 5000,
245
+ });
246
+ return parsePactlSources(json);
247
+ } catch {
248
+ try {
249
+ const out = execSync("pactl list sources short 2>/dev/null", {
250
+ encoding: "utf-8",
251
+ timeout: 5000,
252
+ });
253
+ return parsePactlSourcesShort(out);
254
+ } catch {
255
+ return [];
256
+ }
257
+ }
258
+ }
259
+ return [];
260
+ }
261
+
262
+ export function buildRecordArgs(backend, mic) {
263
+ if (backend === "pulseaudio") return ["-t", "pulseaudio", mic || "default"];
264
+ if (backend === "coreaudio" && mic) return ["-t", "coreaudio", mic];
265
+ return ["-d"];
266
+ }
267
+
268
+ // ---- Recording state and control ----
269
+
270
+ let soxProc = null;
271
+ let soxStderr = "";
272
+ let recording = false;
273
+ let processing = false;
274
+
275
+ // ---- Conversation-mode hooks (wired by index.js) ----
276
+ // When voice conversation is active, the plain record/submit keys defer to the
277
+ // conversation loop instead of running the one-shot pipeline.
278
+
279
+ let conversationHooks = { isActive: null, onKey: null };
280
+
281
+ export function __setConversationHooks(hooks) {
282
+ conversationHooks = { ...conversationHooks, ...hooks };
283
+ }
284
+
285
+ function conversationActive() {
286
+ try {
287
+ return conversationHooks.isActive?.() === true;
288
+ } catch {
289
+ return false;
290
+ }
291
+ }
292
+
293
+ // Live notes owns the microphone continuously while recording; one-shot and
294
+ // conversation-mode STT must stay out of the way while it is active.
295
+
296
+ let liveNotesHooks = { isActive: null };
297
+
298
+ export function __setLiveNotesHooks(hooks) {
299
+ liveNotesHooks = { ...liveNotesHooks, ...hooks };
300
+ }
301
+
302
+ function liveNotesActive() {
303
+ try {
304
+ return liveNotesHooks.isActive?.() === true;
305
+ } catch {
306
+ return false;
307
+ }
308
+ }
309
+
310
+ // Read-only guards for other modes (live notes) to check before starting,
311
+ // and for one-shot/conversation to refuse while live notes owns the mic.
312
+
313
+ export function isSttBusy() {
314
+ return recording || processing;
315
+ }
316
+
317
+ // ---- Streaming dictation mode (stage 2) ----
318
+ // `sttMode: "batch"` (default) preserves the one-shot pipeline exactly.
319
+ // `sttMode: "streaming"` selects live local dictation (no LLM, zero quota).
320
+
321
+ let streamingDictationActive = false;
322
+
323
+ export function isStreamingActive() {
324
+ return streamingDictationActive;
325
+ }
326
+
327
+ export function __setStreamingActive(v) {
328
+ streamingDictationActive = !!v;
329
+ }
330
+
331
+ export const STREAMING_MODE_DEFAULTS = {
332
+ windowMs: 10000,
333
+ cadenceMs: 1000,
334
+ };
335
+
336
+ export function resolveSttMode(opts) {
337
+ return opts?.sttMode === "streaming" ? "streaming" : "batch";
338
+ }
339
+
340
+ export function resolveStreamTimings(opts) {
341
+ const windowMs =
342
+ Number(opts?.sttStreamWindowMs) > 0
343
+ ? Math.floor(Number(opts.sttStreamWindowMs))
344
+ : STREAMING_MODE_DEFAULTS.windowMs;
345
+ const cadenceMs =
346
+ Number(opts?.sttStreamStepMs) > 0
347
+ ? Math.floor(Number(opts.sttStreamStepMs))
348
+ : STREAMING_MODE_DEFAULTS.cadenceMs;
349
+ return { windowMs, cadenceMs };
350
+ }
351
+
352
+ // Warm-lease policy for streaming dictation (stage-2 follow-up): the shared
353
+ // whisper-server lease stays loaded across finalize/cancel and is released
354
+ // only when the server itself is at fault, on model/language change, or on
355
+ // plugin unload. Per-dictation failures (capture/mic) preserve the lease.
356
+
357
+ // Error codes where the mic/capture failed but the server is healthy.
358
+ const STREAMING_PER_DICTATION_CODES = new Set([
359
+ "CAPTURE_FAILED",
360
+ "CAPTURE_SPAWN_ENOENT",
361
+ "CAPTURE_SPAWN_FAILED",
362
+ "CAPTURE_SPAWN_ERROR",
363
+ "CAPTURE_EXITED",
364
+ "AUDIO_SPOOL_OVERFLOW",
365
+ ]);
366
+
367
+ export function isStreamingServerFault(err) {
368
+ if (!err) return false;
369
+ const code = typeof err === "string" ? null : err?.code || null;
370
+ if (code && STREAMING_PER_DICTATION_CODES.has(code)) return false;
371
+ const message = typeof err === "string" ? err : err?.message || "";
372
+ if (/^capture|^sox|^mic/i.test(message || "")) return false;
373
+ // Any explicit error object/string beyond per-dictation capture faults is
374
+ // treated as server-side (NOT_READY, START_TIMEOUT, PORT_IN_USE,
375
+ // SERVER_BINARY_MISSING, BAD_STATUS, REQUEST_FAILED, STOP_NOT_READY,
376
+ // STOP_FINAL_TIMEOUT, TRANSCRIBE_*). Absence of error means healthy.
377
+ return true;
378
+ }
379
+
380
+ export function streamModelKey(modelPath, language, port) {
381
+ const base = `${modelPath || ""}::${language || "auto"}`;
382
+ // The bound whisper-server port is part of the warm-lease identity: two
383
+ // TUIs are separate processes (no shared state), but within one process
384
+ // leases on different ports must not be mistaken for each other. Omitted
385
+ // port keeps the legacy "model::language" form.
386
+ return port === undefined || port === null ? base : `${base}::${port}`;
387
+ }
388
+
389
+ export function getSttApiConfig() {
390
+ return sttApiEndpoint && sttApiModel
391
+ ? { endpoint: sttApiEndpoint, model: sttApiModel, apiKeyEnv: sttApiKeyEnv }
392
+ : null;
393
+ }
394
+
395
+ export function getTmpDir() {
396
+ return tmpDir;
397
+ }
398
+
399
+ // ---- Sticky recording toast (Option A: polling to simulate persistence) ----
400
+ // ---- Sticky processing toast (Transcribing... / Normalizing...) ----
401
+ // Single-shot toasts (duration 3000ms) expire long before whisper/LLM finish,
402
+ // leaving a silent gap where the user thinks STT died. Poll like the recording
403
+ // toast so the stage stays visible until it completes.
404
+
405
+ let processingToastTimer = null;
406
+ let processingToastFn = null;
407
+ let processingMessage = "";
408
+ const PROCESSING_TOAST_DURATION = 3000;
409
+ const PROCESSING_TOAST_INTERVAL = 2500;
410
+
411
+ let recordingToastTimer = null;
412
+ let recordingToastFn = null;
413
+ const RECORDING_TOAST_BASE = "● Recording";
414
+ const RECORDING_TOAST_DURATION = 3000;
415
+ const RECORDING_TOAST_INTERVAL = 2500;
416
+ // Live wave + timer config (PR2)
417
+ let recordingStartMs = null;
418
+ let prevWaveLevels = [];
419
+ let noiseFloorRms = null;
420
+ let calibrationEndMs = null;
421
+ const WAVE_CHARS = ["▁", "▂", "▃", "▄", "▅", "▆", "▇", "█"];
422
+ const WAVE_LEN = 16;
423
+ const WAVE_INTERVAL = 35;
424
+ const WAVE_DURATION = 70;
425
+ let useWave = true;
426
+
427
+ export function __setRecordingToastFn(fn) {
428
+ recordingToastFn = fn;
429
+ }
430
+
431
+ export function __clearRecordingToastState() {
432
+ if (recordingToastTimer) {
433
+ clearInterval(recordingToastTimer);
434
+ recordingToastTimer = null;
435
+ }
436
+ recordingToastFn = null;
437
+ recordingStartMs = null;
438
+ prevWaveLevels = [];
439
+ noiseFloorRms = null;
440
+ calibrationEndMs = null;
441
+ }
442
+
443
+ export function __setUseWave(v) {
444
+ useWave = !!v;
445
+ }
446
+
447
+ export function isRecordingToastActive() {
448
+ return recordingToastTimer !== null;
449
+ }
450
+
451
+ export function __setProcessingToastFn(fn) {
452
+ processingToastFn = fn;
453
+ }
454
+
455
+ export function __clearProcessingToastState() {
456
+ if (processingToastTimer) {
457
+ clearInterval(processingToastTimer);
458
+ processingToastTimer = null;
459
+ }
460
+ processingToastFn = null;
461
+ processingMessage = "";
462
+ }
463
+
464
+ export function isProcessingToastActive() {
465
+ return processingToastTimer !== null;
466
+ }
467
+
468
+ export function showProcessingToast(message) {
469
+ clearProcessingToast();
470
+ processingMessage = message;
471
+ if (!processingToastFn) return;
472
+ processingToastFn({
473
+ message: processingMessage,
474
+ variant: "info",
475
+ duration: PROCESSING_TOAST_DURATION,
476
+ });
477
+ processingToastTimer = setInterval(() => {
478
+ if (processingToastFn) {
479
+ processingToastFn({
480
+ message: processingMessage,
481
+ variant: "info",
482
+ duration: PROCESSING_TOAST_DURATION,
483
+ });
484
+ }
485
+ }, PROCESSING_TOAST_INTERVAL);
486
+ }
487
+
488
+ export function updateProcessingToast(message) {
489
+ if (!isProcessingToastActive()) {
490
+ showProcessingToast(message);
491
+ return;
492
+ }
493
+ processingMessage = message;
494
+ if (processingToastFn) {
495
+ processingToastFn({
496
+ message: processingMessage,
497
+ variant: "info",
498
+ duration: PROCESSING_TOAST_DURATION,
499
+ });
500
+ }
501
+ }
502
+
503
+ export function clearProcessingToast() {
504
+ if (processingToastTimer) {
505
+ clearInterval(processingToastTimer);
506
+ processingToastTimer = null;
507
+ }
508
+ }
509
+
510
+ // ---- Sticky streaming status (never-silent dictation) ----
511
+ // Streaming dictation previously showed one 3s toast on start, then silence
512
+ // until the next partial/finalize - long cold loads and slow ticks read as
513
+ // crashes. This reuses the polling-toast pattern above: one sticky status,
514
+ // re-emitted before the single-shot duration expires, with an elapsed timer
515
+ // so every state (loading / live / finalizing) stays visibly alive.
516
+ // Partial preview toasts (🎙 ...) continue alongside, not instead.
517
+ // Privacy-safe: elapsed time + state words only, never audio paths/secrets.
518
+
519
+ let streamingToastTimer = null;
520
+ let streamingToastFn = null;
521
+ let streamingToastBase = "";
522
+ let streamingToastStartMs = null;
523
+ let streamingLastPartialMs = null;
524
+ let streamingSlowAfterMs = 2000;
525
+ const STREAMING_TOAST_DURATION = 3000;
526
+ const STREAMING_TOAST_INTERVAL = 2500;
527
+
528
+ export function __setStreamingToastFn(fn) {
529
+ streamingToastFn = fn;
530
+ }
531
+
532
+ export function __clearStreamingToastState() {
533
+ if (streamingToastTimer) {
534
+ clearInterval(streamingToastTimer);
535
+ streamingToastTimer = null;
536
+ }
537
+ streamingToastFn = null;
538
+ streamingToastBase = "";
539
+ streamingToastStartMs = null;
540
+ streamingLastPartialMs = null;
541
+ }
542
+
543
+ export function isStreamingToastActive() {
544
+ return streamingToastTimer !== null;
545
+ }
546
+
547
+ function buildStreamingMessage() {
548
+ const elapsed =
549
+ streamingToastStartMs === null ? "00:00" : formatElapsed(Date.now() - streamingToastStartMs);
550
+ let message = `${streamingToastBase} · ${elapsed}`;
551
+ // Slow-tick note: no partial for ~2x cadence while live - the mic is fine,
552
+ // the inference is just slow. Trivially detectable via last-partial time.
553
+ if (
554
+ streamingLastPartialMs !== null &&
555
+ Date.now() - streamingLastPartialMs > streamingSlowAfterMs
556
+ ) {
557
+ message += " (listening…)";
558
+ }
559
+ return message;
560
+ }
561
+
562
+ function emitStreamingToast() {
563
+ if (!streamingToastFn) return;
564
+ streamingToastFn({
565
+ message: buildStreamingMessage(),
566
+ variant: "info",
567
+ duration: STREAMING_TOAST_DURATION,
568
+ });
569
+ }
570
+
571
+ export function showStreamingToast(message, { slowAfterMs } = {}) {
572
+ clearStreamingToast();
573
+ streamingToastBase = message;
574
+ streamingToastStartMs = Date.now();
575
+ streamingLastPartialMs = null;
576
+ if (Number(slowAfterMs) > 0) streamingSlowAfterMs = Math.floor(Number(slowAfterMs));
577
+ if (!streamingToastFn) return;
578
+ emitStreamingToast();
579
+ streamingToastTimer = setInterval(() => {
580
+ emitStreamingToast();
581
+ }, STREAMING_TOAST_INTERVAL);
582
+ }
583
+
584
+ export function updateStreamingToast(message) {
585
+ if (!isStreamingToastActive()) {
586
+ showStreamingToast(message);
587
+ return;
588
+ }
589
+ streamingToastBase = message;
590
+ emitStreamingToast();
591
+ }
592
+
593
+ export function noteStreamingPartial() {
594
+ streamingLastPartialMs = Date.now();
595
+ }
596
+
597
+ export function clearStreamingToast() {
598
+ if (streamingToastTimer) {
599
+ clearInterval(streamingToastTimer);
600
+ streamingToastTimer = null;
601
+ }
602
+ streamingToastBase = "";
603
+ streamingToastStartMs = null;
604
+ streamingLastPartialMs = null;
605
+ }
606
+
607
+ export function formatElapsed(ms) {
608
+ const totalSec = Math.floor(ms / 1000);
609
+ const m = String(Math.floor(totalSec / 60)).padStart(2, "0");
610
+ const s = String(totalSec % 60).padStart(2, "0");
611
+ return `${m}:${s}`;
612
+ }
613
+
614
+ export function rmsToChar(rms) {
615
+ if (rms < 0.008) return WAVE_CHARS[0];
616
+ if (rms < 0.018) return WAVE_CHARS[1];
617
+ if (rms < 0.035) return WAVE_CHARS[2];
618
+ if (rms < 0.06) return WAVE_CHARS[3];
619
+ if (rms < 0.1) return WAVE_CHARS[4];
620
+ if (rms < 0.16) return WAVE_CHARS[5];
621
+ if (rms < 0.24) return WAVE_CHARS[6];
622
+ return WAVE_CHARS[7];
623
+ }
624
+
625
+ export function rmsToLevel(rms) {
626
+ if (rms < 0.008) return 0;
627
+ if (rms < 0.018) return 1;
628
+ if (rms < 0.035) return 2;
629
+ if (rms < 0.06) return 3;
630
+ if (rms < 0.1) return 4;
631
+ if (rms < 0.16) return 5;
632
+ if (rms < 0.24) return 6;
633
+ return 7;
634
+ }
635
+
636
+ export function computeWaveChar() {
637
+ // kept for tests / fallback single-char
638
+ const wavFile = path.join(tmpDir, WAV_FILENAME);
639
+ try {
640
+ const data = fs.readFileSync(wavFile);
641
+ if (data.length <= 44) return WAVE_CHARS[0];
642
+ const windowSize = 2560;
643
+ const start = Math.max(44, data.length - windowSize);
644
+ const slice = data.subarray(start);
645
+ const samples = Math.floor(slice.length / 2);
646
+ if (samples === 0) return WAVE_CHARS[0];
647
+ let sum = 0;
648
+ for (let i = 0; i < slice.length - 1; i += 2) {
649
+ const s = slice.readInt16LE(i);
650
+ const n = s / 32768;
651
+ sum += n * n;
652
+ }
653
+ const rms = Math.sqrt(sum / samples);
654
+ if (rms < 0.003) return WAVE_CHARS[0];
655
+ return rmsToChar(rms);
656
+ } catch {
657
+ return WAVE_CHARS[0];
658
+ }
659
+ }
660
+
661
+ export function getWaveString() {
662
+ // Fixed-column snapshot: each column = noise-gated RMS of sub-slice of last ~0.08s window
663
+ const wavFile = path.join(tmpDir, WAV_FILENAME);
664
+ try {
665
+ let slice;
666
+ try {
667
+ const fd = fs.openSync(wavFile, "r");
668
+ const stat = fs.fstatSync(fd);
669
+ const fileSize = stat.size;
670
+ if (fileSize <= 44) {
671
+ fs.closeSync(fd);
672
+ prevWaveLevels = Array(WAVE_LEN).fill(0);
673
+ return WAVE_CHARS[0].repeat(WAVE_LEN);
674
+ }
675
+ const windowSize = 1024; // ~0.032s — ultra instant for toast
676
+ const readSize = Math.min(windowSize, fileSize - 44);
677
+ const buf = Buffer.alloc(readSize);
678
+ fs.readSync(fd, buf, 0, readSize, fileSize - readSize);
679
+ fs.closeSync(fd);
680
+ slice = buf;
681
+ } catch {
682
+ const data = fs.readFileSync(wavFile);
683
+ if (data.length <= 44) {
684
+ prevWaveLevels = Array(WAVE_LEN).fill(0);
685
+ return WAVE_CHARS[0].repeat(WAVE_LEN);
686
+ }
687
+ const windowSize = 1024;
688
+ const start = Math.max(44, data.length - windowSize);
689
+ slice = data.subarray(start);
690
+ }
691
+ const chunkBytes = Math.max(2, Math.floor(slice.length / WAVE_LEN));
692
+ const rawRms = [];
693
+ const levels = [];
694
+ for (let c = 0; c < WAVE_LEN; c++) {
695
+ const off = c * chunkBytes;
696
+ const chunk = slice.subarray(off, Math.min(off + chunkBytes, slice.length));
697
+ const samples = Math.floor(chunk.length / 2);
698
+ if (samples === 0) {
699
+ rawRms.push(0);
700
+ continue;
701
+ }
702
+ let sum = 0;
703
+ for (let i = 0; i < chunk.length - 1; i += 2) {
704
+ const s = chunk.readInt16LE(i);
705
+ const n = s / 32768;
706
+ sum += n * n;
707
+ }
708
+ const rms = Math.sqrt(sum / samples);
709
+ rawRms.push(rms);
710
+ }
711
+ // calibration: first 120ms estimates fan noise floor, but still show live wave
712
+ const now = Date.now();
713
+ if (calibrationEndMs && now < calibrationEndMs) {
714
+ const minRms = Math.min(...rawRms);
715
+ if (noiseFloorRms === null || minRms < noiseFloorRms) noiseFloorRms = minRms;
716
+ }
717
+ if (noiseFloorRms === null) noiseFloorRms = Math.min(...rawRms);
718
+ else {
719
+ // slowly adapt noise floor down (fan may vary) but never jump up quickly
720
+ const minRms = Math.min(...rawRms);
721
+ // if we see a consistently lower min for 500ms, drift down 5%
722
+ noiseFloorRms = Math.min(noiseFloorRms, minRms);
723
+ noiseFloorRms = noiseFloorRms * 0.995 + minRms * 0.005;
724
+ }
725
+ // console.log(`[wave] noiseFloor=${noiseFloorRms?.toFixed(4)} rawMin=${Math.min(...rawRms).toFixed(4)} rawMax=${Math.max(...rawRms).toFixed(4)}`);
726
+ for (const rms of rawRms) {
727
+ const effective = Math.max(0, rms - noiseFloorRms * 1.15);
728
+ // gate: very small effective still flat — fan residue
729
+ if (effective < 0.005) levels.push(0);
730
+ else levels.push(rmsToLevel(effective * 1.6)); // 1.6x gain makes voice pop higher
731
+ }
732
+ // instant attack, very fast decay (0.2) → fan doesn't linger, voice explicit
733
+ if (prevWaveLevels.length !== WAVE_LEN) prevWaveLevels = Array(WAVE_LEN).fill(0);
734
+ const smoothed = levels.map((lvl, i) => {
735
+ const prev = prevWaveLevels[i] || 0;
736
+ if (lvl > prev) return lvl; // instant rise
737
+ const decayed = Math.floor(prev * 0.2);
738
+ return lvl > decayed ? lvl : decayed;
739
+ });
740
+ prevWaveLevels = smoothed.slice();
741
+ return smoothed.map((l) => WAVE_CHARS[l]).join("");
742
+ } catch {
743
+ prevWaveLevels = Array(WAVE_LEN).fill(0);
744
+ return WAVE_CHARS[0].repeat(WAVE_LEN);
745
+ }
746
+ }
747
+
748
+ let currentRecordKeybind = "<leader>[";
749
+ export function __setRecordKeybind(kb) {
750
+ if (kb) currentRecordKeybind = kb.replace(/^<leader>/, "leader+");
751
+ }
752
+
753
+ // Temporary stop-hint override for voice conversation mode, whose key differs
754
+ // from stt.record. Pass null to restore the previous hint.
755
+
756
+ let savedRecordKeybind = null;
757
+ export function __setStopHint(hint) {
758
+ if (hint) {
759
+ if (savedRecordKeybind === null) savedRecordKeybind = currentRecordKeybind;
760
+ __setRecordKeybind(hint);
761
+ } else if (savedRecordKeybind !== null) {
762
+ currentRecordKeybind = savedRecordKeybind;
763
+ savedRecordKeybind = null;
764
+ }
765
+ }
766
+ export function buildRecordingMessage() {
767
+ // When the conversation loop overrides the stop hint, the toast says so
768
+ // explicitly - otherwise users wonder why it names a key they did not press.
769
+ const mode = savedRecordKeybind !== null ? " (conversation)" : "";
770
+ if (!useWave || recordingStartMs === null)
771
+ return `${RECORDING_TOAST_BASE}${mode} · ${currentRecordKeybind} to stop`;
772
+ const elapsed = formatElapsed(Date.now() - recordingStartMs);
773
+ const wave = getWaveString();
774
+ return `${RECORDING_TOAST_BASE}${mode} ${wave} ${elapsed} · ${currentRecordKeybind} to stop`;
775
+ }
776
+
777
+ export function showRecordingToast() {
778
+ clearRecordingToast();
779
+ if (!recordingToastFn) return;
780
+ recordingStartMs = Date.now();
781
+ prevWaveLevels = Array(WAVE_LEN).fill(0);
782
+ noiseFloorRms = null;
783
+ calibrationEndMs = Date.now() + 120;
784
+ const interval = useWave ? WAVE_INTERVAL : RECORDING_TOAST_INTERVAL;
785
+ const duration = useWave ? WAVE_DURATION : RECORDING_TOAST_DURATION;
786
+ recordingToastFn({
787
+ message: buildRecordingMessage(),
788
+ variant: "info",
789
+ duration,
790
+ });
791
+ recordingToastTimer = setInterval(() => {
792
+ if (recordingToastFn) {
793
+ recordingToastFn({
794
+ message: buildRecordingMessage(),
795
+ variant: "info",
796
+ duration,
797
+ });
798
+ }
799
+ }, interval);
800
+ }
801
+
802
+ export function clearRecordingToast() {
803
+ if (recordingToastTimer) {
804
+ clearInterval(recordingToastTimer);
805
+ recordingToastTimer = null;
806
+ }
807
+ recordingStartMs = null;
808
+ noiseFloorRms = null;
809
+ calibrationEndMs = null;
810
+ }
811
+
812
+ function forceKillSox(logger) {
813
+ if (soxProc) {
814
+ try {
815
+ process.kill(soxProc.pid, "SIGKILL");
816
+ logger?.log("STT", `Killed sox pid=${soxProc.pid}`, "debug");
817
+ } catch {}
818
+ soxProc = null;
819
+ }
820
+ try {
821
+ execSync("pkill -9 -f 'sox.*opencode-stt'", { stdio: "ignore" });
822
+ } catch {}
823
+ }
824
+
825
+ function startRecording(kv, backend, toast, logger, trimSilence = true) {
826
+ if (soxProc) {
827
+ logger?.log("STT", "Start recording skipped: sox already running", "debug");
828
+ return;
829
+ }
830
+
831
+ const wavFile = path.join(tmpDir, WAV_FILENAME);
832
+ forceKillSox(logger);
833
+ try {
834
+ fs.unlinkSync(wavFile);
835
+ } catch {}
836
+
837
+ soxStderr = "";
838
+ const mic = kv.get("stt.mic", "") || null;
839
+ const inputArgs = buildRecordArgs(backend, mic);
840
+ const spawnT0 = Date.now();
841
+ logger?.log(
842
+ "STT",
843
+ `Starting recording backend=${backend} mic=${mic || "system default"}`,
844
+ "debug",
845
+ );
846
+
847
+ const silenceArgs = trimSilence ? ["silence", "1", "0.05", "1%"] : [];
848
+ soxProc = spawn(
849
+ "sox",
850
+ [...inputArgs, "-r", "16000", "-c", "1", "-b", "16", wavFile, ...silenceArgs],
851
+ {
852
+ stdio: ["ignore", "ignore", "pipe"],
853
+ detached: false,
854
+ },
855
+ );
856
+ logger?.log("STT", `sox spawned pid=${soxProc?.pid} after ${Date.now() - spawnT0}ms`, "debug");
857
+
858
+ let firstStderrLogged = false;
859
+ soxProc.stderr.on("data", (chunk) => {
860
+ if (!firstStderrLogged) {
861
+ firstStderrLogged = true;
862
+ logger?.log("STT", `sox first output after ${Date.now() - spawnT0}ms`, "debug");
863
+ }
864
+ soxStderr += chunk.toString();
865
+ });
866
+
867
+ soxProc.on("error", (err) => {
868
+ soxProc = null;
869
+ logger?.log("STT", `Recording failed: ${err.message}`, "error");
870
+ if (recording) {
871
+ recording = false;
872
+ clearRecordingToast();
873
+ toast(`Recording failed: ${err.message}${audioFailureSuffix(backend)}`, "error");
874
+ }
875
+ });
876
+
877
+ soxProc.on("exit", (code) => {
878
+ soxProc = null;
879
+ logger?.log(
880
+ "STT",
881
+ `sox exited code=${code} stderr=${soxStderr.trim()}`,
882
+ code === 0 || code === null ? "debug" : "warn",
883
+ );
884
+ if (recording && !processing) {
885
+ recording = false;
886
+ clearRecordingToast();
887
+ if (code !== 0 && code !== null) {
888
+ const errLine = soxStderr.trim().split("\n").pop();
889
+ toast(
890
+ `Recording error: ${errLine || `sox exited (code=${code})`}${audioFailureSuffix(backend)}`,
891
+ "error",
892
+ );
893
+ } else {
894
+ toast("Recording stopped unexpectedly", "warning");
895
+ }
896
+ }
897
+ });
898
+
899
+ recording = true;
900
+ }
901
+
902
+ function stopRecording(logger) {
903
+ logger?.log("STT", "Stopping recording", "debug");
904
+ if (soxProc) soxProc.kill("SIGINT");
905
+ }
906
+
907
+ async function waitForSoxExit(logger, timeoutMs = 2000) {
908
+ const start = Date.now();
909
+ while (soxProc && Date.now() - start < timeoutMs) {
910
+ await new Promise((r) => setTimeout(r, 100));
911
+ }
912
+ if (soxProc) {
913
+ logger?.log("STT", "sox did not stop before timeout", "warn");
914
+ forceKillSox(logger);
915
+ }
916
+ }
917
+
918
+ function getModelName(kv) {
919
+ const model = kv.get("stt.model", DEFAULT_MODEL);
920
+ return MODELS[model] ? model : DEFAULT_MODEL;
921
+ }
922
+
923
+ export function getModelPath(kv) {
924
+ return path.join(getModelsDir(), MODELS[getModelName(kv)].file);
925
+ }
926
+
927
+ export function getLanguage(kv) {
928
+ return kv.get("stt.language") || sttDefaultLanguage;
929
+ }
930
+
931
+ export function buildWhisperArgs(modelPath, wavFile, language) {
932
+ return ["-m", modelPath, "-f", wavFile, "-l", language || DEFAULT_LANGUAGE, "-np", "-nt"];
933
+ }
934
+
935
+ // Transcribe an arbitrary WAV file with local whisper-cli. Generalized out of
936
+ // the one-shot `transcribe()` path so live notes can transcribe its own
937
+ // per-chunk WAV files with the same model/language resolution and error
938
+ // handling, without depending on the one-shot recording's tmpDir/state.
939
+
940
+ export function transcribeFileLocal(wavFile, modelPath, language, logger) {
941
+ logger?.log(
942
+ "STT",
943
+ `Local transcription requested model=${modelPath} language=${language}`,
944
+ "debug",
945
+ );
946
+ if (!fs.existsSync(modelPath)) {
947
+ logger?.log("STT", `Whisper model missing: ${modelPath}`, "error");
948
+ return Promise.resolve({
949
+ error: `Model not found: ${modelPath}. Download from huggingface.co/ggerganov/whisper.cpp`,
950
+ });
951
+ }
952
+ if (!fs.existsSync(wavFile)) {
953
+ logger?.log("STT", `Recording file missing: ${wavFile}`, "error");
954
+ return Promise.resolve({ error: "No recording file - sox may have failed to capture audio" });
955
+ }
956
+ if (fs.statSync(wavFile).size <= 44) {
957
+ logger?.log("STT", `Recording file empty: ${wavFile}`, "warn");
958
+ return Promise.resolve({ error: "Recording is empty - no audio captured" });
959
+ }
960
+
961
+ return new Promise((resolve) => {
962
+ let stdout = "";
963
+ let stderr = "";
964
+ const proc = spawn("whisper-cli", buildWhisperArgs(modelPath, wavFile, language), {
965
+ stdio: ["ignore", "pipe", "pipe"],
966
+ });
967
+ logger?.log("STT", `Started whisper-cli pid=${proc.pid}`, "debug");
968
+
969
+ proc.stdout.on("data", (chunk) => {
970
+ stdout += chunk.toString();
971
+ });
972
+ proc.stderr.on("data", (chunk) => {
973
+ stderr += chunk.toString();
974
+ });
975
+
976
+ const timer = setTimeout(() => {
977
+ proc.kill("SIGKILL");
978
+ logger?.log("STT", "whisper-cli timed out after 60s", "error");
979
+ resolve({ error: "Transcription timed out (60s)" });
980
+ }, 60000);
981
+
982
+ proc.on("error", (err) => {
983
+ clearTimeout(timer);
984
+ logger?.log("STT", `whisper-cli error: ${err.message}`, "error");
985
+ resolve({ error: `Transcription failed: ${err.message}` });
986
+ });
987
+
988
+ proc.on("exit", (code) => {
989
+ clearTimeout(timer);
990
+ // whisper-cli exits 0 even for an unknown language, printing the error to
991
+ // stderr instead; surface it rather than reporting "no speech detected".
992
+ const langError = stderr.match(/error: unknown language '([^']+)'/);
993
+ if (langError) {
994
+ logger?.log("STT", `whisper-cli rejected language: ${langError[1]}`, "error");
995
+ resolve({ error: `Unknown whisper language: ${langError[1]}` });
996
+ return;
997
+ }
998
+ if (code !== 0) {
999
+ logger?.log("STT", `whisper-cli exited code=${code} stderr=${stderr.trim()}`, "error");
1000
+ resolve({ error: stderr.trim().split("\n").pop() || `whisper-cli exited (code=${code})` });
1001
+ return;
1002
+ }
1003
+ logger?.log("STT", `Local transcription succeeded stdoutChars=${stdout.length}`, "debug");
1004
+ resolve({
1005
+ text: stdout
1006
+ .replace(/\[.*?\]/g, "")
1007
+ .replace(/\(.*?\)/g, "")
1008
+ .replace(/\s+/g, " ")
1009
+ .trim(),
1010
+ });
1011
+ });
1012
+ });
1013
+ }
1014
+
1015
+ function transcribe(kv, logger) {
1016
+ const wavFile = path.join(tmpDir, WAV_FILENAME);
1017
+ return transcribeFileLocal(wavFile, getModelPath(kv), getLanguage(kv), logger);
1018
+ }
1019
+
1020
+ export const STT_SYSTEM_PROMPT = `You are a DETERMINISTIC speech-to-text normalizer for a coding assistant CLI. You are NOT a chatbot. You NEVER answer, explain, greet back, or introduce yourself.
1021
+
1022
+ Task: Return ONLY what the user meant to say, cleaned up for use as their message. Fix what the speech recognizer misheard, not just punctuation. No reply. No extra sentences.
1023
+
1024
+ Strict rules:
1025
+ - Output is the user's utterance, cleaned. It is NOT a response to the user.
1026
+ - Fix punctuation, capitalization, grammar.
1027
+ - Remove filler words (um, uh, like, you know, etc.).
1028
+ - Keep technical terms, file names, and code references exact.
1029
+ - If the user is dictating code, format it appropriately.
1030
+ - INTERPRET misheard words: speech recognition often produces acoustically-close-but-wrong words ("they face" for "database", "bait" for "page"). Replace a word or short phrase ONLY when ALL of these hold: (1) it sounds similar to the replacement, (2) the <knowledge> conversation context or the domain lists below support the replacement, (3) the sentence makes clearly more sense with it. Otherwise leave the original words untouched.
1031
+ - Use the <knowledge> conversation context to resolve ambiguous names, pronouns, references, AND misheard content words (that function, the file, it; e.g. Cristina uses he/him if the conversation says so) - never copy unrelated facts from the context into the output.
1032
+ - Output ONLY the cleaned text, nothing else. No quotes, no prefixes, no suffixes.
1033
+ - Do NOT add greetings, introductions, offers to help, questions, or commentary.
1034
+ - Do NOT expand short inputs. Keep output length close to input length (within ~30%); do not add paragraphs or new requests the user did not make.
1035
+ - If input is a short greeting/noise like "hey", "hello", "hi", "hey there", output exactly that greeting capitalized with a period (e.g., "hey" -> "Hey.") — DO NOT expand to "Hello! I am an AI..." or add self-introduction.
1036
+
1037
+ Examples (follow exactly):
1038
+ Input: "hey"
1039
+ Output: Hey.
1040
+
1041
+ Input: "hello there"
1042
+ Output: Hello there.
1043
+
1044
+ Input: "umm add tests for the a sink user service"
1045
+ Output: Add tests for the async user service.
1046
+
1047
+ Input: "check the locks for the doc container"
1048
+ Output: Check the logs for the Docker container.
1049
+
1050
+ Input: "craft a notion bait tracking our work items"
1051
+ Output: Craft a Notion page tracking our work items.
1052
+
1053
+ Input: "check the notion they face for the next phase target"
1054
+ Output: Check the Notion database for the next phase target.
1055
+
1056
+ If you cannot normalize, return the input trimmed.
1057
+
1058
+ CRITICAL DOMAIN CORRECTIONS - Fix common STT homophone errors in software engineering contexts:
1059
+ - "locks" -> "logs" (unless explicitly talking about mutexes/concurrency)
1060
+ - "note" / "no" -> "node"
1061
+ - "app and" -> "append"
1062
+ - "sink" -> "sync"
1063
+ - "a sink" -> "async"
1064
+ - "doc" / "talker" -> "docker"
1065
+ - "cash" -> "cache"
1066
+ - "rap" -> "wrap"
1067
+ - "Jason" -> "JSON"
1068
+ - "get" -> "Git"
1069
+ - "react" -> "React"
1070
+ - "types creep" / "type script" -> "TypeScript"
1071
+ - "bite" -> "byte"
1072
+ - "string" -> "String"
1073
+ - "int" -> "Int"
1074
+ - "bullion" -> "boolean"
1075
+
1076
+ WORKFLOW VOCABULARY - the user works with these tools; prefer these terms when the transcript sounds similar:
1077
+ - "bait" / "paid" -> "page" (in Notion contexts)
1078
+ - "they face" / "data base" -> "database"
1079
+ - "phase target" - keep as-is (not "face target")
1080
+ - "master" - keep as-is (git branch, not "muster")
1081
+ - "commit" - keep as-is (not "comet")
1082
+ - "notion" - keep as-is (not "motion" / "ocean")
1083
+
1084
+ Rely heavily on context to fix words that sound similar to programming terminology.`;
1085
+
1086
+ // Strict mode: the old transcribe-exactly behavior (no reinterpretation).
1087
+ // Opt in via "sttNormalizeMode": "strict".
1088
+
1089
+ export const STT_SYSTEM_PROMPT_STRICT = `You are a DETERMINISTIC speech-to-text normalizer for a coding assistant CLI. You are NOT a chatbot. You NEVER answer, explain, greet back, or introduce yourself.
1090
+
1091
+ Task: Return ONLY the cleaned version of the user's spoken words — exactly what they said, with punctuation and homophone fixes. No reply. No extra sentences.
1092
+
1093
+ Strict rules:
1094
+ - Output is the user's utterance, cleaned. It is NOT a response to the user.
1095
+ - Fix punctuation, capitalization, grammar minimally.
1096
+ - Remove filler words (um, uh, like, you know, etc.).
1097
+ - Keep technical terms, file names, and code references exact.
1098
+ - If the user is dictating code, format it appropriately.
1099
+ - Use the <knowledge> conversation context only to resolve ambiguous names, pronouns, and references (that function, the file, it; e.g. Cristina uses he/him if the conversation says so) — do not invent new content and never copy facts from the context into the output.
1100
+ - Output ONLY the cleaned text, nothing else. No quotes, no prefixes, no suffixes.
1101
+ - Do NOT add greetings, introductions, offers to help, questions, or commentary.
1102
+ - Do NOT expand short inputs. Keep output length close to input length; do not add paragraphs.
1103
+ - If input is a short greeting/noise like "hey", "hello", "hi", "hey there", output exactly that greeting capitalized with a period (e.g., "hey" -> "Hey.") — DO NOT expand to "Hello! I am an AI..." or add self-introduction.
1104
+
1105
+ Examples (follow exactly):
1106
+ Input: "hey"
1107
+ Output: Hey.
1108
+
1109
+ Input: "hello there"
1110
+ Output: Hello there.
1111
+
1112
+ Input: "umm add tests for the a sink user service"
1113
+ Output: Add tests for the async user service.
1114
+
1115
+ Input: "check the locks for the doc container"
1116
+ Output: Check the logs for the Docker container.
1117
+
1118
+ If you cannot normalize, return the input trimmed.
1119
+
1120
+ CRITICAL DOMAIN CORRECTIONS - Fix common STT homophone errors in software engineering contexts:
1121
+ - "locks" -> "logs" (unless explicitly talking about mutexes/concurrency)
1122
+ - "note" / "no" -> "node"
1123
+ - "app and" -> "append"
1124
+ - "sink" -> "sync"
1125
+ - "a sink" -> "async"
1126
+ - "doc" / "talker" -> "docker"
1127
+ - "cash" -> "cache"
1128
+ - "rap" -> "wrap"
1129
+ - "Jason" -> "JSON"
1130
+ - "get" -> "Git"
1131
+ - "react" -> "React"
1132
+ - "types creep" / "type script" -> "TypeScript"
1133
+ - "bite" -> "byte"
1134
+ - "string" -> "String"
1135
+ - "int" -> "Int"
1136
+ - "bullion" -> "boolean"
1137
+
1138
+ Rely heavily on context to fix words that sound similar to programming terminology.`;
1139
+
1140
+ export function selectSttSystemPrompt(mode, custom) {
1141
+ if (custom) return custom;
1142
+ return mode === "strict" ? STT_SYSTEM_PROMPT_STRICT : STT_SYSTEM_PROMPT;
1143
+ }
1144
+
1145
+ // Live notes: cleans ONE chunk of a continuous recording (meeting, lecture)
1146
+ // into readable prose. Different job than STT_SYSTEM_PROMPT's "message to
1147
+ // submit" framing - this is note-taking, so it must not shorten, summarize,
1148
+ // or address anyone; it stays close to what was actually said.
1149
+
1150
+ export const NOTES_SYSTEM_PROMPT = `You are a DETERMINISTIC transcript cleaner for a live meeting/lecture note-taking tool. You are NOT a chatbot. You NEVER answer, summarize, or add commentary.
1151
+
1152
+ Task: Clean up ONE short segment of a continuous spoken recording (already split from a longer session) into readable, accurate prose for a written transcript.
1153
+
1154
+ Rules:
1155
+ - Fix punctuation, capitalization, and speech-recognition mistakes (words that sound similar to the correct word), using the <knowledge> context (the tail of the immediately preceding segment) when it helps.
1156
+ - Preserve facts, names, numbers, and technical terms exactly. Do not translate the spoken language.
1157
+ - Remove filler words (um, uh, like, you know) and false starts/stutters, but keep everything else the speaker said - do NOT shorten, paraphrase, or summarize. This is a transcript, not a summary.
1158
+ - Output length must stay close to the input length (within ~20%).
1159
+ - If the segment is empty, pure noise, or a speech-recognizer hallucination on silence (e.g. "Thank you for watching", "[Music]", "Subscribe", "Thanks for watching!"), output nothing.
1160
+ - Output ONLY the cleaned segment text. No labels, no quotes, no timestamps, no commentary.
1161
+
1162
+ If you cannot clean it, return the input trimmed.`;
1163
+
1164
+ // whisper.cpp loves to "hear" YouTube outros on quiet/noise - English AND
1165
+ // Vietnamese training data leaks through. Real lecture speech is rarely an
1166
+ // exact outro CTA, so phrase-level matches are safe to drop.
1167
+ const NOTES_HALLUCINATION_PATTERNS = [
1168
+ /thank(s| you) for watching/i,
1169
+ /thanks for watching/i,
1170
+ /please (subscribe|like and subscribe)/i,
1171
+ /see you (next time|in the next video)/i,
1172
+ /^\[.*?(music|applause|noise|silence|blank_audio).*?\]$/i,
1173
+ // Vietnamese YouTube-outro hallucinations (seen verbatim in real sessions)
1174
+ /hãy subscribe cho kênh/i,
1175
+ /đăng k[ýy] kênh/i,
1176
+ /ủng hộ kênh/i,
1177
+ /ghiền mì gõ/i,
1178
+ /đừng quên (like|đăng k[ýy]|nhấn)/i,
1179
+ /nhấn chuông/i,
1180
+ /video hấp dẫn/i,
1181
+ /cảm ơn .* (xem|theo dõi)/i,
1182
+ /các bạn hãy .* kênh của mình nhé/i,
1183
+ ];
1184
+
1185
+ export function isLikelyWhisperHallucination(text) {
1186
+ const t = (text || "").trim();
1187
+ if (!t || t.length > 140) return false;
1188
+ return NOTES_HALLUCINATION_PATTERNS.some((re) => re.test(t));
1189
+ }
1190
+
1191
+ const STT_HALLUCINATION_PATTERNS = [
1192
+ /I am (an )?AI/i,
1193
+ /as an AI/i,
1194
+ /language model/i,
1195
+ /nice to meet you/i,
1196
+ /how can I help/i,
1197
+ /hello! I am/i,
1198
+ /I'm here to help/i,
1199
+ ];
1200
+
1201
+ function isHallucinated(raw, normalized) {
1202
+ if (!normalized) return false;
1203
+ if (STT_HALLUCINATION_PATTERNS.some((re) => re.test(normalized))) return true;
1204
+ // blow-up: short input -> long output is hallucination (e.g., "hey" -> paragraph)
1205
+ if (raw.length <= 20 && normalized.length > raw.length * 4 + 20) return true;
1206
+ if (raw.length <= 10 && normalized.split(/\s+/).length > 10) return true;
1207
+ return false;
1208
+ }
1209
+
1210
+ function fallbackForRaw(raw) {
1211
+ return raw.trim();
1212
+ }
1213
+
1214
+ export async function normalizeTranscription(
1215
+ complete,
1216
+ rawText,
1217
+ contextBlock,
1218
+ systemPrompt,
1219
+ logger,
1220
+ ) {
1221
+ logger?.log(
1222
+ "STT",
1223
+ `Normalizing transcription chars=${rawText.length} contextChars=${contextBlock?.length || 0}`,
1224
+ "debug",
1225
+ );
1226
+ const prompt = contextBlock
1227
+ ? `<knowledge>\n${contextBlock}\n</knowledge>\n<task>Clean up this speech-to-text transcription:\n\n${rawText}</task>`
1228
+ : `Clean up this speech-to-text transcription:\n\n${rawText}`;
1229
+ const result = await Promise.race([
1230
+ complete({
1231
+ system: systemPrompt,
1232
+ prompt,
1233
+ // Spoken turns are short; a tight cap bounds worst-case latency so a
1234
+ // rambling model cannot stall the turn (overrides global maxTokens).
1235
+ config: { temperature: 0, maxTokens: 1024 },
1236
+ }),
1237
+ new Promise((resolve) =>
1238
+ setTimeout(
1239
+ () =>
1240
+ resolve({
1241
+ text: null,
1242
+ error: `Normalization timed out after ${sttNormalizeTimeoutMs}ms`,
1243
+ timedOut: true,
1244
+ }),
1245
+ sttNormalizeTimeoutMs,
1246
+ ),
1247
+ ),
1248
+ ]);
1249
+ if (result?.timedOut) {
1250
+ logger?.log("STT", `Normalization timeout after ${sttNormalizeTimeoutMs}ms, using raw`, "warn");
1251
+ }
1252
+ if (result.text && isHallucinated(rawText, result.text)) {
1253
+ logger?.log(
1254
+ "STT",
1255
+ `Hallucination detected rawLen=${rawText.length} normalizedLen=${result.text.length} -> fallback`,
1256
+ "warn",
1257
+ );
1258
+ return { text: fallbackForRaw(rawText) };
1259
+ }
1260
+ return result;
1261
+ }
1262
+
1263
+ async function getApiModels(logger) {
1264
+ if (!sttApiEndpoint) return [];
1265
+ try {
1266
+ const url = sttApiEndpoint.endsWith("/")
1267
+ ? `${sttApiEndpoint}models`
1268
+ : `${sttApiEndpoint}/models`;
1269
+ const headers = {};
1270
+ if (sttApiKeyEnv && process.env[sttApiKeyEnv]) {
1271
+ headers["Authorization"] = "Bearer " + process.env[sttApiKeyEnv];
1272
+ }
1273
+ const resp = await fetch(url, { headers, signal: AbortSignal.timeout(5000) });
1274
+ logger?.log("STT", `Fetched STT API models status=${resp.status}`, resp.ok ? "debug" : "warn");
1275
+ if (!resp.ok) return [];
1276
+ const data = await resp.json();
1277
+ return (data.data || [])
1278
+ .filter((m) => m.id && /whisper/i.test(m.id))
1279
+ .map((m) => ({ value: m.id, label: m.id }));
1280
+ } catch (err) {
1281
+ logger?.log("STT", `Failed to fetch STT API models: ${err.message}`, "error");
1282
+ return [];
1283
+ }
1284
+ }
1285
+
1286
+ export async function transcribeApiFile(wavFile, endpoint, model, apiKeyEnv, logger) {
1287
+ if (!endpoint || !model) {
1288
+ logger?.log("STT", "STT API transcription skipped: API not configured", "warn");
1289
+ return { error: "STT API not configured" };
1290
+ }
1291
+ logger?.log("STT", `STT API transcription requested model=${model}`, "debug");
1292
+
1293
+ if (!fs.existsSync(wavFile)) {
1294
+ logger?.log("STT", `Recording file missing: ${wavFile}`, "error");
1295
+ return { error: "No recording file - sox may have failed to capture audio" };
1296
+ }
1297
+ if (fs.statSync(wavFile).size <= 44) {
1298
+ logger?.log("STT", `Recording file empty: ${wavFile}`, "warn");
1299
+ return { error: "Recording is empty - no audio captured" };
1300
+ }
1301
+
1302
+ try {
1303
+ const audioBuffer = await fs.promises.readFile(wavFile);
1304
+ const apiKey = apiKeyEnv ? process.env[apiKeyEnv] : null;
1305
+ const useOpenRouterFormat = isOpenRouterEndpoint(endpoint);
1306
+
1307
+ const url = endpoint.endsWith("/")
1308
+ ? `${endpoint}audio/transcriptions`
1309
+ : `${endpoint}/audio/transcriptions`;
1310
+
1311
+ const request = useOpenRouterFormat
1312
+ ? buildOpenRouterTranscriptionRequest(model, audioBuffer, apiKey)
1313
+ : buildMultipartTranscriptionRequest(model, audioBuffer, apiKey);
1314
+
1315
+ const resp = await fetch(url, {
1316
+ method: "POST",
1317
+ headers: request.headers,
1318
+ body: request.body,
1319
+ signal: AbortSignal.timeout(60000),
1320
+ });
1321
+ logger?.log("STT", `STT API response status=${resp.status}`, resp.ok ? "debug" : "error");
1322
+
1323
+ if (!resp.ok) {
1324
+ const responseBody = await resp.text();
1325
+ let msg = `STT API error ${resp.status}`;
1326
+ try {
1327
+ const err = JSON.parse(responseBody);
1328
+ msg = err?.error?.message || msg;
1329
+ } catch {}
1330
+ return { error: msg };
1331
+ }
1332
+
1333
+ let data;
1334
+ try {
1335
+ data = await resp.json();
1336
+ } catch (err) {
1337
+ logger?.log("STT", `STT API returned invalid JSON: ${err.message}`, "error");
1338
+ return { error: `STT API returned invalid JSON: ${err.message}` };
1339
+ }
1340
+ logger?.log("STT", `STT API transcription succeeded chars=${data.text?.length || 0}`, "debug");
1341
+ return { text: data.text?.trim() || "" };
1342
+ } catch (err) {
1343
+ logger?.log("STT", `STT API request failed: ${err.message}`, "error");
1344
+ if (err.name === "TimeoutError" || err.name === "AbortError") {
1345
+ return { error: "STT API request timed out (60s)" };
1346
+ }
1347
+ return { error: `STT API request failed: ${err.message}` };
1348
+ }
1349
+ }
1350
+
1351
+ async function transcribeApi(kv, logger) {
1352
+ const wavFile = path.join(tmpDir, WAV_FILENAME);
1353
+ const model = kv.get("stt.api.model") || sttApiModel;
1354
+ return transcribeApiFile(wavFile, sttApiEndpoint, model, sttApiKeyEnv, logger);
1355
+ }
1356
+
1357
+ export function insertIntoFocusedInput(renderer, text, submit = false) {
1358
+ const focused = renderer?.currentFocusedRenderable;
1359
+ if (!focused || typeof focused.insertText !== "function") return false;
1360
+
1361
+ try {
1362
+ focused.insertText(text);
1363
+ if (submit && typeof focused.submit === "function") focused.submit();
1364
+ return true;
1365
+ } catch {
1366
+ return false;
1367
+ }
1368
+ }
1369
+
1370
+ async function appendTranscription(client, renderer, text, submit) {
1371
+ if (insertIntoFocusedInput(renderer, text, submit)) {
1372
+ return;
1373
+ }
1374
+
1375
+ let appendResult = await client.tui.appendPrompt({ body: { text } });
1376
+
1377
+ if (appendResult?.error?.data?.message === "Expected object, got undefined") {
1378
+ appendResult = await client.tui.appendPrompt({ text });
1379
+ }
1380
+
1381
+ if (appendResult?.error) {
1382
+ throw new Error(
1383
+ `appendPrompt failed: ${appendResult.error.data?.message || appendResult.error.name}`,
1384
+ );
1385
+ }
1386
+
1387
+ if (submit) {
1388
+ await client.tui.submitPrompt();
1389
+ }
1390
+ }
1391
+
1392
+ // Transcribe the finished recording and normalize it, without touching the
1393
+ // prompt. Leaves the sticky "Normalizing..." toast up on success so the caller
1394
+ // can submit/speak without a silent gap; error/empty paths clear it and reset
1395
+ // the flags themselves.
1396
+
1397
+ async function transcribeTurn(kv, complete, client, api, toast, systemPrompt, logger) {
1398
+ processing = true;
1399
+ clearRecordingToast();
1400
+ logger?.log("STT", "Turn transcription started", "debug");
1401
+ stopRecording(logger);
1402
+ await waitForSoxExit(logger);
1403
+
1404
+ try {
1405
+ showProcessingToast("Transcribing...");
1406
+ const tTranscribe = Date.now();
1407
+ const result = sttApiEndpoint ? await transcribeApi(kv, logger) : await transcribe(kv, logger);
1408
+ const transcribeMs = Date.now() - tTranscribe;
1409
+
1410
+ if (result.error) {
1411
+ clearProcessingToast();
1412
+ processing = false;
1413
+ recording = false;
1414
+ logger?.log("STT", `Transcription failed: ${result.error}`, "error");
1415
+ toast(result.error, "error");
1416
+ return { error: result.error };
1417
+ }
1418
+ if (!result.text) {
1419
+ clearProcessingToast();
1420
+ processing = false;
1421
+ recording = false;
1422
+ logger?.log("STT", "Transcription produced no text", "warn");
1423
+ toast("No speech detected", "warning");
1424
+ return { text: null, empty: true };
1425
+ }
1426
+
1427
+ updateProcessingToast("Normalizing...");
1428
+ const tContext = Date.now();
1429
+ const sessionID = currentRouteSessionID(api);
1430
+ const pref = prefetchedContext;
1431
+ prefetchedContext = null;
1432
+ const { sessionTitle, recent } =
1433
+ pref && Date.now() - pref.at < 60000 && pref.sessionID === sessionID
1434
+ ? await pref.promise
1435
+ : await fetchContextParts(client, api);
1436
+ const contextMs = Date.now() - tContext;
1437
+ const withTurns = needsContext(result.text);
1438
+ const contextBlock = [
1439
+ sessionTitle ? `[session: "${sessionTitle}"]` : null,
1440
+ withTurns && recent ? recent : null,
1441
+ ]
1442
+ .filter(Boolean)
1443
+ .join("\n");
1444
+ const tNormalize = Date.now();
1445
+ const llmResult = await normalizeTranscription(
1446
+ complete,
1447
+ result.text,
1448
+ contextBlock,
1449
+ systemPrompt,
1450
+ logger,
1451
+ );
1452
+ const normalizeMs = Date.now() - tNormalize;
1453
+
1454
+ if (!llmResult.text) {
1455
+ clearProcessingToast();
1456
+ processing = false;
1457
+ recording = false;
1458
+ logger?.log("STT", `Normalization failed, using raw input: ${llmResult.error}`, "warn");
1459
+ toast(`Normalization failed, using raw input: ${llmResult.error}`, "warning");
1460
+ return { text: result.text, fallback: true };
1461
+ }
1462
+
1463
+ logger?.log(
1464
+ "STT",
1465
+ `Turn timings transcribeMs=${transcribeMs} contextMs=${contextMs} contextTurns=${withTurns ? 1 : 0} normalizeMs=${normalizeMs} rawChars=${result.text.length}`,
1466
+ "debug",
1467
+ );
1468
+ return { text: llmResult.text };
1469
+ } catch (err) {
1470
+ clearProcessingToast();
1471
+ processing = false;
1472
+ recording = false;
1473
+ logger?.log("STT", `Turn transcription error: ${err.message}`, "error");
1474
+ toast(`STT error: ${err.message}`, "error");
1475
+ return { error: err.message };
1476
+ }
1477
+ }
1478
+
1479
+ async function submitTurnText(client, renderer, toast, text, logger) {
1480
+ try {
1481
+ await appendTranscription(client, renderer, text, true);
1482
+ clearProcessingToast();
1483
+ logger?.log("STT", `Turn submitted chars=${text.length}`, "debug");
1484
+ toast("Transcription submitted", "success");
1485
+ return { text };
1486
+ } catch (err) {
1487
+ clearProcessingToast();
1488
+ logger?.log("STT", `Turn submit failed: ${err.message}`, "error");
1489
+ toast(`STT error: ${err.message}`, "error");
1490
+ return { error: err.message };
1491
+ } finally {
1492
+ processing = false;
1493
+ recording = false;
1494
+ }
1495
+ }
1496
+
1497
+ async function appendTurnText(client, renderer, toast, text, logger) {
1498
+ try {
1499
+ await appendTranscription(client, renderer, text, false);
1500
+ clearProcessingToast();
1501
+ logger?.log("STT", `Turn appended chars=${text.length}`, "debug");
1502
+ toast("Transcription added to prompt", "success");
1503
+ return { text };
1504
+ } catch (err) {
1505
+ clearProcessingToast();
1506
+ logger?.log("STT", `Turn append failed: ${err.message}`, "error");
1507
+ toast(`STT error: ${err.message}`, "error");
1508
+ return { error: err.message };
1509
+ } finally {
1510
+ processing = false;
1511
+ recording = false;
1512
+ }
1513
+ }
1514
+
1515
+ // Drop a transcribed turn without submitting (e.g. voice stop phrase).
1516
+ // Clears the sticky toast left up by transcribeTurn and resets the flags.
1517
+
1518
+ function discardTurn() {
1519
+ clearProcessingToast();
1520
+ processing = false;
1521
+ recording = false;
1522
+ }
1523
+
1524
+ // Cancel an in-progress recording without transcribing.
1525
+
1526
+ function cancelRecording(logger) {
1527
+ recording = false;
1528
+ clearRecordingToast();
1529
+ forceKillSox(logger);
1530
+ }
1531
+
1532
+ async function doTranscribePipeline(
1533
+ kv,
1534
+ complete,
1535
+ client,
1536
+ api,
1537
+ toast,
1538
+ systemPrompt,
1539
+ submit = false,
1540
+ logger,
1541
+ renderer,
1542
+ ) {
1543
+ logger?.log("STT", `Pipeline started submit=${submit}`, "debug");
1544
+ const turn = await transcribeTurn(kv, complete, client, api, toast, systemPrompt, logger);
1545
+ if (!turn.text) return;
1546
+ if (submit) {
1547
+ await submitTurnText(client, renderer, toast, turn.text, logger);
1548
+ } else {
1549
+ await appendTurnText(client, renderer, toast, turn.text, logger);
1550
+ }
1551
+ }
1552
+
1553
+ // ---- Public API for TUI plugin ----
1554
+
1555
+ export function registerSTT(api, kv, complete, prompts, opts, logger, deps = {}) {
1556
+ if (deps.isConversationActive || deps.onConversationKey) {
1557
+ __setConversationHooks({
1558
+ isActive: deps.isConversationActive,
1559
+ onKey: deps.onConversationKey,
1560
+ });
1561
+ }
1562
+ if (deps.isLiveNotesActive) {
1563
+ __setLiveNotesHooks({ isActive: deps.isLiveNotesActive });
1564
+ }
1565
+ const client = api.client;
1566
+ const renderer = opts?.focusMode === "primary" ? null : api.renderer;
1567
+ const systemPrompt = selectSttSystemPrompt(sttNormalizeMode, prompts?.stt);
1568
+ const backend = detectAudioBackend();
1569
+ logger?.log("STT", `Audio backend=${backend}`, "debug");
1570
+ function toast(message, variant = "info") {
1571
+ api.ui.toast({ message, variant, duration: 3000 });
1572
+ }
1573
+
1574
+ __setRecordingToastFn((input) => api.ui.toast(input));
1575
+ __setProcessingToastFn((input) => api.ui.toast(input));
1576
+ __setStreamingToastFn((input) => api.ui.toast(input));
1577
+ api.lifecycle?.onDispose?.(() => {
1578
+ __clearRecordingToastState();
1579
+ __clearProcessingToastState();
1580
+ __clearStreamingToastState();
1581
+ });
1582
+
1583
+ if (opts?.sttEndpoint) {
1584
+ sttApiEndpoint = opts.sttEndpoint;
1585
+ sttApiModel = opts.sttModel || "whisper-large-v3-turbo";
1586
+ sttApiKeyEnv = opts.sttApiKeyEnv || null;
1587
+ logger?.log(
1588
+ "STT",
1589
+ `Configured STT API endpoint=${sttApiEndpoint} model=${sttApiModel}`,
1590
+ "debug",
1591
+ );
1592
+ }
1593
+
1594
+ if (opts?.sttLanguage) {
1595
+ sttDefaultLanguage = opts.sttLanguage;
1596
+ }
1597
+ logger?.log("STT", `Default language=${sttDefaultLanguage}`, "debug");
1598
+
1599
+ if (opts?.sttNormalizeMode === "strict") {
1600
+ sttNormalizeMode = "strict";
1601
+ }
1602
+ logger?.log("STT", `Normalize mode=${sttNormalizeMode}`, "debug");
1603
+
1604
+ if (Number(opts?.sttContextMessages) > 0) {
1605
+ sttContextMessages = Math.floor(Number(opts.sttContextMessages));
1606
+ }
1607
+ if (Number(opts?.sttContextChars) > 0) {
1608
+ sttContextChars = Math.floor(Number(opts.sttContextChars));
1609
+ }
1610
+ if (Number(opts?.sttNormalizeTimeoutMs) > 0) {
1611
+ sttNormalizeTimeoutMs = Math.floor(Number(opts.sttNormalizeTimeoutMs));
1612
+ }
1613
+ logger?.log(
1614
+ "STT",
1615
+ `Normalize context messages=${sttContextMessages} chars=${sttContextChars}`,
1616
+ "debug",
1617
+ );
1618
+
1619
+ tmpDir = opts?.tmpDir || "/tmp";
1620
+ try {
1621
+ fs.mkdirSync(tmpDir, { recursive: true });
1622
+ } catch (err) {
1623
+ logger?.log("STT", `Failed to create tmpDir ${tmpDir}: ${err.message}`, "warn");
1624
+ }
1625
+ logger?.log("STT", `STT temp dir=${tmpDir}`, "debug");
1626
+
1627
+ const DEFAULT_KEYBINDS = {
1628
+ "stt.record": "<leader>[",
1629
+ };
1630
+ function kb(value) {
1631
+ const kb = opts?.keybinds;
1632
+ if (!kb || typeof kb !== "object" || Array.isArray(kb)) return DEFAULT_KEYBINDS[value];
1633
+ if (!Object.prototype.hasOwnProperty.call(kb, value)) return DEFAULT_KEYBINDS[value];
1634
+ const v = kb[value];
1635
+ if (!v || v === "none") return undefined;
1636
+ return v;
1637
+ }
1638
+ __setRecordKeybind(kb("stt.record") || "<leader>[");
1639
+
1640
+ // Start a recording for the voice-conversation loop. Returns false when busy.
1641
+
1642
+ function startLoopRecording() {
1643
+ if (recording || processing || streamingDictationActive) return false;
1644
+ startRecording(kv, backend, toast, logger, opts?.trimSilence);
1645
+ if (recording) {
1646
+ showRecordingToast();
1647
+ // Fetch title + recent turns while the user speaks, so the context is
1648
+ // usually ready before the turn ends instead of on the critical path.
1649
+ prefetchContextParts(client, api);
1650
+ }
1651
+ return recording;
1652
+ }
1653
+
1654
+ // ---- Streaming dictation session (stage 2, local-only, no LLM) ----
1655
+ const sttMode = resolveSttMode(opts);
1656
+ const streamTimings = resolveStreamTimings(opts);
1657
+ logger?.log(
1658
+ "STT",
1659
+ `Mode=${sttMode} streamWindowMs=${streamTimings.windowMs} streamStepMs=${streamTimings.cadenceMs}`,
1660
+ "debug",
1661
+ );
1662
+ let streamController = null;
1663
+ let streamEditor = null;
1664
+ let streamTarget = null;
1665
+ let streamStarting = false;
1666
+ // Warm server lease retained across dictations: the idle controller (mic
1667
+ // fully stopped, server still loaded) parks here between sessions. At most
1668
+ // one warm lease ever exists; the next start reuses it when the model key
1669
+ // matches, or tears it down first when the model/language changed.
1670
+ let warmStreamController = null;
1671
+ let warmStreamModelKey = null;
1672
+
1673
+ function warmControllerPort(controller) {
1674
+ try {
1675
+ return controller?.getServerPort?.() ?? undefined;
1676
+ } catch {
1677
+ return undefined;
1678
+ }
1679
+ }
1680
+
1681
+ function currentStreamModelKey(port) {
1682
+ return streamModelKey(getModelPath(kv), getLanguage(kv), port);
1683
+ }
1684
+
1685
+ function disposeWarmStreamController() {
1686
+ if (warmStreamController) {
1687
+ const c = warmStreamController;
1688
+ warmStreamController = null;
1689
+ warmStreamModelKey = null;
1690
+ try {
1691
+ const maybe = c.dispose?.();
1692
+ if (maybe && typeof maybe.catch === "function") maybe.catch(() => {});
1693
+ } catch {}
1694
+ }
1695
+ }
1696
+
1697
+ function retainWarmStreamController(controller) {
1698
+ // Single-lease invariant: never park a second handle alongside one.
1699
+ if (warmStreamController && warmStreamController !== controller) {
1700
+ try {
1701
+ const maybe = controller.dispose?.();
1702
+ if (maybe && typeof maybe.catch === "function") maybe.catch(() => {});
1703
+ } catch {}
1704
+ return;
1705
+ }
1706
+ warmStreamController = controller;
1707
+ warmStreamModelKey = currentStreamModelKey(warmControllerPort(controller));
1708
+ }
1709
+
1710
+ function isStreaming() {
1711
+ return streamingDictationActive;
1712
+ }
1713
+
1714
+ async function startStreamingDictation() {
1715
+ if (streamingDictationActive || streamStarting) return false;
1716
+ if (recording || processing) {
1717
+ toast("STT busy, please wait...");
1718
+ return false;
1719
+ }
1720
+ if (liveNotesActive()) {
1721
+ toast("Live notes recording - stop it first (/voice-notes-stop)", "warning");
1722
+ return false;
1723
+ }
1724
+ if (conversationActive()) {
1725
+ toast("Conversation mode is on - use the conversation key to exit", "warning");
1726
+ return false;
1727
+ }
1728
+ streamStarting = true;
1729
+ // Model/language change tears down the warm lease first so the next
1730
+ // start loads the new model instead of reusing the stale server. The
1731
+ // key carries the bound server port, so a lease on a different port is
1732
+ // never mistaken for a matching one.
1733
+ if (
1734
+ warmStreamController &&
1735
+ warmStreamModelKey !== currentStreamModelKey(warmControllerPort(warmStreamController))
1736
+ ) {
1737
+ disposeWarmStreamController();
1738
+ }
1739
+ let controller = null;
1740
+ let reusedWarm = false;
1741
+ try {
1742
+ const { createStreamingController } = await import("./streaming-stt.js");
1743
+ const { createStreamingEditorAdapter } = await import("./streaming-editor.js");
1744
+ const focused = renderer?.currentFocusedRenderable || null;
1745
+ const target = focused && typeof focused.insertText === "function" ? focused : null;
1746
+ const editor = createStreamingEditorAdapter({
1747
+ toast: (message, variant) => api.ui.toast({ message, variant, duration: 2500 }),
1748
+ getFocused: () => renderer?.currentFocusedRenderable || null,
1749
+ });
1750
+ editor.begin(target);
1751
+ const sessionCallbacks = (ed) => ({
1752
+ onPartial: (p) => {
1753
+ try {
1754
+ noteStreamingPartial();
1755
+ } catch {}
1756
+ try {
1757
+ ed.applyPartial(p);
1758
+ } catch {}
1759
+ },
1760
+ onStateChange: (s) => {
1761
+ try {
1762
+ if (s === "streaming") {
1763
+ updateStreamingToast("● Streaming dictation — listening");
1764
+ }
1765
+ } catch {}
1766
+ },
1767
+ onError: (e) => {
1768
+ logger?.log("STT", `Streaming error: ${e?.code || ""} ${e?.message || ""}`, "error");
1769
+ try {
1770
+ clearStreamingToast();
1771
+ } catch {}
1772
+ toast(
1773
+ `Streaming STT error: ${e?.message || "unknown"}${e?.hasFallbackAudio ? " (audio kept)" : ""}`,
1774
+ "error",
1775
+ );
1776
+ },
1777
+ });
1778
+ if (warmStreamController) {
1779
+ controller = warmStreamController;
1780
+ warmStreamController = null;
1781
+ warmStreamModelKey = null;
1782
+ reusedWarm = true;
1783
+ // Rebind to the new session's editor: the warm lease stays, but
1784
+ // partials/errors must never reach the previous session's editor.
1785
+ try {
1786
+ controller.updateCallbacks?.(sessionCallbacks(editor));
1787
+ } catch {}
1788
+ } else {
1789
+ controller = createStreamingController({
1790
+ windowMs: streamTimings.windowMs,
1791
+ cadenceMs: streamTimings.cadenceMs,
1792
+ model: { modelPath: getModelPath(kv), language: getLanguage(kv) },
1793
+ logger,
1794
+ ...sessionCallbacks(editor),
1795
+ // Test seam (stage 3 e2e): deterministic capture/transcriber/clock
1796
+ // without a mic. Absent in production; the default path is
1797
+ // unchanged when deps.streamingControllerOpts is unset.
1798
+ ...deps.streamingControllerOpts,
1799
+ });
1800
+ }
1801
+ streamEditor = editor;
1802
+ streamTarget = target;
1803
+ streamController = controller;
1804
+ if (!controller.start()) {
1805
+ streamEditor = null;
1806
+ streamTarget = null;
1807
+ streamController = null;
1808
+ // start() refused (already running): park the warm handle back so
1809
+ // the lease is never dropped on a refusal path.
1810
+ if (reusedWarm) retainWarmStreamController(controller);
1811
+ return false;
1812
+ }
1813
+ __setStreamingActive(true);
1814
+ // Sticky, never silent: covers the cold model load ("Loading…") until
1815
+ // the controller reports streaming, then live capture with elapsed
1816
+ // time. Partial preview toasts continue alongside via the editor.
1817
+ showStreamingToast("Loading speech model…", {
1818
+ slowAfterMs: Math.max(2000, streamTimings.cadenceMs * 2),
1819
+ });
1820
+ logger?.log("STT", "Streaming dictation started", "debug");
1821
+ return true;
1822
+ } catch (err) {
1823
+ logger?.log("STT", `Streaming start failed: ${err?.message || err}`, "error");
1824
+ try {
1825
+ clearStreamingToast();
1826
+ } catch {}
1827
+ toast(`Streaming STT failed to start: ${err?.message || err}`, "error");
1828
+ streamEditor = null;
1829
+ streamTarget = null;
1830
+ streamController = null;
1831
+ if (reusedWarm && controller) {
1832
+ // Setup failed after taking the warm handle: park it back so the
1833
+ // lease survives a transient editor/import failure.
1834
+ retainWarmStreamController(controller);
1835
+ } else if (controller) {
1836
+ // Fresh handle that never became a session: release any partial
1837
+ // lease so a failed start never leaks a server ref.
1838
+ try {
1839
+ const maybe = controller.dispose?.();
1840
+ if (maybe && typeof maybe.catch === "function") maybe.catch(() => {});
1841
+ } catch {}
1842
+ }
1843
+ return false;
1844
+ } finally {
1845
+ streamStarting = false;
1846
+ }
1847
+ }
1848
+
1849
+ // Finalize: stop the controller (single tail transcription, no LLM),
1850
+ // commit the editor range, then insert EXACTLY once via the focused-input
1851
+ // helper. Never falls through to primary-chat appendPrompt. When `submit`
1852
+ // is set, submits ONLY via the captured target's own submit().
1853
+ // Warm-lease semantics: the mic is fully stopped by stop(), but the
1854
+ // server lease is RETAINED across dictations (parked as the single warm
1855
+ // handle) and released only when the stop result proves the server itself
1856
+ // is at fault, on model/language change, or on plugin unload.
1857
+ async function finalizeStreamingDictation({ submit = false } = {}) {
1858
+ if (!streamingDictationActive || !streamController) return null;
1859
+ const controller = streamController;
1860
+ const editor = streamEditor;
1861
+ const target = streamTarget;
1862
+ streamController = null;
1863
+ streamEditor = null;
1864
+ streamTarget = null;
1865
+ __setStreamingActive(false);
1866
+ // Never silent between the second keypress and the insert: the stop-tail
1867
+ // transcription can take ~a second while the screen otherwise freezes.
1868
+ try {
1869
+ updateStreamingToast("Finalizing…");
1870
+ } catch {}
1871
+ let done = null;
1872
+ try {
1873
+ done = await controller.stop();
1874
+ } catch (err) {
1875
+ logger?.log("STT", `Streaming stop failed: ${err?.message || err}`, "error");
1876
+ try {
1877
+ clearStreamingToast();
1878
+ } catch {}
1879
+ toast(`Streaming stop failed: ${err?.message || err}`, "error");
1880
+ // Mid-stop throw: preserve the lease (transient/per-dictation until
1881
+ // proven otherwise); the next start re-probes server readiness.
1882
+ retainWarmStreamController(controller);
1883
+ return null;
1884
+ }
1885
+ // The stop-tail resolved: the sticky status served its purpose (no
1886
+ // silent gap during the wait). Clear before the terminal toasts below.
1887
+ try {
1888
+ clearStreamingToast();
1889
+ } catch {}
1890
+ if (done?.error && isStreamingServerFault(done.error)) {
1891
+ // Server at fault (not ready / transcribe failure): release the lease
1892
+ // so the next start cold-loads from a clean handle instead of reusing
1893
+ // a dead server.
1894
+ toast(`Streaming stopped with error: ${done.error?.message || done.error}`, "warning");
1895
+ try {
1896
+ editor?.finalize((done?.text || editor?.getTranscript?.() || "").trim());
1897
+ } catch {}
1898
+ try {
1899
+ await controller.dispose?.();
1900
+ } catch {}
1901
+ return { text: (done?.text || "").trim(), error: done.error };
1902
+ }
1903
+ const text = (done?.text || editor?.getTranscript?.() || "").trim();
1904
+ if (done?.error) {
1905
+ toast(`Streaming stopped with error: ${done.error?.message || done.error}`, "warning");
1906
+ }
1907
+ // editor.finalize already writes the owned range for live-capable
1908
+ // targets (inserted:true) — inserting again would duplicate. Only the
1909
+ // fallback path (no range API) needs the single insert below, routed
1910
+ // through the captured target, never the current focus.
1911
+ let fin = null;
1912
+ try {
1913
+ fin = editor?.finalize(text);
1914
+ } catch {}
1915
+ // Every path below retains the warm lease: the mic stopped, the server
1916
+ // stays loaded for the next dictation.
1917
+ if (!text) {
1918
+ toast("No speech detected", "warning");
1919
+ retainWarmStreamController(controller);
1920
+ return { text: "" };
1921
+ }
1922
+ if (fin?.inserted) {
1923
+ logger?.log("STT", `Streaming finalized via owned range chars=${text.length}`, "debug");
1924
+ if (submit || autoSubmit) {
1925
+ if (target && typeof target.submit === "function") {
1926
+ try {
1927
+ target.submit();
1928
+ toast("Streaming transcription submitted", "success");
1929
+ } catch (err) {
1930
+ toast(`Streaming submit failed: ${err?.message || err}`, "error");
1931
+ }
1932
+ } else {
1933
+ toast("Target cannot submit - text inserted, not submitted", "warning");
1934
+ }
1935
+ } else {
1936
+ toast("Streaming transcription added", "success");
1937
+ }
1938
+ retainWarmStreamController(controller);
1939
+ return { text, inserted: true };
1940
+ }
1941
+ // Fallback session: single insert into the captured target only. A
1942
+ // missing/unusable target retains the transcript via toast instead of
1943
+ // redirecting into chat or the current focus.
1944
+ let inserted = false;
1945
+ if (target && typeof target.insertText === "function") {
1946
+ try {
1947
+ target.insertText(text);
1948
+ inserted = true;
1949
+ } catch {
1950
+ inserted = false;
1951
+ }
1952
+ } else {
1953
+ inserted = insertIntoFocusedInput(renderer, text, false);
1954
+ }
1955
+ if (!inserted) {
1956
+ logger?.log("STT", "Streaming finalize: no editable field, transcript retained", "warn");
1957
+ toast("No editable field - dictated text kept (not inserted elsewhere)", "warning");
1958
+ retainWarmStreamController(controller);
1959
+ return { text, inserted: false };
1960
+ }
1961
+ logger?.log("STT", `Streaming finalized chars=${text.length}`, "debug");
1962
+ if (submit || autoSubmit) {
1963
+ if (target && typeof target.submit === "function") {
1964
+ try {
1965
+ target.submit();
1966
+ toast("Streaming transcription submitted", "success");
1967
+ } catch (err) {
1968
+ toast(`Streaming submit failed: ${err?.message || err}`, "error");
1969
+ }
1970
+ } else {
1971
+ toast("Target cannot submit - text inserted, not submitted", "warning");
1972
+ }
1973
+ } else {
1974
+ toast("Streaming transcription added", "success");
1975
+ }
1976
+ retainWarmStreamController(controller);
1977
+ return { text, inserted: true };
1978
+ }
1979
+
1980
+ // Cancel: drop the session and remove ONLY plugin-owned dictated text.
1981
+ // Fallback path inserted nothing, so there is nothing to remove. The mic
1982
+ // is fully stopped by cancel(); the server lease is retained warm for the
1983
+ // next dictation (user cancel is per-dictation, never a server fault).
1984
+ async function cancelStreamingDictation() {
1985
+ if (!streamingDictationActive || !streamController) return false;
1986
+ const controller = streamController;
1987
+ const editor = streamEditor;
1988
+ streamController = null;
1989
+ streamEditor = null;
1990
+ streamTarget = null;
1991
+ __setStreamingActive(false);
1992
+ try {
1993
+ await controller.cancel();
1994
+ } catch {}
1995
+ let removed = 0;
1996
+ try {
1997
+ removed = editor?.cancel?.()?.removedChars || 0;
1998
+ } catch {}
1999
+ retainWarmStreamController(controller);
2000
+ logger?.log("STT", `Streaming cancelled removedChars=${removed}`, "debug");
2001
+ try {
2002
+ clearStreamingToast();
2003
+ } catch {}
2004
+ toast("Streaming dictation cancelled");
2005
+ return true;
2006
+ }
2007
+
2008
+ api.lifecycle?.onDispose?.(() => {
2009
+ if (streamController) {
2010
+ const c = streamController;
2011
+ streamController = null;
2012
+ streamEditor = null;
2013
+ streamTarget = null;
2014
+ __setStreamingActive(false);
2015
+ try {
2016
+ const maybe = c.dispose?.();
2017
+ if (maybe && typeof maybe.catch === "function") maybe.catch(() => {});
2018
+ } catch {}
2019
+ }
2020
+ try {
2021
+ clearStreamingToast();
2022
+ } catch {}
2023
+ // Plugin unload releases the parked warm lease: the model must not stay
2024
+ // loaded after the plugin is gone.
2025
+ disposeWarmStreamController();
2026
+ });
2027
+
2028
+ const controller = {
2029
+ isRecording: () => recording,
2030
+ isProcessing: () => processing,
2031
+ isStreaming,
2032
+ start: startLoopRecording,
2033
+ setStopHint: (hint) => __setStopHint(hint),
2034
+ cancel: () => cancelRecording(logger),
2035
+ discard: () => discardTurn(),
2036
+ transcribeTurn: () => transcribeTurn(kv, complete, client, api, toast, systemPrompt, logger),
2037
+ submitTurnText: (text) => submitTurnText(client, renderer, toast, text, logger),
2038
+ startStreaming: startStreamingDictation,
2039
+ finalizeStreaming: finalizeStreamingDictation,
2040
+ cancelStreaming: cancelStreamingDictation,
2041
+ };
2042
+
2043
+ // One-shot auto-submit: speak, press again, and the text submits itself
2044
+ // without landing in the box first. Off by default (append-then-edit).
2045
+ const autoSubmit = opts?.sttAutoSubmit === true;
2046
+
2047
+ const commands = [
2048
+ {
2049
+ title: sttApiEndpoint ? "STT: record/transcribe (API)" : "STT: record/transcribe",
2050
+ value: "stt.record",
2051
+ category: "opencode-voice",
2052
+ description:
2053
+ sttApiEndpoint && autoSubmit
2054
+ ? "Toggle recording; press again to stop, transcribe via API and submit"
2055
+ : sttApiEndpoint
2056
+ ? "Toggle recording; press again to stop and transcribe via API"
2057
+ : autoSubmit
2058
+ ? "Toggle recording; press again to stop, transcribe and submit"
2059
+ : "Toggle recording; press again to stop and transcribe",
2060
+ ...(kb("stt.record") ? { keybind: kb("stt.record") } : {}),
2061
+ slash: { name: "stt-record" },
2062
+ onSelect() {
2063
+ // Streaming mode: toggle streaming record/finalize. Batch default
2064
+ // below is behavior-identical to before.
2065
+ if (sttMode === "streaming") {
2066
+ if (streamingDictationActive) {
2067
+ void finalizeStreamingDictation({ submit: false });
2068
+ return;
2069
+ }
2070
+ if (liveNotesActive()) {
2071
+ toast("Live notes recording - stop it first (/voice-notes-stop)", "warning");
2072
+ return;
2073
+ }
2074
+ if (conversationActive()) {
2075
+ toast("Conversation mode is on - use the conversation key to exit", "warning");
2076
+ return;
2077
+ }
2078
+ if (processing || recording) {
2079
+ toast("STT busy, please wait...");
2080
+ return;
2081
+ }
2082
+ void startStreamingDictation();
2083
+ return;
2084
+ }
2085
+ if (streamingDictationActive) {
2086
+ toast("Streaming dictation active - use /stt-stop to cancel first", "warning");
2087
+ return;
2088
+ }
2089
+ if (liveNotesActive()) {
2090
+ toast("Live notes recording - stop it first (/voice-notes-stop)", "warning");
2091
+ return;
2092
+ }
2093
+ if (conversationActive()) {
2094
+ conversationHooks.onKey?.("record");
2095
+ return;
2096
+ }
2097
+ if (processing) {
2098
+ toast("STT busy, please wait...");
2099
+ return;
2100
+ }
2101
+ if (recording) {
2102
+ clearRecordingToast();
2103
+ toast("Stopping, transcribing...");
2104
+ doTranscribePipeline(
2105
+ kv,
2106
+ complete,
2107
+ client,
2108
+ api,
2109
+ toast,
2110
+ systemPrompt,
2111
+ autoSubmit,
2112
+ logger,
2113
+ renderer,
2114
+ );
2115
+ } else {
2116
+ startLoopRecording();
2117
+ }
2118
+ },
2119
+ },
2120
+ {
2121
+ title: sttApiEndpoint ? "STT: submit recording (API)" : "STT: submit recording",
2122
+ value: "stt.submit",
2123
+ category: "opencode-voice",
2124
+ description: sttApiEndpoint
2125
+ ? "Stop recording, transcribe via API, and submit prompt"
2126
+ : "Stop recording, transcribe, and submit prompt",
2127
+ ...(kb("stt.submit") ? { keybind: kb("stt.submit") } : {}),
2128
+ slash: { name: "stt-submit" },
2129
+ onSelect() {
2130
+ // Streaming mode: finalize then submit ONLY via the captured
2131
+ // target's own submit(). Never falls through to primary-chat submit.
2132
+ if (sttMode === "streaming") {
2133
+ if (!streamingDictationActive) {
2134
+ toast("No streaming dictation in progress", "warning");
2135
+ return;
2136
+ }
2137
+ void finalizeStreamingDictation({ submit: true });
2138
+ return;
2139
+ }
2140
+ if (streamingDictationActive) {
2141
+ toast("Streaming dictation active - use /stt-stop to cancel first", "warning");
2142
+ return;
2143
+ }
2144
+ if (liveNotesActive()) {
2145
+ toast("Live notes recording - stop it first (/voice-notes-stop)", "warning");
2146
+ return;
2147
+ }
2148
+ if (conversationActive()) {
2149
+ conversationHooks.onKey?.("submit");
2150
+ return;
2151
+ }
2152
+ if (processing) {
2153
+ toast("STT busy, please wait...");
2154
+ return;
2155
+ }
2156
+ if (!recording) {
2157
+ toast("No recording in progress", "warning");
2158
+ return;
2159
+ }
2160
+ clearRecordingToast();
2161
+ toast("Stopping, transcribing...");
2162
+ doTranscribePipeline(
2163
+ kv,
2164
+ complete,
2165
+ client,
2166
+ api,
2167
+ toast,
2168
+ systemPrompt,
2169
+ true,
2170
+ logger,
2171
+ renderer,
2172
+ );
2173
+ },
2174
+ },
2175
+ {
2176
+ title: "STT: cancel recording",
2177
+ value: "stt.stop",
2178
+ category: "opencode-voice",
2179
+ description: "Cancel current recording",
2180
+ slash: { name: "stt-stop" },
2181
+ onSelect() {
2182
+ // Streaming mode: cancel and remove ONLY plugin-owned dictation
2183
+ // text, never user-typed text.
2184
+ if (sttMode === "streaming" || streamingDictationActive) {
2185
+ if (streamingDictationActive) {
2186
+ void cancelStreamingDictation();
2187
+ return;
2188
+ }
2189
+ if (liveNotesActive()) {
2190
+ toast("Live notes recording - use /voice-notes-stop", "warning");
2191
+ return;
2192
+ }
2193
+ if (conversationActive()) {
2194
+ toast("Conversation mode is on - use the conversation key to exit", "warning");
2195
+ return;
2196
+ }
2197
+ return;
2198
+ }
2199
+ if (liveNotesActive()) {
2200
+ toast("Live notes recording - use /voice-notes-stop", "warning");
2201
+ return;
2202
+ }
2203
+ if (conversationActive()) {
2204
+ toast("Conversation mode is on - use the conversation key to exit", "warning");
2205
+ return;
2206
+ }
2207
+ if (recording) {
2208
+ cancelRecording(logger);
2209
+ logger?.log("STT", "Recording cancelled", "debug");
2210
+ toast("Recording cancelled");
2211
+ }
2212
+ },
2213
+ },
2214
+ {
2215
+ title: sttApiEndpoint ? "STT: select model (API)" : "STT: select model",
2216
+ value: "stt.model",
2217
+ category: "opencode-voice",
2218
+ description: sttApiEndpoint ? "Choose whisper model via API" : "Choose whisper model",
2219
+ slash: { name: "stt-model" },
2220
+ async onSelect() {
2221
+ if (sttApiEndpoint) {
2222
+ const current = kv.get("stt.api.model") || sttApiModel;
2223
+ const apiModels = await getApiModels(logger);
2224
+ const options = apiModels.length > 0 ? apiModels : [{ value: current, label: current }];
2225
+ api.ui.dialog.replace(() =>
2226
+ api.ui.DialogSelect({
2227
+ title: "Select whisper model (API)",
2228
+ current,
2229
+ options: options.map((m) => ({
2230
+ title: m.label,
2231
+ value: m.value,
2232
+ onSelect() {
2233
+ kv.set("stt.api.model", m.value);
2234
+ toast(`Whisper API model: ${m.label}`);
2235
+ api.ui.dialog.clear();
2236
+ },
2237
+ })),
2238
+ }),
2239
+ );
2240
+ } else {
2241
+ const current = getModelName(kv);
2242
+ api.ui.dialog.replace(() =>
2243
+ api.ui.DialogSelect({
2244
+ title: "Select whisper model",
2245
+ current,
2246
+ options: Object.entries(MODELS).map(([key, v]) => ({
2247
+ title: v.label,
2248
+ value: key,
2249
+ onSelect() {
2250
+ kv.set("stt.model", key);
2251
+ // Model change tears down the parked warm server lease so
2252
+ // the next dictation loads the new model (refused
2253
+ // mid-dictation by the start guards; idle teardown here).
2254
+ if (!streamingDictationActive) disposeWarmStreamController();
2255
+ toast(`Whisper model: ${v.label}`);
2256
+ api.ui.dialog.clear();
2257
+ },
2258
+ })),
2259
+ }),
2260
+ );
2261
+ }
2262
+ },
2263
+ },
2264
+ {
2265
+ title: "STT: select language",
2266
+ value: "stt.language",
2267
+ category: "opencode-voice",
2268
+ description: "Choose transcription language (local whisper-cli only)",
2269
+ slash: { name: "stt-language" },
2270
+ onSelect() {
2271
+ const current = getLanguage(kv);
2272
+ api.ui.dialog.replace(() =>
2273
+ api.ui.DialogSelect({
2274
+ title: "Select transcription language",
2275
+ current,
2276
+ options: Object.entries(LANGUAGES).map(([key, v]) => ({
2277
+ title: v.label,
2278
+ value: key,
2279
+ onSelect() {
2280
+ kv.set("stt.language", key);
2281
+ // Language is part of the server key: drop the parked warm
2282
+ // lease so the next dictation starts the new configuration.
2283
+ if (!streamingDictationActive) disposeWarmStreamController();
2284
+ toast(`Whisper language: ${v.label}`);
2285
+ api.ui.dialog.clear();
2286
+ },
2287
+ })),
2288
+ }),
2289
+ );
2290
+ },
2291
+ },
2292
+ {
2293
+ title: "STT: select microphone",
2294
+ value: "stt.mic",
2295
+ category: "opencode-voice",
2296
+ description: "Choose audio input device",
2297
+ slash: { name: "stt-mic" },
2298
+ onSelect() {
2299
+ const current = kv.get("stt.mic", "");
2300
+ const devices = listInputDevices(backend);
2301
+ if (devices.length === 0) {
2302
+ const serverOk = backend !== "pulseaudio" || pulseServerHealth().ok;
2303
+ const hint = buildAudioHint({ backend, serverOk, isWsl: isWSL() });
2304
+ logger?.log("STT", `No input devices: ${hint}`, "warn");
2305
+ toast(hint);
2306
+ return;
2307
+ }
2308
+ api.ui.dialog.replace(() =>
2309
+ api.ui.DialogSelect({
2310
+ title: "Select microphone",
2311
+ current,
2312
+ options: [
2313
+ {
2314
+ title: "System default",
2315
+ value: "",
2316
+ onSelect() {
2317
+ kv.set("stt.mic", "");
2318
+ toast("Mic: system default");
2319
+ api.ui.dialog.clear();
2320
+ },
2321
+ },
2322
+ ...devices.map((d) => ({
2323
+ title: d.label,
2324
+ value: d.name,
2325
+ onSelect() {
2326
+ kv.set("stt.mic", d.name);
2327
+ toast(`Mic: ${d.label}`);
2328
+ api.ui.dialog.clear();
2329
+ },
2330
+ })),
2331
+ ],
2332
+ }),
2333
+ );
2334
+ },
2335
+ },
2336
+ ];
2337
+
2338
+ return { commands, controller };
2339
+ }