@bojackduy/opencode-voice 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +729 -0
- package/index.js +124 -0
- package/lib/audio-chunker.js +231 -0
- package/lib/audio-enhance.js +172 -0
- package/lib/conversation.js +528 -0
- package/lib/live-notes.js +622 -0
- package/lib/llm-client.js +304 -0
- package/lib/logger.js +18 -0
- package/lib/notes-writer.js +224 -0
- package/lib/session.js +102 -0
- package/lib/streaming-editor.js +308 -0
- package/lib/streaming-stt.js +1322 -0
- package/lib/streaming-transcript.js +236 -0
- package/lib/stt.js +2339 -0
- package/lib/tts.js +671 -0
- package/lib/voice-model.js +122 -0
- package/lib/whisper-server.js +471 -0
- package/package.json +47 -0
|
@@ -0,0 +1,622 @@
|
|
|
1
|
+
// Live voice notes: continuous meeting/lecture capture with background
|
|
2
|
+
// transcription. The mic is never paused to wait for whisper or the LLM, so
|
|
3
|
+
// you can talk for as long as you want.
|
|
4
|
+
//
|
|
5
|
+
// Two independent lanes:
|
|
6
|
+
//
|
|
7
|
+
// capture lane - one persistent sox process streams raw PCM; the
|
|
8
|
+
// audio-chunker splits it into self-contained WAV chunks
|
|
9
|
+
// on natural pauses (or a forced max-duration cut)
|
|
10
|
+
// processing lane - a 2-stage pipeline (transcribe, then normalize+write)
|
|
11
|
+
// processes chunks one at a time per stage, but the two
|
|
12
|
+
// stages overlap: chunk N+1 transcribes while chunk N is
|
|
13
|
+
// still being normalized. Chunks are appended to the
|
|
14
|
+
// live Markdown/JSONL files strictly in recording order,
|
|
15
|
+
// even if a stage resolves out of order.
|
|
16
|
+
//
|
|
17
|
+
// Local whisper transcription uses a persistent whisper-server (loads the
|
|
18
|
+
// model once) when available, falling back to per-chunk whisper-cli
|
|
19
|
+
// (reloads the model every call - fine for occasional use, but would fall
|
|
20
|
+
// behind a long meeting) or the configured STT API endpoint.
|
|
21
|
+
|
|
22
|
+
import fs from "node:fs";
|
|
23
|
+
import path from "node:path";
|
|
24
|
+
import { spawn } from "node:child_process";
|
|
25
|
+
import {
|
|
26
|
+
NOTES_SYSTEM_PROMPT,
|
|
27
|
+
buildRecordArgs,
|
|
28
|
+
detectAudioBackend,
|
|
29
|
+
formatElapsed,
|
|
30
|
+
getLanguage,
|
|
31
|
+
getModelPath,
|
|
32
|
+
getSttApiConfig,
|
|
33
|
+
getTmpDir,
|
|
34
|
+
isLikelyWhisperHallucination,
|
|
35
|
+
isStreamingActive,
|
|
36
|
+
isSttBusy,
|
|
37
|
+
normalizeTranscription,
|
|
38
|
+
transcribeApiFile,
|
|
39
|
+
transcribeFileLocal,
|
|
40
|
+
} from "./stt.js";
|
|
41
|
+
import { createPcmChunker, wrapPcmAsWav } from "./audio-chunker.js";
|
|
42
|
+
import { enhanceWavFile } from "./audio-enhance.js";
|
|
43
|
+
import { buildSessionBaseName, createNotesWriter } from "./notes-writer.js";
|
|
44
|
+
import { acquireSharedWhisperServer } from "./whisper-server.js";
|
|
45
|
+
|
|
46
|
+
// Single-concurrency FIFO worker. Each stage processes its queue strictly in
|
|
47
|
+
// arrival order but never waits for the OTHER stage, so stage A can start
|
|
48
|
+
// chunk N+1 while stage B is still finishing chunk N.
|
|
49
|
+
|
|
50
|
+
function createStage(name, logger, worker) {
|
|
51
|
+
const queue = [];
|
|
52
|
+
let running = false;
|
|
53
|
+
|
|
54
|
+
async function pump() {
|
|
55
|
+
if (running) return;
|
|
56
|
+
running = true;
|
|
57
|
+
while (queue.length > 0) {
|
|
58
|
+
const item = queue.shift();
|
|
59
|
+
try {
|
|
60
|
+
await worker(item);
|
|
61
|
+
} catch (err) {
|
|
62
|
+
logger?.log("VOICE", `Live notes ${name} stage error: ${err.message}`, "error");
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
running = false;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
return {
|
|
69
|
+
push(item) {
|
|
70
|
+
queue.push(item);
|
|
71
|
+
pump();
|
|
72
|
+
},
|
|
73
|
+
size() {
|
|
74
|
+
return queue.length + (running ? 1 : 0);
|
|
75
|
+
},
|
|
76
|
+
};
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
function waitForProcessExit(getProc, timeoutMs = 2000) {
|
|
80
|
+
return new Promise((resolve) => {
|
|
81
|
+
const start = Date.now();
|
|
82
|
+
const check = () => {
|
|
83
|
+
const proc = getProc();
|
|
84
|
+
if (!proc || Date.now() - start >= timeoutMs) {
|
|
85
|
+
if (proc) {
|
|
86
|
+
try {
|
|
87
|
+
process.kill(proc.pid, "SIGKILL");
|
|
88
|
+
} catch {}
|
|
89
|
+
}
|
|
90
|
+
resolve();
|
|
91
|
+
return;
|
|
92
|
+
}
|
|
93
|
+
setTimeout(check, 100);
|
|
94
|
+
};
|
|
95
|
+
check();
|
|
96
|
+
});
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
export function registerLiveNotes(api, kv, complete, opts, logger, deps = {}) {
|
|
100
|
+
function toast(message, variant = "info") {
|
|
101
|
+
api.ui.toast({ message, variant, duration: 3000 });
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
const workspaceDir = api.state?.path?.directory || process.cwd();
|
|
105
|
+
const notesDirOpt = opts?.notesDir || "voice-notes";
|
|
106
|
+
const chunkMaxMs =
|
|
107
|
+
Number(opts?.notesChunkMaxSeconds) > 0 ? Number(opts.notesChunkMaxSeconds) * 1000 : 20000;
|
|
108
|
+
const silenceMs = Number(opts?.notesSilenceMs) > 0 ? Number(opts.notesSilenceMs) : 700;
|
|
109
|
+
const overlapMs = Number(opts?.notesOverlapMs) >= 0 ? Number(opts.notesOverlapMs) : 400;
|
|
110
|
+
const minChunkMs =
|
|
111
|
+
Number(opts?.notesMinChunkSeconds) > 0 ? Number(opts.notesMinChunkSeconds) * 1000 : 3000;
|
|
112
|
+
// Room noise-floor calibration: learn the silence threshold from the first
|
|
113
|
+
// moments of mic audio instead of assuming 0.02 fits every room. An
|
|
114
|
+
// explicit notesSilenceRms disables it and is used as-is.
|
|
115
|
+
const explicitSilenceRms = Number(opts?.notesSilenceRms) > 0 ? Number(opts.notesSilenceRms) : 0;
|
|
116
|
+
const calibrationMs =
|
|
117
|
+
explicitSilenceRms > 0
|
|
118
|
+
? 0
|
|
119
|
+
: Number(opts?.notesCalibrationMs) >= 0
|
|
120
|
+
? Number(opts.notesCalibrationMs)
|
|
121
|
+
: 1500;
|
|
122
|
+
const doNormalize = opts?.notesNormalize !== false;
|
|
123
|
+
const keepAudio = opts?.notesKeepAudio === true;
|
|
124
|
+
const useWhisperServer = opts?.notesUseWhisperServer !== false;
|
|
125
|
+
// Voice-only preprocessing before whisper (far-field AGC). Disable with
|
|
126
|
+
// notesEnhance:false if the mic is already close/loud.
|
|
127
|
+
const doEnhance = opts?.notesEnhance !== false;
|
|
128
|
+
const enhanceTargetRms =
|
|
129
|
+
Number(opts?.notesEnhanceTargetRms) > 0 ? Number(opts.notesEnhanceTargetRms) : 0.1;
|
|
130
|
+
const enhanceMaxGainDb =
|
|
131
|
+
Number(opts?.notesEnhanceMaxGainDb) >= 0 ? Number(opts.notesEnhanceMaxGainDb) : 24;
|
|
132
|
+
|
|
133
|
+
let active = false;
|
|
134
|
+
let finishing = false;
|
|
135
|
+
let soxProc = null;
|
|
136
|
+
let chunker = null;
|
|
137
|
+
let notesWriter = null;
|
|
138
|
+
let scratchDir = null;
|
|
139
|
+
let recordingDir = null;
|
|
140
|
+
let recordingBaseName = null;
|
|
141
|
+
let recordingStartedAtMs = 0;
|
|
142
|
+
let useApiMode = false;
|
|
143
|
+
// Shared whisper-server lease (stage 2 unification): live-notes and
|
|
144
|
+
// streaming dictation share one owned server per model/language/port via
|
|
145
|
+
// the whisper-server rendezvous instead of racing for the port. The lease
|
|
146
|
+
// is acquired at session start and released at session end; a second
|
|
147
|
+
// claimant on the same port gets explicit PORT_IN_USE, never a kill.
|
|
148
|
+
let whisperServerLease = null;
|
|
149
|
+
let whisperServerClient = null;
|
|
150
|
+
let whisperServerReadyPromise = null;
|
|
151
|
+
|
|
152
|
+
function releaseWhisperServerLease() {
|
|
153
|
+
whisperServerClient = null;
|
|
154
|
+
if (whisperServerLease) {
|
|
155
|
+
try {
|
|
156
|
+
whisperServerLease.release();
|
|
157
|
+
} catch {}
|
|
158
|
+
whisperServerLease = null;
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
let statusTimer = null;
|
|
162
|
+
|
|
163
|
+
async function transcribeLocalChunk(wavPath) {
|
|
164
|
+
if (useApiMode) {
|
|
165
|
+
const apiCfg = getSttApiConfig();
|
|
166
|
+
return transcribeApiFile(wavPath, apiCfg.endpoint, apiCfg.model, apiCfg.apiKeyEnv, logger);
|
|
167
|
+
}
|
|
168
|
+
if (whisperServerReadyPromise) {
|
|
169
|
+
const ready = await whisperServerReadyPromise;
|
|
170
|
+
if (ready && whisperServerClient?.isRunning()) {
|
|
171
|
+
const result = await whisperServerClient.transcribeFile(wavPath);
|
|
172
|
+
if (!result.error) return result;
|
|
173
|
+
logger?.log(
|
|
174
|
+
"VOICE",
|
|
175
|
+
`whisper-server chunk failed, falling back to whisper-cli: ${result.error}`,
|
|
176
|
+
"warn",
|
|
177
|
+
);
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
return transcribeFileLocal(wavPath, getModelPath(kv), getLanguage(kv), logger);
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
const stageA = createStage("transcribe", logger, async (chunk) => {
|
|
184
|
+
const wavPath = path.join(scratchDir, `chunk-${chunk.seq}.wav`);
|
|
185
|
+
fs.writeFileSync(wavPath, wrapPcmAsWav(chunk.pcm, { sampleRate: 16000 }));
|
|
186
|
+
|
|
187
|
+
// Voice-only preprocessing: bring quiet far-field speech up to a
|
|
188
|
+
// healthy level before whisper sees it. Never loses audio - on any
|
|
189
|
+
// failure the original recording is transcribed as-is.
|
|
190
|
+
let audio = { enhanced: false, reason: "disabled" };
|
|
191
|
+
if (doEnhance) {
|
|
192
|
+
audio = await enhanceWavFile(
|
|
193
|
+
wavPath,
|
|
194
|
+
chunk.pcm,
|
|
195
|
+
{ targetRms: enhanceTargetRms, maxGainDb: enhanceMaxGainDb },
|
|
196
|
+
logger,
|
|
197
|
+
);
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
const t0 = Date.now();
|
|
201
|
+
const result = await transcribeLocalChunk(wavPath);
|
|
202
|
+
const transcribeMs = Date.now() - t0;
|
|
203
|
+
|
|
204
|
+
if (keepAudio) {
|
|
205
|
+
try {
|
|
206
|
+
const audioDir = path.join(recordingDir, `${recordingBaseName}-audio`);
|
|
207
|
+
fs.mkdirSync(audioDir, { recursive: true });
|
|
208
|
+
fs.renameSync(
|
|
209
|
+
wavPath,
|
|
210
|
+
path.join(audioDir, `chunk-${String(chunk.seq).padStart(5, "0")}.wav`),
|
|
211
|
+
);
|
|
212
|
+
} catch {}
|
|
213
|
+
} else {
|
|
214
|
+
try {
|
|
215
|
+
fs.unlinkSync(wavPath);
|
|
216
|
+
} catch {}
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
stageB.push({
|
|
220
|
+
chunk,
|
|
221
|
+
raw: result.text || "",
|
|
222
|
+
error: result.error || null,
|
|
223
|
+
transcribeMs,
|
|
224
|
+
audio,
|
|
225
|
+
});
|
|
226
|
+
});
|
|
227
|
+
|
|
228
|
+
// Consecutive identical transcripts are almost always whisper
|
|
229
|
+
// hallucinating on quiet/noise (e.g. the same YouTube outro 7x in a row),
|
|
230
|
+
// not a professor repeating themselves word-for-word.
|
|
231
|
+
let repeatCount = 0;
|
|
232
|
+
let lastRawText = "";
|
|
233
|
+
|
|
234
|
+
const stageB = createStage("normalize", logger, async (item) => {
|
|
235
|
+
const { chunk } = item;
|
|
236
|
+
const base = {
|
|
237
|
+
seq: chunk.seq,
|
|
238
|
+
startMs: chunk.startMs,
|
|
239
|
+
endMs: chunk.endMs,
|
|
240
|
+
forced: chunk.forced,
|
|
241
|
+
audio: item.audio || { enhanced: false, reason: "unknown" },
|
|
242
|
+
};
|
|
243
|
+
|
|
244
|
+
if (item.error) {
|
|
245
|
+
repeatCount = 0;
|
|
246
|
+
lastRawText = "";
|
|
247
|
+
notesWriter.appendChunk({ ...base, raw: "", normalized: "", error: item.error });
|
|
248
|
+
logger?.log(
|
|
249
|
+
"VOICE",
|
|
250
|
+
`Live notes chunk ${chunk.seq} transcription failed: ${item.error}`,
|
|
251
|
+
"warn",
|
|
252
|
+
);
|
|
253
|
+
return;
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
const raw = item.raw;
|
|
257
|
+
if (raw) {
|
|
258
|
+
repeatCount = raw === lastRawText ? repeatCount + 1 : 1;
|
|
259
|
+
lastRawText = raw;
|
|
260
|
+
} else {
|
|
261
|
+
repeatCount = 0;
|
|
262
|
+
lastRawText = "";
|
|
263
|
+
}
|
|
264
|
+
if (!raw || isLikelyWhisperHallucination(raw) || repeatCount >= 3) {
|
|
265
|
+
notesWriter.appendChunk({
|
|
266
|
+
...base,
|
|
267
|
+
raw,
|
|
268
|
+
normalized: "",
|
|
269
|
+
skipped: !raw ? "empty" : repeatCount >= 3 ? "repeated" : "hallucination",
|
|
270
|
+
transcribeMs: item.transcribeMs,
|
|
271
|
+
});
|
|
272
|
+
return;
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
if (!doNormalize) {
|
|
276
|
+
notesWriter.appendChunk({ ...base, raw, normalized: raw, transcribeMs: item.transcribeMs });
|
|
277
|
+
return;
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
const contextBlock = (notesWriter.lastText() || "").slice(-300) || null;
|
|
281
|
+
const t0 = Date.now();
|
|
282
|
+
const llmResult = await normalizeTranscription(
|
|
283
|
+
complete,
|
|
284
|
+
raw,
|
|
285
|
+
contextBlock,
|
|
286
|
+
NOTES_SYSTEM_PROMPT,
|
|
287
|
+
logger,
|
|
288
|
+
);
|
|
289
|
+
const normalizeMs = Date.now() - t0;
|
|
290
|
+
notesWriter.appendChunk({
|
|
291
|
+
...base,
|
|
292
|
+
raw,
|
|
293
|
+
normalized: llmResult.text || raw,
|
|
294
|
+
transcribeMs: item.transcribeMs,
|
|
295
|
+
normalizeMs,
|
|
296
|
+
error: llmResult.error || null,
|
|
297
|
+
});
|
|
298
|
+
});
|
|
299
|
+
|
|
300
|
+
let calibrationLogged = false;
|
|
301
|
+
function routeChunk(chunk) {
|
|
302
|
+
if (!calibrationLogged && chunker?.isCalibrated?.() === true) {
|
|
303
|
+
calibrationLogged = true;
|
|
304
|
+
logger?.log(
|
|
305
|
+
"VOICE",
|
|
306
|
+
`Live notes room calibrated noiseFloor=${chunker.getNoiseFloor?.()?.toFixed(4)} silenceRms=${chunker.getSilenceThreshold?.()?.toFixed(4)}`,
|
|
307
|
+
"debug",
|
|
308
|
+
);
|
|
309
|
+
}
|
|
310
|
+
if (!chunk.hasSpeech) {
|
|
311
|
+
notesWriter.appendChunk({
|
|
312
|
+
seq: chunk.seq,
|
|
313
|
+
startMs: chunk.startMs,
|
|
314
|
+
endMs: chunk.endMs,
|
|
315
|
+
forced: chunk.forced,
|
|
316
|
+
raw: "",
|
|
317
|
+
normalized: "",
|
|
318
|
+
skipped: "silence",
|
|
319
|
+
});
|
|
320
|
+
return;
|
|
321
|
+
}
|
|
322
|
+
stageA.push(chunk);
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
function statusMessage() {
|
|
326
|
+
const elapsed = formatElapsed(Date.now() - recordingStartedAtMs);
|
|
327
|
+
const written = notesWriter ? notesWriter.entryCount() : 0;
|
|
328
|
+
const queued = stageA.size() + stageB.size();
|
|
329
|
+
const lagMs = notesWriter
|
|
330
|
+
? Math.max(0, Date.now() - recordingStartedAtMs - notesWriter.lastWrittenEndMs())
|
|
331
|
+
: 0;
|
|
332
|
+
return `● Live notes ${elapsed} · written ${written} · queued ${queued} · lag ${Math.round(lagMs / 1000)}s`;
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
function showStatusToast() {
|
|
336
|
+
clearStatusToast();
|
|
337
|
+
toast(statusMessage());
|
|
338
|
+
statusTimer = setInterval(() => {
|
|
339
|
+
if (active) toast(statusMessage());
|
|
340
|
+
}, 2500);
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
function clearStatusToast() {
|
|
344
|
+
if (statusTimer) {
|
|
345
|
+
clearInterval(statusTimer);
|
|
346
|
+
statusTimer = null;
|
|
347
|
+
}
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
function startCapture(backend) {
|
|
351
|
+
const mic = kv.get("stt.mic", "") || null;
|
|
352
|
+
const inputArgs = buildRecordArgs(backend, mic);
|
|
353
|
+
// Test seam: deterministic capture without a mic (`deps.spawn`).
|
|
354
|
+
// Production passes nothing and uses the real spawn.
|
|
355
|
+
const spawnFn = deps.spawn ?? spawn;
|
|
356
|
+
let stderr = "";
|
|
357
|
+
try {
|
|
358
|
+
soxProc = spawnFn(
|
|
359
|
+
"sox",
|
|
360
|
+
[...inputArgs, "-r", "16000", "-c", "1", "-b", "16", "-t", "raw", "-"],
|
|
361
|
+
{
|
|
362
|
+
stdio: ["ignore", "pipe", "pipe"],
|
|
363
|
+
},
|
|
364
|
+
);
|
|
365
|
+
} catch (err) {
|
|
366
|
+
logger?.log("VOICE", `Live notes failed to start capture: ${err.message}`, "error");
|
|
367
|
+
return false;
|
|
368
|
+
}
|
|
369
|
+
|
|
370
|
+
soxProc.stdout.on("data", (buf) => {
|
|
371
|
+
const ready = chunker.push(buf);
|
|
372
|
+
for (const chunk of ready) routeChunk(chunk);
|
|
373
|
+
});
|
|
374
|
+
soxProc.stderr.on("data", (chunk) => {
|
|
375
|
+
stderr += chunk.toString();
|
|
376
|
+
});
|
|
377
|
+
soxProc.on("error", (err) => {
|
|
378
|
+
soxProc = null;
|
|
379
|
+
logger?.log("VOICE", `Live notes capture process error: ${err.message}`, "error");
|
|
380
|
+
// Terminal capture failure (e.g. sox binary missing): same stop/save
|
|
381
|
+
// path as an unexpected exit, otherwise `active` stays true with the
|
|
382
|
+
// status timer running while no chunks are ever saved.
|
|
383
|
+
if (active && !finishing) {
|
|
384
|
+
toast("Live notes: microphone failed, saving what we have", "error");
|
|
385
|
+
stopCommand();
|
|
386
|
+
}
|
|
387
|
+
});
|
|
388
|
+
soxProc.on("exit", (code) => {
|
|
389
|
+
soxProc = null;
|
|
390
|
+
if (active && code !== 0 && code !== null) {
|
|
391
|
+
logger?.log(
|
|
392
|
+
"VOICE",
|
|
393
|
+
`Live notes capture exited code=${code} stderr=${stderr.trim()}`,
|
|
394
|
+
"error",
|
|
395
|
+
);
|
|
396
|
+
toast("Live notes: microphone stopped unexpectedly, saving what we have", "error");
|
|
397
|
+
stopCommand();
|
|
398
|
+
}
|
|
399
|
+
});
|
|
400
|
+
return true;
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
function waitForDrain() {
|
|
404
|
+
return new Promise((resolve) => {
|
|
405
|
+
const check = () => {
|
|
406
|
+
if (stageA.size() === 0 && stageB.size() === 0) resolve();
|
|
407
|
+
else setTimeout(check, 200);
|
|
408
|
+
};
|
|
409
|
+
check();
|
|
410
|
+
});
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
async function drainAndSave() {
|
|
414
|
+
finishing = true;
|
|
415
|
+
await waitForDrain();
|
|
416
|
+
releaseWhisperServerLease();
|
|
417
|
+
const paths = await notesWriter.close();
|
|
418
|
+
notesWriter = null;
|
|
419
|
+
finishing = false;
|
|
420
|
+
logger?.log("VOICE", `Live notes saved md=${paths.mdPath} jsonl=${paths.jsonlPath}`, "debug");
|
|
421
|
+
toast(`Live notes saved: ${path.relative(workspaceDir, paths.mdPath)}`, "success");
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
async function startNotes() {
|
|
425
|
+
if (active || finishing) {
|
|
426
|
+
toast(
|
|
427
|
+
finishing ? "Live notes still saving the last session" : "Live notes already recording",
|
|
428
|
+
"warning",
|
|
429
|
+
);
|
|
430
|
+
return;
|
|
431
|
+
}
|
|
432
|
+
if (isSttBusy() || isStreamingActive()) {
|
|
433
|
+
toast("STT busy - finish or cancel it first", "warning");
|
|
434
|
+
return;
|
|
435
|
+
}
|
|
436
|
+
if (deps.isConversationActive?.()) {
|
|
437
|
+
toast("Voice conversation is on - exit it first", "warning");
|
|
438
|
+
return;
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
const startedAt = new Date();
|
|
442
|
+
recordingDir = path.isAbsolute(notesDirOpt)
|
|
443
|
+
? notesDirOpt
|
|
444
|
+
: path.join(workspaceDir, notesDirOpt);
|
|
445
|
+
recordingBaseName = buildSessionBaseName(startedAt, opts?.notesTitle);
|
|
446
|
+
const language = getLanguage(kv);
|
|
447
|
+
const modelPath = getModelPath(kv);
|
|
448
|
+
|
|
449
|
+
try {
|
|
450
|
+
notesWriter = createNotesWriter({
|
|
451
|
+
dir: recordingDir,
|
|
452
|
+
baseName: recordingBaseName,
|
|
453
|
+
startedAt: startedAt.toISOString(),
|
|
454
|
+
language,
|
|
455
|
+
model: path.basename(modelPath),
|
|
456
|
+
});
|
|
457
|
+
} catch (err) {
|
|
458
|
+
toast(`Could not create notes file: ${err.message}`, "error");
|
|
459
|
+
return;
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
scratchDir = path.join(getTmpDir(), "opencode-voice-notes");
|
|
463
|
+
try {
|
|
464
|
+
fs.mkdirSync(scratchDir, { recursive: true });
|
|
465
|
+
} catch (err) {
|
|
466
|
+
logger?.log("VOICE", `Failed to create notes scratch dir: ${err.message}`, "warn");
|
|
467
|
+
}
|
|
468
|
+
|
|
469
|
+
chunker = createPcmChunker({
|
|
470
|
+
minChunkMs,
|
|
471
|
+
maxChunkMs: chunkMaxMs,
|
|
472
|
+
silenceMs,
|
|
473
|
+
overlapMs,
|
|
474
|
+
...(explicitSilenceRms > 0 ? { silenceRmsThreshold: explicitSilenceRms } : {}),
|
|
475
|
+
calibrationMs,
|
|
476
|
+
});
|
|
477
|
+
if (chunker.isCalibrated?.() === false) {
|
|
478
|
+
logger?.log("VOICE", "Live notes calibrating room noise floor...", "debug");
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
useApiMode = getSttApiConfig() !== null;
|
|
482
|
+
releaseWhisperServerLease();
|
|
483
|
+
whisperServerReadyPromise = null;
|
|
484
|
+
if (!useApiMode && useWhisperServer) {
|
|
485
|
+
whisperServerLease = acquireSharedWhisperServer({ modelPath, language, logger });
|
|
486
|
+
whisperServerClient = whisperServerLease.client;
|
|
487
|
+
whisperServerReadyPromise = whisperServerClient.start();
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
active = true;
|
|
491
|
+
recordingStartedAtMs = Date.now();
|
|
492
|
+
deps.tts?.setLiveNotesActive?.(true);
|
|
493
|
+
deps.tts?.stop?.();
|
|
494
|
+
|
|
495
|
+
const backend = detectAudioBackend();
|
|
496
|
+
if (!startCapture(backend)) {
|
|
497
|
+
active = false;
|
|
498
|
+
deps.tts?.setLiveNotesActive?.(false);
|
|
499
|
+
releaseWhisperServerLease();
|
|
500
|
+
await notesWriter.close();
|
|
501
|
+
notesWriter = null;
|
|
502
|
+
toast("Failed to start microphone for live notes", "error");
|
|
503
|
+
return;
|
|
504
|
+
}
|
|
505
|
+
|
|
506
|
+
showStatusToast();
|
|
507
|
+
logger?.log(
|
|
508
|
+
"VOICE",
|
|
509
|
+
`Live notes started dir=${recordingDir} base=${recordingBaseName}`,
|
|
510
|
+
"debug",
|
|
511
|
+
);
|
|
512
|
+
toast(`Live notes started: ${recordingBaseName}.md`, "success");
|
|
513
|
+
}
|
|
514
|
+
|
|
515
|
+
async function stopCommand() {
|
|
516
|
+
if (finishing) {
|
|
517
|
+
toast("Live notes already saving", "warning");
|
|
518
|
+
return;
|
|
519
|
+
}
|
|
520
|
+
if (!active) {
|
|
521
|
+
toast("Live notes not recording", "warning");
|
|
522
|
+
return;
|
|
523
|
+
}
|
|
524
|
+
// Claim the drain BEFORE the first await (SoX exit wait): startNotes
|
|
525
|
+
// must not replace notesWriter/chunker/lease mid-drain and corrupt
|
|
526
|
+
// sessions.
|
|
527
|
+
finishing = true;
|
|
528
|
+
active = false;
|
|
529
|
+
deps.tts?.setLiveNotesActive?.(false);
|
|
530
|
+
clearStatusToast();
|
|
531
|
+
if (soxProc) {
|
|
532
|
+
try {
|
|
533
|
+
soxProc.kill("SIGINT");
|
|
534
|
+
} catch {}
|
|
535
|
+
}
|
|
536
|
+
await waitForProcessExit(() => soxProc);
|
|
537
|
+
const finalChunk = chunker?.flush();
|
|
538
|
+
if (finalChunk) routeChunk(finalChunk);
|
|
539
|
+
|
|
540
|
+
toast(`Draining live notes - ${stageA.size() + stageB.size()} chunk(s) remaining`);
|
|
541
|
+
await drainAndSave();
|
|
542
|
+
}
|
|
543
|
+
|
|
544
|
+
function cancelCommand() {
|
|
545
|
+
if (!active) {
|
|
546
|
+
toast("Live notes not recording", "warning");
|
|
547
|
+
return;
|
|
548
|
+
}
|
|
549
|
+
// Same claim as stopCommand: block startNotes for the background drain.
|
|
550
|
+
finishing = true;
|
|
551
|
+
active = false;
|
|
552
|
+
deps.tts?.setLiveNotesActive?.(false);
|
|
553
|
+
clearStatusToast();
|
|
554
|
+
if (soxProc) {
|
|
555
|
+
try {
|
|
556
|
+
soxProc.kill("SIGINT");
|
|
557
|
+
} catch {}
|
|
558
|
+
}
|
|
559
|
+
toast("Live notes stopping - saving remaining chunks in the background", "info");
|
|
560
|
+
(async () => {
|
|
561
|
+
await waitForProcessExit(() => soxProc);
|
|
562
|
+
const finalChunk = chunker?.flush();
|
|
563
|
+
if (finalChunk) routeChunk(finalChunk);
|
|
564
|
+
await drainAndSave();
|
|
565
|
+
})();
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
function isActive() {
|
|
569
|
+
return active;
|
|
570
|
+
}
|
|
571
|
+
|
|
572
|
+
api.lifecycle?.onDispose?.(() => {
|
|
573
|
+
clearStatusToast();
|
|
574
|
+
if (active) cancelCommand();
|
|
575
|
+
else releaseWhisperServerLease();
|
|
576
|
+
});
|
|
577
|
+
|
|
578
|
+
const commands = [
|
|
579
|
+
{
|
|
580
|
+
title: "Voice notes: start",
|
|
581
|
+
value: "voice.notes.start",
|
|
582
|
+
category: "opencode-voice",
|
|
583
|
+
description: "Start continuous live-notes recording with background transcription",
|
|
584
|
+
slash: { name: "voice-notes-start" },
|
|
585
|
+
onSelect() {
|
|
586
|
+
startNotes();
|
|
587
|
+
},
|
|
588
|
+
},
|
|
589
|
+
{
|
|
590
|
+
title: "Voice notes: stop",
|
|
591
|
+
value: "voice.notes.stop",
|
|
592
|
+
category: "opencode-voice",
|
|
593
|
+
description: "Stop live notes, flush the remaining audio, and save",
|
|
594
|
+
slash: { name: "voice-notes-stop" },
|
|
595
|
+
onSelect() {
|
|
596
|
+
stopCommand();
|
|
597
|
+
},
|
|
598
|
+
},
|
|
599
|
+
{
|
|
600
|
+
title: "Voice notes: cancel",
|
|
601
|
+
value: "voice.notes.cancel",
|
|
602
|
+
category: "opencode-voice",
|
|
603
|
+
description: "Stop capturing immediately; finish saving in the background",
|
|
604
|
+
slash: { name: "voice-notes-cancel" },
|
|
605
|
+
onSelect() {
|
|
606
|
+
cancelCommand();
|
|
607
|
+
},
|
|
608
|
+
},
|
|
609
|
+
{
|
|
610
|
+
title: "Voice notes: status",
|
|
611
|
+
value: "voice.notes.status",
|
|
612
|
+
category: "opencode-voice",
|
|
613
|
+
description: "Show live-notes recording status",
|
|
614
|
+
slash: { name: "voice-notes-status" },
|
|
615
|
+
onSelect() {
|
|
616
|
+
toast(active ? statusMessage() : "Live notes not recording");
|
|
617
|
+
},
|
|
618
|
+
},
|
|
619
|
+
];
|
|
620
|
+
|
|
621
|
+
return { commands, controller: { isActive } };
|
|
622
|
+
}
|