@bojackduy/opencode-voice 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +729 -0
- package/index.js +124 -0
- package/lib/audio-chunker.js +231 -0
- package/lib/audio-enhance.js +172 -0
- package/lib/conversation.js +528 -0
- package/lib/live-notes.js +622 -0
- package/lib/llm-client.js +304 -0
- package/lib/logger.js +18 -0
- package/lib/notes-writer.js +224 -0
- package/lib/session.js +102 -0
- package/lib/streaming-editor.js +308 -0
- package/lib/streaming-stt.js +1322 -0
- package/lib/streaming-transcript.js +236 -0
- package/lib/stt.js +2339 -0
- package/lib/tts.js +671 -0
- package/lib/voice-model.js +122 -0
- package/lib/whisper-server.js +471 -0
- package/package.json +47 -0
package/index.js
ADDED
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
// opencode-voice: Speech-to-text and text-to-speech for OpenCode.
|
|
2
|
+
//
|
|
3
|
+
// STT: Record voice via sox, transcribe with whisper-cpp, normalize with
|
|
4
|
+
// an OpenAI-compatible LLM, append to the TUI prompt.
|
|
5
|
+
//
|
|
6
|
+
// TTS: Auto-speak assistant responses (or read on demand) via Piper,
|
|
7
|
+
// with LLM normalization for natural speech.
|
|
8
|
+
//
|
|
9
|
+
// Prerequisites:
|
|
10
|
+
// STT: brew install whisper-cpp sox
|
|
11
|
+
// TTS: Piper binary on PATH, voice models at ~/.local/share/piper-voices/
|
|
12
|
+
//
|
|
13
|
+
// Configuration via tui.json plugin options:
|
|
14
|
+
// ["opencode-voice", { "endpoint": "...", "model": "...", "apiKeyEnv": "..." }]
|
|
15
|
+
//
|
|
16
|
+
// Runtime state (model, mic, voice, tts mode) persisted via api.kv.
|
|
17
|
+
//
|
|
18
|
+
// Commands (palette + slash; default shortcuts use rare leader combos to avoid collisions):
|
|
19
|
+
// /stt-record - start/stop recording + transcribe (default: <leader>[ = ctrl+x, [)
|
|
20
|
+
// /stt-submit - stop recording + transcribe + submit (palette-only)
|
|
21
|
+
// /stt-stop - cancel recording (palette-only)
|
|
22
|
+
// /stt-model - select whisper model
|
|
23
|
+
// /stt-language - select transcription language
|
|
24
|
+
// /stt-mic - select microphone
|
|
25
|
+
// /tts-speak - read last response aloud (default: <leader>] = ctrl+x, ])
|
|
26
|
+
// /tts-mode - toggle auto TTS on/off (palette-only)
|
|
27
|
+
// /tts-stop - stop playback (default: <leader>; , also palette)
|
|
28
|
+
// /tts-voice - select TTS voice
|
|
29
|
+
// /voice-conversation - toggle hands-free voice conversation (default: <leader>v = ctrl+x, v)
|
|
30
|
+
// /voice-conversation-stop - exit voice conversation mode (palette-only)
|
|
31
|
+
// /voice-notes-start - start continuous live-notes recording (palette-only)
|
|
32
|
+
// /voice-notes-stop - stop live notes, flush, and save (palette-only)
|
|
33
|
+
// /voice-notes-cancel - stop live notes immediately, save in background (palette-only)
|
|
34
|
+
// /voice-notes-status - show live-notes recording status (palette-only)
|
|
35
|
+
// All also palette-accessible via Ctrl+P or /slash. Override via plugin options `keybinds`:
|
|
36
|
+
// { "keybinds": { "stt.record": "ctrl+r", "tts.speak-last": "none", "voice.conversation": "none" } }
|
|
37
|
+
// Weird keys [ ] ; were chosen because opencode doesn't use them and shift variants were ignored in terminals.
|
|
38
|
+
// <leader>v is free in opencode defaults (c/e/s/m/a/y/u/r/h etc. are taken) and mnemonic for voice.
|
|
39
|
+
|
|
40
|
+
import fs from "node:fs";
|
|
41
|
+
import os from "node:os";
|
|
42
|
+
import { registerSTT } from "./lib/stt.js";
|
|
43
|
+
import { registerTTS } from "./lib/tts.js";
|
|
44
|
+
import { registerConversation } from "./lib/conversation.js";
|
|
45
|
+
import { registerLiveNotes } from "./lib/live-notes.js";
|
|
46
|
+
import { registerVoiceModel, resolveVoiceProviderModel } from "./lib/voice-model.js";
|
|
47
|
+
import { createClient } from "./lib/llm-client.js";
|
|
48
|
+
import { createLogger } from "./lib/logger.js";
|
|
49
|
+
|
|
50
|
+
function loadPromptFile(filePath, logger, name) {
|
|
51
|
+
if (!filePath) return null;
|
|
52
|
+
const resolved = filePath.replace(/^~(?=\/|$)/, os.homedir());
|
|
53
|
+
try {
|
|
54
|
+
const prompt = fs.readFileSync(resolved, "utf-8").trim() || null;
|
|
55
|
+
logger?.log(
|
|
56
|
+
"plugin",
|
|
57
|
+
prompt ? `Loaded ${name} prompt: ${resolved}` : `Ignored empty ${name} prompt: ${resolved}`,
|
|
58
|
+
"debug",
|
|
59
|
+
);
|
|
60
|
+
return prompt;
|
|
61
|
+
} catch (err) {
|
|
62
|
+
logger?.log("Plugin", `Failed to load ${name} prompt ${resolved}: ${err.message}`, "warn");
|
|
63
|
+
return null;
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
export default {
|
|
68
|
+
id: "opencode-voice",
|
|
69
|
+
tui: async (api, options) => {
|
|
70
|
+
const { kv } = api;
|
|
71
|
+
const logger = createLogger(api.client);
|
|
72
|
+
logger.log("plugin", "Initializing", "debug");
|
|
73
|
+
// Session-scoped gateways (opencode.ai/zen/go) require x-opencode-session.
|
|
74
|
+
// Read the live route at call time so normalize works in any session.
|
|
75
|
+
// The voice-selected provider (/voice-model) fills endpoint/model only
|
|
76
|
+
// when the explicit options leave them out.
|
|
77
|
+
const { complete } = createClient(
|
|
78
|
+
options,
|
|
79
|
+
logger,
|
|
80
|
+
() => {
|
|
81
|
+
const route = api?.route?.current;
|
|
82
|
+
return route?.name === "session" ? route?.params?.sessionID : undefined;
|
|
83
|
+
},
|
|
84
|
+
() => resolveVoiceProviderModel(api, kv),
|
|
85
|
+
);
|
|
86
|
+
|
|
87
|
+
const prompts = {
|
|
88
|
+
stt: loadPromptFile(options?.sttPrompt, logger, "STT"),
|
|
89
|
+
ttsAuto: loadPromptFile(options?.ttsAutoPrompt, logger, "TTS auto"),
|
|
90
|
+
ttsManual: loadPromptFile(options?.ttsManualPrompt, logger, "TTS manual"),
|
|
91
|
+
};
|
|
92
|
+
|
|
93
|
+
const shared = {};
|
|
94
|
+
const stt = registerSTT(api, kv, complete, prompts, options, logger, {
|
|
95
|
+
isConversationActive: () => shared.conversation?.isActive() === true,
|
|
96
|
+
onConversationKey: (source) => shared.conversation?.onKey(source),
|
|
97
|
+
isLiveNotesActive: () => shared.liveNotes?.isActive() === true,
|
|
98
|
+
});
|
|
99
|
+
const tts = registerTTS(api, kv, complete, prompts, options, logger, {
|
|
100
|
+
isConversationActive: () => shared.conversation?.isActive() === true,
|
|
101
|
+
onConversationStopKey: () => shared.conversation?.onTtsStop() === true,
|
|
102
|
+
});
|
|
103
|
+
const conversation = registerConversation(api, options, logger, {
|
|
104
|
+
stt: stt.controller,
|
|
105
|
+
tts: tts.controller,
|
|
106
|
+
isLiveNotesActive: () => shared.liveNotes?.isActive() === true,
|
|
107
|
+
});
|
|
108
|
+
shared.conversation = conversation.controller;
|
|
109
|
+
const liveNotes = registerLiveNotes(api, kv, complete, options, logger, {
|
|
110
|
+
tts: tts.controller,
|
|
111
|
+
isConversationActive: () => shared.conversation?.isActive() === true,
|
|
112
|
+
});
|
|
113
|
+
shared.liveNotes = liveNotes.controller;
|
|
114
|
+
const voiceModel = registerVoiceModel(api, kv, options, logger);
|
|
115
|
+
|
|
116
|
+
api.command.register(() => [
|
|
117
|
+
...stt.commands,
|
|
118
|
+
...tts.commands,
|
|
119
|
+
...conversation.commands,
|
|
120
|
+
...liveNotes.commands,
|
|
121
|
+
...voiceModel.commands,
|
|
122
|
+
]);
|
|
123
|
+
},
|
|
124
|
+
};
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
// Continuous-capture audio chunker for live voice notes.
|
|
2
|
+
//
|
|
3
|
+
// Consumes a continuous stream of raw PCM16LE mono samples (as produced by
|
|
4
|
+
// `sox ... -t raw -`) and splits it into self-contained chunks the
|
|
5
|
+
// transcription lane can process independently, without ever pausing
|
|
6
|
+
// capture. Two triggers close a chunk:
|
|
7
|
+
//
|
|
8
|
+
// - natural: enough speech has accumulated and a silence gap follows (a
|
|
9
|
+
// real pause in the conversation - the ideal cut point)
|
|
10
|
+
// - forced: the chunk hit its max duration regardless of silence (bounds
|
|
11
|
+
// worst-case processing latency during continuous speech)
|
|
12
|
+
//
|
|
13
|
+
// Forced cuts carry a short audio overlap into the next chunk so a word is
|
|
14
|
+
// never fully lost mid-cut; the caller (notes-writer) dedupes the
|
|
15
|
+
// overlapping words from the transcribed text.
|
|
16
|
+
|
|
17
|
+
const SAMPLE_RATE = 16000;
|
|
18
|
+
const BYTES_PER_SAMPLE = 2;
|
|
19
|
+
const FRAME_MS = 20;
|
|
20
|
+
const FRAME_BYTES = (SAMPLE_RATE / 1000) * FRAME_MS * BYTES_PER_SAMPLE; // 640
|
|
21
|
+
|
|
22
|
+
export const CHUNKER_FRAME_MS = FRAME_MS;
|
|
23
|
+
export const CHUNKER_FRAME_BYTES = FRAME_BYTES;
|
|
24
|
+
|
|
25
|
+
const DEFAULTS = {
|
|
26
|
+
sampleRate: SAMPLE_RATE,
|
|
27
|
+
minChunkMs: 3000,
|
|
28
|
+
maxChunkMs: 20000,
|
|
29
|
+
silenceMs: 700,
|
|
30
|
+
overlapMs: 400,
|
|
31
|
+
silenceRmsThreshold: 0.02,
|
|
32
|
+
minSpeechMsToKeep: 300,
|
|
33
|
+
// Auto-learn the room noise floor from the first `calibrationMs` of audio
|
|
34
|
+
// (usually mic hiss / HVAC before anyone speaks) instead of assuming the
|
|
35
|
+
// fixed threshold fits every room. 0 disables. An explicit
|
|
36
|
+
// silenceRmsThreshold always wins and skips calibration.
|
|
37
|
+
calibrationMs: 0,
|
|
38
|
+
};
|
|
39
|
+
|
|
40
|
+
// Threshold = 2.5x the room floor, clamped so a very quiet room never drops
|
|
41
|
+
// into the mic's own noise and a loud room never eats quiet speech.
|
|
42
|
+
export const CALIBRATION_MULTIPLIER = 2.5;
|
|
43
|
+
export const CALIBRATION_MIN_THRESHOLD = 0.008;
|
|
44
|
+
export const CALIBRATION_MAX_THRESHOLD = 0.03;
|
|
45
|
+
|
|
46
|
+
export function calibrateSilenceThreshold(noiseFloorRms) {
|
|
47
|
+
const t = (noiseFloorRms || 0) * CALIBRATION_MULTIPLIER;
|
|
48
|
+
return Math.min(CALIBRATION_MAX_THRESHOLD, Math.max(CALIBRATION_MIN_THRESHOLD, t));
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* RMS (0..1) of a buffer of PCM16LE mono samples. Returns 0 for an empty or
|
|
53
|
+
* odd-length buffer.
|
|
54
|
+
*/
|
|
55
|
+
export function pcm16Rms(buffer) {
|
|
56
|
+
const usable = buffer.length - (buffer.length % 2);
|
|
57
|
+
if (usable <= 0) return 0;
|
|
58
|
+
let sum = 0;
|
|
59
|
+
const samples = usable / 2;
|
|
60
|
+
for (let i = 0; i < usable; i += 2) {
|
|
61
|
+
const n = buffer.readInt16LE(i) / 32768;
|
|
62
|
+
sum += n * n;
|
|
63
|
+
}
|
|
64
|
+
return Math.sqrt(sum / samples);
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Build a standard 44-byte PCM WAV header for the given data length.
|
|
69
|
+
*/
|
|
70
|
+
export function buildWavHeader(dataLength, options = {}) {
|
|
71
|
+
const sampleRate = options.sampleRate ?? SAMPLE_RATE;
|
|
72
|
+
const channels = options.channels ?? 1;
|
|
73
|
+
const bitsPerSample = options.bitsPerSample ?? 16;
|
|
74
|
+
const blockAlign = channels * (bitsPerSample / 8);
|
|
75
|
+
const byteRate = sampleRate * blockAlign;
|
|
76
|
+
|
|
77
|
+
const header = Buffer.alloc(44);
|
|
78
|
+
header.write("RIFF", 0, "ascii");
|
|
79
|
+
header.writeUInt32LE(36 + dataLength, 4);
|
|
80
|
+
header.write("WAVE", 8, "ascii");
|
|
81
|
+
header.write("fmt ", 12, "ascii");
|
|
82
|
+
header.writeUInt32LE(16, 16);
|
|
83
|
+
header.writeUInt16LE(1, 20); // PCM
|
|
84
|
+
header.writeUInt16LE(channels, 22);
|
|
85
|
+
header.writeUInt32LE(sampleRate, 24);
|
|
86
|
+
header.writeUInt32LE(byteRate, 28);
|
|
87
|
+
header.writeUInt16LE(blockAlign, 32);
|
|
88
|
+
header.writeUInt16LE(bitsPerSample, 34);
|
|
89
|
+
header.write("data", 36, "ascii");
|
|
90
|
+
header.writeUInt32LE(dataLength, 40);
|
|
91
|
+
return header;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Wrap raw PCM16LE mono samples into a self-contained WAV buffer.
|
|
96
|
+
*/
|
|
97
|
+
export function wrapPcmAsWav(pcmBuffer, options = {}) {
|
|
98
|
+
return Buffer.concat([buildWavHeader(pcmBuffer.length, options), pcmBuffer]);
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* Create a stateful chunker. Feed it raw PCM16LE mono bytes via `push()` as
|
|
103
|
+
* they arrive from the capture process; it returns any chunks that became
|
|
104
|
+
* ready to transcribe. Call `flush()` once when capture stops to get the
|
|
105
|
+
* final partial chunk.
|
|
106
|
+
*
|
|
107
|
+
* Chunk shape: { seq, pcm, startMs, endMs, durationMs, forced, hasSpeech }
|
|
108
|
+
*/
|
|
109
|
+
export function createPcmChunker(options = {}) {
|
|
110
|
+
const cfg = { ...DEFAULTS, ...options };
|
|
111
|
+
const frameBytes = ((cfg.sampleRate / 1000) * FRAME_MS * BYTES_PER_SAMPLE) | 0;
|
|
112
|
+
const overlapFrameCount = Math.max(0, Math.round(cfg.overlapMs / FRAME_MS));
|
|
113
|
+
// Explicit threshold wins; otherwise learn the room for calibrationMs.
|
|
114
|
+
const autoCalibrate = !(options.silenceRmsThreshold > 0) && cfg.calibrationMs > 0;
|
|
115
|
+
|
|
116
|
+
let frames = [];
|
|
117
|
+
let leftover = Buffer.alloc(0);
|
|
118
|
+
let accMs = 0;
|
|
119
|
+
let speechMs = 0;
|
|
120
|
+
let silenceRun = 0;
|
|
121
|
+
let startMs = 0;
|
|
122
|
+
let totalElapsedMs = 0;
|
|
123
|
+
let seq = 0;
|
|
124
|
+
let calibrated = !autoCalibrate;
|
|
125
|
+
let calibrationRms = [];
|
|
126
|
+
let noiseFloorRms = null;
|
|
127
|
+
|
|
128
|
+
function finalizeChunk(forced) {
|
|
129
|
+
const pcm = Buffer.concat(frames);
|
|
130
|
+
const endMs = totalElapsedMs;
|
|
131
|
+
const chunk = {
|
|
132
|
+
seq: seq++,
|
|
133
|
+
pcm,
|
|
134
|
+
startMs,
|
|
135
|
+
endMs,
|
|
136
|
+
durationMs: endMs - startMs,
|
|
137
|
+
forced,
|
|
138
|
+
hasSpeech: speechMs >= cfg.minSpeechMsToKeep,
|
|
139
|
+
};
|
|
140
|
+
|
|
141
|
+
const keepFrames = forced ? frames.slice(-overlapFrameCount) : [];
|
|
142
|
+
frames = keepFrames.map((f) => Buffer.from(f));
|
|
143
|
+
accMs = frames.length * FRAME_MS;
|
|
144
|
+
speechMs = 0;
|
|
145
|
+
silenceRun = 0;
|
|
146
|
+
startMs = endMs - accMs;
|
|
147
|
+
|
|
148
|
+
return chunk;
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
function push(buffer) {
|
|
152
|
+
const ready = [];
|
|
153
|
+
let combined = leftover.length > 0 ? Buffer.concat([leftover, buffer]) : buffer;
|
|
154
|
+
const frameCount = Math.floor(combined.length / frameBytes);
|
|
155
|
+
leftover = combined.subarray(frameCount * frameBytes);
|
|
156
|
+
|
|
157
|
+
for (let i = 0; i < frameCount; i++) {
|
|
158
|
+
const frame = combined.subarray(i * frameBytes, (i + 1) * frameBytes);
|
|
159
|
+
const frameRms = pcm16Rms(frame);
|
|
160
|
+
|
|
161
|
+
// While calibrating, collect room-tone levels but don't cut on them -
|
|
162
|
+
// the fixed default threshold is meaningless before we know the room.
|
|
163
|
+
if (!calibrated) {
|
|
164
|
+
calibrationRms.push(frameRms);
|
|
165
|
+
frames.push(frame);
|
|
166
|
+
accMs += FRAME_MS;
|
|
167
|
+
totalElapsedMs += FRAME_MS;
|
|
168
|
+
if (totalElapsedMs >= cfg.calibrationMs) {
|
|
169
|
+
const sorted = [...calibrationRms].sort((a, b) => a - b);
|
|
170
|
+
noiseFloorRms = sorted[Math.floor(sorted.length / 2)] ?? 0;
|
|
171
|
+
cfg.silenceRmsThreshold = calibrateSilenceThreshold(noiseFloorRms);
|
|
172
|
+
calibrationRms = [];
|
|
173
|
+
calibrated = true;
|
|
174
|
+
}
|
|
175
|
+
continue;
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
const isSilence = frameRms < cfg.silenceRmsThreshold;
|
|
179
|
+
|
|
180
|
+
frames.push(frame);
|
|
181
|
+
accMs += FRAME_MS;
|
|
182
|
+
totalElapsedMs += FRAME_MS;
|
|
183
|
+
if (isSilence) silenceRun += FRAME_MS;
|
|
184
|
+
else {
|
|
185
|
+
silenceRun = 0;
|
|
186
|
+
speechMs += FRAME_MS;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
const forcedReady = accMs >= cfg.maxChunkMs;
|
|
190
|
+
const naturalReady =
|
|
191
|
+
!forcedReady &&
|
|
192
|
+
accMs >= cfg.minChunkMs &&
|
|
193
|
+
speechMs >= cfg.minSpeechMsToKeep &&
|
|
194
|
+
silenceRun >= cfg.silenceMs;
|
|
195
|
+
|
|
196
|
+
if (forcedReady || naturalReady) {
|
|
197
|
+
ready.push(finalizeChunk(forcedReady));
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
return ready;
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
function flush() {
|
|
205
|
+
if (accMs <= 0) return null;
|
|
206
|
+
return finalizeChunk(false);
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
function reset() {
|
|
210
|
+
frames = [];
|
|
211
|
+
leftover = Buffer.alloc(0);
|
|
212
|
+
accMs = 0;
|
|
213
|
+
speechMs = 0;
|
|
214
|
+
silenceRun = 0;
|
|
215
|
+
startMs = 0;
|
|
216
|
+
totalElapsedMs = 0;
|
|
217
|
+
seq = 0;
|
|
218
|
+
calibrated = !autoCalibrate;
|
|
219
|
+
calibrationRms = [];
|
|
220
|
+
noiseFloorRms = null;
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
return {
|
|
224
|
+
push,
|
|
225
|
+
flush,
|
|
226
|
+
reset,
|
|
227
|
+
getSilenceThreshold: () => cfg.silenceRmsThreshold,
|
|
228
|
+
getNoiseFloor: () => noiseFloorRms,
|
|
229
|
+
isCalibrated: () => calibrated,
|
|
230
|
+
};
|
|
231
|
+
}
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
// Voice-only preprocessing for live notes (and anything else that feeds
|
|
2
|
+
// quiet far-field audio to whisper).
|
|
3
|
+
//
|
|
4
|
+
// A classroom professor 3-5m from a laptop mic produces RMS ~0.01-0.02 -
|
|
5
|
+
// right at the chunker silence threshold and far below what whisper.cpp was
|
|
6
|
+
// trained on. Human ears compensate with automatic gain control; whisper
|
|
7
|
+
// does not, so it hallucinates (Vietnamese YouTube outros) or mistranscribes
|
|
8
|
+
// instead. This module applies the missing AGC in software before each chunk
|
|
9
|
+
// reaches whisper:
|
|
10
|
+
//
|
|
11
|
+
// 1. measure the chunk (RMS + peak, same math as audio-chunker.js)
|
|
12
|
+
// 2. compute the gain needed to bring speech up to a healthy target level
|
|
13
|
+
// 3. run sox: highpass (room rumble/HVAC out) + gain with limiter (speech
|
|
14
|
+
// up, loud chunks untouched, never clips)
|
|
15
|
+
//
|
|
16
|
+
// Enhancement is per-chunk (not in the live capture chain) so the mic path
|
|
17
|
+
// stays untouched and a failed enhance can never lose audio - on any sox
|
|
18
|
+
// error the original file is kept as-is.
|
|
19
|
+
|
|
20
|
+
import fs from "node:fs";
|
|
21
|
+
import { spawn } from "node:child_process";
|
|
22
|
+
import { pcm16Rms } from "./audio-chunker.js";
|
|
23
|
+
|
|
24
|
+
export const DEFAULT_ENHANCE_TARGET_RMS = 0.1;
|
|
25
|
+
export const DEFAULT_ENHANCE_MAX_GAIN_DB = 24;
|
|
26
|
+
export const DEFAULT_ENHANCE_HIGHPASS_HZ = 80;
|
|
27
|
+
// Chunks quieter than this are essentially silence - gaining them up only
|
|
28
|
+
// amplifies noise into hallucination fuel, so they pass through untouched.
|
|
29
|
+
export const DEFAULT_ENHANCE_MIN_RMS = 0.004;
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* RMS + peak (both 0..1) of a PCM16LE mono buffer. Pure.
|
|
33
|
+
*/
|
|
34
|
+
export function measurePcmStats(pcmBuffer) {
|
|
35
|
+
const rms = pcm16Rms(pcmBuffer);
|
|
36
|
+
let peak = 0;
|
|
37
|
+
const usable = pcmBuffer.length - (pcmBuffer.length % 2);
|
|
38
|
+
for (let i = 0; i < usable; i += 2) {
|
|
39
|
+
const a = Math.abs(pcmBuffer.readInt16LE(i) / 32768);
|
|
40
|
+
if (a > peak) peak = a;
|
|
41
|
+
}
|
|
42
|
+
return { rms, peak };
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export function rmsToDb(rms) {
|
|
46
|
+
if (!rms || rms <= 0) return -Infinity;
|
|
47
|
+
return 20 * Math.log10(rms);
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Gain in dB needed to bring `rms` up to `targetRms`. Never negative
|
|
52
|
+
* (loud chunks stay as they are) and never above `maxGainDb`. Pure.
|
|
53
|
+
*/
|
|
54
|
+
export function computeGainDb(rms, options = {}) {
|
|
55
|
+
const targetRms = options.targetRms ?? DEFAULT_ENHANCE_TARGET_RMS;
|
|
56
|
+
const maxGainDb = options.maxGainDb ?? DEFAULT_ENHANCE_MAX_GAIN_DB;
|
|
57
|
+
if (!rms || rms <= 0 || rms >= targetRms) return 0;
|
|
58
|
+
return Math.min(maxGainDb, 20 * Math.log10(targetRms / rms));
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* sox effect chain: strip sub-voice rumble, then adaptive gain with the
|
|
63
|
+
* limiter engaged so an underestimated peak can never clip. Pure.
|
|
64
|
+
*/
|
|
65
|
+
export function buildEnhanceArgs(gainDb, options = {}) {
|
|
66
|
+
const highpassHz = options.highpassHz ?? DEFAULT_ENHANCE_HIGHPASS_HZ;
|
|
67
|
+
return ["highpass", String(highpassHz), "gain", "-l", gainDb.toFixed(1)];
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* Run sox without blocking the event loop. Enhancement runs on the
|
|
72
|
+
* processing-lane stage (not the capture lane), but a synchronous 15s spawn
|
|
73
|
+
* would still stall every timer/callback sharing the loop — including the
|
|
74
|
+
* stdout drain of a live capture — so the pipe can fill and stall SoX.
|
|
75
|
+
*/
|
|
76
|
+
function runSoxAsync(args, timeoutMs) {
|
|
77
|
+
return new Promise((resolve) => {
|
|
78
|
+
let proc;
|
|
79
|
+
try {
|
|
80
|
+
proc = spawn("sox", args);
|
|
81
|
+
} catch (err) {
|
|
82
|
+
resolve({ error: err, status: null, stderr: "" });
|
|
83
|
+
return;
|
|
84
|
+
}
|
|
85
|
+
let stderr = "";
|
|
86
|
+
const timer = setTimeout(() => {
|
|
87
|
+
try {
|
|
88
|
+
proc.kill("SIGKILL");
|
|
89
|
+
} catch {}
|
|
90
|
+
}, timeoutMs);
|
|
91
|
+
proc.stderr?.on("data", (chunk) => {
|
|
92
|
+
stderr += chunk.toString();
|
|
93
|
+
});
|
|
94
|
+
proc.on("error", (err) => {
|
|
95
|
+
clearTimeout(timer);
|
|
96
|
+
resolve({ error: err, status: null, stderr });
|
|
97
|
+
});
|
|
98
|
+
proc.on("close", (code) => {
|
|
99
|
+
clearTimeout(timer);
|
|
100
|
+
resolve({ error: null, status: code, stderr });
|
|
101
|
+
});
|
|
102
|
+
});
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Enhance one chunk WAV file in place for whisper. `pcm` is the raw chunk
|
|
107
|
+
* samples (used for measurement so we don't have to re-read the file).
|
|
108
|
+
* Returns stats for the JSONL sidecar; `enhanced:false` means the original
|
|
109
|
+
* file was kept untouched (quiet-as-silence, already loud, or sox failed).
|
|
110
|
+
* Async (never blocks the event loop); on any sox failure the original file
|
|
111
|
+
* is kept as-is so enhancement can never lose audio.
|
|
112
|
+
*/
|
|
113
|
+
export async function enhanceWavFile(wavPath, pcm, options = {}, logger) {
|
|
114
|
+
const targetRms = options.targetRms ?? DEFAULT_ENHANCE_TARGET_RMS;
|
|
115
|
+
const maxGainDb = options.maxGainDb ?? DEFAULT_ENHANCE_MAX_GAIN_DB;
|
|
116
|
+
const minRms = options.minRms ?? DEFAULT_ENHANCE_MIN_RMS;
|
|
117
|
+
|
|
118
|
+
const before = measurePcmStats(pcm);
|
|
119
|
+
const noEnhance = (reason) => ({
|
|
120
|
+
enhanced: false,
|
|
121
|
+
reason,
|
|
122
|
+
rmsBefore: before.rms,
|
|
123
|
+
peakBefore: before.peak,
|
|
124
|
+
gainDb: 0,
|
|
125
|
+
});
|
|
126
|
+
|
|
127
|
+
if (before.rms < minRms) return noEnhance("too-quiet");
|
|
128
|
+
const gainDb = computeGainDb(before.rms, { targetRms, maxGainDb });
|
|
129
|
+
if (gainDb <= 0) return noEnhance("already-loud");
|
|
130
|
+
|
|
131
|
+
const tmpPath = `${wavPath}.enh.wav`;
|
|
132
|
+
const args = [wavPath, "-b", "16", tmpPath, ...buildEnhanceArgs(gainDb, options)];
|
|
133
|
+
let res;
|
|
134
|
+
try {
|
|
135
|
+
res = await runSoxAsync(args, 15000);
|
|
136
|
+
} catch (err) {
|
|
137
|
+
logger?.log("VOICE", `Enhance spawn failed, keeping original: ${err.message}`, "warn");
|
|
138
|
+
return noEnhance("sox-spawn-failed");
|
|
139
|
+
}
|
|
140
|
+
if (res.error || res.status !== 0) {
|
|
141
|
+
try {
|
|
142
|
+
fs.unlinkSync(tmpPath);
|
|
143
|
+
} catch {}
|
|
144
|
+
logger?.log(
|
|
145
|
+
"VOICE",
|
|
146
|
+
`Enhance failed, keeping original: ${(res.error?.message || (res.stderr || "").trim() || `exit ${res.status}`).slice(0, 200)}`,
|
|
147
|
+
"warn",
|
|
148
|
+
);
|
|
149
|
+
return noEnhance(res.error ? "sox-spawn-failed" : "sox-failed");
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
let after;
|
|
153
|
+
try {
|
|
154
|
+
const out = fs.readFileSync(tmpPath);
|
|
155
|
+
after = measurePcmStats(out.subarray(Math.min(44, out.length)));
|
|
156
|
+
fs.renameSync(tmpPath, wavPath);
|
|
157
|
+
} catch (err) {
|
|
158
|
+
try {
|
|
159
|
+
fs.unlinkSync(tmpPath);
|
|
160
|
+
} catch {}
|
|
161
|
+
logger?.log("VOICE", `Enhance replace failed, keeping original: ${err.message}`, "warn");
|
|
162
|
+
return noEnhance("replace-failed");
|
|
163
|
+
}
|
|
164
|
+
return {
|
|
165
|
+
enhanced: true,
|
|
166
|
+
gainDb,
|
|
167
|
+
rmsBefore: before.rms,
|
|
168
|
+
peakBefore: before.peak,
|
|
169
|
+
rmsAfter: after.rms,
|
|
170
|
+
peakAfter: after.peak,
|
|
171
|
+
};
|
|
172
|
+
}
|