omnirush 0.8.6 → 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/CHANGELOG.md +120 -0
- package/assets/extensions/omnirush/agents-lib.ts +501 -63
- package/assets/extensions/omnirush/agents.ts +140 -115
- package/assets/extensions/omnirush/bgshell-lib.ts +432 -0
- package/assets/extensions/omnirush/bgshell.ts +392 -0
- package/assets/extensions/omnirush/capture/workspace-collector.ts +86 -22
- package/assets/extensions/omnirush/collector.ts +275 -43
- package/assets/extensions/omnirush/commands.ts +2 -0
- package/assets/extensions/omnirush/deliveries.ts +145 -0
- package/assets/extensions/omnirush/guard/UPSTREAM +2 -0
- package/assets/extensions/omnirush/guard/git-command-policy.ts +885 -0
- package/assets/extensions/omnirush/guard-lib.ts +230 -0
- package/assets/extensions/omnirush/guard.ts +340 -0
- package/assets/extensions/omnirush/index.ts +32 -6
- package/assets/extensions/omnirush/mcp.ts +103 -8
- package/assets/extensions/omnirush/memory-lib.ts +516 -0
- package/assets/extensions/omnirush/pi-engine.ts +115 -3
- package/assets/extensions/omnirush/plan-lib.ts +38 -9
- package/assets/extensions/omnirush/plan.ts +85 -30
- package/assets/extensions/omnirush/sota.ts +52 -4
- package/assets/extensions/omnirush/status-lib.ts +3 -0
- package/assets/extensions/omnirush/subagent-marker.ts +21 -0
- package/assets/extensions/omnirush/subagents-lib.ts +623 -0
- package/assets/extensions/omnirush/subagents.ts +305 -0
- package/assets/extensions/omnirush/swarm-lib.ts +142 -0
- package/assets/extensions/omnirush/swarm.ts +95 -0
- package/assets/extensions/omnirush/voice/capture.ts +502 -0
- package/assets/extensions/omnirush/voice/core/UPSTREAM +16 -0
- package/assets/extensions/omnirush/voice/core/file-source.ts +70 -0
- package/assets/extensions/omnirush/voice/core/index.ts +21 -0
- package/assets/extensions/omnirush/voice/core/keyterms.ts +117 -0
- package/assets/extensions/omnirush/voice/core/resample.ts +63 -0
- package/assets/extensions/omnirush/voice/core/segmenter.ts +231 -0
- package/assets/extensions/omnirush/voice/core/session.ts +403 -0
- package/assets/extensions/omnirush/voice/core/text.ts +81 -0
- package/assets/extensions/omnirush/voice/core/transcriber.ts +135 -0
- package/assets/extensions/omnirush/voice/core/types.ts +102 -0
- package/assets/extensions/omnirush/voice/core/wav.ts +95 -0
- package/assets/extensions/omnirush/voice/keys.ts +435 -0
- package/assets/extensions/omnirush/voice/kitty.ts +64 -0
- package/assets/extensions/omnirush/voice/pvrecorder-worker.cjs +43 -0
- package/assets/extensions/omnirush/voice/settings.ts +67 -0
- package/assets/extensions/omnirush/voice.ts +838 -0
- package/assets/extensions/omnirush/yolo-lib.ts +80 -0
- package/assets/extensions/omnirush/yolo.ts +85 -0
- package/package.json +7 -3
- package/scripts/brand-engine.js +526 -0
- package/scripts/build-all-packages.py +29 -1
- package/scripts/smoke-packages.py +33 -1
- package/src/bin.js +232 -33
- package/src/compat.js +272 -0
- package/src/lib.js +64 -0
- package/src/sessions.js +222 -0
- package/scripts/patch-pi-branding.js +0 -251
|
@@ -0,0 +1,502 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Microphone capture for the CLI: 16 kHz mono s16le PCM, in memory only.
|
|
3
|
+
*
|
|
4
|
+
* Recorder chain (first one that opens wins):
|
|
5
|
+
* 1. @picovoice/pvrecorder-node (optional dependency, Apache-2.0, no
|
|
6
|
+
* access key): native, prebuilt for macOS x64/arm64, Windows
|
|
7
|
+
* x64/arm64, Linux x64 and Raspberry Pi. Runs in a worker thread
|
|
8
|
+
* because its read() blocks.
|
|
9
|
+
* 2. System tools writing raw PCM to stdout (never to a file):
|
|
10
|
+
* Linux pw-record → parecord → arecord; SoX `rec` (Linux/macOS);
|
|
11
|
+
* ffmpeg last (pulse/alsa, avfoundation, dshow).
|
|
12
|
+
* 3. Nothing works → one exact install command for this OS.
|
|
13
|
+
*
|
|
14
|
+
* OMNIRUSH_VOICE_INPUT_FILE=<wav> replaces the microphone with the file,
|
|
15
|
+
* paced in real time (tests and e2e); the file is only read.
|
|
16
|
+
*
|
|
17
|
+
* Audio is never written anywhere: tool output is piped, frames are handed
|
|
18
|
+
* to the caller and not retained here.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import { spawn, type ChildProcess } from "node:child_process";
|
|
22
|
+
import fs from "node:fs";
|
|
23
|
+
import path from "node:path";
|
|
24
|
+
import { createRequire } from "node:module";
|
|
25
|
+
import { rms, VOICE_SAMPLE_RATE as SAMPLE_RATE, VoiceError, WavFileSource, voiceInputFileFromEnv, type AudioSource } from "./core";
|
|
26
|
+
|
|
27
|
+
/** A core AudioSource with a name for messages ("pvrecorder", "pw-record", "file", …). */
|
|
28
|
+
export interface PcmSource extends AudioSource {
|
|
29
|
+
readonly name: string;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export interface Candidate {
|
|
33
|
+
name: string;
|
|
34
|
+
kind: "native" | "tool";
|
|
35
|
+
command?: string;
|
|
36
|
+
args?: string[];
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export class RecorderUnavailableError extends Error {
|
|
40
|
+
constructor(
|
|
41
|
+
message: string,
|
|
42
|
+
readonly hint: string | null,
|
|
43
|
+
readonly tried: Array<{ name: string; reason: string }>,
|
|
44
|
+
) {
|
|
45
|
+
super(message);
|
|
46
|
+
this.name = "RecorderUnavailableError";
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
// ---------------------------------------------------------------------------
|
|
51
|
+
// Environment checks
|
|
52
|
+
|
|
53
|
+
export interface EnvProbe {
|
|
54
|
+
env: NodeJS.ProcessEnv;
|
|
55
|
+
platform: NodeJS.Platform;
|
|
56
|
+
exists: (p: string) => boolean;
|
|
57
|
+
readText: (p: string) => string | null;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export function realProbe(): EnvProbe {
|
|
61
|
+
return {
|
|
62
|
+
env: process.env,
|
|
63
|
+
platform: process.platform,
|
|
64
|
+
exists: (p) => {
|
|
65
|
+
try {
|
|
66
|
+
fs.accessSync(p);
|
|
67
|
+
return true;
|
|
68
|
+
} catch {
|
|
69
|
+
return false;
|
|
70
|
+
}
|
|
71
|
+
},
|
|
72
|
+
readText: (p) => {
|
|
73
|
+
try {
|
|
74
|
+
return fs.readFileSync(p, "utf8");
|
|
75
|
+
} catch {
|
|
76
|
+
return null;
|
|
77
|
+
}
|
|
78
|
+
},
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/** WSL1 / WSL2 / not WSL. */
|
|
83
|
+
export function wslVersion(probe: EnvProbe): 0 | 1 | 2 {
|
|
84
|
+
if (probe.platform !== "linux") return 0;
|
|
85
|
+
const version = (probe.readText("/proc/version") || "").toLowerCase();
|
|
86
|
+
if (!version.includes("microsoft")) return 0;
|
|
87
|
+
return version.includes("wsl2") || probe.exists("/run/WSL") || !!probe.env.WSL_INTEROP ? 2 : 1;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
function hasSoundServer(probe: EnvProbe): boolean {
|
|
91
|
+
const env = probe.env;
|
|
92
|
+
if (env.PULSE_SERVER) return true;
|
|
93
|
+
const runtime = env.XDG_RUNTIME_DIR;
|
|
94
|
+
if (runtime && (probe.exists(path.join(runtime, "pulse", "native")) || probe.exists(path.join(runtime, "pipewire-0")))) return true;
|
|
95
|
+
return false;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Why a microphone can't be used here (remote session, cloud workspace,
|
|
100
|
+
* no audio device), or null when capture may work.
|
|
101
|
+
*/
|
|
102
|
+
export function remoteReason(probe: EnvProbe): string | null {
|
|
103
|
+
const env = probe.env;
|
|
104
|
+
if (env.OMNIRUSH_VOICE_INPUT_FILE) return null;
|
|
105
|
+
if (env.SSH_CONNECTION || env.SSH_CLIENT || env.SSH_TTY) {
|
|
106
|
+
return "this is an SSH session: the microphone is on your own machine, not on the host — run omnirush locally to dictate";
|
|
107
|
+
}
|
|
108
|
+
if (env.CODESPACES === "true" || env.GITPOD_WORKSPACE_ID || env.CLOUD_SHELL === "true" || env.DEVPOD === "true" || env.AWS_CLOUD9_USER) {
|
|
109
|
+
return "this is a cloud workspace with no microphone — run omnirush locally to dictate";
|
|
110
|
+
}
|
|
111
|
+
if (probe.platform === "linux") {
|
|
112
|
+
const wsl = wslVersion(probe);
|
|
113
|
+
if (wsl === 1) return "WSL 1 has no audio input — run omnirush in native Windows (or WSL 2 with WSLg)";
|
|
114
|
+
if (wsl === 2) return null; // WSLg exposes PulseAudio
|
|
115
|
+
if (!probe.exists("/dev/snd") && !hasSoundServer(probe)) {
|
|
116
|
+
const container = probe.exists("/.dockerenv") || probe.exists("/run/.containerenv");
|
|
117
|
+
return container
|
|
118
|
+
? "no audio device is available in this container"
|
|
119
|
+
: "no audio device is available in this environment (no /dev/snd and no PulseAudio/PipeWire)";
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
return null;
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* Linux: the native recorder goes through ALSA, which prints its errors
|
|
127
|
+
* straight to the terminal (fd 2) when there is no sound card — so it is
|
|
128
|
+
* only tried when /proc/asound/cards lists one.
|
|
129
|
+
*/
|
|
130
|
+
export function nativeRecorderAllowed(probe: EnvProbe): boolean {
|
|
131
|
+
if (probe.platform !== "linux") return true;
|
|
132
|
+
const cards = probe.readText("/proc/asound/cards") || "";
|
|
133
|
+
return /^\s*\d+\s*\[/m.test(cards);
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/** One exact command that gets a recorder working on this OS. */
|
|
137
|
+
export function installHint(probe: EnvProbe): string {
|
|
138
|
+
if (probe.platform === "darwin") return "brew install sox";
|
|
139
|
+
if (probe.platform === "win32") return "npm install -g omnirush (reinstalls the bundled recorder), or: winget install ffmpeg";
|
|
140
|
+
if (wslVersion(probe) === 2) return "sudo apt install sox libsox-fmt-pulse";
|
|
141
|
+
const osRelease = probe.readText("/etc/os-release") || "";
|
|
142
|
+
const field = (name: string) => (new RegExp(`^${name}=["']?([^"'\\n]*)`, "m").exec(osRelease)?.[1] ?? "").toLowerCase();
|
|
143
|
+
const ids = `${field("ID")} ${field("ID_LIKE")}`;
|
|
144
|
+
if (/\b(debian|ubuntu)\b/.test(ids)) return "sudo apt install pulseaudio-utils (or: sudo apt install alsa-utils)";
|
|
145
|
+
if (/\b(fedora|rhel|centos)\b/.test(ids)) return "sudo dnf install pipewire-utils";
|
|
146
|
+
if (/\barch\b/.test(ids)) return "sudo pacman -S pipewire (or: sudo pacman -S alsa-utils)";
|
|
147
|
+
if (/\b(suse|opensuse)\b/.test(ids)) return "sudo zypper install pipewire-tools";
|
|
148
|
+
if (/\balpine\b/.test(ids)) return "sudo apk add alsa-utils";
|
|
149
|
+
return "install PipeWire (pw-record), PulseAudio (parecord) or ALSA (arecord) command-line tools, or SoX";
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
// ---------------------------------------------------------------------------
|
|
153
|
+
// Candidates
|
|
154
|
+
|
|
155
|
+
const PCM_OUT = ["-ac", "1", "-ar", String(SAMPLE_RATE), "-f", "s16le", "-"];
|
|
156
|
+
|
|
157
|
+
/** The recorder chain for a platform, in order. */
|
|
158
|
+
export function recorderCandidates(platform: NodeJS.Platform, withNative = true): Candidate[] {
|
|
159
|
+
const out: Candidate[] = [];
|
|
160
|
+
if (withNative) out.push({ name: "pvrecorder", kind: "native" });
|
|
161
|
+
const ffmpeg = (args: string[]): Candidate => ({ name: "ffmpeg", kind: "tool", command: "ffmpeg", args: ["-hide_banner", "-loglevel", "error", "-nostdin", ...args, ...PCM_OUT] });
|
|
162
|
+
const rec: Candidate = { name: "rec", kind: "tool", command: "rec", args: ["-q", "--buffer", "1024", "-t", "raw", "-r", String(SAMPLE_RATE), "-e", "signed", "-b", "16", "-c", "1", "-"] };
|
|
163
|
+
if (platform === "linux") {
|
|
164
|
+
out.push(
|
|
165
|
+
{ name: "pw-record", kind: "tool", command: "pw-record", args: ["--rate", String(SAMPLE_RATE), "--channels", "1", "--format", "s16", "-"] },
|
|
166
|
+
{ name: "parecord", kind: "tool", command: "parecord", args: ["--raw", `--rate=${SAMPLE_RATE}`, "--channels=1", "--format=s16le"] },
|
|
167
|
+
{ name: "arecord", kind: "tool", command: "arecord", args: ["-f", "S16_LE", "-r", String(SAMPLE_RATE), "-c", "1", "-t", "raw", "-q", "-"] },
|
|
168
|
+
rec,
|
|
169
|
+
ffmpeg(["-f", "pulse", "-i", "default"]),
|
|
170
|
+
{ ...ffmpeg(["-f", "alsa", "-i", "default"]), name: "ffmpeg-alsa" },
|
|
171
|
+
);
|
|
172
|
+
} else if (platform === "darwin") {
|
|
173
|
+
out.push(rec, ffmpeg(["-f", "avfoundation", "-i", ":0"]));
|
|
174
|
+
} else if (platform === "win32") {
|
|
175
|
+
out.push(ffmpeg(["-f", "dshow", "-i", "audio=default"]));
|
|
176
|
+
} else {
|
|
177
|
+
out.push(rec, ffmpeg(["-f", "pulse", "-i", "default"]));
|
|
178
|
+
}
|
|
179
|
+
return out;
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
// ---------------------------------------------------------------------------
|
|
183
|
+
// Sources
|
|
184
|
+
|
|
185
|
+
/** Raw s16le bytes → Int16Array, carrying an odd trailing byte to the next chunk. */
|
|
186
|
+
export function createPcmDecoder(): (chunk: Uint8Array) => Int16Array {
|
|
187
|
+
let carry: Uint8Array | null = null;
|
|
188
|
+
return (chunk) => {
|
|
189
|
+
let bytes = chunk;
|
|
190
|
+
if (carry) {
|
|
191
|
+
const merged = new Uint8Array(carry.length + chunk.length);
|
|
192
|
+
merged.set(carry);
|
|
193
|
+
merged.set(chunk, carry.length);
|
|
194
|
+
bytes = merged;
|
|
195
|
+
carry = null;
|
|
196
|
+
}
|
|
197
|
+
const even = bytes.length - (bytes.length % 2);
|
|
198
|
+
if (even < bytes.length) carry = bytes.slice(even);
|
|
199
|
+
const out = new Int16Array(even / 2);
|
|
200
|
+
const view = new DataView(bytes.buffer, bytes.byteOffset, even);
|
|
201
|
+
for (let i = 0; i < out.length; i++) out[i] = view.getInt16(i * 2, true);
|
|
202
|
+
return out;
|
|
203
|
+
};
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
export interface ToolSourceOptions {
|
|
207
|
+
spawnImpl?: typeof spawn;
|
|
208
|
+
/** How long to wait for the first audio before accepting a silent-but-alive tool. */
|
|
209
|
+
probeMs?: number;
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/** A command-line recorder piping raw PCM on stdout. */
|
|
213
|
+
export class ToolSource implements PcmSource {
|
|
214
|
+
readonly name: string;
|
|
215
|
+
private child: ChildProcess | null = null;
|
|
216
|
+
private stopped = false;
|
|
217
|
+
|
|
218
|
+
constructor(
|
|
219
|
+
private readonly candidate: Candidate,
|
|
220
|
+
private readonly o: ToolSourceOptions = {},
|
|
221
|
+
) {
|
|
222
|
+
this.name = candidate.name;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
start(onPcm: (pcm: Int16Array) => void, onEnd?: (error?: VoiceError) => void): Promise<void> {
|
|
226
|
+
const onError = (error: Error) => onEnd?.(new VoiceError("capture_failed", error.message));
|
|
227
|
+
const spawnImpl = this.o.spawnImpl ?? spawn;
|
|
228
|
+
const probeMs = this.o.probeMs ?? 400;
|
|
229
|
+
return new Promise((resolve, reject) => {
|
|
230
|
+
let settled = false;
|
|
231
|
+
let stderr = "";
|
|
232
|
+
const decode = createPcmDecoder();
|
|
233
|
+
const settle = (error?: Error) => {
|
|
234
|
+
if (settled) return;
|
|
235
|
+
settled = true;
|
|
236
|
+
clearTimeout(timer);
|
|
237
|
+
if (error) reject(error);
|
|
238
|
+
else resolve();
|
|
239
|
+
};
|
|
240
|
+
let child: ChildProcess;
|
|
241
|
+
try {
|
|
242
|
+
child = spawnImpl(this.candidate.command!, this.candidate.args ?? [], { stdio: ["ignore", "pipe", "pipe"], windowsHide: true });
|
|
243
|
+
} catch (error: any) {
|
|
244
|
+
settle(Object.assign(new Error(`${this.name}: ${error?.message ?? error}`), { code: error?.code }));
|
|
245
|
+
return;
|
|
246
|
+
}
|
|
247
|
+
this.child = child;
|
|
248
|
+
const timer = setTimeout(() => settle(), probeMs);
|
|
249
|
+
child.on("error", (error: any) => {
|
|
250
|
+
const wrapped = Object.assign(new Error(error?.code === "ENOENT" ? `${this.name} is not installed` : `${this.name}: ${error?.message ?? error}`), { code: error?.code });
|
|
251
|
+
if (!settled) settle(wrapped);
|
|
252
|
+
else if (!this.stopped) onError(wrapped);
|
|
253
|
+
});
|
|
254
|
+
child.stderr?.on("data", (chunk: Buffer) => {
|
|
255
|
+
if (stderr.length < 2048) stderr += chunk.toString("utf8");
|
|
256
|
+
});
|
|
257
|
+
child.stdout?.on("data", (chunk: Buffer) => {
|
|
258
|
+
if (this.stopped) return;
|
|
259
|
+
const pcm = decode(new Uint8Array(chunk.buffer, chunk.byteOffset, chunk.byteLength));
|
|
260
|
+
settle();
|
|
261
|
+
if (pcm.length) onPcm(pcm);
|
|
262
|
+
});
|
|
263
|
+
child.on("exit", (code, signal) => {
|
|
264
|
+
if (this.stopped) return;
|
|
265
|
+
const detail = stderr.trim().split("\n")[0] || (signal ? `signal ${signal}` : `exit code ${code}`);
|
|
266
|
+
const error = new Error(`${this.name} stopped (${detail})`);
|
|
267
|
+
if (!settled) settle(error);
|
|
268
|
+
else onError(error);
|
|
269
|
+
});
|
|
270
|
+
});
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
stop(): void {
|
|
274
|
+
this.stopped = true;
|
|
275
|
+
const child = this.child;
|
|
276
|
+
this.child = null;
|
|
277
|
+
if (child && child.exitCode === null) {
|
|
278
|
+
try {
|
|
279
|
+
child.kill("SIGTERM");
|
|
280
|
+
} catch {
|
|
281
|
+
/* already gone */
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
/**
|
|
288
|
+
* OMNIRUSH_VOICE_INPUT_FILE: the core's WavFileSource (real-time paced).
|
|
289
|
+
* The file is read here, once; it is never written.
|
|
290
|
+
*/
|
|
291
|
+
export class FileSource implements PcmSource {
|
|
292
|
+
readonly name = "file";
|
|
293
|
+
private inner: WavFileSource | null = null;
|
|
294
|
+
|
|
295
|
+
constructor(private readonly file: string) {}
|
|
296
|
+
|
|
297
|
+
async start(onPcm: (pcm: Int16Array) => void, onEnd?: (error?: VoiceError) => void): Promise<void> {
|
|
298
|
+
let bytes: Uint8Array;
|
|
299
|
+
try {
|
|
300
|
+
bytes = new Uint8Array(fs.readFileSync(this.file));
|
|
301
|
+
} catch (error: any) {
|
|
302
|
+
throw new VoiceError("capture_failed", `OMNIRUSH_VOICE_INPUT_FILE: ${error?.message ?? error}`);
|
|
303
|
+
}
|
|
304
|
+
this.inner = new WavFileSource(bytes);
|
|
305
|
+
await this.inner.start(onPcm, onEnd);
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
stop(): void {
|
|
309
|
+
this.inner?.stop();
|
|
310
|
+
}
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
/** The package root the launcher hands over (optional native deps live there). */
|
|
314
|
+
export function packageRoot(env: NodeJS.ProcessEnv = process.env): string | null {
|
|
315
|
+
const root = (env.OMNIRUSH_PACKAGE_ROOT || "").trim();
|
|
316
|
+
return root || null;
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
/** Can @picovoice/pvrecorder-node be loaded on this machine? (Loads it on the main thread once.) */
|
|
320
|
+
export function nativeRecorderPath(env: NodeJS.ProcessEnv = process.env): string | null {
|
|
321
|
+
const root = packageRoot(env);
|
|
322
|
+
if (!root) return null;
|
|
323
|
+
try {
|
|
324
|
+
const require = createRequire(path.join(root, "package.json"));
|
|
325
|
+
return require.resolve("@picovoice/pvrecorder-node");
|
|
326
|
+
} catch {
|
|
327
|
+
return null;
|
|
328
|
+
}
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
/** pvrecorder in a worker thread (its read() blocks until a frame is ready). */
|
|
332
|
+
export class NativeSource implements PcmSource {
|
|
333
|
+
readonly name = "pvrecorder";
|
|
334
|
+
private worker: any = null;
|
|
335
|
+
private stopped = false;
|
|
336
|
+
|
|
337
|
+
constructor(
|
|
338
|
+
private readonly modulePath: string,
|
|
339
|
+
private readonly workerFile: string,
|
|
340
|
+
private readonly deviceIndex = -1,
|
|
341
|
+
) {}
|
|
342
|
+
|
|
343
|
+
async start(onPcm: (pcm: Int16Array) => void, onEnd?: (error?: VoiceError) => void): Promise<void> {
|
|
344
|
+
const onError = (error: Error) => onEnd?.(new VoiceError("capture_failed", error.message));
|
|
345
|
+
const { Worker } = await import("node:worker_threads");
|
|
346
|
+
const worker = new Worker(this.workerFile, { workerData: { modulePath: this.modulePath, deviceIndex: this.deviceIndex } });
|
|
347
|
+
this.worker = worker;
|
|
348
|
+
await new Promise<void>((resolve, reject) => {
|
|
349
|
+
const timeout = setTimeout(() => reject(new Error("pvrecorder did not start")), 3000);
|
|
350
|
+
worker.on("message", (message: any) => {
|
|
351
|
+
if (message?.type === "started") {
|
|
352
|
+
clearTimeout(timeout);
|
|
353
|
+
resolve();
|
|
354
|
+
} else if (message?.type === "error") {
|
|
355
|
+
clearTimeout(timeout);
|
|
356
|
+
if (!this.stopped) {
|
|
357
|
+
const error = new Error(`pvrecorder: ${message.message}`);
|
|
358
|
+
reject(error);
|
|
359
|
+
onError(error);
|
|
360
|
+
}
|
|
361
|
+
} else if (message?.type === "pcm" && !this.stopped) {
|
|
362
|
+
onPcm(new Int16Array(message.pcm));
|
|
363
|
+
}
|
|
364
|
+
});
|
|
365
|
+
worker.on("error", (error: Error) => {
|
|
366
|
+
clearTimeout(timeout);
|
|
367
|
+
reject(error);
|
|
368
|
+
if (!this.stopped) onError(error);
|
|
369
|
+
});
|
|
370
|
+
});
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
stop(): void {
|
|
374
|
+
this.stopped = true;
|
|
375
|
+
const worker = this.worker;
|
|
376
|
+
this.worker = null;
|
|
377
|
+
if (!worker) return;
|
|
378
|
+
try {
|
|
379
|
+
worker.postMessage("stop");
|
|
380
|
+
} catch {
|
|
381
|
+
/* gone */
|
|
382
|
+
}
|
|
383
|
+
setTimeout(() => worker.terminate().catch?.(() => undefined), 500).unref?.();
|
|
384
|
+
}
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
// ---------------------------------------------------------------------------
|
|
388
|
+
// Opening the first working recorder
|
|
389
|
+
|
|
390
|
+
export interface OpenOptions {
|
|
391
|
+
probe?: EnvProbe;
|
|
392
|
+
spawnImpl?: typeof spawn;
|
|
393
|
+
/** Override the native recorder (tests): a factory or null to skip it. */
|
|
394
|
+
native?: (() => PcmSource) | null;
|
|
395
|
+
candidates?: Candidate[];
|
|
396
|
+
probeMs?: number;
|
|
397
|
+
workerFile?: string;
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
export interface OpenedRecorder {
|
|
401
|
+
source: PcmSource;
|
|
402
|
+
tried: Array<{ name: string; reason: string }>;
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
function defaultWorkerFile(): string | null {
|
|
406
|
+
const root = packageRoot();
|
|
407
|
+
const candidates = [
|
|
408
|
+
root ? path.join(root, "assets", "extensions", "omnirush", "voice", "pvrecorder-worker.cjs") : null,
|
|
409
|
+
(() => {
|
|
410
|
+
try {
|
|
411
|
+
return path.join(path.dirname(new URL(import.meta.url).pathname), "pvrecorder-worker.cjs");
|
|
412
|
+
} catch {
|
|
413
|
+
return null;
|
|
414
|
+
}
|
|
415
|
+
})(),
|
|
416
|
+
];
|
|
417
|
+
for (const file of candidates) if (file && fs.existsSync(file)) return file;
|
|
418
|
+
return null;
|
|
419
|
+
}
|
|
420
|
+
|
|
421
|
+
/**
|
|
422
|
+
* Starts the first recorder that delivers audio. `onPcm`/`onError` are
|
|
423
|
+
* wired to whichever wins. Throws RecorderUnavailableError (with the
|
|
424
|
+
* install hint) when none does.
|
|
425
|
+
*/
|
|
426
|
+
export async function openRecorder(
|
|
427
|
+
onPcm: (pcm: Int16Array) => void,
|
|
428
|
+
onEnd: (error?: VoiceError) => void,
|
|
429
|
+
options: OpenOptions = {},
|
|
430
|
+
): Promise<OpenedRecorder> {
|
|
431
|
+
const probe = options.probe ?? realProbe();
|
|
432
|
+
const tried: Array<{ name: string; reason: string }> = [];
|
|
433
|
+
const file = voiceInputFileFromEnv(probe.env);
|
|
434
|
+
if (file) {
|
|
435
|
+
const source = new FileSource(file);
|
|
436
|
+
await source.start(onPcm, onEnd);
|
|
437
|
+
return { source, tried };
|
|
438
|
+
}
|
|
439
|
+
const reason = remoteReason(probe);
|
|
440
|
+
if (reason) throw new RecorderUnavailableError(`Voice can't use a microphone here: ${reason}.`, null, tried);
|
|
441
|
+
const candidates = options.candidates ?? recorderCandidates(probe.platform);
|
|
442
|
+
for (const candidate of candidates) {
|
|
443
|
+
let source: PcmSource | null = null;
|
|
444
|
+
if (candidate.kind === "native") {
|
|
445
|
+
if (options.native === null) continue;
|
|
446
|
+
if (options.native) source = options.native();
|
|
447
|
+
else if (!nativeRecorderAllowed(probe)) {
|
|
448
|
+
tried.push({ name: candidate.name, reason: "no ALSA sound card" });
|
|
449
|
+
continue;
|
|
450
|
+
}
|
|
451
|
+
else {
|
|
452
|
+
const modulePath = nativeRecorderPath(probe.env);
|
|
453
|
+
const workerFile = options.workerFile ?? defaultWorkerFile();
|
|
454
|
+
if (!modulePath || !workerFile) {
|
|
455
|
+
tried.push({ name: candidate.name, reason: "not installed" });
|
|
456
|
+
continue;
|
|
457
|
+
}
|
|
458
|
+
source = new NativeSource(modulePath, workerFile);
|
|
459
|
+
}
|
|
460
|
+
} else {
|
|
461
|
+
source = new ToolSource(candidate, { spawnImpl: options.spawnImpl, probeMs: options.probeMs });
|
|
462
|
+
}
|
|
463
|
+
try {
|
|
464
|
+
await source.start(onPcm, onEnd);
|
|
465
|
+
return { source, tried };
|
|
466
|
+
} catch (error: any) {
|
|
467
|
+
source.stop();
|
|
468
|
+
tried.push({ name: candidate.name, reason: String(error?.message ?? error) });
|
|
469
|
+
}
|
|
470
|
+
}
|
|
471
|
+
const installedButFailed = tried.filter((t) => !/not installed|no ALSA sound card/.test(t.reason));
|
|
472
|
+
if (installedButFailed.length) {
|
|
473
|
+
const detail = installedButFailed.map((t) => `${t.name}: ${t.reason}`).join("; ");
|
|
474
|
+
throw new RecorderUnavailableError(
|
|
475
|
+
`No microphone could be opened (${detail}). Check that an input device is connected and that this terminal may use the microphone.`,
|
|
476
|
+
null,
|
|
477
|
+
tried,
|
|
478
|
+
);
|
|
479
|
+
}
|
|
480
|
+
const hint = installHint(probe);
|
|
481
|
+
throw new RecorderUnavailableError(`No audio recorder found. Install one with: ${hint}`, hint, tried);
|
|
482
|
+
}
|
|
483
|
+
|
|
484
|
+
/** Records `ms` of audio and returns the highest RMS seen (the /voice mic check). */
|
|
485
|
+
export async function testCapture(ms: number, options: OpenOptions = {}): Promise<{ maxRms: number; recorder: string }> {
|
|
486
|
+
let maxRms = 0;
|
|
487
|
+
let failure: Error | null = null;
|
|
488
|
+
const opened = await openRecorder(
|
|
489
|
+
(pcm) => {
|
|
490
|
+
const value = rms(pcm);
|
|
491
|
+
if (value > maxRms) maxRms = value;
|
|
492
|
+
},
|
|
493
|
+
(error) => {
|
|
494
|
+
if (error) failure = error;
|
|
495
|
+
},
|
|
496
|
+
options,
|
|
497
|
+
);
|
|
498
|
+
await new Promise((resolve) => setTimeout(resolve, ms));
|
|
499
|
+
opened.source.stop();
|
|
500
|
+
if (failure) throw failure;
|
|
501
|
+
return { maxRms, recorder: opened.source.name };
|
|
502
|
+
}
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
Copied from omnirush-ai/omnirush-gui @ 1850b71 (branch feat/voice-mode, PR #20;
|
|
2
|
+
apps/app/src/app/lib/voice-core/, the desktop app's shared voice core), unchanged
|
|
3
|
+
except for the CLI seam marked "CLI:" in the code:
|
|
4
|
+
|
|
5
|
+
session.ts finalize() clears its 45 s deadline timer once the last upload
|
|
6
|
+
is back (otherwise every finished recording kept a timer alive
|
|
7
|
+
for 45 s; node --test waited on it)
|
|
8
|
+
everything else unchanged
|
|
9
|
+
|
|
10
|
+
The CLI's own pieces live one level up: capture.ts (recorder chain, AudioSource
|
|
11
|
+
implementations, remote/no-device checks), keys.ts + kitty.ts (push-to-talk
|
|
12
|
+
key handling), settings.ts, and ../voice.ts (the /voice command and prompt
|
|
13
|
+
editor integration).
|
|
14
|
+
|
|
15
|
+
Resync: copy the folder again from the desktop repo, re-apply the seam (or drop
|
|
16
|
+
it once upstream has it), then run `npm test` (test/voice-*.test.js).
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import { resampleTo16k } from "./resample";
|
|
2
|
+
import { VOICE_SAMPLE_RATE, VoiceError, type AudioSource } from "./types";
|
|
3
|
+
import { decodeWav } from "./wav";
|
|
4
|
+
|
|
5
|
+
/** The test hook: a WAV file played in place of the microphone. */
|
|
6
|
+
export const VOICE_INPUT_FILE_ENV = "OMNIRUSH_VOICE_INPUT_FILE";
|
|
7
|
+
|
|
8
|
+
export function voiceInputFileFromEnv(env: Record<string, string | undefined>): string | null {
|
|
9
|
+
const value = env[VOICE_INPUT_FILE_ENV]?.trim();
|
|
10
|
+
return value ? value : null;
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* An AudioSource that plays a WAV file (any PCM format, resampled to 16 kHz
|
|
15
|
+
* mono) in chunks, paced in real time unless `realtime` is false. The caller
|
|
16
|
+
* reads the file; this module never touches a file system.
|
|
17
|
+
*/
|
|
18
|
+
export class WavFileSource implements AudioSource {
|
|
19
|
+
private readonly pcm: Int16Array;
|
|
20
|
+
private timer: ReturnType<typeof setInterval> | null = null;
|
|
21
|
+
private stopped = false;
|
|
22
|
+
|
|
23
|
+
constructor(wavBytes: Uint8Array, private readonly options: { realtime?: boolean; chunkMs?: number } = {}) {
|
|
24
|
+
let decoded;
|
|
25
|
+
try {
|
|
26
|
+
decoded = decodeWav(wavBytes);
|
|
27
|
+
} catch (error) {
|
|
28
|
+
throw new VoiceError("capture_failed", `The voice input file could not be read: ${error instanceof Error ? error.message : String(error)}`);
|
|
29
|
+
}
|
|
30
|
+
this.pcm = resampleTo16k(decoded.samples, decoded.sampleRate);
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
get durationMs(): number {
|
|
34
|
+
return (this.pcm.length / VOICE_SAMPLE_RATE) * 1000;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
async start(onFrames: (frames: Int16Array) => void, onEnd?: (error?: VoiceError) => void): Promise<void> {
|
|
38
|
+
this.stopped = false;
|
|
39
|
+
const chunkMs = this.options.chunkMs ?? 100;
|
|
40
|
+
const chunk = Math.round((VOICE_SAMPLE_RATE * chunkMs) / 1000);
|
|
41
|
+
let offset = 0;
|
|
42
|
+
const step = () => {
|
|
43
|
+
if (this.stopped) return false;
|
|
44
|
+
if (offset >= this.pcm.length) {
|
|
45
|
+
this.stop();
|
|
46
|
+
onEnd?.();
|
|
47
|
+
return false;
|
|
48
|
+
}
|
|
49
|
+
onFrames(this.pcm.slice(offset, offset + chunk));
|
|
50
|
+
offset += chunk;
|
|
51
|
+
return true;
|
|
52
|
+
};
|
|
53
|
+
if (this.options.realtime === false) {
|
|
54
|
+
// Delivered after start() resolves, as a microphone would.
|
|
55
|
+
setTimeout(() => {
|
|
56
|
+
while (step()) {
|
|
57
|
+
// drain
|
|
58
|
+
}
|
|
59
|
+
}, 0);
|
|
60
|
+
return;
|
|
61
|
+
}
|
|
62
|
+
this.timer = setInterval(step, chunkMs);
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
stop(): void {
|
|
66
|
+
this.stopped = true;
|
|
67
|
+
if (this.timer) clearInterval(this.timer);
|
|
68
|
+
this.timer = null;
|
|
69
|
+
}
|
|
70
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* voice-core: dictation without a UI. Capture-agnostic (any AudioSource
|
|
3
|
+
* delivering 16 kHz mono Int16), dependency-free TypeScript that runs in a
|
|
4
|
+
* browser renderer and in Node 20+, so the CLI can vendor this folder as is.
|
|
5
|
+
*
|
|
6
|
+
* source → Segmenter (energy VAD, pause-cut) → encodeWav → Transcriber
|
|
7
|
+
* → ordered stitch → keyterm post-correction → text
|
|
8
|
+
*
|
|
9
|
+
* Audio is disposable: silence is dropped as it arrives, each speech
|
|
10
|
+
* segment's PCM and WAV are released once its text (or final failure) is
|
|
11
|
+
* known, and nothing is ever written to disk.
|
|
12
|
+
*/
|
|
13
|
+
export * from "./types";
|
|
14
|
+
export { encodeWav, decodeWav, type DecodedWav } from "./wav";
|
|
15
|
+
export { Downsampler, resampleTo16k } from "./resample";
|
|
16
|
+
export { LevelMeter, Segmenter, levelFromRms, rms, type Segment, type SegmenterOptions } from "./segmenter";
|
|
17
|
+
export { countWords, isCjk, isLikelyHallucination, padInsertion, stitch } from "./text";
|
|
18
|
+
export { DEV_KEYTERMS, applyKeyterms, buildKeyterms, splitIdentifier, type KeytermContext } from "./keyterms";
|
|
19
|
+
export { RemoteTranscriber, errorCode, multipartBody, voiceErrorFromResponse, type RemoteTranscriberOptions } from "./transcriber";
|
|
20
|
+
export { CircuitBreaker, VoiceSession, type VoiceSessionOptions } from "./session";
|
|
21
|
+
export { VOICE_INPUT_FILE_ENV, WavFileSource, voiceInputFileFromEnv } from "./file-source";
|