omnirush 0.8.6 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/assets/CHANGELOG.md +85 -0
  2. package/assets/extensions/omnirush/agents-lib.ts +134 -13
  3. package/assets/extensions/omnirush/agents.ts +51 -107
  4. package/assets/extensions/omnirush/bgshell-lib.ts +400 -0
  5. package/assets/extensions/omnirush/bgshell.ts +392 -0
  6. package/assets/extensions/omnirush/collector.ts +72 -11
  7. package/assets/extensions/omnirush/commands.ts +2 -0
  8. package/assets/extensions/omnirush/deliveries.ts +145 -0
  9. package/assets/extensions/omnirush/guard/UPSTREAM +2 -0
  10. package/assets/extensions/omnirush/guard/git-command-policy.ts +885 -0
  11. package/assets/extensions/omnirush/guard-lib.ts +230 -0
  12. package/assets/extensions/omnirush/guard.ts +340 -0
  13. package/assets/extensions/omnirush/index.ts +12 -0
  14. package/assets/extensions/omnirush/pi-engine.ts +65 -1
  15. package/assets/extensions/omnirush/sota.ts +52 -4
  16. package/assets/extensions/omnirush/status-lib.ts +3 -0
  17. package/assets/extensions/omnirush/subagents-lib.ts +623 -0
  18. package/assets/extensions/omnirush/subagents.ts +305 -0
  19. package/assets/extensions/omnirush/swarm-lib.ts +142 -0
  20. package/assets/extensions/omnirush/swarm.ts +95 -0
  21. package/assets/extensions/omnirush/voice/capture.ts +502 -0
  22. package/assets/extensions/omnirush/voice/core/UPSTREAM +16 -0
  23. package/assets/extensions/omnirush/voice/core/file-source.ts +70 -0
  24. package/assets/extensions/omnirush/voice/core/index.ts +21 -0
  25. package/assets/extensions/omnirush/voice/core/keyterms.ts +117 -0
  26. package/assets/extensions/omnirush/voice/core/resample.ts +63 -0
  27. package/assets/extensions/omnirush/voice/core/segmenter.ts +231 -0
  28. package/assets/extensions/omnirush/voice/core/session.ts +403 -0
  29. package/assets/extensions/omnirush/voice/core/text.ts +81 -0
  30. package/assets/extensions/omnirush/voice/core/transcriber.ts +135 -0
  31. package/assets/extensions/omnirush/voice/core/types.ts +102 -0
  32. package/assets/extensions/omnirush/voice/core/wav.ts +95 -0
  33. package/assets/extensions/omnirush/voice/keys.ts +435 -0
  34. package/assets/extensions/omnirush/voice/kitty.ts +64 -0
  35. package/assets/extensions/omnirush/voice/pvrecorder-worker.cjs +43 -0
  36. package/assets/extensions/omnirush/voice/settings.ts +67 -0
  37. package/assets/extensions/omnirush/voice.ts +838 -0
  38. package/assets/extensions/omnirush/yolo-lib.ts +80 -0
  39. package/assets/extensions/omnirush/yolo.ts +85 -0
  40. package/package.json +7 -3
  41. package/scripts/brand-engine.js +526 -0
  42. package/scripts/build-all-packages.py +29 -1
  43. package/scripts/smoke-packages.py +32 -1
  44. package/src/bin.js +205 -33
  45. package/src/compat.js +272 -0
  46. package/src/lib.js +64 -0
  47. package/scripts/patch-pi-branding.js +0 -251
@@ -0,0 +1,117 @@
1
+ /** Coding vocabulary sent as a hint with every recording. */
2
+ export const DEV_KEYTERMS = [
3
+ "OmniRush",
4
+ "TypeScript",
5
+ "JavaScript",
6
+ "JSON",
7
+ "YAML",
8
+ "OAuth",
9
+ "GitHub",
10
+ "localhost",
11
+ "regex",
12
+ "gRPC",
13
+ "API",
14
+ "CLI",
15
+ "MCP",
16
+ "npm",
17
+ "pnpm",
18
+ "worktree",
19
+ "subagent",
20
+ "async",
21
+ "README",
22
+ ] as const;
23
+
24
+ const MAX_TERMS = 50;
25
+ const MAX_CHARS = 1024;
26
+
27
+ /** `feat/voice-mode_v2`, `useVoiceDictation`, `voice-core.ts` → their words. */
28
+ export function splitIdentifier(value: string): string[] {
29
+ return value
30
+ .replace(/\.[a-z0-9]{1,5}$/i, "")
31
+ .replace(/([a-z0-9])([A-Z])/g, "$1 $2")
32
+ .split(/[^\p{L}\p{N}]+/u)
33
+ .filter((word) => word.length >= 3 && !/^\d+$/.test(word));
34
+ }
35
+
36
+ export type KeytermContext = {
37
+ /** The project folder or repository name. */
38
+ repo?: string | null;
39
+ branch?: string | null;
40
+ /** Recently used file paths or names. */
41
+ files?: readonly string[];
42
+ extra?: readonly string[];
43
+ };
44
+
45
+ /**
46
+ * At most 50 terms and 1024 characters: the dev list, then the repository,
47
+ * branch and file names (whole and split into words), de-duplicated
48
+ * case-insensitively, first occurrence wins.
49
+ */
50
+ export function buildKeyterms(context: KeytermContext = {}): string[] {
51
+ const candidates: string[] = [...DEV_KEYTERMS, ...(context.extra ?? [])];
52
+ if (context.repo?.trim()) candidates.push(context.repo.trim(), ...splitIdentifier(context.repo));
53
+ if (context.branch?.trim() && !/^(main|master|HEAD)$/.test(context.branch.trim())) candidates.push(...splitIdentifier(context.branch));
54
+ for (const file of context.files ?? []) {
55
+ const name = file.split(/[\\/]/).pop() ?? "";
56
+ if (!name) continue;
57
+ candidates.push(name, ...splitIdentifier(name));
58
+ }
59
+ const seen = new Set<string>();
60
+ const out: string[] = [];
61
+ let chars = 0;
62
+ for (const raw of candidates) {
63
+ const term = raw.trim();
64
+ const key = term.toLowerCase();
65
+ if (!term || term.length > 64 || seen.has(key)) continue;
66
+ if (out.length >= MAX_TERMS || chars + term.length + 1 > MAX_CHARS) break;
67
+ seen.add(key);
68
+ out.push(term);
69
+ chars += term.length + 1;
70
+ }
71
+ return out;
72
+ }
73
+
74
+ /** Spoken forms of written dev terms. Conservative: only phrases no one says for anything else. */
75
+ const SPOKEN_FORMS: ReadonlyArray<[RegExp, string]> = [
76
+ [/\bo(?:h)?[ -]auth\b/gi, "OAuth"],
77
+ [/\bgit[ -]?hub\b/gi, "GitHub"],
78
+ [/\bget[ -]hub\b/gi, "GitHub"],
79
+ [/\btype[ -]script\b/gi, "TypeScript"],
80
+ [/\bjava[ -]script\b/gi, "JavaScript"],
81
+ [/\bnode[ .]?js\b/gi, "Node.js"],
82
+ [/\bnext[ .]?js\b/gi, "Next.js"],
83
+ [/\blocal[ -]host\b/gi, "localhost"],
84
+ [/\bwork[ -]tree(s?)\b/gi, "worktree$1"],
85
+ [/\bsub[ -]agent(s?)\b/gi, "subagent$1"],
86
+ [/\bread[ -]me\b/gi, "README"],
87
+ [/\bp[ -]?npm\b/gi, "pnpm"],
88
+ [/\bomni[ -]rush\b/gi, "OmniRush"],
89
+ ];
90
+
91
+ const EXTENSIONS = "ts|tsx|js|jsx|mjs|cjs|py|rs|go|md|json|yaml|yml|toml|css|html|sh|sql|txt";
92
+ const DOT_EXTENSION = new RegExp(`\\b([\\p{L}\\p{N}_-]+) dot (${EXTENSIONS})\\b`, "giu");
93
+
94
+ function escapeRegExp(value: string): string {
95
+ return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
96
+ }
97
+
98
+ /**
99
+ * Light post-correction of a transcript: spoken dev terms to their written
100
+ * form, "auth dot ts" to "auth.ts", and project terms with inner capitals
101
+ * ("omni rush" / "omnirush" → "OmniRush") restored from the keyterms.
102
+ */
103
+ export function applyKeyterms(text: string, keyterms: readonly string[] = []): string {
104
+ let out = text;
105
+ for (const [pattern, replacement] of SPOKEN_FORMS) out = out.replace(pattern, replacement);
106
+ out = out.replace(DOT_EXTENSION, (_match, name: string, extension: string) => `${name}.${extension.toLowerCase()}`);
107
+ for (const term of keyterms) {
108
+ // Only terms whose casing carries meaning, never plain words.
109
+ if (term.length < 4 || term === term.toLowerCase() || /[^\p{L}\p{N}]/u.test(term)) continue;
110
+ const words = splitIdentifier(term);
111
+ // The words must spell the whole term ("gRPC" splits to "RPC": skipped).
112
+ if (words.join("") !== term) continue;
113
+ const spoken = words.map(escapeRegExp).join("[ -]?");
114
+ out = out.replace(new RegExp(`\\b${spoken}\\b`, "gi"), term);
115
+ }
116
+ return out;
117
+ }
@@ -0,0 +1,63 @@
1
+ import { VOICE_SAMPLE_RATE } from "./types";
2
+
3
+ function toInt16(value: number): number {
4
+ const clamped = Math.max(-1, Math.min(1, value));
5
+ return clamped < 0 ? Math.round(clamped * 32768) : Math.round(clamped * 32767);
6
+ }
7
+
8
+ /**
9
+ * A streaming converter from float audio at `inputRate` to 16 kHz Int16.
10
+ * Downsampling averages every input sample that falls inside an output
11
+ * sample's window (a box filter: cheap, and enough against aliasing for
12
+ * speech); upsampling interpolates linearly. State carries across calls, so
13
+ * chunk boundaries leave no clicks.
14
+ */
15
+ export class Downsampler {
16
+ private readonly ratio: number;
17
+ /** Position of the next output sample, in input samples from the start of the pending chunk. */
18
+ private position = 0;
19
+ private carry: Float32Array = new Float32Array(0);
20
+
21
+ constructor(readonly inputRate: number, readonly outputRate = VOICE_SAMPLE_RATE) {
22
+ if (!(inputRate > 0) || !(outputRate > 0)) throw new Error("Sample rates must be positive");
23
+ this.ratio = inputRate / outputRate;
24
+ }
25
+
26
+ push(input: Float32Array): Int16Array {
27
+ if (this.inputRate === this.outputRate) {
28
+ const out = new Int16Array(input.length);
29
+ for (let index = 0; index < input.length; index += 1) out[index] = toInt16(input[index]!);
30
+ return out;
31
+ }
32
+ const data = new Float32Array(this.carry.length + input.length);
33
+ data.set(this.carry);
34
+ data.set(input, this.carry.length);
35
+ const out: number[] = [];
36
+ if (this.ratio >= 1) {
37
+ while (this.position + this.ratio <= data.length) {
38
+ const start = Math.floor(this.position);
39
+ const end = Math.floor(this.position + this.ratio);
40
+ let sum = 0;
41
+ for (let index = start; index < end; index += 1) sum += data[index]!;
42
+ out.push(toInt16(sum / Math.max(1, end - start)));
43
+ this.position += this.ratio;
44
+ }
45
+ } else {
46
+ while (this.position + 1 < data.length) {
47
+ const index = Math.floor(this.position);
48
+ const fraction = this.position - index;
49
+ out.push(toInt16(data[index]! * (1 - fraction) + data[index + 1]! * fraction));
50
+ this.position += this.ratio;
51
+ }
52
+ }
53
+ const consumed = Math.floor(this.position);
54
+ this.carry = data.slice(consumed);
55
+ this.position -= consumed;
56
+ return Int16Array.from(out);
57
+ }
58
+ }
59
+
60
+ /** Whole-buffer conversion (file input). */
61
+ export function resampleTo16k(samples: Float32Array, inputRate: number): Int16Array {
62
+ return new Downsampler(inputRate).push(samples);
63
+ }
@@ -0,0 +1,231 @@
1
+ import { VOICE_SAMPLE_RATE } from "./types";
2
+
3
+ /** RMS of Int16 samples, normalized to 0..1. */
4
+ export function rms(samples: Int16Array): number {
5
+ if (samples.length === 0) return 0;
6
+ let sum = 0;
7
+ for (let index = 0; index < samples.length; index += 1) {
8
+ const value = samples[index]! / 32768;
9
+ sum += value * value;
10
+ }
11
+ return Math.sqrt(sum / samples.length);
12
+ }
13
+
14
+ /** A display level 0..1 for a normalized RMS: square-root scaled so quiet speech still moves the bars. */
15
+ export function levelFromRms(value: number): number {
16
+ return Math.min(1, Math.sqrt((value * 32768) / 2000));
17
+ }
18
+
19
+ /** The rolling 16-bar level history the waveform draws. */
20
+ export class LevelMeter {
21
+ private readonly history: number[];
22
+
23
+ constructor(readonly bars = 16) {
24
+ this.history = new Array<number>(bars).fill(0);
25
+ }
26
+
27
+ push(value: number): void {
28
+ this.history.push(levelFromRms(value));
29
+ if (this.history.length > this.bars) this.history.shift();
30
+ }
31
+
32
+ levels(): number[] {
33
+ return [...this.history];
34
+ }
35
+ }
36
+
37
+ export type Segment = {
38
+ index: number;
39
+ /** 16 kHz mono PCM, including a short lead-in before the first speech frame. */
40
+ samples: Int16Array;
41
+ startMs: number;
42
+ endMs: number;
43
+ /** Frames classified as speech, in ms. */
44
+ speechMs: number;
45
+ /** Highest frame RMS in the segment, 0..1. */
46
+ peakRms: number;
47
+ /** The speech threshold when the segment closed, for the hallucination filter. */
48
+ threshold: number;
49
+ };
50
+
51
+ export type SegmenterOptions = {
52
+ sampleRate?: number;
53
+ /** Analysis frame length. */
54
+ frameMs?: number;
55
+ /** A pause at least this long ends a segment. */
56
+ pauseMs?: number;
57
+ /** A segment is cut here even mid-speech. */
58
+ maxSegmentMs?: number;
59
+ /** A segment shorter than this is held open across a normal pause (short segments transcribe badly). */
60
+ minSegmentMs?: number;
61
+ /** A pause this long ends even a short segment. */
62
+ longPauseMs?: number;
63
+ /** Less speech than this (a click, a cough) is dropped as noise. */
64
+ minSpeechMs?: number;
65
+ /** Audio kept before the first speech frame, so onsets are not clipped. */
66
+ prerollMs?: number;
67
+ /** Silence kept after the last speech frame. */
68
+ tailMs?: number;
69
+ /** Absolute speech floor (normalized RMS); the adaptive threshold never goes below it. */
70
+ minThreshold?: number;
71
+ /** Speech is this many times above the running noise floor. */
72
+ noiseMultiplier?: number;
73
+ };
74
+
75
+ const DEFAULTS: Required<SegmenterOptions> = {
76
+ sampleRate: VOICE_SAMPLE_RATE,
77
+ frameMs: 20,
78
+ pauseMs: 600,
79
+ maxSegmentMs: 12_000,
80
+ minSegmentMs: 1_000,
81
+ longPauseMs: 1_500,
82
+ minSpeechMs: 120,
83
+ prerollMs: 200,
84
+ tailMs: 250,
85
+ minThreshold: 0.01,
86
+ noiseMultiplier: 3,
87
+ };
88
+
89
+ /**
90
+ * Energy VAD with an adaptive noise floor that splits a PCM stream into
91
+ * speech segments at pauses. Silence outside a segment is discarded as it
92
+ * arrives: a silent recording produces no segment at all, so nothing is ever
93
+ * uploaded for it.
94
+ */
95
+ export class Segmenter {
96
+ readonly options: Required<SegmenterOptions>;
97
+ private readonly frameSize: number;
98
+ private pendingFrame: Int16Array = new Int16Array(0);
99
+ private noiseFloor = 0.004;
100
+ private frames: Int16Array[] = [];
101
+ private frameRms: number[] = [];
102
+ /** Frames before the first speech of the next segment, capped at the preroll. */
103
+ private preroll: Int16Array[] = [];
104
+ private inSegment = false;
105
+ private speechFrames = 0;
106
+ private silenceRun = 0;
107
+ private segmentStartFrame = 0;
108
+ private framesSeen = 0;
109
+ private nextIndex = 0;
110
+ /** Highest frame RMS of the whole stream. */
111
+ maxRms = 0;
112
+ /** Speech frames of the whole stream (including dropped noise bursts). */
113
+ totalSpeechFrames = 0;
114
+
115
+ constructor(options: SegmenterOptions = {}) {
116
+ this.options = { ...DEFAULTS, ...options };
117
+ this.frameSize = Math.round((this.options.sampleRate * this.options.frameMs) / 1000);
118
+ }
119
+
120
+ get threshold(): number {
121
+ return Math.max(this.options.minThreshold, this.noiseFloor * this.options.noiseMultiplier);
122
+ }
123
+
124
+ /** Feeds PCM; returns segments that closed. `onFrame` sees every analysis frame's RMS (the level meter). */
125
+ push(samples: Int16Array, onFrame?: (frameRms: number) => void): Segment[] {
126
+ const closed: Segment[] = [];
127
+ let data = samples;
128
+ if (this.pendingFrame.length > 0) {
129
+ data = new Int16Array(this.pendingFrame.length + samples.length);
130
+ data.set(this.pendingFrame);
131
+ data.set(samples, this.pendingFrame.length);
132
+ }
133
+ let offset = 0;
134
+ for (; offset + this.frameSize <= data.length; offset += this.frameSize) {
135
+ const frame = data.slice(offset, offset + this.frameSize);
136
+ const value = rms(frame);
137
+ onFrame?.(value);
138
+ const segment = this.frame(frame, value);
139
+ if (segment) closed.push(segment);
140
+ }
141
+ this.pendingFrame = data.slice(offset);
142
+ return closed;
143
+ }
144
+
145
+ /** Ends the stream: the open segment, if it holds enough speech. */
146
+ flush(): Segment | null {
147
+ this.pendingFrame = new Int16Array(0);
148
+ if (!this.inSegment) return null;
149
+ return this.close(true);
150
+ }
151
+
152
+ private frame(frame: Int16Array, value: number): Segment | null {
153
+ this.framesSeen += 1;
154
+ this.maxRms = Math.max(this.maxRms, value);
155
+ const speech = value > this.threshold;
156
+ if (speech) this.totalSpeechFrames += 1;
157
+ else this.noiseFloor = Math.min(0.05, Math.max(0.001, this.noiseFloor * 0.95 + value * 0.05));
158
+ const { frameMs } = this.options;
159
+
160
+ if (!this.inSegment) {
161
+ if (!speech) {
162
+ this.preroll.push(frame);
163
+ if (this.preroll.length * frameMs > this.options.prerollMs) this.preroll.shift();
164
+ return null;
165
+ }
166
+ this.inSegment = true;
167
+ this.frames = [...this.preroll, frame];
168
+ this.frameRms = [...this.preroll.map(rms), value];
169
+ this.segmentStartFrame = this.framesSeen - this.frames.length;
170
+ this.preroll = [];
171
+ this.speechFrames = 1;
172
+ this.silenceRun = 0;
173
+ return null;
174
+ }
175
+
176
+ this.frames.push(frame);
177
+ this.frameRms.push(value);
178
+ if (speech) {
179
+ this.speechFrames += 1;
180
+ this.silenceRun = 0;
181
+ } else {
182
+ this.silenceRun += 1;
183
+ }
184
+ const lengthMs = this.frames.length * frameMs;
185
+ const silenceMs = this.silenceRun * frameMs;
186
+ if (lengthMs >= this.options.maxSegmentMs) return this.close(false);
187
+ if (silenceMs >= this.options.longPauseMs) return this.close(true);
188
+ if (silenceMs >= this.options.pauseMs && lengthMs - silenceMs >= this.options.minSegmentMs) return this.close(true);
189
+ return null;
190
+ }
191
+
192
+ private close(trimSilence: boolean): Segment | null {
193
+ const { frameMs } = this.options;
194
+ let frames = this.frames;
195
+ let frameRms = this.frameRms;
196
+ if (trimSilence && this.silenceRun > 0) {
197
+ const keep = Math.ceil(this.options.tailMs / frameMs);
198
+ const drop = Math.max(0, this.silenceRun - keep);
199
+ frames = frames.slice(0, frames.length - drop);
200
+ frameRms = frameRms.slice(0, frameRms.length - drop);
201
+ // The trimmed silence may lead into the next segment's preroll.
202
+ this.preroll = this.frames.slice(this.frames.length - Math.min(drop, Math.ceil(this.options.prerollMs / frameMs)));
203
+ } else {
204
+ this.preroll = [];
205
+ }
206
+ const speechMs = this.speechFrames * frameMs;
207
+ const startFrame = this.segmentStartFrame;
208
+ this.inSegment = false;
209
+ this.frames = [];
210
+ this.frameRms = [];
211
+ this.speechFrames = 0;
212
+ this.silenceRun = 0;
213
+ if (speechMs < this.options.minSpeechMs) return null;
214
+ const length = frames.reduce((total, frame) => total + frame.length, 0);
215
+ const samples = new Int16Array(length);
216
+ let at = 0;
217
+ for (const frame of frames) {
218
+ samples.set(frame, at);
219
+ at += frame.length;
220
+ }
221
+ return {
222
+ index: this.nextIndex++,
223
+ samples,
224
+ startMs: startFrame * frameMs,
225
+ endMs: (startFrame + frames.length) * frameMs,
226
+ speechMs,
227
+ peakRms: Math.max(0, ...frameRms),
228
+ threshold: this.threshold,
229
+ };
230
+ }
231
+ }