omnirush 0.8.6 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/assets/CHANGELOG.md +85 -0
  2. package/assets/extensions/omnirush/agents-lib.ts +134 -13
  3. package/assets/extensions/omnirush/agents.ts +51 -107
  4. package/assets/extensions/omnirush/bgshell-lib.ts +400 -0
  5. package/assets/extensions/omnirush/bgshell.ts +392 -0
  6. package/assets/extensions/omnirush/collector.ts +72 -11
  7. package/assets/extensions/omnirush/commands.ts +2 -0
  8. package/assets/extensions/omnirush/deliveries.ts +145 -0
  9. package/assets/extensions/omnirush/guard/UPSTREAM +2 -0
  10. package/assets/extensions/omnirush/guard/git-command-policy.ts +885 -0
  11. package/assets/extensions/omnirush/guard-lib.ts +230 -0
  12. package/assets/extensions/omnirush/guard.ts +340 -0
  13. package/assets/extensions/omnirush/index.ts +12 -0
  14. package/assets/extensions/omnirush/pi-engine.ts +65 -1
  15. package/assets/extensions/omnirush/sota.ts +52 -4
  16. package/assets/extensions/omnirush/status-lib.ts +3 -0
  17. package/assets/extensions/omnirush/subagents-lib.ts +623 -0
  18. package/assets/extensions/omnirush/subagents.ts +305 -0
  19. package/assets/extensions/omnirush/swarm-lib.ts +142 -0
  20. package/assets/extensions/omnirush/swarm.ts +95 -0
  21. package/assets/extensions/omnirush/voice/capture.ts +502 -0
  22. package/assets/extensions/omnirush/voice/core/UPSTREAM +16 -0
  23. package/assets/extensions/omnirush/voice/core/file-source.ts +70 -0
  24. package/assets/extensions/omnirush/voice/core/index.ts +21 -0
  25. package/assets/extensions/omnirush/voice/core/keyterms.ts +117 -0
  26. package/assets/extensions/omnirush/voice/core/resample.ts +63 -0
  27. package/assets/extensions/omnirush/voice/core/segmenter.ts +231 -0
  28. package/assets/extensions/omnirush/voice/core/session.ts +403 -0
  29. package/assets/extensions/omnirush/voice/core/text.ts +81 -0
  30. package/assets/extensions/omnirush/voice/core/transcriber.ts +135 -0
  31. package/assets/extensions/omnirush/voice/core/types.ts +102 -0
  32. package/assets/extensions/omnirush/voice/core/wav.ts +95 -0
  33. package/assets/extensions/omnirush/voice/keys.ts +435 -0
  34. package/assets/extensions/omnirush/voice/kitty.ts +64 -0
  35. package/assets/extensions/omnirush/voice/pvrecorder-worker.cjs +43 -0
  36. package/assets/extensions/omnirush/voice/settings.ts +67 -0
  37. package/assets/extensions/omnirush/voice.ts +838 -0
  38. package/assets/extensions/omnirush/yolo-lib.ts +80 -0
  39. package/assets/extensions/omnirush/yolo.ts +85 -0
  40. package/package.json +7 -3
  41. package/scripts/brand-engine.js +526 -0
  42. package/scripts/build-all-packages.py +29 -1
  43. package/scripts/smoke-packages.py +32 -1
  44. package/src/bin.js +205 -33
  45. package/src/compat.js +272 -0
  46. package/src/lib.js +64 -0
  47. package/scripts/patch-pi-branding.js +0 -251
@@ -0,0 +1,623 @@
1
+ // subagents-lib — the sub-agent model and effort picker (as the desktop app's
2
+ // Settings > Preferences > Model "Sub-agents: model" / "Sub-agent effort"
3
+ // and its composer "Sub-agents" menu): which omnirush.ai model and effort
4
+ // the spawn_agents children run on, at every nesting layer, and the fallback
5
+ // to the main agent's model when that model cannot serve them.
6
+ //
7
+ // Who does what:
8
+ // - `/subagents` (subagents.ts) reads and writes the setting, kept in
9
+ // <omnirush dir>/settings.json under "subagents"; --subagent-model /
10
+ // --subagent-effort and OMNIRUSH_SUBAGENT_MODEL / _EFFORT override it for
11
+ // one run. Untouched ("same" for both) a child runs on the delegating
12
+ // agent's model and the main agent's effort.
13
+ // - spawn_agents (agents.ts) resolves every task against it
14
+ // (resolveSubagentModel) and starts the child with that --model and
15
+ // --thinking; a task's own `model` wins over the picked one.
16
+ // - A picked model that is not in the account's catalog (the list the
17
+ // launcher fetched from GET /v1/models), or that the gateway refused in
18
+ // the last five minutes, resolves to the main model instead (a
19
+ // "selection" fallback, noted on the sub-agent's result and trace title).
20
+ // - A child whose model differs from the main one gets the main model in
21
+ // its environment (OMNIRUSH_SUBAGENT_FALLBACK_MODEL / _EFFORT); its
22
+ // gateway guard (sota.ts) sends a request the gateway refuses
23
+ // (model_unavailable and friends at once, 429/503 after one more try)
24
+ // again on the main model, marks the model refused for five minutes
25
+ // (a file every process of this machine reads) and records the model
26
+ // that really answered on the message.
27
+ //
28
+ // Pure helpers plus small synchronous file helpers; node:test covers them.
29
+
30
+ import { mkdirSync, readFileSync, renameSync, writeFileSync } from "node:fs";
31
+ import { randomUUID } from "node:crypto";
32
+ import path from "node:path";
33
+
34
+ /** Every effort, lowest first (the gateway's spellings; pi says "off" for none). */
35
+ export const EFFORTS = ["none", "minimal", "low", "medium", "high", "xhigh", "max"] as const;
36
+ export type Effort = (typeof EFFORTS)[number];
37
+ const EFFORT_RANK: Record<string, number> = Object.fromEntries(EFFORTS.map((effort, index) => [effort, index]));
38
+
39
+ /** The Codex route's levels (Astra, Sol): the backend catalog's REASONING_LEVELS. */
40
+ export const CODEX_LEVELS: readonly string[] = ["low", "high", "xhigh", "max"];
41
+ /** The Muse relay's levels (MUSE_EFFORTS). */
42
+ export const MUSE_LEVELS: readonly string[] = ["minimal", "low", "medium", "high", "xhigh"];
43
+
44
+ export const PROVIDER_ID = "omnirush";
45
+ /** How long a model the gateway refused for sub-agents is skipped. */
46
+ export const REFUSAL_COOLDOWN_MS = 5 * 60_000;
47
+ export const SETTINGS_FILE = "settings.json";
48
+ export const REFUSALS_FILE = "subagent-model-refusals.json";
49
+ export const CATALOG_FILE = "model-catalog.json";
50
+
51
+ /** Child environment: the main model a sub-agent falls back to, its effort, and the main session's model/effort for nested layers. */
52
+ export const ENV_FALLBACK_MODEL = "OMNIRUSH_SUBAGENT_FALLBACK_MODEL";
53
+ export const ENV_FALLBACK_EFFORT = "OMNIRUSH_SUBAGENT_FALLBACK_EFFORT";
54
+ export const ENV_MAIN_MODEL = "OMNIRUSH_SUBAGENT_MAIN_MODEL";
55
+ export const ENV_MAIN_EFFORT = "OMNIRUSH_SUBAGENT_MAIN_EFFORT";
56
+ /** The setting for one run (and, handed down, for every nested layer). */
57
+ export const ENV_SETTING_MODEL = "OMNIRUSH_SUBAGENT_MODEL";
58
+ export const ENV_SETTING_EFFORT = "OMNIRUSH_SUBAGENT_EFFORT";
59
+ /** A child reports a gateway fallback on stderr with this prefix and a JSON object. */
60
+ export const FALLBACK_MARKER = "omnirush:subagent-model-fallback ";
61
+
62
+ const MODEL_ID = /^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$/;
63
+
64
+ export interface SubagentSetting {
65
+ /** An omnirush.ai catalog model id; null = the delegating agent's model. */
66
+ model: string | null;
67
+ /** An effort; null = the main agent's effort (the nearest level the model offers). */
68
+ effort: Effort | null;
69
+ }
70
+
71
+ export const SAME: SubagentSetting = Object.freeze({ model: null, effort: null }) as SubagentSetting;
72
+
73
+ export interface CatalogModel {
74
+ id: string;
75
+ name: string;
76
+ default: boolean;
77
+ /** The efforts the gateway takes for it (reasoning_levels). */
78
+ levels: string[];
79
+ }
80
+
81
+ function isRecord(value: unknown): value is Record<string, any> {
82
+ return typeof value === "object" && value !== null && !Array.isArray(value);
83
+ }
84
+
85
+ /** An effort from any spelling ("off" and "ultra" too); null when unknown. */
86
+ export function parseEffort(value: unknown): Effort | null {
87
+ if (typeof value !== "string") return null;
88
+ const raw = value.trim().toLowerCase();
89
+ const effort = raw === "ultra" ? "max" : raw === "off" ? "none" : raw;
90
+ return (EFFORTS as readonly string[]).includes(effort) ? (effort as Effort) : null;
91
+ }
92
+
93
+ /** pi's --thinking spelling of an effort. */
94
+ export function piThinking(effort: string): string {
95
+ return effort === "none" ? "off" : effort;
96
+ }
97
+
98
+ export function isModelId(value: unknown): value is string {
99
+ return typeof value === "string" && MODEL_ID.test(value.trim());
100
+ }
101
+
102
+ /**
103
+ * A task's `model[:effort]` (the spawn_agents spelling, e.g. "gpt-6-sol:high",
104
+ * "muse-spark-1.1:low"): the model id and the effort, when the suffix is one.
105
+ */
106
+ export function splitModelEffort(value: string): { model: string; effort: Effort | null } {
107
+ const text = value.trim();
108
+ const at = text.lastIndexOf(":");
109
+ if (at > 0) {
110
+ const effort = parseEffort(text.slice(at + 1));
111
+ if (effort) return { model: text.slice(0, at), effort };
112
+ }
113
+ return { model: text, effort: null };
114
+ }
115
+
116
+ /** A stored or typed setting; anything unusable reads as "same as main". */
117
+ export function sanitizeSetting(raw: unknown): SubagentSetting {
118
+ if (!isRecord(raw)) return { ...SAME };
119
+ const model = isModelId(raw.model) && !/^(same|main|default|inherit)$/i.test(raw.model.trim()) ? raw.model.trim() : null;
120
+ return { model, effort: parseEffort(raw.effort) };
121
+ }
122
+
123
+ // --- the catalog --------------------------------------------------------------
124
+
125
+ /** The account's models from a GET /v1/models payload, in its order; null when unusable. */
126
+ export function catalogFromPayload(payload: unknown): CatalogModel[] | null {
127
+ const data = isRecord(payload) && Array.isArray(payload.data) ? payload.data : null;
128
+ if (!data) return null;
129
+ const out: CatalogModel[] = [];
130
+ for (const entry of data) {
131
+ if (!isRecord(entry) || !isModelId(entry.id)) continue;
132
+ const id = entry.id.trim();
133
+ if (out.some((model) => model.id === id)) continue;
134
+ const levels = Array.isArray(entry.reasoning_levels)
135
+ ? EFFORTS.filter((effort) => entry.reasoning_levels.some((level: unknown) => parseEffort(level) === effort))
136
+ : [...staticLevels(id)];
137
+ out.push({
138
+ id,
139
+ name: typeof entry.display_name === "string" && entry.display_name.trim() ? entry.display_name.trim() : id,
140
+ default: entry.default === true,
141
+ levels,
142
+ });
143
+ }
144
+ return out.length > 0 ? out : null;
145
+ }
146
+
147
+ /** The catalog the launcher cached for this run (<omnirush dir>/model-catalog.json); null when there is none. */
148
+ export function loadCatalog(dir: string): CatalogModel[] | null {
149
+ try {
150
+ const cached = JSON.parse(readFileSync(path.join(dir, CATALOG_FILE), "utf8"));
151
+ return catalogFromPayload(isRecord(cached) ? cached.payload : null);
152
+ } catch {
153
+ return null;
154
+ }
155
+ }
156
+
157
+ /** The levels a model's route takes when the catalog does not say. */
158
+ export function staticLevels(id: string): readonly string[] {
159
+ return /muse/i.test(id) ? MUSE_LEVELS : CODEX_LEVELS;
160
+ }
161
+
162
+ export function levelsFor(catalog: readonly CatalogModel[] | null, id: string): string[] {
163
+ const entry = catalog?.find((model) => model.id === id);
164
+ return entry ? [...entry.levels] : [...staticLevels(id)];
165
+ }
166
+
167
+ export function displayName(catalog: readonly CatalogModel[] | null, id: string): string {
168
+ return catalog?.find((model) => model.id === id)?.name ?? id;
169
+ }
170
+
171
+ /**
172
+ * The effort a model offers that is closest to `effort` (ties go to the
173
+ * higher one); null when the model offers none or the effort is unknown.
174
+ */
175
+ export function nearestEffort(effort: string | null | undefined, levels: readonly string[]): string | null {
176
+ const wanted = parseEffort(effort);
177
+ if (!wanted || levels.length === 0) return null;
178
+ if (levels.includes(wanted)) return wanted;
179
+ const rank = EFFORT_RANK[wanted];
180
+ let best: string | null = null;
181
+ let bestDistance = Number.POSITIVE_INFINITY;
182
+ for (const level of levels) {
183
+ const levelRank = EFFORT_RANK[level];
184
+ if (levelRank === undefined) continue;
185
+ const distance = Math.abs(levelRank - rank);
186
+ if (distance < bestDistance || (distance === bestDistance && best !== null && levelRank > EFFORT_RANK[best])) {
187
+ best = level;
188
+ bestDistance = distance;
189
+ }
190
+ }
191
+ return best;
192
+ }
193
+
194
+ /** The efforts to offer for a picked model (all the catalog's when the model is "same"). */
195
+ export function effortOptions(catalog: readonly CatalogModel[] | null, model: string | null): string[] {
196
+ const offered = model
197
+ ? levelsFor(catalog, model)
198
+ : (catalog ?? []).flatMap((entry) => entry.levels);
199
+ const pool = offered.length > 0 ? offered : [...CODEX_LEVELS, ...MUSE_LEVELS];
200
+ return EFFORTS.filter((effort) => pool.includes(effort));
201
+ }
202
+
203
+ /**
204
+ * The setting after a change: an effort the newly picked model does not offer
205
+ * goes back to "same as main" rather than silently changing level (as the
206
+ * desktop's nextSubagentSetting).
207
+ */
208
+ export function nextSetting(
209
+ catalog: readonly CatalogModel[] | null,
210
+ current: SubagentSetting,
211
+ patch: Partial<SubagentSetting>,
212
+ ): SubagentSetting {
213
+ const model = patch.model !== undefined ? patch.model : current.model;
214
+ const effort = patch.effort !== undefined ? patch.effort : current.effort;
215
+ return { model, effort: effort && effortOptions(catalog, model).includes(effort) ? effort : null };
216
+ }
217
+
218
+ // --- the setting on disk, and per run ------------------------------------------
219
+
220
+ export function readSetting(dir: string): SubagentSetting {
221
+ try {
222
+ const raw = JSON.parse(readFileSync(path.join(dir, SETTINGS_FILE), "utf8"));
223
+ return sanitizeSetting(isRecord(raw) ? raw.subagents : null);
224
+ } catch {
225
+ return { ...SAME };
226
+ }
227
+ }
228
+
229
+ /** Keeps the file's other keys; atomic (temp file + rename). */
230
+ export function writeSetting(dir: string, setting: SubagentSetting): SubagentSetting {
231
+ const clean = sanitizeSetting(setting);
232
+ const file = path.join(dir, SETTINGS_FILE);
233
+ let current: Record<string, unknown> = {};
234
+ try {
235
+ const parsed = JSON.parse(readFileSync(file, "utf8"));
236
+ if (isRecord(parsed)) current = parsed;
237
+ } catch {
238
+ /* fresh file */
239
+ }
240
+ mkdirSync(dir, { recursive: true, mode: 0o700 });
241
+ const next = { ...current, subagents: { model: clean.model, effort: clean.effort } };
242
+ const tmp = `${file}.${randomUUID()}.tmp`;
243
+ writeFileSync(tmp, `${JSON.stringify(next, null, 2)}\n`, { mode: 0o600 });
244
+ renameSync(tmp, file);
245
+ return clean;
246
+ }
247
+
248
+ export type SettingSource = "flag" | "env" | "settings" | "default";
249
+
250
+ /**
251
+ * The setting for this run: a --subagent-model / --subagent-effort flag, else
252
+ * OMNIRUSH_SUBAGENT_MODEL / _EFFORT, else the saved one, field by field.
253
+ * "same" (or an empty value) in a flag or the environment means "same as
254
+ * main" and still overrides the saved field.
255
+ */
256
+ export function effectiveSetting(input: {
257
+ saved: SubagentSetting;
258
+ env?: NodeJS.ProcessEnv;
259
+ flags?: { model?: unknown; effort?: unknown };
260
+ }): { setting: SubagentSetting; source: { model: SettingSource; effort: SettingSource } } {
261
+ const env = input.env ?? {};
262
+ const pick = <T>(flag: unknown, envValue: unknown, saved: T | null, parse: (value: string) => T | null) => {
263
+ if (typeof flag === "string" && flag.trim()) return { value: parse(flag.trim()), source: "flag" as SettingSource };
264
+ if (typeof envValue === "string" && envValue.trim()) return { value: parse(envValue.trim()), source: "env" as SettingSource };
265
+ return { value: saved, source: (saved ? "settings" : "default") as SettingSource };
266
+ };
267
+ const model = pick(input.flags?.model, env[ENV_SETTING_MODEL], input.saved.model, (value) => sanitizeSetting({ model: value }).model);
268
+ const effort = pick(input.flags?.effort, env[ENV_SETTING_EFFORT], input.saved.effort, (value) => parseEffort(value));
269
+ return { setting: { model: model.value, effort: effort.value }, source: { model: model.source, effort: effort.source } };
270
+ }
271
+
272
+ // --- refusals (shared by every process on this machine) -------------------------
273
+
274
+ type Refusals = Record<string, { until: number; reason: string }>;
275
+
276
+ function readRefusals(dir: string): Refusals {
277
+ try {
278
+ const parsed = JSON.parse(readFileSync(path.join(dir, REFUSALS_FILE), "utf8"));
279
+ return isRecord(parsed) ? (parsed as Refusals) : {};
280
+ } catch {
281
+ return {};
282
+ }
283
+ }
284
+
285
+ export function markRefused(dir: string, model: string, reason: string, now: number = Date.now(), cooldownMs: number = REFUSAL_COOLDOWN_MS): void {
286
+ try {
287
+ const refusals = readRefusals(dir);
288
+ for (const [id, entry] of Object.entries(refusals)) {
289
+ if (!isRecord(entry) || typeof entry.until !== "number" || entry.until <= now) delete refusals[id];
290
+ }
291
+ refusals[model] = { until: now + cooldownMs, reason };
292
+ mkdirSync(dir, { recursive: true, mode: 0o700 });
293
+ const file = path.join(dir, REFUSALS_FILE);
294
+ const tmp = `${file}.${randomUUID()}.tmp`;
295
+ writeFileSync(tmp, JSON.stringify(refusals), { mode: 0o600 });
296
+ renameSync(tmp, file);
297
+ } catch {
298
+ /* the cooldown is an optimisation: the gateway fallback still applies */
299
+ }
300
+ }
301
+
302
+ export function isRefused(dir: string, model: string, now: number = Date.now()): { reason: string } | null {
303
+ const entry = readRefusals(dir)[model];
304
+ return isRecord(entry) && typeof entry.until === "number" && entry.until > now
305
+ ? { reason: typeof entry.reason === "string" ? entry.reason : "refused" }
306
+ : null;
307
+ }
308
+
309
+ // --- resolution ----------------------------------------------------------------
310
+
311
+ export type FallbackReason = "not_in_catalog" | "refused";
312
+
313
+ export interface SubagentFallback {
314
+ requested: string;
315
+ requestedName: string;
316
+ used: string;
317
+ usedName: string;
318
+ reason: FallbackReason | string;
319
+ }
320
+
321
+ export interface SubagentResolution {
322
+ /** The model the child runs on; null = leave the child to pi's own default (no omnirush model known). */
323
+ model: string | null;
324
+ /** The effort (gateway spelling); null = the model's default (no --thinking). */
325
+ effort: string | null;
326
+ /** The picked model could not be used: the main one runs instead. */
327
+ fallback?: SubagentFallback;
328
+ /** The main model (and effort) the child's gateway guard moves to when the gateway refuses `model`. */
329
+ gatewayFallback?: { model: string; effort: string | null };
330
+ }
331
+
332
+ export interface AgentModel {
333
+ /** pi provider id ("omnirush" for the gateway's models). */
334
+ provider: string | null;
335
+ model: string | null;
336
+ /** pi thinking level or gateway effort. */
337
+ effort: string | null;
338
+ }
339
+
340
+ /**
341
+ * What one sub-agent runs on.
342
+ *
343
+ * `delegating` is the agent calling spawn_agents (its model is what "same as
344
+ * main" keeps); `main` is the main session's model and effort (handed down
345
+ * to nested layers through the environment), the one a picked model falls
346
+ * back to. `explicit` is the task's own `model`, which wins over the picked
347
+ * one and is not checked against the catalog (the gateway decides).
348
+ */
349
+ export function resolveSubagentModel(input: {
350
+ setting: SubagentSetting;
351
+ catalog: readonly CatalogModel[] | null;
352
+ delegating: AgentModel;
353
+ main?: AgentModel | null;
354
+ explicit?: string | null;
355
+ refused?: (model: string) => boolean;
356
+ }): SubagentResolution {
357
+ const { setting, catalog, delegating } = input;
358
+ const main: AgentModel = input.main && input.main.model ? input.main : delegating;
359
+ const mainIsGateway = main.provider === PROVIDER_ID && Boolean(main.model);
360
+ const wantedEffort = setting.effort ?? parseEffort(main.effort) ?? null;
361
+ const effortOn = (model: string): string | null => nearestEffort(wantedEffort, levelsFor(catalog, model));
362
+
363
+ const withGatewayFallback = (resolution: SubagentResolution): SubagentResolution => {
364
+ if (
365
+ resolution.model &&
366
+ mainIsGateway &&
367
+ main.model !== resolution.model &&
368
+ (!catalog || catalog.some((entry) => entry.id === main.model))
369
+ ) {
370
+ resolution.gatewayFallback = { model: main.model!, effort: effortOn(main.model!) };
371
+ }
372
+ return resolution;
373
+ };
374
+
375
+ // The task's own model[:effort]: an effort it names wins over the picked
376
+ // and the parent's (mapped to the nearest level the model offers).
377
+ const explicit = input.explicit && isModelId(input.explicit) ? splitModelEffort(input.explicit) : null;
378
+ if (explicit && isModelId(explicit.model)) {
379
+ const effort = explicit.effort ? nearestEffort(explicit.effort, levelsFor(catalog, explicit.model)) ?? explicit.effort : effortOn(explicit.model);
380
+ return withGatewayFallback({ model: explicit.model, effort });
381
+ }
382
+
383
+ if (setting.model) {
384
+ const inCatalog = !catalog || catalog.some((entry) => entry.id === setting.model);
385
+ const reason: FallbackReason | null = !inCatalog ? "not_in_catalog" : input.refused?.(setting.model) ? "refused" : null;
386
+ if (reason && mainIsGateway && main.model !== setting.model) {
387
+ return {
388
+ model: main.model!,
389
+ effort: effortOn(main.model!),
390
+ fallback: {
391
+ requested: setting.model,
392
+ requestedName: displayName(catalog, setting.model),
393
+ used: main.model!,
394
+ usedName: displayName(catalog, main.model!),
395
+ reason,
396
+ },
397
+ };
398
+ }
399
+ return withGatewayFallback({ model: setting.model, effort: effortOn(setting.model) });
400
+ }
401
+
402
+ // "Same as main": the delegating agent's own model.
403
+ if (delegating.provider !== PROVIDER_ID || !delegating.model) {
404
+ return { model: null, effort: null };
405
+ }
406
+ return withGatewayFallback({ model: delegating.model, effort: effortOn(delegating.model) });
407
+ }
408
+
409
+ /** A readable note for a sub-agent that ran on the main model instead of the picked one. */
410
+ export function fallbackNote(fallback: { requestedName?: string; requested: string; usedName?: string; used: string; reason: string }): string {
411
+ const why = fallback.reason === "not_in_catalog"
412
+ ? "is not available to this account"
413
+ : fallback.reason === "refused"
414
+ ? "was refused by omnirush.ai recently"
415
+ : `was refused by omnirush.ai (${fallback.reason})`;
416
+ return `ran on ${fallback.usedName ?? fallback.used}: ${fallback.requestedName ?? fallback.requested} ${why}`;
417
+ }
418
+
419
+ // --- the gateway side (a child's guard) -----------------------------------------
420
+
421
+ /** Refusals of the picked model that move the request to the main model at once. */
422
+ const MODEL_REFUSED = new Set([
423
+ "model_unavailable",
424
+ "model_not_allowed",
425
+ "model_not_found",
426
+ "model_input_not_supported",
427
+ "reasoning_effort_not_allowed",
428
+ "unsupported_model_endpoint",
429
+ "muse_relay_not_configured",
430
+ "model_upstream_auth_failed",
431
+ "model_upstream_misconfigured",
432
+ ]);
433
+ /** Busy or down: the picked model is tried once more, then the request moves. */
434
+ const MODEL_BUSY = new Set([
435
+ "model_concurrency_limited",
436
+ "model_upstream_unavailable",
437
+ "internal_proxy_unavailable",
438
+ "provider_unavailable",
439
+ "provider_rate_limited",
440
+ "capacity_unavailable",
441
+ "circuit_open",
442
+ "maintenance",
443
+ "draining",
444
+ "auth_unavailable",
445
+ "token_capacity",
446
+ "gateway_overload",
447
+ "queue_timeout",
448
+ ]);
449
+ /** Account-wide refusals: another model would be refused the same way. */
450
+ const ACCOUNT_REFUSED = new Set([
451
+ "daily_grant_exhausted",
452
+ "grant_check_unavailable",
453
+ "account_inactive",
454
+ "consent_required",
455
+ "consent_version_outdated",
456
+ "omnirush_account_required",
457
+ "model_request_too_large",
458
+ ]);
459
+ const BUSY_STATUSES = new Set([429, 502, 503, 504]);
460
+
461
+ export type RefusalMove = "now" | "after_retry" | "never";
462
+
463
+ /** The gateway's error code in a refused response body (detail, error, error.code or code). */
464
+ export function refusalCode(bodyText: string): string | null {
465
+ let payload: unknown;
466
+ try {
467
+ payload = JSON.parse(bodyText);
468
+ } catch {
469
+ return null;
470
+ }
471
+ if (!isRecord(payload)) return null;
472
+ const nested = isRecord(payload.error) ? payload.error : null;
473
+ const text = (value: unknown) => (typeof value === "string" && value.trim() ? value.trim() : null);
474
+ const code = text(payload.detail) ?? text(payload.error) ?? text(nested?.code) ?? text(payload.code) ?? (isRecord(payload.detail) ? text(payload.detail.code) : null);
475
+ return code ? code.replace(/^upstream_/, "") : null;
476
+ }
477
+
478
+ /** Whether a refused sub-agent request moves to the main model, at once or after one more try. */
479
+ export function classifyRefusal(status: number, bodyText: string): { move: RefusalMove; reason: string } {
480
+ const code = refusalCode(bodyText);
481
+ const reason = code ?? `http_${status}`;
482
+ if (status === 401) return { move: "never", reason };
483
+ if (code && ACCOUNT_REFUSED.has(code)) return { move: "never", reason };
484
+ if (code && MODEL_REFUSED.has(code)) return { move: "now", reason };
485
+ if ((code && MODEL_BUSY.has(code)) || BUSY_STATUSES.has(status)) return { move: "after_retry", reason };
486
+ return { move: "never", reason };
487
+ }
488
+
489
+ /** The model a Responses request body names. */
490
+ export function requestModel(body: unknown): string | null {
491
+ if (typeof body !== "string") return null;
492
+ try {
493
+ const parsed = JSON.parse(body);
494
+ return isRecord(parsed) && typeof parsed.model === "string" ? parsed.model : null;
495
+ } catch {
496
+ return null;
497
+ }
498
+ }
499
+
500
+ /** The same request on another model: its effort replaced (or dropped for the model's default). */
501
+ export function withModel(body: string, model: string, effort: string | null): string | null {
502
+ let parsed: unknown;
503
+ try {
504
+ parsed = JSON.parse(body);
505
+ } catch {
506
+ return null;
507
+ }
508
+ if (!isRecord(parsed)) return null;
509
+ const { reasoning_effort: _legacy, ...rest } = parsed;
510
+ const reasoning = isRecord(rest.reasoning) ? { ...rest.reasoning } : null;
511
+ if (reasoning) delete reasoning.effort;
512
+ const nextReasoning = effort ? { ...(reasoning ?? {}), effort: effort === "none" ? "none" : effort } : reasoning;
513
+ return JSON.stringify({ ...rest, model, ...(nextReasoning ? { reasoning: nextReasoning } : {}) });
514
+ }
515
+
516
+ // --- a child's gateway fallback (wired into sota.ts's fetch) ---------------------
517
+
518
+ /** The request's route after a fallback: the model the SOTA guard now expects, and where it moved. */
519
+ export interface FallbackRoute {
520
+ expected: string;
521
+ movedTo: { model: string; effort: string | null; reason: string } | null;
522
+ }
523
+
524
+ /** Models this process moved away from: requested model -> until (epoch ms). */
525
+ const movedAway = new Map<string, number>();
526
+
527
+ function pause(ms: number, signal?: AbortSignal | null): Promise<void> {
528
+ return new Promise((resolve) => {
529
+ if (signal?.aborted) return resolve();
530
+ const done = () => {
531
+ clearTimeout(timer);
532
+ signal?.removeEventListener("abort", done);
533
+ resolve();
534
+ };
535
+ const timer = setTimeout(done, ms);
536
+ signal?.addEventListener("abort", done, { once: true });
537
+ });
538
+ }
539
+
540
+ function abortError(signal: AbortSignal): unknown {
541
+ return signal.reason ?? Object.assign(new Error("This operation was aborted"), { name: "AbortError" });
542
+ }
543
+
544
+ /**
545
+ * Wrap a sub-agent's fetch: when the gateway refuses the picked model before
546
+ * answering (model_unavailable and friends at once; 429/502/503/504 or a busy
547
+ * code after one more try), the same request goes to the main model instead,
548
+ * so the task continues rather than failing. The model is then skipped for
549
+ * REFUSAL_COOLDOWN_MS — by this process (its later steps go straight to the
550
+ * main model) and, through the refusals file, by new sub-agents. Account-wide
551
+ * refusals and every other error pass through untouched.
552
+ */
553
+ export function subagentFallbackFetch(
554
+ baseFetch: (input: any, init?: any) => Promise<Response>,
555
+ options: {
556
+ model: string;
557
+ effort: string | null;
558
+ dir: string;
559
+ route: FallbackRoute;
560
+ retryDelayMs?: number;
561
+ now?: () => number;
562
+ /** Told once per requested model and process (the parent reads it from stderr). */
563
+ report?: (event: { requested: string; used: string; effort: string | null; reason: string }) => void;
564
+ },
565
+ ): (input: any, init?: any) => Promise<Response> {
566
+ const now = options.now ?? Date.now;
567
+ const reported = new Set<string>();
568
+ return async (input: any, init?: any) => {
569
+ const body = init?.body;
570
+ const requested = requestModel(body);
571
+ if (typeof body !== "string" || !requested || requested === options.model) return baseFetch(input, init);
572
+ const signal: AbortSignal | undefined = init?.signal ?? undefined;
573
+
574
+ const move = async (reason: string): Promise<Response | null> => {
575
+ const moved = withModel(body, options.model, options.effort);
576
+ if (moved === null) return null;
577
+ options.route.expected = options.model;
578
+ options.route.movedTo = { model: options.model, effort: options.effort, reason };
579
+ if ((movedAway.get(requested) ?? 0) <= now()) {
580
+ movedAway.set(requested, now() + REFUSAL_COOLDOWN_MS);
581
+ markRefused(options.dir, requested, reason, now());
582
+ }
583
+ if (!reported.has(requested)) {
584
+ reported.add(requested);
585
+ try {
586
+ options.report?.({ requested, used: options.model, effort: options.effort, reason });
587
+ } catch {
588
+ /* reporting never breaks the request */
589
+ }
590
+ }
591
+ return baseFetch(input, { ...init, body: moved });
592
+ };
593
+
594
+ // Refused a moment ago in this process: straight to the main model.
595
+ if ((movedAway.get(requested) ?? 0) > now()) {
596
+ return (await move("refused")) ?? baseFetch(input, init);
597
+ }
598
+
599
+ let response = await baseFetch(input, init);
600
+ if (response.ok || response.status === 401) return response;
601
+ let refusal = classifyRefusal(response.status, await response.clone().text().catch(() => ""));
602
+ if (refusal.move === "never") return response;
603
+ if (refusal.move === "after_retry") {
604
+ await response.body?.cancel().catch(() => undefined);
605
+ await pause(options.retryDelayMs ?? 1_500, signal);
606
+ if (signal?.aborted) throw abortError(signal);
607
+ const again = await baseFetch(input, init);
608
+ if (again.ok || again.status === 401) return again;
609
+ refusal = classifyRefusal(again.status, await again.clone().text().catch(() => ""));
610
+ if (refusal.move === "never") return again;
611
+ response = again;
612
+ }
613
+ const moved = await move(refusal.reason);
614
+ if (!moved) return response;
615
+ await response.body?.cancel().catch(() => undefined);
616
+ return moved;
617
+ };
618
+ }
619
+
620
+ /** For tests: forget this process's cooldowns. */
621
+ export function resetFallbackMemory(): void {
622
+ movedAway.clear();
623
+ }