omnirush 0.8.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -41,7 +41,7 @@ const SpawnAgentsParams = Type.Object({
41
41
  }),
42
42
  task: Type.String({ description: "Self-contained task description for the child agent (it sees ONLY this, not your conversation)" }),
43
43
  model: Type.Optional(Type.String({
44
- description: 'Cross-model child: gateway model id for THIS task (e.g. "muse-spark-1.3", "muse-spark-1.2-contributor" for cheap swarm workers, "gpt-6-astra", "gpt-5.6-sol"). Omit to inherit this session\'s model',
44
+ description: 'Cross-model child: gateway model id for THIS task (e.g. "muse-spark-1.3", "muse-spark-1.2-contributor" for cheap swarm workers, "meta-muse-spark", "muse-spark-1.1", "gpt-6-astra", "gpt-6-sol", "gpt-5.6-sol"). Omit to inherit this session\'s model',
45
45
  })),
46
46
  }),
47
47
  { description: "Tasks to delegate; they all run in parallel", minItems: 1 },
@@ -315,6 +315,10 @@ export default function (pi: any) {
315
315
  lastMessageId: string | undefined;
316
316
  toolStarts: Map<string, number>;
317
317
  seenFiles: Map<string, string>;
318
+ /** Prompts sent in this process with the model and effort they went with, until their user entry is found. */
319
+ pendingPrompts: Array<{ at: number; model: { providerID: string; modelID: string } | null; variant: string | null }>;
320
+ /** User entry id -> what its prompt was sent with (pi-engine promptSettings). */
321
+ promptSettings: Map<string, { model: { providerID: string; modelID: string } | null; variant: string | null }>;
318
322
  } | null = null;
319
323
  let settling: Promise<void> = Promise.resolve();
320
324
 
@@ -338,7 +342,18 @@ export default function (pi: any) {
338
342
 
339
343
  const messagesOf = (state: NonNullable<typeof session>): EngineMessage[] => {
340
344
  const entries = state.ctx?.sessionManager?.getBranch?.() ?? [];
341
- return engineMessagesFromEntries(entries, { sessionId: state.id, agent: ROOT_AGENT, cwd: state.root, toolStarts: state.toolStarts });
345
+ // Each prompt of this process is the first user entry written at or
346
+ // after it went out (pi stamps the message when it appends it).
347
+ for (const pending of state.pendingPrompts.splice(0)) {
348
+ const entry = entries.find((candidate: any) =>
349
+ candidate?.type === "message" && candidate.message?.role === "user" && typeof candidate.id === "string"
350
+ && !state.promptSettings.has(candidate.id) && Number(candidate.message.timestamp ?? Date.parse(candidate.timestamp)) >= pending.at - 1_000);
351
+ if (entry) state.promptSettings.set(entry.id, { model: pending.model, variant: pending.variant });
352
+ // Not written yet (read right as the prompt went out); a prompt that
353
+ // never wrote a user entry (handled by an extension) is let go.
354
+ else if (pending.at > Date.now() - 60_000) state.pendingPrompts.push(pending);
355
+ }
356
+ return engineMessagesFromEntries(entries, { sessionId: state.id, agent: ROOT_AGENT, cwd: state.root, toolStarts: state.toolStarts, promptSettings: state.promptSettings });
342
357
  };
343
358
 
344
359
  /**
@@ -351,7 +366,7 @@ export default function (pi: any) {
351
366
  if (!id) return null;
352
367
  if (session?.id !== id) {
353
368
  const root = ctx?.cwd || process.cwd();
354
- session = { id, root, ctx, started: false, running: false, lastMessageId: undefined, toolStarts: new Map(), seenFiles: new Map() };
369
+ session = { id, root, ctx, started: false, running: false, lastMessageId: undefined, toolStarts: new Map(), seenFiles: new Map(), pendingPrompts: [], promptSettings: new Map() };
355
370
  }
356
371
  session.ctx = ctx;
357
372
  // Idempotent: a session already started in this process keeps its segment.
@@ -496,6 +511,11 @@ export default function (pi: any) {
496
511
  if (!state) return;
497
512
  state.running = true;
498
513
  const model = ctx?.model;
514
+ state.pendingPrompts.push({
515
+ at: Date.now(),
516
+ model: model && typeof model.provider === "string" && typeof model.id === "string" ? { providerID: model.provider, modelID: model.id } : null,
517
+ variant: typeof ctx?.thinkingLevel === "string" && ctx.thinkingLevel !== "off" ? ctx.thinkingLevel : null,
518
+ });
499
519
  const images = Array.isArray(event?.images) ? event.images : [];
500
520
  // The prompt as the desktop records its engine request: text, attached
501
521
  // files (inline bytes replaced by a marker), model, variant, agent and
@@ -74,6 +74,12 @@ export type EngineConvertOptions = {
74
74
  initialModel?: { providerID: string; modelID: string } | null;
75
75
  /** The session's thinking level before its first thinking_level_change entry, if known. */
76
76
  initialThinkingLevel?: string | null;
77
+ /**
78
+ * What a prompt was really sent with, by its user entry id, as the live
79
+ * session knew it: pi records no model or thinking-level change for the
80
+ * `--model <id>:<effort>` flag of a resumed session.
81
+ */
82
+ promptSettings?: ReadonlyMap<string, { model?: { providerID: string; modelID: string } | null; variant?: string | null }>;
77
83
  };
78
84
 
79
85
  /** The shell-command marker the engine writes on a user message holding a command the user ran (public_trace.py `_USER_SHELL_TEXT`). */
@@ -265,8 +271,12 @@ export function engineMessagesFromEntries(entries: readonly PiEntry[], options:
265
271
  let lastUserId: string | null = null;
266
272
  const variant = () => (thinkingLevel && thinkingLevel !== "off" ? thinkingLevel : null);
267
273
 
268
- const userMessage = (id: string, created: number | null, parts: Array<Record<string, unknown>>, index: number, extraInfo: Record<string, unknown> = {}): void => {
269
- const messageModel = nextAssistantModel.get(index) ?? model ?? null;
274
+ // The effort of the prompt the current turn answers (a live setting when known).
275
+ let turnVariant: string | null = null;
276
+ const userMessage = (id: string, created: number | null, parts: Array<Record<string, unknown>>, index: number, extraInfo: Record<string, unknown> = {}): string | null => {
277
+ const live = typeof entries[index]?.id === "string" ? options.promptSettings?.get(entries[index]!.id as string) : undefined;
278
+ const messageModel = nextAssistantModel.get(index) ?? live?.model ?? model ?? null;
279
+ const messageVariant = live && live.variant !== undefined ? live.variant : variant();
270
280
  const info: Record<string, unknown> = {
271
281
  id,
272
282
  sessionID: sessionId,
@@ -274,7 +284,7 @@ export function engineMessagesFromEntries(entries: readonly PiEntry[], options:
274
284
  time: { created: created ?? 0 },
275
285
  agent,
276
286
  ...(messageModel ? { model: { providerID: messageModel.providerID, modelID: messageModel.modelID } } : {}),
277
- ...(variant() ? { variant: variant() } : {}),
287
+ ...(messageVariant ? { variant: messageVariant } : {}),
278
288
  ...extraInfo,
279
289
  };
280
290
  out.push({
@@ -282,6 +292,7 @@ export function engineMessagesFromEntries(entries: readonly PiEntry[], options:
282
292
  parts: parts.map((part, partIndex) => ({ id: partId(id, partIndex), sessionID: sessionId, messageID: id, ...part })),
283
293
  });
284
294
  lastUserId = id;
295
+ return messageVariant ?? null;
285
296
  };
286
297
 
287
298
  entries.forEach((entry, index) => {
@@ -339,7 +350,7 @@ export function engineMessagesFromEntries(entries: readonly PiEntry[], options:
339
350
  switch (message.role) {
340
351
  case "user": {
341
352
  const parts = contentParts(message.content);
342
- if (parts.length > 0) userMessage(id, created, parts, index);
353
+ if (parts.length > 0) turnVariant = userMessage(id, created, parts, index);
343
354
  return;
344
355
  }
345
356
  case "custom": {
@@ -463,7 +474,7 @@ export function engineMessagesFromEntries(entries: readonly PiEntry[], options:
463
474
  model = model ?? { providerID: message.provider, modelID: message.model };
464
475
  }
465
476
  const error = assistantError(message);
466
- const effort = typeof message.providerThinkingLevel === "string" && message.providerThinkingLevel ? message.providerThinkingLevel : variant();
477
+ const effort = typeof message.providerThinkingLevel === "string" && message.providerThinkingLevel ? message.providerThinkingLevel : turnVariant ?? variant();
467
478
  out.push({
468
479
  info: {
469
480
  id,
@@ -57,7 +57,7 @@ const STRICT_EXIT_CODE = 3;
57
57
  const CATALOG_MODELS = [
58
58
  {
59
59
  id: "gpt-6-astra",
60
- name: "GPT-6 Astra",
60
+ name: "GPT 6 Astra",
61
61
  api: "openai-responses",
62
62
  reasoning: true,
63
63
  input: ["text", "image"],
@@ -65,6 +65,16 @@ const CATALOG_MODELS = [
65
65
  contextWindow: 400000,
66
66
  compat: { supportsMaxOutputTokens: false },
67
67
  },
68
+ {
69
+ id: "gpt-6-sol",
70
+ name: "GPT 6 Sol",
71
+ api: "openai-responses",
72
+ reasoning: true,
73
+ input: ["text", "image"],
74
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
75
+ contextWindow: 400000,
76
+ compat: { supportsMaxOutputTokens: false },
77
+ },
68
78
  {
69
79
  id: "gpt-5.6-sol",
70
80
  name: "GPT-5.6 Sol",
@@ -8,7 +8,7 @@
8
8
  "models": [
9
9
  {
10
10
  "id": "gpt-6-astra",
11
- "name": "GPT-6 Astra",
11
+ "name": "GPT 6 Astra",
12
12
  "reasoning": true,
13
13
  "input": ["text", "image"],
14
14
  "cost": { "input": 10, "output": 50, "cacheRead": 2.5, "cacheWrite": 12.5 },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "omnirush",
3
- "version": "0.8.0",
3
+ "version": "0.8.1",
4
4
  "description": "Omnirush \u2014 free daily tokens for the most powerful coding model on earth.",
5
5
  "license": "Apache-2.0",
6
6
  "type": "module",
package/src/bin.js CHANGED
@@ -18,12 +18,14 @@
18
18
  // OMNIRUSH_GATEWAY_URL provider base URL override ({origin}/v1)
19
19
  // OMNIRUSH_DIR state dir override (default ~/.omnirush)
20
20
  // OMNIRUSH_MODEL model override for plain runs (default
21
- // gpt-6-astra; gpt-5.6-sol also available; user
21
+ // gpt-6-astra; gpt-6-sol, gpt-5.6-sol and the Muse
22
+ // models also available; user
22
23
  // --model still wins)
23
24
  //
24
25
  // Agent runs default to `--provider omnirush --model gpt-6-astra`; the
25
- // catalog also carries gpt-5.6-sol (`--model gpt-5.6-sol` or
26
- // OMNIRUSH_MODEL=gpt-5.6-sol). The sota extension watches every response,
26
+ // catalog also carries gpt-6-sol, gpt-5.6-sol, meta-muse-spark and
27
+ // muse-spark-1.1 (`--model gpt-6-sol` or OMNIRUSH_MODEL=gpt-6-sol), refreshed
28
+ // from the gateway's GET /v1/models at launch. The sota extension watches every response,
27
29
  // refreshes the device token single-flight on 401 (retry once), and
28
30
  // warns on stderr if the gateway serves a different model than
29
31
  // requested; `--strict-sota` makes that violation fatal (exit 3).
@@ -39,6 +41,7 @@ import { fileURLToPath } from "node:url";
39
41
 
40
42
  import {
41
43
  DEFAULT_ORIGIN,
44
+ catalogFromGateway,
42
45
  modelsConfig,
43
46
  resolveGatewayUrl,
44
47
  withModelDefaults,
@@ -147,7 +150,53 @@ function installDefaultThinkingLevel() {
147
150
  // Merge our provider into the user's models.json without clobbering
148
151
  // unrelated providers. Ours is authoritative: this file is how the
149
152
  // product ships its provider. Other providers untouched.
150
- function installModelsJson(gatewayUrl) {
153
+ /**
154
+ * The model list for this run, as the desktop app gets it: the gateway's
155
+ * catalog (GET <gateway>/models, bounded to a couple of seconds), cached in
156
+ * the state dir so an offline start still shows the last list; the shipped
157
+ * list (src/lib.js MODELS) when neither is available. New models then
158
+ * appear without a CLI release. OMNIRUSH_STATIC_MODELS=1 skips the fetch.
159
+ */
160
+ async function launchModels(gatewayUrl, accessToken) {
161
+ const cacheFile = path.join(OMNI_DIR, "model-catalog.json");
162
+ const base = String(gatewayUrl).replace(/\/+$/, "");
163
+ if (process.env.OMNIRUSH_STATIC_MODELS !== "1" && accessToken) {
164
+ try {
165
+ const response = await fetch(`${base}/models`, {
166
+ headers: { Authorization: `Bearer ${accessToken}`, Accept: "application/json" },
167
+ signal: AbortSignal.timeout(2500),
168
+ });
169
+ if (response.ok) {
170
+ const payload = await response.json();
171
+ const models = catalogFromGateway(payload);
172
+ if (models) {
173
+ try {
174
+ fs.writeFileSync(cacheFile, JSON.stringify({ gateway: base, fetched_at: new Date().toISOString(), payload }), { mode: 0o600 });
175
+ } catch {
176
+ /* the cache is a convenience */
177
+ }
178
+ return models;
179
+ }
180
+ } else {
181
+ await response.body?.cancel().catch(() => undefined);
182
+ }
183
+ } catch {
184
+ /* offline, slow or refused: the cached or shipped list */
185
+ }
186
+ }
187
+ try {
188
+ const cached = JSON.parse(fs.readFileSync(cacheFile, "utf8"));
189
+ if (cached && cached.gateway === base) {
190
+ const models = catalogFromGateway(cached.payload);
191
+ if (models) return models;
192
+ }
193
+ } catch {
194
+ /* no cache yet */
195
+ }
196
+ return undefined;
197
+ }
198
+
199
+ function installModelsJson(gatewayUrl, models) {
151
200
  let current = {};
152
201
  try {
153
202
  current = JSON.parse(fs.readFileSync(PI_MODELS_JSON, "utf8"));
@@ -157,7 +206,7 @@ function installModelsJson(gatewayUrl) {
157
206
  if (typeof current !== "object" || current === null) current = {};
158
207
  current.providers = {
159
208
  ...(current.providers || {}),
160
- ...modelsConfig({ gatewayUrl }).providers,
209
+ ...modelsConfig({ gatewayUrl, ...(models ? { models } : {}) }).providers,
161
210
  };
162
211
  fs.writeFileSync(PI_MODELS_JSON, JSON.stringify(current, null, 2) + "\n");
163
212
  }
@@ -505,7 +554,7 @@ async function cmdDoctor() {
505
554
  /* update check is advisory only */
506
555
  }
507
556
 
508
- installModelsJson(gateway);
557
+ installModelsJson(gateway, await launchModels(gateway, null));
509
558
  installExtensions();
510
559
  try {
511
560
  installManagedTools({ agentDir: PI_AGENT_DIR, runtimeRoot: path.resolve(__dirname, "..", ".runtime") });
@@ -732,7 +781,7 @@ function spawnAgent(cmd, args, opts) {
732
781
  async function cmdRun() {
733
782
  ensureDirs();
734
783
  const auth = requireAuth();
735
- installModelsJson(agentGatewayUrl(auth));
784
+ installModelsJson(agentGatewayUrl(auth), await launchModels(agentGatewayUrl(auth), auth.accessToken));
736
785
  installExtensions();
737
786
  const entry = piEntry();
738
787
  const { cmd, args } = await runtimeFor(entry);
@@ -774,7 +823,8 @@ Usage:
774
823
  omnirush --version show the omnirush version
775
824
 
776
825
  Options:
777
- --model <model[:effort]> model override (gpt-6-astra, gpt-5.6-sol;
826
+ --model <model[:effort]> model override (gpt-6-astra, gpt-6-sol,
827
+ gpt-5.6-sol, meta-muse-spark, muse-spark-1.1;
778
828
  effort: minimal..max)
779
829
  --thinking <level> reasoning effort override
780
830
  --strict-sota exit non-zero if the served model is not
@@ -792,7 +842,8 @@ Session commands (inside the agent):
792
842
 
793
843
  Environment:
794
844
  OMNIRUSH_ORIGIN manager origin override
795
- OMNIRUSH_MODEL default model override (e.g. gpt-5.6-sol)
845
+ OMNIRUSH_MODEL default model override (e.g. gpt-6-sol)
846
+ OMNIRUSH_STATIC_MODELS=1 use the shipped model list (no GET /v1/models)
796
847
  OMNIRUSH_DIR state dir override (default ~/.omnirush)
797
848
  OMNIRUSH_DEBUG=1 verbose diagnostics (raw gateway errors on failure)
798
849
 
@@ -816,7 +867,7 @@ else {
816
867
  // duplicated and pi admin subcommands pass through untouched.
817
868
  ensureDirs();
818
869
  const auth = requireAuth();
819
- installModelsJson(agentGatewayUrl(auth));
870
+ installModelsJson(agentGatewayUrl(auth), await launchModels(agentGatewayUrl(auth), auth.accessToken));
820
871
  installExtensions();
821
872
  const entry = piEntry();
822
873
  const { cmd: bin, args } = await runtimeFor(entry);
package/src/lib.js CHANGED
@@ -68,7 +68,7 @@ const VALUE_FLAGS = new Set([
68
68
  /**
69
69
  * Model defaults for plain `omnirush` runs. OMNIRUSH_MODEL overrides the
70
70
  * default model — any catalog id below, optionally with a thinking level
71
- * like "gpt-6-astra:high" or "gpt-5.6-sol:xhigh".
71
+ * like "gpt-6-astra:high" or "gpt-6-sol:xhigh".
72
72
  */
73
73
  export function resolveModelDefaults(env = process.env) {
74
74
  return {
@@ -77,8 +77,11 @@ export function resolveModelDefaults(env = process.env) {
77
77
  };
78
78
  }
79
79
 
80
- // The product's shipped catalog: gpt-6-astra (default), gpt-5.6-sol and
81
- // the Muse models. Live-verified 2026-09-22 against the production
80
+ // The product's shipped catalog, in the desktop app's order (the backend
81
+ // catalog, GET /v1/models): gpt-6-astra (default), gpt-6-sol, gpt-5.6-sol,
82
+ // then the Muse models. gpt-6-sol added 2026-09-26 (the gateway serves it
83
+ // as gpt-6-sol). At launch the list is refreshed from GET /v1/models
84
+ // (catalogFromGateway); this one is the fallback. Live-verified 2026-09-22 against the production
82
85
  // gateway: GET /v1/models lists both ids and POST /v1/responses with
83
86
  // model "gpt-5.6-sol" is served as gpt-5.6-sol (x-omnirush-model
84
87
  // response header + the model named in every SSE frame), so the client
@@ -87,7 +90,7 @@ export function resolveModelDefaults(env = process.env) {
87
90
  export const MODELS = [
88
91
  {
89
92
  id: "gpt-6-astra",
90
- name: "GPT-6 Astra",
93
+ name: "GPT 6 Astra",
91
94
  reasoning: true,
92
95
  input: ["text", "image"],
93
96
  cost: { input: 10, output: 50, cacheRead: 2.5, cacheWrite: 12.5 },
@@ -105,6 +108,25 @@ export const MODELS = [
105
108
  max: "max",
106
109
  },
107
110
  },
111
+ {
112
+ id: "gpt-6-sol",
113
+ name: "GPT 6 Sol",
114
+ reasoning: true,
115
+ input: ["text", "image"],
116
+ // No published list pricing: pi shows zeros (as for gpt-5.6-sol).
117
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
118
+ contextWindow: 400000,
119
+ compat: { supportsMaxOutputTokens: false },
120
+ // Same Codex route as astra: the gateway accepts every effort level.
121
+ thinkingLevelMap: {
122
+ minimal: "minimal",
123
+ low: "low",
124
+ medium: "medium",
125
+ high: "high",
126
+ xhigh: "xhigh",
127
+ max: "max",
128
+ },
129
+ },
108
130
  {
109
131
  id: "gpt-5.6-sol",
110
132
  name: "GPT-5.6 Sol",
@@ -205,6 +227,51 @@ export const MODELS = [
205
227
  },
206
228
  ];
207
229
 
230
+ /** Every effort the gateway accepts from a client (backend catalog ACCEPTED_EFFORTS, pi spellings). */
231
+ const GATEWAY_EFFORTS = ["minimal", "low", "medium", "high", "xhigh", "max"];
232
+
233
+ /**
234
+ * The model list from the gateway's catalog (GET /v1/models, the list the
235
+ * desktop app shows), in its order, for pi's models.json: a model the CLI
236
+ * ships keeps its entry (pricing, effort map) under the gateway's display
237
+ * name; a model it does not know gets a generic entry from the catalog's
238
+ * fields (context limit, image input, reasoning). Shipped models the
239
+ * gateway does not list (sub-agent-only Muse ids) follow, so
240
+ * `spawn_agents` can still name them. Null when the payload is not a
241
+ * usable catalog: the caller keeps the shipped list.
242
+ */
243
+ export function catalogFromGateway(payload, shipped = MODELS) {
244
+ const data = payload && Array.isArray(payload.data) ? payload.data : null;
245
+ if (!data) return null;
246
+ const known = new Map(shipped.map((model) => [model.id, model]));
247
+ const listed = [];
248
+ for (const entry of data) {
249
+ const id = entry && typeof entry.id === "string" ? entry.id.trim() : "";
250
+ if (!/^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$/.test(id) || listed.some((model) => model.id === id)) continue;
251
+ const name = typeof entry.display_name === "string" && entry.display_name.trim() ? entry.display_name.trim() : null;
252
+ const base = known.get(id);
253
+ if (base) {
254
+ listed.push({ ...base, ...(name ? { name } : {}) });
255
+ continue;
256
+ }
257
+ const caps = entry.capabilities && typeof entry.capabilities === "object" ? entry.capabilities : {};
258
+ const context = entry.limits && Number.isSafeInteger(entry.limits.context) && entry.limits.context > 0 ? entry.limits.context : 400000;
259
+ const reasoning = caps.reasoning !== false;
260
+ listed.push({
261
+ id,
262
+ name: name ?? id,
263
+ reasoning,
264
+ input: caps.image_input === false ? ["text"] : ["text", "image"],
265
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
266
+ contextWindow: context,
267
+ compat: { supportsMaxOutputTokens: false },
268
+ ...(reasoning ? { thinkingLevelMap: Object.fromEntries(GATEWAY_EFFORTS.map((level) => [level, level])) } : {}),
269
+ });
270
+ }
271
+ if (listed.length === 0) return null;
272
+ return [...listed, ...shipped.filter((model) => !listed.some((entry) => entry.id === model.id)).map((model) => ({ ...model }))];
273
+ }
274
+
208
275
  /**
209
276
  * Prepend `--provider`/`--model` defaults for agent runs, without
210
277
  * duplicating flags the user already specified. Consumption rules mirror
@@ -284,7 +351,7 @@ export function withModelDefaults(args, defaults) {
284
351
  * `compat.supportsMaxOutputTokens: false` and ships no maxTokens — pi would
285
352
  * otherwise clamp a default and break every request.
286
353
  */
287
- export function modelsConfig({ gatewayUrl } = {}) {
354
+ export function modelsConfig({ gatewayUrl, models = MODELS } = {}) {
288
355
  return {
289
356
  providers: {
290
357
  [DEFAULT_PROVIDER]: {
@@ -292,7 +359,7 @@ export function modelsConfig({ gatewayUrl } = {}) {
292
359
  baseUrl: gatewayUrl ?? DEFAULT_GATEWAY_URL,
293
360
  api: "openai-responses",
294
361
  apiKey: "$OMNIRUSH_TOKEN",
295
- models: MODELS.map((model) => ({ ...model })),
362
+ models: models.map((model) => ({ ...model })),
296
363
  },
297
364
  },
298
365
  };