omnirush 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -25,8 +25,10 @@ export const CHILD_TIMEOUT_MS = 10 * 60_000;
25
25
  export const CHILD_KILL_GRACE_MS = 5_000;
26
26
  /** Default number of children running at once. */
27
27
  export const MAX_CONCURRENT_CHILDREN = 4;
28
+ /** Upper bound on children running at once (swarms). */
29
+ export const MAX_CONCURRENCY_CEILING = 12;
28
30
  /** Upper bound on tasks per spawn_agents call. */
29
- export const MAX_TASKS_PER_CALL = 8;
31
+ export const MAX_TASKS_PER_CALL = 12;
30
32
  /** Per-child output cap in the structured result (bytes, UTF-8). */
31
33
  export const CHILD_OUTPUT_CAP_BYTES = 50 * 1024;
32
34
 
@@ -64,7 +66,7 @@ export const ROLE_PRESETS: Record<AgentRole, RolePreset> = {
64
66
  description: "Web research via web_fetch: read documentation pages, articles and API responses.",
65
67
  systemPrompt: [
66
68
  "You are a researcher subagent with web access.",
67
- "Use the web_fetch tool to read documentation, articles and API responses (HTTPS URLs; the tool strips pages to readable text).",
69
+ "Use the web_search tool to find pages (returns titles, URLs and snippets) and web_fetch to read the promising ones (HTTPS; pages stripped to text).",
68
70
  "Cross-check claims across more than one page when it matters, and prefer primary sources (official docs, spec pages) over blog summaries.",
69
71
  "Report a compact synthesis with the source URLs you actually used — not a list of everything you fetched.",
70
72
  ].join(" "),
@@ -104,13 +106,22 @@ export function childInvocation(args: string[]): {
104
106
  /**
105
107
  * Build the child's argv for one task: JSON print mode with the role's
106
108
  * system prompt appended via a temp file path (the caller writes and
107
- * cleans up that file — see withRolePromptFile).
109
+ * cleans up that file — see withRolePromptFile). `model` selects a
110
+ * cross-model child (pi's --provider/--model flags); the provider is
111
+ * always our gateway extension.
108
112
  */
109
- export function buildChildArgs(task: string, promptFilePath: string | null): string[] {
113
+ export function buildChildArgs(
114
+ task: string,
115
+ promptFilePath: string | null,
116
+ model?: string,
117
+ ): string[] {
110
118
  const args: string[] = ["--mode", "json", "-p"];
111
119
  if (promptFilePath) {
112
120
  args.push("--append-system-prompt", promptFilePath);
113
121
  }
122
+ if (model && model.trim()) {
123
+ args.push("--provider", "omnirush", "--model", model.trim());
124
+ }
114
125
  args.push(`Task: ${task}`);
115
126
  return args;
116
127
  }
@@ -137,6 +148,10 @@ export async function withRolePromptFile(
137
148
  export interface ChildTask {
138
149
  role: AgentRole;
139
150
  task: string;
151
+ /** Cross-model children: gateway model id the child runs on
152
+ * (e.g. "muse-spark-1.3" for cheap swarm workers under an astra
153
+ * parent). Undefined = inherit the parent's model. */
154
+ model?: string;
140
155
  }
141
156
 
142
157
  export type ChildStatus = "completed" | "failed" | "timeout" | "cancelled";
@@ -144,6 +159,8 @@ export type ChildStatus = "completed" | "failed" | "timeout" | "cancelled";
144
159
  export interface ChildResult {
145
160
  role: AgentRole;
146
161
  task: string;
162
+ /** The gateway model the child ran on (cross-model children). */
163
+ model?: string;
147
164
  status: ChildStatus;
148
165
  /** Process exit code (null when killed by a signal or still unknown). */
149
166
  exitCode: number | null;
@@ -181,6 +198,7 @@ export function buildChildResult(
181
198
  return {
182
199
  role: task.role,
183
200
  task: task.task,
201
+ ...(task.model ? { model: task.model } : {}),
184
202
  status: input.status ?? (input.exitCode === 0 ? "completed" : "failed"),
185
203
  exitCode: input.exitCode,
186
204
  output: capped,
@@ -266,7 +284,7 @@ export async function runChildAgent(
266
284
  const spawnImpl = options.spawnImpl ?? nodeSpawn;
267
285
 
268
286
  return withRolePromptFile(task.role, async (promptFile) => {
269
- const args = buildChildArgs(task.task, promptFile);
287
+ const args = buildChildArgs(task.task, promptFile, task.model);
270
288
  const invocation = childInvocation(args);
271
289
 
272
290
  return await new Promise<ChildResult>((resolvePromise) => {
@@ -412,7 +430,7 @@ export function renderChildResults(results: ChildResult[]): string {
412
430
  const succeeded = results.filter((result) => result.status === "completed").length;
413
431
  const sections = results.map((result) => {
414
432
  const minutes = Math.round((result.durationMs / 60_000) * 10) / 10;
415
- const header = `### ${result.role} — ${result.status} (${minutes} min${result.outputTruncated ? ", output capped" : ""})`;
433
+ const header = `### ${result.role}${result.model ? ` [${result.model}]` : ""} — ${result.status} (${minutes} min${result.outputTruncated ? ", output capped" : ""})`;
416
434
  const meta: string[] = [`task: ${result.task}`];
417
435
  if (result.error) meta.push(`error: ${result.error}`);
418
436
  return `${header}\n${meta.join("\n")}\n\n${result.output || "(no output)"}`;
@@ -28,6 +28,7 @@ import {
28
28
  CHILD_TIMEOUT_MS,
29
29
  mapWithConcurrency,
30
30
  MAX_CONCURRENT_CHILDREN,
31
+ MAX_CONCURRENCY_CEILING,
31
32
  MAX_TASKS_PER_CALL,
32
33
  renderChildResults,
33
34
  runChildAgent,
@@ -39,9 +40,12 @@ const SpawnAgentsParams = Type.Object({
39
40
  tasks: Type.Array(
40
41
  Type.Object({
41
42
  role: StringEnum(AGENT_ROLES, {
42
- description: "code-searcher: read-only codebase exploration; researcher-web: web research via web_fetch; general-worker: full-tool implementation work",
43
+ description: "code-searcher: read-only codebase exploration; researcher-web: web research via web_search/web_fetch; general-worker: full-tool implementation work",
43
44
  }),
44
45
  task: Type.String({ description: "Self-contained task description for the child agent (it sees ONLY this, not your conversation)" }),
46
+ model: Type.Optional(Type.String({
47
+ description: 'Cross-model child: gateway model id for THIS task (e.g. "muse-spark-1.3", "muse-spark-1.2-contributor" for cheap swarm workers, "gpt-6-astra", "gpt-5.6-sol"). Omit to inherit this session\'s model',
48
+ })),
45
49
  }),
46
50
  { description: "Tasks to delegate; they run in parallel", minItems: 1, maxItems: MAX_TASKS_PER_CALL },
47
51
  ),
@@ -50,6 +54,11 @@ const SpawnAgentsParams = Type.Object({
50
54
  minimum: 0.1,
51
55
  maximum: 60,
52
56
  })),
57
+ max_parallel: Type.Optional(Type.Number({
58
+ description: `Children running at once (default ${MAX_CONCURRENT_CHILDREN}, max ${MAX_CONCURRENCY_CEILING}). Swarm fan-outs on cheap models (muse) can raise this`,
59
+ minimum: 1,
60
+ maximum: MAX_CONCURRENCY_CEILING,
61
+ })),
53
62
  });
54
63
 
55
64
  interface AgentRegistryEntry {
@@ -70,8 +79,9 @@ export default function (pi: ExtensionAPI) {
70
79
  description: [
71
80
  "Delegate tasks to parallel subagents with isolated context windows.",
72
81
  "Each child is a headless agent run in this workspace with a role preset:",
73
- "code-searcher (read-only exploration), researcher-web (web research via web_fetch), general-worker (full tools).",
74
- `Children run in parallel (max ${MAX_CONCURRENT_CHILDREN} at once, ${MAX_TASKS_PER_CALL} tasks per call) and each gets its own timeout (default ${CHILD_TIMEOUT_MS / 60_000} min).`,
82
+ "code-searcher (read-only exploration), researcher-web (web research via web_search + web_fetch), general-worker (full tools).",
83
+ `Children run in parallel (default ${MAX_CONCURRENT_CHILDREN} at once, up to ${MAX_CONCURRENCY_CEILING}; ${MAX_TASKS_PER_CALL} tasks per call) and each gets its own timeout (default ${CHILD_TIMEOUT_MS / 60_000} min).`,
84
+ 'Each task can run on a DIFFERENT model via its "model" field — e.g. cheap muse workers for wide swarm sweeps under an astra parent.',
75
85
  "Every task description must be SELF-CONTAINED: the child cannot see this conversation.",
76
86
  "Use for parallelizable work: broad code surveys, independent research questions, independent implementation chunks.",
77
87
  ].join(" "),
@@ -88,7 +98,15 @@ export default function (pi: ExtensionAPI) {
88
98
  if (!raw || typeof raw.task !== "string" || !raw.task.trim()) {
89
99
  throw new Error("invalid tasks: every entry needs a non-empty task string");
90
100
  }
91
- tasks.push({ role: raw.role, task: raw.task.trim() });
101
+ const model = typeof raw.model === "string" ? raw.model.trim() : "";
102
+ if (raw.model !== undefined && !model) {
103
+ throw new Error("invalid tasks: model must be a non-empty gateway model id");
104
+ }
105
+ tasks.push({
106
+ role: raw.role,
107
+ task: raw.task.trim(),
108
+ ...(model ? { model } : {}),
109
+ });
92
110
  }
93
111
  if (tasks.length === 0) throw new Error("no tasks given");
94
112
  if (tasks.length > MAX_TASKS_PER_CALL) {
@@ -100,6 +118,12 @@ export default function (pi: ExtensionAPI) {
100
118
  const timeoutMinutes = typeof params.timeout_minutes === "number" && params.timeout_minutes > 0
101
119
  ? params.timeout_minutes
102
120
  : undefined;
121
+ const concurrency = Math.min(
122
+ MAX_CONCURRENCY_CEILING,
123
+ Math.max(1, typeof params.max_parallel === "number" && params.max_parallel > 0
124
+ ? Math.floor(params.max_parallel)
125
+ : MAX_CONCURRENT_CHILDREN),
126
+ );
103
127
 
104
128
  const entries = registry.get(parentSessionId) ?? [];
105
129
  registry.set(parentSessionId, entries);
@@ -117,7 +141,7 @@ export default function (pi: ExtensionAPI) {
117
141
 
118
142
  const results = await mapWithConcurrency<ChildTask, ChildResult>(
119
143
  tasks,
120
- MAX_CONCURRENT_CHILDREN,
144
+ concurrency,
121
145
  async (task, index) => {
122
146
  const result = await runChildAgent(task, {
123
147
  cwd,
@@ -1520,8 +1520,17 @@ export class WorkspaceCollector {
1520
1520
  }
1521
1521
  this.enqueue(state, async () => {
1522
1522
  if (state.budgetExhausted || state.finished) return;
1523
+ // The JOURNAL is what holds unshipped change records: a scan
1524
+ // captures edits and advances lastSignature in the same breath,
1525
+ // so by the time this debounced flush runs (it can queue behind a
1526
+ // long multi-part baseline) the workspace signature often equals
1527
+ // lastSignature again and a signature-only guard would skip
1528
+ // forever, stranding the journal until session end. Ship whenever
1529
+ // unacked journal records exist; the signature is only the
1530
+ // trigger for captures that have not been journalled yet.
1531
+ const unackedJournal = state.journalAppendedBytes > state.journalAckOffset;
1523
1532
  const signature = await workspaceSignature(state.root);
1524
- if (!signature || signature === state.lastSignature) return;
1533
+ if (!unackedJournal && (!signature || signature === state.lastSignature)) return;
1525
1534
  state.lastSignature = signature;
1526
1535
  await this.uploadWorkspace(state, "change");
1527
1536
  });
@@ -15,6 +15,7 @@ import sota from "./sota";
15
15
  import agents from "./agents";
16
16
  import plan from "./plan";
17
17
  import webfetch from "./webfetch";
18
+ import websearch from "./websearch";
18
19
 
19
20
  export default function (pi: ExtensionAPI) {
20
21
  commands(pi as any);
@@ -23,6 +24,7 @@ export default function (pi: ExtensionAPI) {
23
24
  collector(pi as any);
24
25
  mcp(pi as any);
25
26
  webfetch(pi as any);
27
+ websearch(pi as any);
26
28
  plan(pi as any);
27
29
  agents(pi as any);
28
30
  }
@@ -144,6 +144,9 @@ export interface FetchReadableOptions {
144
144
  maxBytes?: number;
145
145
  timeoutMs?: number;
146
146
  signal?: AbortSignal;
147
+ /** Override the User-Agent (some endpoints, e.g. DuckDuckGo's HTML
148
+ * results, serve an anomaly page to unknown crawlers). */
149
+ userAgent?: string;
147
150
  }
148
151
 
149
152
  /**
@@ -175,7 +178,7 @@ export async function fetchReadableText(
175
178
  // identify honestly (traces show the agent anyway).
176
179
  Accept: "text/html,application/json,text/*;q=0.9,*/*;q=0.1",
177
180
  "Accept-Language": "en-US,en;q=0.9",
178
- "User-Agent": "omnirush-webfetch/1.0 (+https://omnirush.ai)",
181
+ "User-Agent": options.userAgent ?? "omnirush-webfetch/1.0 (+https://omnirush.ai)",
179
182
  },
180
183
  });
181
184
  const contentType = String(response.headers.get("content-type") ?? "");
@@ -0,0 +1,144 @@
1
+ // web_search internals — pure, injectable, node:test covered.
2
+ //
3
+ // Keyless web search via DuckDuckGo's no-JS HTML endpoint
4
+ // (html.duckduckgo.com/html/?q=...). The fetch itself reuses
5
+ // webfetch-lib's bounded fetchReadableText (size cap + timeout +
6
+ // browser-ish headers); this module owns the result parsing: extract
7
+ // title/URL/snippet triples, unwrap DDG's /l/?uddg=<encoded> redirect
8
+ // links, strip tags, decode entities. No API key, no JS rendering.
9
+
10
+ import { decodeHtmlEntities, fetchReadableText } from "./webfetch-lib";
11
+
12
+ /** Hard cap on results returned per query. */
13
+ export const WEB_SEARCH_MAX_RESULTS = 10;
14
+
15
+ /** The no-JS HTML results endpoint. */
16
+ export const WEB_SEARCH_ENDPOINT = "https://html.duckduckgo.com/html/";
17
+
18
+ export interface WebSearchResult {
19
+ title: string;
20
+ url: string;
21
+ snippet: string;
22
+ }
23
+
24
+ export type WebSearchOutcome =
25
+ | { ok: true; query: string; results: WebSearchResult[] }
26
+ | { ok: false; query: string; error: string };
27
+
28
+ /** Build the search URL for one query. */
29
+ export function buildSearchUrl(query: string): string {
30
+ return `${WEB_SEARCH_ENDPOINT}?q=${encodeURIComponent(String(query ?? "").trim())}`;
31
+ }
32
+
33
+ /** Decode entities, strip tags (AFTER decoding — snippets carry
34
+ * entity-encoded markup like &lt;b&gt;), collapse whitespace. */
35
+ function cleanText(html: string): string {
36
+ return decodeHtmlEntities(String(html ?? ""))
37
+ .replace(/<[^>]*>/g, " ")
38
+ .replace(/\s+/g, " ")
39
+ .trim();
40
+ }
41
+
42
+ /** Unwrap DDG redirect links (`.../l/?uddg=<encoded>`) to the target URL. */
43
+ export function unwrapDdgHref(href: string): string | null {
44
+ const raw = String(href ?? "").trim();
45
+ if (!raw) return null;
46
+ const absolute = raw.startsWith("//") ? `https:${raw}` : raw;
47
+ try {
48
+ const url = new URL(absolute);
49
+ const uddg = url.searchParams.get("uddg");
50
+ if (uddg) {
51
+ const target = decodeURIComponent(uddg);
52
+ // Ad/track links point at DDG itself — drop them.
53
+ if (/^https?:\/\/(www\.)?duckduckgo\.com\//i.test(target)) return null;
54
+ return target;
55
+ }
56
+ if (/duckduckgo\.com\/y\.js/i.test(absolute)) return null;
57
+ if (url.hostname.includes("duckduckgo.com")) return null;
58
+ return absolute;
59
+ } catch {
60
+ return null;
61
+ }
62
+ }
63
+
64
+ /**
65
+ * Parse the DDG HTML results page into title/url/snippet triples, in
66
+ * page order. Tolerates markup drift: a result needs at least a title
67
+ * anchor and a resolvable URL.
68
+ */
69
+ export function parseDdgResults(html: string): WebSearchResult[] {
70
+ const results: WebSearchResult[] = [];
71
+ const snippetByHref = new Map<string, string>();
72
+ const anchorRe =
73
+ /<a\b[^>]*class="(?:[^"]*\s)?result__(a|snippet)(?:\s[^"]*)?"[^>]*href="([^"]*)"[^>]*>([\s\S]*?)<\/a>/gi;
74
+ for (const match of String(html ?? "").matchAll(anchorRe)) {
75
+ const kind = match[1];
76
+ const href = unwrapDdgHref(match[2]);
77
+ const text = cleanText(match[3]);
78
+ if (kind === "snippet") {
79
+ // Snippets carry the result URL too; remember the text for the
80
+ // result anchor that follows with the same href.
81
+ if (href) snippetByHref.set(href, text);
82
+ continue;
83
+ }
84
+ if (!href || !text) continue;
85
+ const seen = results.find((r) => r.url === href);
86
+ if (seen) continue;
87
+ results.push({
88
+ title: text,
89
+ url: href,
90
+ snippet: snippetByHref.get(href) ?? "",
91
+ });
92
+ }
93
+ // Attach snippets that appeared after their anchor (page order varies).
94
+ if (snippetByHref.size > 0) {
95
+ for (const result of results) {
96
+ if (!result.snippet) result.snippet = snippetByHref.get(result.url) ?? "";
97
+ }
98
+ }
99
+ return results;
100
+ }
101
+
102
+ /** Render results as compact numbered text for the model. */
103
+ export function renderSearchResults(query: string, results: WebSearchResult[]): string {
104
+ if (results.length === 0) {
105
+ return `web_search: no results for "${query}"`;
106
+ }
107
+ const lines = results.map((result, index) => {
108
+ const parts = [`${index + 1}. ${result.title}`, ` ${result.url}`];
109
+ if (result.snippet) parts.push(` ${result.snippet}`);
110
+ return parts.join("\n");
111
+ });
112
+ return `web_search: ${results.length} result(s) for "${query}"\n\n${lines.join("\n\n")}`;
113
+ }
114
+
115
+ /** Run one search: fetch the endpoint and parse. Injectable fetchImpl for tests. */
116
+ export async function webSearch(
117
+ query: string,
118
+ options: {
119
+ count?: number;
120
+ signal?: AbortSignal;
121
+ fetchImpl?: typeof fetchReadableText;
122
+ } = {},
123
+ ): Promise<WebSearchOutcome> {
124
+ const trimmed = String(query ?? "").trim();
125
+ if (!trimmed) return { ok: false, query: trimmed, error: "empty query" };
126
+ const count = Math.min(
127
+ WEB_SEARCH_MAX_RESULTS,
128
+ Math.max(1, typeof options.count === "number" && options.count > 0 ? Math.floor(options.count) : 8),
129
+ );
130
+ const fetchImpl = options.fetchImpl ?? fetchReadableText;
131
+ const outcome = await fetchImpl(buildSearchUrl(trimmed), {
132
+ signal: options.signal,
133
+ // DDG serves an anomaly page (HTTP 202, zero results) to unknown
134
+ // crawlers; the search endpoint needs a browser UA.
135
+ userAgent: "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36",
136
+ });
137
+ if (!outcome.ok) {
138
+ return { ok: false, query: trimmed, error: outcome.error };
139
+ }
140
+ // Parse the RAW body — outcome.text is HTML-stripped and has no
141
+ // anchors left for the parser.
142
+ const results = parseDdgResults(outcome.raw).slice(0, count);
143
+ return { ok: true, query: trimmed, results };
144
+ }
@@ -0,0 +1,60 @@
1
+ // web_search — a built-in Omnirush tool: query -> ranked results
2
+ // (title + URL + snippet) from DuckDuckGo's no-JS HTML endpoint.
3
+ // No API key, no JS rendering; the bounded fetch lives in
4
+ // webfetch-lib, the parsing in websearch-lib (pure, unit-tested).
5
+ // Pair with web_fetch to read the promising pages.
6
+
7
+ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
8
+ import { Type } from "typebox";
9
+
10
+ import {
11
+ renderSearchResults,
12
+ WEB_SEARCH_MAX_RESULTS,
13
+ webSearch,
14
+ } from "./websearch-lib";
15
+
16
+ const WebSearchParams = Type.Object({
17
+ query: Type.String({ description: "The search query (web-search-engine syntax works: quoted phrases, site: filters, -exclusions)" }),
18
+ count: Type.Optional(Type.Number({
19
+ description: `Max results to return (default 8, max ${WEB_SEARCH_MAX_RESULTS})`,
20
+ minimum: 1,
21
+ maximum: WEB_SEARCH_MAX_RESULTS,
22
+ })),
23
+ });
24
+
25
+ export default function (pi: ExtensionAPI) {
26
+ pi.registerTool({
27
+ name: "web_search",
28
+ label: "Web Search",
29
+ description: [
30
+ "Search the web and return ranked results (title, URL, snippet).",
31
+ "No API key; results come from DuckDuckGo's HTML endpoint.",
32
+ "Use web_fetch on a result's URL to read the full page.",
33
+ `Responses are capped at ${WEB_SEARCH_MAX_RESULTS} results and share web_fetch's 30s timeout.`,
34
+ ].join(" "),
35
+ promptSnippet: "web_search: search the web (titles, URLs, snippets — pair with web_fetch)",
36
+ promptGuidelines: [
37
+ "Use web_search when you need to FIND pages (you do not know the URL yet); use web_fetch when you do.",
38
+ 'Engine syntax works in the query: quoted phrases, site:example.com filters and -excluded-terms.',
39
+ ],
40
+ parameters: WebSearchParams,
41
+
42
+ async execute(_toolCallId, params, signal) {
43
+ const outcome = await webSearch(params.query, {
44
+ count: typeof params.count === "number" ? params.count : undefined,
45
+ signal,
46
+ });
47
+ if (!outcome.ok) {
48
+ throw new Error(outcome.error);
49
+ }
50
+ return {
51
+ content: [{ type: "text", text: renderSearchResults(outcome.query, outcome.results) }],
52
+ details: {
53
+ query: outcome.query,
54
+ count: outcome.results.length,
55
+ results: outcome.results,
56
+ },
57
+ };
58
+ },
59
+ });
60
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "omnirush",
3
- "version": "0.6.0",
3
+ "version": "0.7.0",
4
4
  "description": "Omnirush \u2014 free daily tokens for the most powerful coding model on earth.",
5
5
  "license": "Apache-2.0",
6
6
  "type": "module",