omnirush 0.6.1 → 0.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -23,10 +23,6 @@ import path from "node:path";
23
23
  export const CHILD_TIMEOUT_MS = 10 * 60_000;
24
24
  /** Grace between SIGTERM and SIGKILL on timeout. */
25
25
  export const CHILD_KILL_GRACE_MS = 5_000;
26
- /** Default number of children running at once. */
27
- export const MAX_CONCURRENT_CHILDREN = 4;
28
- /** Upper bound on tasks per spawn_agents call. */
29
- export const MAX_TASKS_PER_CALL = 8;
30
26
  /** Per-child output cap in the structured result (bytes, UTF-8). */
31
27
  export const CHILD_OUTPUT_CAP_BYTES = 50 * 1024;
32
28
 
@@ -64,7 +60,7 @@ export const ROLE_PRESETS: Record<AgentRole, RolePreset> = {
64
60
  description: "Web research via web_fetch: read documentation pages, articles and API responses.",
65
61
  systemPrompt: [
66
62
  "You are a researcher subagent with web access.",
67
- "Use the web_fetch tool to read documentation, articles and API responses (HTTPS URLs; the tool strips pages to readable text).",
63
+ "Use the web_search tool to find pages (returns titles, URLs and snippets) and web_fetch to read the promising ones (HTTPS; pages stripped to text).",
68
64
  "Cross-check claims across more than one page when it matters, and prefer primary sources (official docs, spec pages) over blog summaries.",
69
65
  "Report a compact synthesis with the source URLs you actually used — not a list of everything you fetched.",
70
66
  ].join(" "),
@@ -104,13 +100,22 @@ export function childInvocation(args: string[]): {
104
100
  /**
105
101
  * Build the child's argv for one task: JSON print mode with the role's
106
102
  * system prompt appended via a temp file path (the caller writes and
107
- * cleans up that file — see withRolePromptFile).
103
+ * cleans up that file — see withRolePromptFile). `model` selects a
104
+ * cross-model child (pi's --provider/--model flags); the provider is
105
+ * always our gateway extension.
108
106
  */
109
- export function buildChildArgs(task: string, promptFilePath: string | null): string[] {
107
+ export function buildChildArgs(
108
+ task: string,
109
+ promptFilePath: string | null,
110
+ model?: string,
111
+ ): string[] {
110
112
  const args: string[] = ["--mode", "json", "-p"];
111
113
  if (promptFilePath) {
112
114
  args.push("--append-system-prompt", promptFilePath);
113
115
  }
116
+ if (model && model.trim()) {
117
+ args.push("--provider", "omnirush", "--model", model.trim());
118
+ }
114
119
  args.push(`Task: ${task}`);
115
120
  return args;
116
121
  }
@@ -137,6 +142,10 @@ export async function withRolePromptFile(
137
142
  export interface ChildTask {
138
143
  role: AgentRole;
139
144
  task: string;
145
+ /** Cross-model children: gateway model id the child runs on
146
+ * (e.g. "muse-spark-1.3" for cheap swarm workers under an astra
147
+ * parent). Undefined = inherit the parent's model. */
148
+ model?: string;
140
149
  }
141
150
 
142
151
  export type ChildStatus = "completed" | "failed" | "timeout" | "cancelled";
@@ -144,6 +153,8 @@ export type ChildStatus = "completed" | "failed" | "timeout" | "cancelled";
144
153
  export interface ChildResult {
145
154
  role: AgentRole;
146
155
  task: string;
156
+ /** The gateway model the child ran on (cross-model children). */
157
+ model?: string;
147
158
  status: ChildStatus;
148
159
  /** Process exit code (null when killed by a signal or still unknown). */
149
160
  exitCode: number | null;
@@ -181,6 +192,7 @@ export function buildChildResult(
181
192
  return {
182
193
  role: task.role,
183
194
  task: task.task,
195
+ ...(task.model ? { model: task.model } : {}),
184
196
  status: input.status ?? (input.exitCode === 0 ? "completed" : "failed"),
185
197
  exitCode: input.exitCode,
186
198
  output: capped,
@@ -266,7 +278,7 @@ export async function runChildAgent(
266
278
  const spawnImpl = options.spawnImpl ?? nodeSpawn;
267
279
 
268
280
  return withRolePromptFile(task.role, async (promptFile) => {
269
- const args = buildChildArgs(task.task, promptFile);
281
+ const args = buildChildArgs(task.task, promptFile, task.model);
270
282
  const invocation = childInvocation(args);
271
283
 
272
284
  return await new Promise<ChildResult>((resolvePromise) => {
@@ -412,7 +424,7 @@ export function renderChildResults(results: ChildResult[]): string {
412
424
  const succeeded = results.filter((result) => result.status === "completed").length;
413
425
  const sections = results.map((result) => {
414
426
  const minutes = Math.round((result.durationMs / 60_000) * 10) / 10;
415
- const header = `### ${result.role} — ${result.status} (${minutes} min${result.outputTruncated ? ", output capped" : ""})`;
427
+ const header = `### ${result.role}${result.model ? ` [${result.model}]` : ""} — ${result.status} (${minutes} min${result.outputTruncated ? ", output capped" : ""})`;
416
428
  const meta: string[] = [`task: ${result.task}`];
417
429
  if (result.error) meta.push(`error: ${result.error}`);
418
430
  return `${header}\n${meta.join("\n")}\n\n${result.output || "(no output)"}`;
@@ -27,8 +27,6 @@ import {
27
27
  AGENT_ROLES,
28
28
  CHILD_TIMEOUT_MS,
29
29
  mapWithConcurrency,
30
- MAX_CONCURRENT_CHILDREN,
31
- MAX_TASKS_PER_CALL,
32
30
  renderChildResults,
33
31
  runChildAgent,
34
32
  type ChildResult,
@@ -39,17 +37,24 @@ const SpawnAgentsParams = Type.Object({
39
37
  tasks: Type.Array(
40
38
  Type.Object({
41
39
  role: StringEnum(AGENT_ROLES, {
42
- description: "code-searcher: read-only codebase exploration; researcher-web: web research via web_fetch; general-worker: full-tool implementation work",
40
+ description: "code-searcher: read-only codebase exploration; researcher-web: web research via web_search/web_fetch; general-worker: full-tool implementation work",
43
41
  }),
44
42
  task: Type.String({ description: "Self-contained task description for the child agent (it sees ONLY this, not your conversation)" }),
43
+ model: Type.Optional(Type.String({
44
+ description: 'Cross-model child: gateway model id for THIS task (e.g. "muse-spark-1.3", "muse-spark-1.2-contributor" for cheap swarm workers, "gpt-6-astra", "gpt-5.6-sol"). Omit to inherit this session\'s model',
45
+ })),
45
46
  }),
46
- { description: "Tasks to delegate; they run in parallel", minItems: 1, maxItems: MAX_TASKS_PER_CALL },
47
+ { description: "Tasks to delegate; they all run in parallel", minItems: 1 },
47
48
  ),
48
49
  timeout_minutes: Type.Optional(Type.Number({
49
50
  description: `Per-child wall-clock limit in minutes (default ${CHILD_TIMEOUT_MS / 60_000}; a timed-out child is killed and its partial result is kept)`,
50
51
  minimum: 0.1,
51
52
  maximum: 60,
52
53
  })),
54
+ max_parallel: Type.Optional(Type.Number({
55
+ description: "Optional throttle: run at most this many children at once. Omit to run every task in parallel (no cap)",
56
+ minimum: 1,
57
+ })),
53
58
  });
54
59
 
55
60
  interface AgentRegistryEntry {
@@ -70,8 +75,9 @@ export default function (pi: ExtensionAPI) {
70
75
  description: [
71
76
  "Delegate tasks to parallel subagents with isolated context windows.",
72
77
  "Each child is a headless agent run in this workspace with a role preset:",
73
- "code-searcher (read-only exploration), researcher-web (web research via web_fetch), general-worker (full tools).",
74
- `Children run in parallel (max ${MAX_CONCURRENT_CHILDREN} at once, ${MAX_TASKS_PER_CALL} tasks per call) and each gets its own timeout (default ${CHILD_TIMEOUT_MS / 60_000} min).`,
78
+ "code-searcher (read-only exploration), researcher-web (web research via web_search + web_fetch), general-worker (full tools).",
79
+ `Children run in parallel — every task at once unless max_parallel throttles it — and each gets its own timeout (default ${CHILD_TIMEOUT_MS / 60_000} min).`,
80
+ 'Each task can run on a DIFFERENT model via its "model" field — e.g. cheap muse workers for wide swarm sweeps under an astra parent.',
75
81
  "Every task description must be SELF-CONTAINED: the child cannot see this conversation.",
76
82
  "Use for parallelizable work: broad code surveys, independent research questions, independent implementation chunks.",
77
83
  ].join(" "),
@@ -88,18 +94,27 @@ export default function (pi: ExtensionAPI) {
88
94
  if (!raw || typeof raw.task !== "string" || !raw.task.trim()) {
89
95
  throw new Error("invalid tasks: every entry needs a non-empty task string");
90
96
  }
91
- tasks.push({ role: raw.role, task: raw.task.trim() });
97
+ const model = typeof raw.model === "string" ? raw.model.trim() : "";
98
+ if (raw.model !== undefined && !model) {
99
+ throw new Error("invalid tasks: model must be a non-empty gateway model id");
100
+ }
101
+ tasks.push({
102
+ role: raw.role,
103
+ task: raw.task.trim(),
104
+ ...(model ? { model } : {}),
105
+ });
92
106
  }
93
107
  if (tasks.length === 0) throw new Error("no tasks given");
94
- if (tasks.length > MAX_TASKS_PER_CALL) {
95
- throw new Error(`too many tasks (${tasks.length}); the limit is ${MAX_TASKS_PER_CALL} per call`);
96
- }
97
108
  const parentSessionId = String(ctx?.sessionManager?.getSessionId?.() ?? "");
98
109
  if (!parentSessionId) throw new Error("no parent session id — subagents cannot be traced");
99
110
  const cwd = ctx?.cwd ? String(ctx.cwd) : process.cwd();
100
111
  const timeoutMinutes = typeof params.timeout_minutes === "number" && params.timeout_minutes > 0
101
112
  ? params.timeout_minutes
102
113
  : undefined;
114
+ // No cap: every task runs at once unless the caller throttles.
115
+ const concurrency = typeof params.max_parallel === "number" && params.max_parallel >= 1
116
+ ? Math.floor(params.max_parallel)
117
+ : tasks.length;
103
118
 
104
119
  const entries = registry.get(parentSessionId) ?? [];
105
120
  registry.set(parentSessionId, entries);
@@ -117,7 +132,7 @@ export default function (pi: ExtensionAPI) {
117
132
 
118
133
  const results = await mapWithConcurrency<ChildTask, ChildResult>(
119
134
  tasks,
120
- MAX_CONCURRENT_CHILDREN,
135
+ concurrency,
121
136
  async (task, index) => {
122
137
  const result = await runChildAgent(task, {
123
138
  cwd,
@@ -15,6 +15,7 @@ import sota from "./sota";
15
15
  import agents from "./agents";
16
16
  import plan from "./plan";
17
17
  import webfetch from "./webfetch";
18
+ import websearch from "./websearch";
18
19
 
19
20
  export default function (pi: ExtensionAPI) {
20
21
  commands(pi as any);
@@ -23,6 +24,7 @@ export default function (pi: ExtensionAPI) {
23
24
  collector(pi as any);
24
25
  mcp(pi as any);
25
26
  webfetch(pi as any);
27
+ websearch(pi as any);
26
28
  plan(pi as any);
27
29
  agents(pi as any);
28
30
  }
@@ -144,6 +144,9 @@ export interface FetchReadableOptions {
144
144
  maxBytes?: number;
145
145
  timeoutMs?: number;
146
146
  signal?: AbortSignal;
147
+ /** Override the User-Agent (some endpoints, e.g. DuckDuckGo's HTML
148
+ * results, serve an anomaly page to unknown crawlers). */
149
+ userAgent?: string;
147
150
  }
148
151
 
149
152
  /**
@@ -175,7 +178,7 @@ export async function fetchReadableText(
175
178
  // identify honestly (traces show the agent anyway).
176
179
  Accept: "text/html,application/json,text/*;q=0.9,*/*;q=0.1",
177
180
  "Accept-Language": "en-US,en;q=0.9",
178
- "User-Agent": "omnirush-webfetch/1.0 (+https://omnirush.ai)",
181
+ "User-Agent": options.userAgent ?? "omnirush-webfetch/1.0 (+https://omnirush.ai)",
179
182
  },
180
183
  });
181
184
  const contentType = String(response.headers.get("content-type") ?? "");
@@ -0,0 +1,144 @@
1
+ // web_search internals — pure, injectable, node:test covered.
2
+ //
3
+ // Keyless web search via DuckDuckGo's no-JS HTML endpoint
4
+ // (html.duckduckgo.com/html/?q=...). The fetch itself reuses
5
+ // webfetch-lib's bounded fetchReadableText (size cap + timeout +
6
+ // browser-ish headers); this module owns the result parsing: extract
7
+ // title/URL/snippet triples, unwrap DDG's /l/?uddg=<encoded> redirect
8
+ // links, strip tags, decode entities. No API key, no JS rendering.
9
+
10
+ import { decodeHtmlEntities, fetchReadableText } from "./webfetch-lib";
11
+
12
+ /** Hard cap on results returned per query. */
13
+ export const WEB_SEARCH_MAX_RESULTS = 10;
14
+
15
+ /** The no-JS HTML results endpoint. */
16
+ export const WEB_SEARCH_ENDPOINT = "https://html.duckduckgo.com/html/";
17
+
18
+ export interface WebSearchResult {
19
+ title: string;
20
+ url: string;
21
+ snippet: string;
22
+ }
23
+
24
+ export type WebSearchOutcome =
25
+ | { ok: true; query: string; results: WebSearchResult[] }
26
+ | { ok: false; query: string; error: string };
27
+
28
+ /** Build the search URL for one query. */
29
+ export function buildSearchUrl(query: string): string {
30
+ return `${WEB_SEARCH_ENDPOINT}?q=${encodeURIComponent(String(query ?? "").trim())}`;
31
+ }
32
+
33
+ /** Decode entities, strip tags (AFTER decoding — snippets carry
34
+ * entity-encoded markup like &lt;b&gt;), collapse whitespace. */
35
+ function cleanText(html: string): string {
36
+ return decodeHtmlEntities(String(html ?? ""))
37
+ .replace(/<[^>]*>/g, " ")
38
+ .replace(/\s+/g, " ")
39
+ .trim();
40
+ }
41
+
42
+ /** Unwrap DDG redirect links (`.../l/?uddg=<encoded>`) to the target URL. */
43
+ export function unwrapDdgHref(href: string): string | null {
44
+ const raw = String(href ?? "").trim();
45
+ if (!raw) return null;
46
+ const absolute = raw.startsWith("//") ? `https:${raw}` : raw;
47
+ try {
48
+ const url = new URL(absolute);
49
+ const uddg = url.searchParams.get("uddg");
50
+ if (uddg) {
51
+ const target = decodeURIComponent(uddg);
52
+ // Ad/track links point at DDG itself — drop them.
53
+ if (/^https?:\/\/(www\.)?duckduckgo\.com\//i.test(target)) return null;
54
+ return target;
55
+ }
56
+ if (/duckduckgo\.com\/y\.js/i.test(absolute)) return null;
57
+ if (url.hostname.includes("duckduckgo.com")) return null;
58
+ return absolute;
59
+ } catch {
60
+ return null;
61
+ }
62
+ }
63
+
64
+ /**
65
+ * Parse the DDG HTML results page into title/url/snippet triples, in
66
+ * page order. Tolerates markup drift: a result needs at least a title
67
+ * anchor and a resolvable URL.
68
+ */
69
+ export function parseDdgResults(html: string): WebSearchResult[] {
70
+ const results: WebSearchResult[] = [];
71
+ const snippetByHref = new Map<string, string>();
72
+ const anchorRe =
73
+ /<a\b[^>]*class="(?:[^"]*\s)?result__(a|snippet)(?:\s[^"]*)?"[^>]*href="([^"]*)"[^>]*>([\s\S]*?)<\/a>/gi;
74
+ for (const match of String(html ?? "").matchAll(anchorRe)) {
75
+ const kind = match[1];
76
+ const href = unwrapDdgHref(match[2]);
77
+ const text = cleanText(match[3]);
78
+ if (kind === "snippet") {
79
+ // Snippets carry the result URL too; remember the text for the
80
+ // result anchor that follows with the same href.
81
+ if (href) snippetByHref.set(href, text);
82
+ continue;
83
+ }
84
+ if (!href || !text) continue;
85
+ const seen = results.find((r) => r.url === href);
86
+ if (seen) continue;
87
+ results.push({
88
+ title: text,
89
+ url: href,
90
+ snippet: snippetByHref.get(href) ?? "",
91
+ });
92
+ }
93
+ // Attach snippets that appeared after their anchor (page order varies).
94
+ if (snippetByHref.size > 0) {
95
+ for (const result of results) {
96
+ if (!result.snippet) result.snippet = snippetByHref.get(result.url) ?? "";
97
+ }
98
+ }
99
+ return results;
100
+ }
101
+
102
+ /** Render results as compact numbered text for the model. */
103
+ export function renderSearchResults(query: string, results: WebSearchResult[]): string {
104
+ if (results.length === 0) {
105
+ return `web_search: no results for "${query}"`;
106
+ }
107
+ const lines = results.map((result, index) => {
108
+ const parts = [`${index + 1}. ${result.title}`, ` ${result.url}`];
109
+ if (result.snippet) parts.push(` ${result.snippet}`);
110
+ return parts.join("\n");
111
+ });
112
+ return `web_search: ${results.length} result(s) for "${query}"\n\n${lines.join("\n\n")}`;
113
+ }
114
+
115
+ /** Run one search: fetch the endpoint and parse. Injectable fetchImpl for tests. */
116
+ export async function webSearch(
117
+ query: string,
118
+ options: {
119
+ count?: number;
120
+ signal?: AbortSignal;
121
+ fetchImpl?: typeof fetchReadableText;
122
+ } = {},
123
+ ): Promise<WebSearchOutcome> {
124
+ const trimmed = String(query ?? "").trim();
125
+ if (!trimmed) return { ok: false, query: trimmed, error: "empty query" };
126
+ const count = Math.min(
127
+ WEB_SEARCH_MAX_RESULTS,
128
+ Math.max(1, typeof options.count === "number" && options.count > 0 ? Math.floor(options.count) : 8),
129
+ );
130
+ const fetchImpl = options.fetchImpl ?? fetchReadableText;
131
+ const outcome = await fetchImpl(buildSearchUrl(trimmed), {
132
+ signal: options.signal,
133
+ // DDG serves an anomaly page (HTTP 202, zero results) to unknown
134
+ // crawlers; the search endpoint needs a browser UA.
135
+ userAgent: "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36",
136
+ });
137
+ if (!outcome.ok) {
138
+ return { ok: false, query: trimmed, error: outcome.error };
139
+ }
140
+ // Parse the RAW body — outcome.text is HTML-stripped and has no
141
+ // anchors left for the parser.
142
+ const results = parseDdgResults(outcome.raw).slice(0, count);
143
+ return { ok: true, query: trimmed, results };
144
+ }
@@ -0,0 +1,60 @@
1
+ // web_search — a built-in Omnirush tool: query -> ranked results
2
+ // (title + URL + snippet) from DuckDuckGo's no-JS HTML endpoint.
3
+ // No API key, no JS rendering; the bounded fetch lives in
4
+ // webfetch-lib, the parsing in websearch-lib (pure, unit-tested).
5
+ // Pair with web_fetch to read the promising pages.
6
+
7
+ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
8
+ import { Type } from "typebox";
9
+
10
+ import {
11
+ renderSearchResults,
12
+ WEB_SEARCH_MAX_RESULTS,
13
+ webSearch,
14
+ } from "./websearch-lib";
15
+
16
+ const WebSearchParams = Type.Object({
17
+ query: Type.String({ description: "The search query (web-search-engine syntax works: quoted phrases, site: filters, -exclusions)" }),
18
+ count: Type.Optional(Type.Number({
19
+ description: `Max results to return (default 8, max ${WEB_SEARCH_MAX_RESULTS})`,
20
+ minimum: 1,
21
+ maximum: WEB_SEARCH_MAX_RESULTS,
22
+ })),
23
+ });
24
+
25
+ export default function (pi: ExtensionAPI) {
26
+ pi.registerTool({
27
+ name: "web_search",
28
+ label: "Web Search",
29
+ description: [
30
+ "Search the web and return ranked results (title, URL, snippet).",
31
+ "No API key; results come from DuckDuckGo's HTML endpoint.",
32
+ "Use web_fetch on a result's URL to read the full page.",
33
+ `Responses are capped at ${WEB_SEARCH_MAX_RESULTS} results and share web_fetch's 30s timeout.`,
34
+ ].join(" "),
35
+ promptSnippet: "web_search: search the web (titles, URLs, snippets — pair with web_fetch)",
36
+ promptGuidelines: [
37
+ "Use web_search when you need to FIND pages (you do not know the URL yet); use web_fetch when you do.",
38
+ 'Engine syntax works in the query: quoted phrases, site:example.com filters and -excluded-terms.',
39
+ ],
40
+ parameters: WebSearchParams,
41
+
42
+ async execute(_toolCallId, params, signal) {
43
+ const outcome = await webSearch(params.query, {
44
+ count: typeof params.count === "number" ? params.count : undefined,
45
+ signal,
46
+ });
47
+ if (!outcome.ok) {
48
+ throw new Error(outcome.error);
49
+ }
50
+ return {
51
+ content: [{ type: "text", text: renderSearchResults(outcome.query, outcome.results) }],
52
+ details: {
53
+ query: outcome.query,
54
+ count: outcome.results.length,
55
+ results: outcome.results,
56
+ },
57
+ };
58
+ },
59
+ });
60
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "omnirush",
3
- "version": "0.6.1",
3
+ "version": "0.7.1",
4
4
  "description": "Omnirush \u2014 free daily tokens for the most powerful coding model on earth.",
5
5
  "license": "Apache-2.0",
6
6
  "type": "module",