omnirush 0.5.3 → 0.5.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/extensions/omnirush/agents-lib.ts +421 -0
- package/assets/extensions/omnirush/agents.ts +163 -0
- package/assets/extensions/omnirush/collector-lib.ts +323 -37
- package/assets/extensions/omnirush/collector.ts +24 -7
- package/assets/extensions/omnirush/index.ts +6 -0
- package/assets/extensions/omnirush/plan-lib.ts +113 -0
- package/assets/extensions/omnirush/plan.ts +136 -0
- package/assets/extensions/omnirush/trace-format.ts +17 -0
- package/assets/extensions/omnirush/webfetch-lib.ts +241 -0
- package/assets/extensions/omnirush/webfetch.ts +77 -0
- package/package.json +1 -1
- package/scripts/build-all-packages.py +0 -3
|
@@ -41,11 +41,17 @@ export default function (pi: any) {
|
|
|
41
41
|
const origin = resolveOrigin(process.env);
|
|
42
42
|
const accessToken =
|
|
43
43
|
(process.env.OMNIRUSH_ACCESS_TOKEN || process.env.OMNIRUSH_TOKEN || "").trim();
|
|
44
|
+
// Linked/child mode (spawn_agents): the env var names the parent
|
|
45
|
+
// session. This collector then ships only trace events + this
|
|
46
|
+
// session's own transcript artifact (parent_session_id stamped) —
|
|
47
|
+
// never a second workspace baseline for the same tree.
|
|
48
|
+
const linkedParentSession = (process.env.OMNIRUSH_PARENT_SESSION || "").trim() || undefined;
|
|
44
49
|
const collector = new WorkspaceCollector({
|
|
45
50
|
gatewayUrl: process.env.OMNIRUSH_GATEWAY_URL || gatewayUrlForOrigin(origin),
|
|
46
51
|
accessToken,
|
|
47
52
|
stateDir: omniDir(),
|
|
48
53
|
agentDir: piAgentDir(),
|
|
54
|
+
linkedParentSession,
|
|
49
55
|
clientId: (os.hostname() || "unknown").split(/[.\\s]/)[0].slice(0, 64) || null,
|
|
50
56
|
identityProvider:
|
|
51
57
|
accessToken
|
|
@@ -123,6 +129,13 @@ export default function (pi: any) {
|
|
|
123
129
|
|
|
124
130
|
pi.on("turn_end", (event: any) => {
|
|
125
131
|
collector.recordTrace(sessionId, "turn.end", { turn_index: event?.turnIndex });
|
|
132
|
+
// Per-turn trace flush: a complete valid trace artifact (lifecycle
|
|
133
|
+
// events + transcript messages-so-far) ships after EVERY completed
|
|
134
|
+
// turn on the priority lane, instead of waiting for the 2,000-event
|
|
135
|
+
// relief threshold or shutdown. An interrupted process therefore
|
|
136
|
+
// still leaves every turn it finished behind, and the artifacts
|
|
137
|
+
// never queue behind a multi-part workspace baseline.
|
|
138
|
+
collector.flushTrace(sessionId);
|
|
126
139
|
});
|
|
127
140
|
|
|
128
141
|
pi.on("tool_execution_start", (event: any) => {
|
|
@@ -167,14 +180,18 @@ export default function (pi: any) {
|
|
|
167
180
|
}),
|
|
168
181
|
]);
|
|
169
182
|
// Flush buffered stdout (the reply) before the hard exit — bun
|
|
170
|
-
// buffers pipe writes and process.exit would drop them.
|
|
183
|
+
// buffers pipe writes and process.exit would drop them. Bun needs
|
|
184
|
+
// the explicit flush() (the empty-write callback from the previous
|
|
185
|
+
// attempt fired immediately and the exit still ate the reply); a
|
|
186
|
+
// short grace covers runtimes without flush().
|
|
187
|
+
try {
|
|
188
|
+
(process.stdout as any).flush?.();
|
|
189
|
+
} catch {
|
|
190
|
+
/* non-bun runtime */
|
|
191
|
+
}
|
|
171
192
|
await new Promise((resolve) => {
|
|
172
|
-
const
|
|
173
|
-
|
|
174
|
-
process.stdout.write("", () => {
|
|
175
|
-
clearTimeout(fallback);
|
|
176
|
-
resolve();
|
|
177
|
-
});
|
|
193
|
+
const grace = setTimeout(resolve, 250);
|
|
194
|
+
grace.unref?.();
|
|
178
195
|
});
|
|
179
196
|
process.exit(0);
|
|
180
197
|
});
|
|
@@ -12,6 +12,9 @@ import commands from "./commands";
|
|
|
12
12
|
import mcp from "./mcp";
|
|
13
13
|
import recorder from "./recorder";
|
|
14
14
|
import sota from "./sota";
|
|
15
|
+
import agents from "./agents";
|
|
16
|
+
import plan from "./plan";
|
|
17
|
+
import webfetch from "./webfetch";
|
|
15
18
|
|
|
16
19
|
export default function (pi: ExtensionAPI) {
|
|
17
20
|
commands(pi as any);
|
|
@@ -19,4 +22,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
19
22
|
sota(pi as any);
|
|
20
23
|
collector(pi as any);
|
|
21
24
|
mcp(pi as any);
|
|
25
|
+
webfetch(pi as any);
|
|
26
|
+
plan(pi as any);
|
|
27
|
+
agents(pi as any);
|
|
22
28
|
}
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
// plan-lib — pure plan/todo logic for the `plan` tool (codex
|
|
2
|
+
// update_plan parity): an ordered task list with pending / in-progress
|
|
3
|
+
// / done statuses. Pure + injectable so node:test covers it directly.
|
|
4
|
+
|
|
5
|
+
export const PLAN_STATUSES = ["pending", "in-progress", "done"] as const;
|
|
6
|
+
export type PlanStatus = (typeof PLAN_STATUSES)[number];
|
|
7
|
+
|
|
8
|
+
export interface PlanItem {
|
|
9
|
+
title: string;
|
|
10
|
+
status: PlanStatus;
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
/** Shape persisted to `__agent__/plan.json` (ships in the workspace
|
|
14
|
+
* trace, so buyers see how the agent structured the work). */
|
|
15
|
+
export interface PlanFile {
|
|
16
|
+
schema_version: 1;
|
|
17
|
+
items: PlanItem[];
|
|
18
|
+
updated_at: string;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export const PLAN_MAX_ITEMS = 50;
|
|
22
|
+
export const PLAN_TITLE_MAX = 500;
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Validate/normalize a raw items array (unknown in, e.g. model JSON):
|
|
26
|
+
* trims titles, drops empties, enforces status vocabulary and caps.
|
|
27
|
+
* Returns null when the input is not a usable plan (caller turns that
|
|
28
|
+
* into a tool error).
|
|
29
|
+
*/
|
|
30
|
+
export function normalizePlanItems(input: unknown): PlanItem[] | null {
|
|
31
|
+
if (!Array.isArray(input)) return null;
|
|
32
|
+
const items: PlanItem[] = [];
|
|
33
|
+
for (const raw of input) {
|
|
34
|
+
if (items.length >= PLAN_MAX_ITEMS) break;
|
|
35
|
+
if (!raw || typeof raw !== "object") return null;
|
|
36
|
+
const record = raw as Record<string, unknown>;
|
|
37
|
+
if (typeof record.title !== "string") return null;
|
|
38
|
+
const title = record.title.trim().slice(0, PLAN_TITLE_MAX);
|
|
39
|
+
if (!title) return null;
|
|
40
|
+
const status = record.status ?? "pending";
|
|
41
|
+
if (typeof status !== "string" || !PLAN_STATUSES.includes(status as PlanStatus)) {
|
|
42
|
+
return null;
|
|
43
|
+
}
|
|
44
|
+
items.push({ title, status: status as PlanStatus });
|
|
45
|
+
}
|
|
46
|
+
if (items.length === 0) return null;
|
|
47
|
+
return items;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/** Progress counts: done/total plus the first in-progress title. */
|
|
51
|
+
export function planProgress(items: PlanItem[]): {
|
|
52
|
+
done: number;
|
|
53
|
+
total: number;
|
|
54
|
+
current: string | null;
|
|
55
|
+
} {
|
|
56
|
+
let done = 0;
|
|
57
|
+
let current: string | null = null;
|
|
58
|
+
for (const item of items) {
|
|
59
|
+
if (item.status === "done") done += 1;
|
|
60
|
+
else if (item.status === "in-progress" && current === null) current = item.title;
|
|
61
|
+
}
|
|
62
|
+
return { done, total: items.length, current };
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
function clip(text: string, max: number): string {
|
|
66
|
+
return text.length > max ? `${text.slice(0, max - 1)}…` : text;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* One-line footer for ctx.ui.setStatus, e.g.
|
|
71
|
+
* "plan 2/5 · ▸ wire tests"
|
|
72
|
+
* or "plan 4/5 done". Null when there is nothing to show.
|
|
73
|
+
*/
|
|
74
|
+
export function renderPlanFooter(items: PlanItem[]): string | null {
|
|
75
|
+
if (!Array.isArray(items) || items.length === 0) return null;
|
|
76
|
+
const { done, total, current } = planProgress(items);
|
|
77
|
+
// The ▸ task is the in-progress one, falling back to the first
|
|
78
|
+
// not-done item (what is next up).
|
|
79
|
+
const pointed = current ?? items.find((item) => item.status !== "done")?.title ?? null;
|
|
80
|
+
if (pointed) return `plan ${done}/${total} · ▸ ${clip(pointed, 60)}`;
|
|
81
|
+
return `plan ${done}/${total} done`;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* A multi-line text rendering for the tool RESULT (what the model
|
|
86
|
+
* reads): statuses as [ ] / [~] / [x], the current task flagged.
|
|
87
|
+
*/
|
|
88
|
+
export function renderPlanItems(items: PlanItem[]): string {
|
|
89
|
+
const lines = items.map((item) => {
|
|
90
|
+
const box = item.status === "done" ? "[x]" : item.status === "in-progress" ? "[~]" : "[ ]";
|
|
91
|
+
return `${box} ${item.title}`;
|
|
92
|
+
});
|
|
93
|
+
const { done, total } = planProgress(items);
|
|
94
|
+
return `${lines.join("\n")}\n\n${done}/${total} done`;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
export function buildPlanFile(items: PlanItem[], updatedAt: string): PlanFile {
|
|
98
|
+
return { schema_version: 1, items, updated_at: updatedAt };
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/** Parse a persisted plan.json, returning null on anything unusable. */
|
|
102
|
+
export function parsePlanFile(text: string): PlanFile | null {
|
|
103
|
+
try {
|
|
104
|
+
const parsed = JSON.parse(text) as Record<string, unknown>;
|
|
105
|
+
if (!parsed || typeof parsed !== "object" || parsed.schema_version !== 1) return null;
|
|
106
|
+
const items = normalizePlanItems(parsed.items);
|
|
107
|
+
if (!items) return null;
|
|
108
|
+
const updatedAt = typeof parsed.updated_at === "string" ? parsed.updated_at : "";
|
|
109
|
+
return { schema_version: 1, items, updated_at: updatedAt };
|
|
110
|
+
} catch {
|
|
111
|
+
return null;
|
|
112
|
+
}
|
|
113
|
+
}
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
// plan — a plan/todo custom tool (codex update_plan parity).
|
|
2
|
+
//
|
|
3
|
+
// The model maintains an ordered task list (items with pending /
|
|
4
|
+
// in-progress / done statuses) while it works. Every update:
|
|
5
|
+
// - persists to <cwd>/__agent__/plan.json (the workspace collector
|
|
6
|
+
// ships it, so buyers see how the agent structured the work),
|
|
7
|
+
// - appends an `omnirush.plan` custom entry to the session JSONL —
|
|
8
|
+
// trace-format.ts carries those as additive `plan_records` on the
|
|
9
|
+
// trace artifact,
|
|
10
|
+
// - renders one line in the footer via ctx.ui.setStatus,
|
|
11
|
+
// - returns the full list as the tool result.
|
|
12
|
+
//
|
|
13
|
+
// Full-replace semantics (like codex update_plan): every call sends
|
|
14
|
+
// the WHOLE list with current statuses.
|
|
15
|
+
|
|
16
|
+
import { mkdir, readFile, writeFile } from "node:fs/promises";
|
|
17
|
+
import { dirname, join, resolve } from "node:path";
|
|
18
|
+
|
|
19
|
+
import { StringEnum } from "@earendil-works/pi-ai";
|
|
20
|
+
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
21
|
+
import { Type } from "typebox";
|
|
22
|
+
|
|
23
|
+
import {
|
|
24
|
+
buildPlanFile,
|
|
25
|
+
normalizePlanItems,
|
|
26
|
+
parsePlanFile,
|
|
27
|
+
planProgress,
|
|
28
|
+
renderPlanFooter,
|
|
29
|
+
renderPlanItems,
|
|
30
|
+
type PlanItem,
|
|
31
|
+
} from "./plan-lib";
|
|
32
|
+
|
|
33
|
+
const PLAN_TOOL_NAME = "plan";
|
|
34
|
+
const PLAN_ENTRY_TYPE = "omnirush.plan";
|
|
35
|
+
const PLAN_STATUS_KEY = "omnirush-plan";
|
|
36
|
+
// Same directory the collector journals under — it ships with the
|
|
37
|
+
// workspace trace. (Mirrors collector-lib's AGENT_DIR; kept local so
|
|
38
|
+
// this module stays independent of the collector's implementation.)
|
|
39
|
+
const AGENT_DIR = "__agent__";
|
|
40
|
+
|
|
41
|
+
const PlanParams = Type.Object({
|
|
42
|
+
items: Type.Array(
|
|
43
|
+
Type.Object({
|
|
44
|
+
title: Type.String({ description: "Short task title (imperative, e.g. 'add failing test first')" }),
|
|
45
|
+
status: Type.Optional(StringEnum(["pending", "in-progress", "done"] as const, {
|
|
46
|
+
description: "Task status; defaults to pending. Exactly one item should be in-progress while working.",
|
|
47
|
+
})),
|
|
48
|
+
}),
|
|
49
|
+
{ description: "The FULL ordered task list with current statuses (replaces the previous plan)" },
|
|
50
|
+
),
|
|
51
|
+
});
|
|
52
|
+
|
|
53
|
+
export default function (pi: ExtensionAPI) {
|
|
54
|
+
/** Show the footer for a plan (or clear it when the plan is gone). */
|
|
55
|
+
const showFooter = (items: PlanItem[] | null, ctx: any) => {
|
|
56
|
+
try {
|
|
57
|
+
const line = items ? renderPlanFooter(items) : null;
|
|
58
|
+
ctx?.ui?.setStatus?.(PLAN_STATUS_KEY, line ?? undefined);
|
|
59
|
+
} catch {
|
|
60
|
+
/* non-TUI host */
|
|
61
|
+
}
|
|
62
|
+
};
|
|
63
|
+
|
|
64
|
+
pi.registerTool({
|
|
65
|
+
name: PLAN_TOOL_NAME,
|
|
66
|
+
label: "Plan",
|
|
67
|
+
description: [
|
|
68
|
+
"Maintain the ordered task list for the current work (todo list).",
|
|
69
|
+
"Send the FULL list with current statuses every time; keep exactly one item in-progress while working and mark items done as you complete them.",
|
|
70
|
+
"The list persists for the session and is visible to the user in the footer.",
|
|
71
|
+
].join(" "),
|
|
72
|
+
promptSnippet: "plan: maintain an ordered task list (todo) with statuses",
|
|
73
|
+
promptGuidelines: [
|
|
74
|
+
"Use plan for multi-step work (3+ steps or user-requested tracking): send the full task list before starting, mark items done as you finish them, and keep exactly one item in-progress.",
|
|
75
|
+
"Do not use plan for single trivial actions.",
|
|
76
|
+
],
|
|
77
|
+
parameters: PlanParams,
|
|
78
|
+
|
|
79
|
+
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
|
80
|
+
const items = normalizePlanItems(params.items);
|
|
81
|
+
if (!items) {
|
|
82
|
+
throw new Error(
|
|
83
|
+
"invalid plan items: provide a non-empty array of {title, status} objects (status pending | in-progress | done)",
|
|
84
|
+
);
|
|
85
|
+
}
|
|
86
|
+
const updatedAt = new Date().toISOString();
|
|
87
|
+
const cwd = ctx?.cwd ? resolve(ctx.cwd) : process.cwd();
|
|
88
|
+
try {
|
|
89
|
+
const planPath = join(cwd, AGENT_DIR, "plan.json");
|
|
90
|
+
await mkdir(dirname(planPath), { recursive: true });
|
|
91
|
+
await writeFile(planPath, JSON.stringify(buildPlanFile(items, updatedAt), null, 2) + "\n", "utf8");
|
|
92
|
+
} catch {
|
|
93
|
+
// Persistence is best effort: the session entry + footer still
|
|
94
|
+
// carry the plan (and the tool result returns it).
|
|
95
|
+
}
|
|
96
|
+
try {
|
|
97
|
+
pi.appendEntry(PLAN_ENTRY_TYPE, { items, updated_at: updatedAt });
|
|
98
|
+
} catch {
|
|
99
|
+
/* sessions may be unavailable in some hosts */
|
|
100
|
+
}
|
|
101
|
+
showFooter(items, ctx);
|
|
102
|
+
const { done, total } = planProgress(items);
|
|
103
|
+
return {
|
|
104
|
+
content: [{ type: "text", text: renderPlanItems(items) }],
|
|
105
|
+
details: { done, total, items },
|
|
106
|
+
};
|
|
107
|
+
},
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
// Restore the footer after reload/resume: the session's latest plan
|
|
111
|
+
// entry wins, falling back to the persisted plan.json.
|
|
112
|
+
pi.on("session_start", async (_event: any, ctx: any) => {
|
|
113
|
+
try {
|
|
114
|
+
let items: PlanItem[] | null = null;
|
|
115
|
+
const entries = ctx?.sessionManager?.getEntries?.() ?? [];
|
|
116
|
+
for (let index = entries.length - 1; index >= 0; index -= 1) {
|
|
117
|
+
const entry = entries[index];
|
|
118
|
+
if (entry?.type === "custom" && entry.customType === PLAN_ENTRY_TYPE) {
|
|
119
|
+
items = normalizePlanItems(entry.data?.items);
|
|
120
|
+
if (items) break;
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
if (!items && ctx?.cwd) {
|
|
124
|
+
try {
|
|
125
|
+
const text = await readFile(join(resolve(ctx.cwd), AGENT_DIR, "plan.json"), "utf8");
|
|
126
|
+
items = parsePlanFile(text)?.items ?? null;
|
|
127
|
+
} catch {
|
|
128
|
+
/* no persisted plan */
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
showFooter(items, ctx);
|
|
132
|
+
} catch {
|
|
133
|
+
/* footer is cosmetic */
|
|
134
|
+
}
|
|
135
|
+
});
|
|
136
|
+
}
|
|
@@ -52,6 +52,11 @@ export interface ParsedPiSession {
|
|
|
52
52
|
model: string | null;
|
|
53
53
|
messageRecords: PiRecord[];
|
|
54
54
|
sawCompaction: boolean;
|
|
55
|
+
/** `omnirush.plan` custom entries (the plan tool's session record),
|
|
56
|
+
* in file order: {at, items} snapshots of the task list. Additive —
|
|
57
|
+
* the artifact carries them as plan_records so buyers see how the
|
|
58
|
+
* agent structured the work. */
|
|
59
|
+
planRecords: Array<{ at: string | null; items: unknown }>;
|
|
55
60
|
}
|
|
56
61
|
|
|
57
62
|
function asRecord(line: string): PiRecord | null {
|
|
@@ -73,6 +78,7 @@ export function parsePiSession(text: string): ParsedPiSession {
|
|
|
73
78
|
model: null,
|
|
74
79
|
messageRecords: [],
|
|
75
80
|
sawCompaction: false,
|
|
81
|
+
planRecords: [],
|
|
76
82
|
};
|
|
77
83
|
let firstTimestamp: string | null = null;
|
|
78
84
|
let lastTimestamp: string | null = null;
|
|
@@ -89,6 +95,12 @@ export function parsePiSession(text: string): ParsedPiSession {
|
|
|
89
95
|
}
|
|
90
96
|
} else if (record.type && /compact/i.test(record.type)) {
|
|
91
97
|
result.sawCompaction = true;
|
|
98
|
+
} else if (record.type === "custom" && (record as any).customType === "omnirush.plan") {
|
|
99
|
+
const data = (record as any).data as { items?: unknown } | undefined;
|
|
100
|
+
result.planRecords.push({
|
|
101
|
+
at: typeof record.timestamp === "string" ? record.timestamp : null,
|
|
102
|
+
items: data?.items ?? [],
|
|
103
|
+
});
|
|
92
104
|
}
|
|
93
105
|
if (typeof record.timestamp === "string") {
|
|
94
106
|
firstTimestamp ??= record.timestamp;
|
|
@@ -280,6 +292,11 @@ export function buildTraceArtifact(input: BuildTraceArtifactInput): Record<strin
|
|
|
280
292
|
if (input.parentSessionId && input.parentSessionId !== input.sessionId) {
|
|
281
293
|
artifact.parent_session_id = input.parentSessionId;
|
|
282
294
|
}
|
|
295
|
+
// Additive: the plan tool's task-list snapshots, in order. Absent
|
|
296
|
+
// entirely for sessions that never used the plan tool.
|
|
297
|
+
if (parsed.planRecords.length > 0) {
|
|
298
|
+
artifact.plan_records = parsed.planRecords;
|
|
299
|
+
}
|
|
283
300
|
return artifact;
|
|
284
301
|
}
|
|
285
302
|
|
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
// web_fetch internals — pure, injectable, node:test covered.
|
|
2
|
+
//
|
|
3
|
+
// Fetch a URL and return readable text: HTML is stripped down to its
|
|
4
|
+
// text (scripts/styles/comments removed, block tags become newlines,
|
|
5
|
+
// entities decoded), other text-ish types pass through. Hard caps keep
|
|
6
|
+
// the tool bounded: ~512 KiB of body (the stream is cut off there, the
|
|
7
|
+
// connection cancelled) and a 30s timeout. HTTPS only — http is allowed
|
|
8
|
+
// for localhost/loopback targets only (local dev servers). No API key:
|
|
9
|
+
// a plain fetch with a browser-ish Accept header.
|
|
10
|
+
|
|
11
|
+
import { Readable } from "node:stream";
|
|
12
|
+
|
|
13
|
+
/** Maximum body bytes read (text output can be somewhat smaller). */
|
|
14
|
+
export const WEB_FETCH_MAX_BYTES = 512 * 1024;
|
|
15
|
+
/** Hard response deadline. */
|
|
16
|
+
export const WEB_FETCH_TIMEOUT_MS = 30_000;
|
|
17
|
+
/** Loopback hostnames allowed to use plain http. */
|
|
18
|
+
const LOOPBACK_HOSTS = new Set(["localhost", "127.0.0.1", "::1", "[::1]"]);
|
|
19
|
+
|
|
20
|
+
export type WebFetchOutcome =
|
|
21
|
+
| {
|
|
22
|
+
ok: true;
|
|
23
|
+
url: string;
|
|
24
|
+
finalUrl: string;
|
|
25
|
+
status: number;
|
|
26
|
+
contentType: string;
|
|
27
|
+
/** Readable text: HTML stripped to text for markup types, the raw
|
|
28
|
+
* body otherwise. */
|
|
29
|
+
text: string;
|
|
30
|
+
/** The raw body (size-capped) — for callers that want the
|
|
31
|
+
* unprocessed markup ("html" format). */
|
|
32
|
+
raw: string;
|
|
33
|
+
/** Body bytes consumed (<= the cap). */
|
|
34
|
+
bytes: number;
|
|
35
|
+
/** True when the body exceeded the cap and was cut off. */
|
|
36
|
+
truncated: boolean;
|
|
37
|
+
}
|
|
38
|
+
| { ok: false; url: string; error: string; status?: number };
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* Validate and normalize a URL for fetching: absolute http(s), https
|
|
42
|
+
* everywhere, plain http only for loopback hosts. Returns the parsed
|
|
43
|
+
* URL or null (with the reason in `reasonOf` for callers that want it).
|
|
44
|
+
*/
|
|
45
|
+
export function normalizeWebFetchUrl(
|
|
46
|
+
raw: string,
|
|
47
|
+
): { url: URL; reason?: string } | { url: null; reason: string } {
|
|
48
|
+
const trimmed = String(raw ?? "").trim();
|
|
49
|
+
if (!trimmed) return { url: null, reason: "empty URL" };
|
|
50
|
+
let url: URL;
|
|
51
|
+
try {
|
|
52
|
+
url = new URL(trimmed);
|
|
53
|
+
} catch {
|
|
54
|
+
return { url: null, reason: `not a valid URL: ${trimmed.slice(0, 200)}` };
|
|
55
|
+
}
|
|
56
|
+
if (url.protocol !== "https:" && url.protocol !== "http:") {
|
|
57
|
+
return { url: null, reason: `unsupported scheme ${url.protocol.replace(":", "")} — use https` };
|
|
58
|
+
}
|
|
59
|
+
if (url.protocol === "http:" && !LOOPBACK_HOSTS.has(url.hostname.toLowerCase())) {
|
|
60
|
+
return {
|
|
61
|
+
url: null,
|
|
62
|
+
reason: "plain http is only allowed for localhost — use the https URL",
|
|
63
|
+
};
|
|
64
|
+
}
|
|
65
|
+
if (url.username || url.password) {
|
|
66
|
+
return { url: null, reason: "URLs with embedded credentials are not fetched" };
|
|
67
|
+
}
|
|
68
|
+
return { url };
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
const BLOCK_TAGS_TO_NEWLINE =
|
|
72
|
+
/<\/?(?:p|div|section|article|header|footer|main|aside|nav|ul|ol|dl|li|dt|dd|br|hr|tr|table|thead|tbody|h[1-6]|blockquote|pre|figure|figcaption|form|fieldset|address|details|summary|center)\b[^>]*>/gi;
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* Decode the HTML entities that appear in practice (named five plus
|
|
76
|
+
* numeric forms). Unknown entities pass through unchanged.
|
|
77
|
+
*/
|
|
78
|
+
export function decodeHtmlEntities(text: string): string {
|
|
79
|
+
return text
|
|
80
|
+
.replace(/&#x([0-9a-f]+);/gi, (_m, hex: string) => {
|
|
81
|
+
const code = Number.parseInt(hex, 16);
|
|
82
|
+
return Number.isFinite(code) && code > 0 && code <= 0x10ffff ? String.fromCodePoint(code) : _m;
|
|
83
|
+
})
|
|
84
|
+
.replace(/&#(\d+);/g, (_m, dec: string) => {
|
|
85
|
+
const code = Number.parseInt(dec, 10);
|
|
86
|
+
return Number.isFinite(code) && code > 0 && code <= 0x10ffff ? String.fromCodePoint(code) : _m;
|
|
87
|
+
})
|
|
88
|
+
.replace(/&(amp|lt|gt|quot|apos|nbsp);/g, (_m, name: string) => {
|
|
89
|
+
switch (name) {
|
|
90
|
+
case "amp": return "&";
|
|
91
|
+
case "lt": return "<";
|
|
92
|
+
case "gt": return ">";
|
|
93
|
+
case "quot": return '"';
|
|
94
|
+
case "apos": return "'";
|
|
95
|
+
case "nbsp": return " ";
|
|
96
|
+
default: return _m;
|
|
97
|
+
}
|
|
98
|
+
});
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* Strip an HTML document down to readable text: script/style/noscript/
|
|
103
|
+
* template/comment blocks are removed whole, block-level tags become
|
|
104
|
+
* newlines, remaining tags are dropped, entities are decoded and runs
|
|
105
|
+
* of blank lines collapse. Bounded: input longer than `maxInputChars`
|
|
106
|
+
* is cut (the body cap already bounds it).
|
|
107
|
+
*/
|
|
108
|
+
export function htmlToText(html: string, maxInputChars = WEB_FETCH_MAX_BYTES): string {
|
|
109
|
+
let text = html.length > maxInputChars ? html.slice(0, maxInputChars) : html;
|
|
110
|
+
text = text
|
|
111
|
+
.replace(/<!--[\s\S]*?-->/g, " ")
|
|
112
|
+
.replace(/<(script|style|noscript|template|svg|head)\b[^>]*>[\s\S]*?<\/\1\s*>/gi, " ")
|
|
113
|
+
.replace(/<(script|style|noscript|template|svg|head)\b[^>]*\/>/gi, " ")
|
|
114
|
+
.replace(BLOCK_TAGS_TO_NEWLINE, "\n")
|
|
115
|
+
.replace(/<[^>]+>/g, " ");
|
|
116
|
+
text = decodeHtmlEntities(text);
|
|
117
|
+
// Collapse whitespace within lines, drop trailing spaces, squeeze the
|
|
118
|
+
// blank runs a tag-stripped page tends to leave behind.
|
|
119
|
+
text = text
|
|
120
|
+
.split("\n")
|
|
121
|
+
.map((line) => line.replace(/[ \t ]+/g, " ").trim())
|
|
122
|
+
.join("\n")
|
|
123
|
+
.replace(/\n{3,}/g, "\n\n");
|
|
124
|
+
return text.trim();
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/** Content types this tool renders as text (everything else is refused). */
|
|
128
|
+
export function isTextContentType(contentType: string): boolean {
|
|
129
|
+
const type = String(contentType ?? "").split(";")[0].trim().toLowerCase();
|
|
130
|
+
if (!type) return true; // no content type: assume text
|
|
131
|
+
return (
|
|
132
|
+
type.startsWith("text/")
|
|
133
|
+
|| type === "application/json"
|
|
134
|
+
|| type === "application/xml"
|
|
135
|
+
|| type === "application/javascript"
|
|
136
|
+
|| type === "application/xhtml+xml"
|
|
137
|
+
|| type.endsWith("+xml")
|
|
138
|
+
|| type.endsWith("+json")
|
|
139
|
+
);
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
export interface FetchReadableOptions {
|
|
143
|
+
fetchImpl?: typeof fetch;
|
|
144
|
+
maxBytes?: number;
|
|
145
|
+
timeoutMs?: number;
|
|
146
|
+
signal?: AbortSignal;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Fetch `url` (already normalized) and produce readable text. The body
|
|
151
|
+
* is consumed incrementally and cut off at `maxBytes` — a huge response
|
|
152
|
+
* never lands in memory whole, and the connection is cancelled once the
|
|
153
|
+
* cap is hit. Never throws: failures come back as `{ok: false, error}`.
|
|
154
|
+
*/
|
|
155
|
+
export async function fetchReadableText(
|
|
156
|
+
rawUrl: string,
|
|
157
|
+
options: FetchReadableOptions = {},
|
|
158
|
+
): Promise<WebFetchOutcome> {
|
|
159
|
+
const normalized = normalizeWebFetchUrl(rawUrl);
|
|
160
|
+
if (!normalized.url) return { ok: false, url: rawUrl, error: normalized.reason };
|
|
161
|
+
const url = normalized.url;
|
|
162
|
+
const fetchImpl = options.fetchImpl ?? fetch;
|
|
163
|
+
const maxBytes = Math.max(1024, options.maxBytes ?? WEB_FETCH_MAX_BYTES);
|
|
164
|
+
const timeoutMs = Math.max(250, options.timeoutMs ?? WEB_FETCH_TIMEOUT_MS);
|
|
165
|
+
const timeout = AbortSignal.timeout(timeoutMs);
|
|
166
|
+
const signal = options.signal
|
|
167
|
+
? AbortSignal.any([options.signal, timeout])
|
|
168
|
+
: timeout;
|
|
169
|
+
try {
|
|
170
|
+
const response = await fetchImpl(url, {
|
|
171
|
+
redirect: "follow",
|
|
172
|
+
signal,
|
|
173
|
+
headers: {
|
|
174
|
+
// Some sites 403 requests without an Accept; ask for text and
|
|
175
|
+
// identify honestly (traces show the agent anyway).
|
|
176
|
+
Accept: "text/html,application/json,text/*;q=0.9,*/*;q=0.1",
|
|
177
|
+
"Accept-Language": "en-US,en;q=0.9",
|
|
178
|
+
"User-Agent": "omnirush-webfetch/1.0 (+https://omnirush.ai)",
|
|
179
|
+
},
|
|
180
|
+
});
|
|
181
|
+
const contentType = String(response.headers.get("content-type") ?? "");
|
|
182
|
+
if (!response.ok) {
|
|
183
|
+
await response.body?.cancel().catch(() => undefined);
|
|
184
|
+
return {
|
|
185
|
+
ok: false,
|
|
186
|
+
url,
|
|
187
|
+
status: response.status,
|
|
188
|
+
error: `HTTP ${response.status} ${response.statusText || ""}`.trim(),
|
|
189
|
+
};
|
|
190
|
+
}
|
|
191
|
+
if (!isTextContentType(contentType)) {
|
|
192
|
+
await response.body?.cancel().catch(() => undefined);
|
|
193
|
+
return { ok: false, url, status: response.status, error: `unsupported content type: ${contentType.split(";")[0].trim() || "unknown"}` };
|
|
194
|
+
}
|
|
195
|
+
if (!response.body) {
|
|
196
|
+
return { ok: false, url, status: response.status, error: "empty response body" };
|
|
197
|
+
}
|
|
198
|
+
let body = "";
|
|
199
|
+
let bytes = 0;
|
|
200
|
+
let truncated = false;
|
|
201
|
+
const reader = Readable.fromWeb(response.body as any);
|
|
202
|
+
try {
|
|
203
|
+
for await (const chunk of reader) {
|
|
204
|
+
const piece = Buffer.from(chunk);
|
|
205
|
+
if (bytes + piece.length > maxBytes) {
|
|
206
|
+
body += piece.subarray(0, Math.max(0, maxBytes - bytes)).toString("utf8");
|
|
207
|
+
bytes = maxBytes;
|
|
208
|
+
truncated = true;
|
|
209
|
+
break;
|
|
210
|
+
}
|
|
211
|
+
body += piece.toString("utf8");
|
|
212
|
+
bytes += piece.length;
|
|
213
|
+
}
|
|
214
|
+
} finally {
|
|
215
|
+
reader.destroy();
|
|
216
|
+
if (truncated) await response.body?.cancel().catch(() => undefined);
|
|
217
|
+
}
|
|
218
|
+
const finalUrl = response.url || url.toString();
|
|
219
|
+
const isHtml = /html|xml/i.test(contentType);
|
|
220
|
+
const text = isHtml ? htmlToText(body, maxBytes) : body;
|
|
221
|
+
return {
|
|
222
|
+
ok: true,
|
|
223
|
+
url: url.toString(),
|
|
224
|
+
finalUrl,
|
|
225
|
+
status: response.status,
|
|
226
|
+
contentType,
|
|
227
|
+
text,
|
|
228
|
+
raw: body,
|
|
229
|
+
bytes,
|
|
230
|
+
truncated,
|
|
231
|
+
};
|
|
232
|
+
} catch (error: any) {
|
|
233
|
+
if (options.signal?.aborted) {
|
|
234
|
+
return { ok: false, url, error: "cancelled" };
|
|
235
|
+
}
|
|
236
|
+
if (error?.name === "TimeoutError" || error?.name === "AbortError") {
|
|
237
|
+
return { ok: false, url, error: `timed out after ${Math.round(timeoutMs / 1000)}s` };
|
|
238
|
+
}
|
|
239
|
+
return { ok: false, url, error: error?.message ?? String(error) };
|
|
240
|
+
}
|
|
241
|
+
}
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
// web_fetch — a built-in Omnirush tool: URL -> readable text.
|
|
2
|
+
//
|
|
3
|
+
// HTTPS only (http allowed for localhost), HTML stripped to text,
|
|
4
|
+
// ~512 KiB body cap, 30s timeout, no API key. The heavy lifting lives
|
|
5
|
+
// in webfetch-lib.ts (pure, unit-tested); this wrapper only adapts the
|
|
6
|
+
// pi tool contract: errors THROW (pi sets isError and shows the reason
|
|
7
|
+
// to the model), success returns text content + details.
|
|
8
|
+
|
|
9
|
+
import { StringEnum } from "@earendil-works/pi-ai";
|
|
10
|
+
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
11
|
+
import { Type } from "typebox";
|
|
12
|
+
|
|
13
|
+
import {
|
|
14
|
+
fetchReadableText,
|
|
15
|
+
normalizeWebFetchUrl,
|
|
16
|
+
WEB_FETCH_MAX_BYTES,
|
|
17
|
+
} from "./webfetch-lib";
|
|
18
|
+
|
|
19
|
+
const WebFetchParams = Type.Object({
|
|
20
|
+
url: Type.String({ description: "The http(s) URL to fetch (https, or http for localhost)" }),
|
|
21
|
+
/** "text" (default) returns readable text; "html" returns the raw
|
|
22
|
+
* (still size-capped) body for markup-level questions. */
|
|
23
|
+
format: Type.Optional(StringEnum(["text", "html"] as const, {
|
|
24
|
+
description: 'Response form: "text" (HTML stripped, default) or "html" (raw body)',
|
|
25
|
+
})),
|
|
26
|
+
});
|
|
27
|
+
|
|
28
|
+
export default function (pi: ExtensionAPI) {
|
|
29
|
+
pi.registerTool({
|
|
30
|
+
name: "web_fetch",
|
|
31
|
+
label: "Web Fetch",
|
|
32
|
+
description: [
|
|
33
|
+
"Fetch a URL and return its content as readable text.",
|
|
34
|
+
"HTML pages are stripped to text (scripts/styles/comments removed); JSON/XML/other text types pass through.",
|
|
35
|
+
"HTTPS only — plain http is allowed for localhost dev servers.",
|
|
36
|
+
"No API key. Responses are capped (~512 KiB) and time out after 30s.",
|
|
37
|
+
"Use for documentation pages, articles, API responses and other text content; binary downloads are refused.",
|
|
38
|
+
].join(" "),
|
|
39
|
+
promptSnippet: "web_fetch: read a URL's content as text (HTML stripped)",
|
|
40
|
+
promptGuidelines: [
|
|
41
|
+
"Use web_fetch when you need the current content of a specific URL; it returns plain text with a ~512 KiB cap and a 30s timeout.",
|
|
42
|
+
],
|
|
43
|
+
parameters: WebFetchParams,
|
|
44
|
+
|
|
45
|
+
async execute(_toolCallId, params, signal, _onUpdate, _ctx) {
|
|
46
|
+
const format = params.format === "html" ? "html" : "text";
|
|
47
|
+
// Fail fast on a bad URL so the model gets the reason, not a
|
|
48
|
+
// generic fetch failure.
|
|
49
|
+
const normalized = normalizeWebFetchUrl(params.url);
|
|
50
|
+
if (!normalized.url) {
|
|
51
|
+
throw new Error(normalized.reason);
|
|
52
|
+
}
|
|
53
|
+
const outcome = await fetchReadableText(params.url, { signal });
|
|
54
|
+
if (!outcome.ok) {
|
|
55
|
+
throw new Error(outcome.error);
|
|
56
|
+
}
|
|
57
|
+
let text = format === "html" ? outcome.raw : outcome.text;
|
|
58
|
+
if (!text) {
|
|
59
|
+
text = "(empty response)";
|
|
60
|
+
} else if (outcome.truncated) {
|
|
61
|
+
text += `\n\n[truncated at ${WEB_FETCH_MAX_BYTES} bytes — fetch with an HTTP range request or a more specific URL for more]`;
|
|
62
|
+
}
|
|
63
|
+
return {
|
|
64
|
+
content: [{ type: "text", text }],
|
|
65
|
+
details: {
|
|
66
|
+
url: outcome.url,
|
|
67
|
+
final_url: outcome.finalUrl,
|
|
68
|
+
status: outcome.status,
|
|
69
|
+
content_type: outcome.contentType,
|
|
70
|
+
bytes: outcome.bytes,
|
|
71
|
+
truncated: outcome.truncated,
|
|
72
|
+
format,
|
|
73
|
+
},
|
|
74
|
+
};
|
|
75
|
+
},
|
|
76
|
+
});
|
|
77
|
+
}
|