pi-cloudflare-browser-run 0.1.8 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/package.json +3 -3
- package/src/api.ts +201 -0
- package/src/index.ts +150 -1
package/README.md
CHANGED
|
@@ -16,6 +16,8 @@ the v4 REST API directly.
|
|
|
16
16
|
| `browse` | fetch a public URL, return clean **markdown** text (default; also `screenshot` / `pdf` actions) |
|
|
17
17
|
| `screenshot` | save a PNG of the page locally, returns the file path |
|
|
18
18
|
| `pdf` | save a PDF of the page locally, returns the file path |
|
|
19
|
+
| `crawl` | multi-page crawl via Browser Run `/crawl` (markdown by default; waits for small limits) |
|
|
20
|
+
| `crawl_status` | poll / fetch results for a crawl job id |
|
|
19
21
|
|
|
20
22
|
## Install
|
|
21
23
|
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-cloudflare-browser-run",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "Pi extension: web browsing tools (markdown / screenshot / pdf) powered by Cloudflare Browser Run. Headless Chrome on CF's network
|
|
3
|
+
"version": "0.2.0",
|
|
4
|
+
"description": "Pi extension: web browsing tools (markdown / screenshot / pdf / crawl) powered by Cloudflare Browser Run. Headless Chrome on CF's network \u2014 JS-rendered pages, login sessions, WebMCP-capable.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"engines": {
|
|
7
7
|
"node": ">=20"
|
|
@@ -31,7 +31,7 @@
|
|
|
31
31
|
"devDependencies": {
|
|
32
32
|
"@earendil-works/pi-coding-agent": "^0.86.0",
|
|
33
33
|
"@types/node": "^26.6.2",
|
|
34
|
-
"tsx": "^4.23.
|
|
34
|
+
"tsx": "^4.23.15",
|
|
35
35
|
"typescript": "^7.0.2"
|
|
36
36
|
},
|
|
37
37
|
"repository": {
|
package/src/api.ts
CHANGED
|
@@ -5,6 +5,8 @@
|
|
|
5
5
|
//
|
|
6
6
|
// Endpoints (v4 API):
|
|
7
7
|
// POST /accounts/{ACCOUNT_ID}/browser-rendering/{screenshot|markdown|pdf}
|
|
8
|
+
// POST /accounts/{ACCOUNT_ID}/browser-rendering/crawl
|
|
9
|
+
// GET /accounts/{ACCOUNT_ID}/browser-rendering/crawl/{jobId}
|
|
8
10
|
//
|
|
9
11
|
// Security: URL strictness first — only public http(s), localhost / private
|
|
10
12
|
// / reserved IPs / IPv6 literals / userinfo are rejected (SSRF guard).
|
|
@@ -133,3 +135,202 @@ export async function browserRunAction(config: BrowserRunConfig, action: Action,
|
|
|
133
135
|
return { ok: false, error: e instanceof Error ? e.message : String(e) };
|
|
134
136
|
}
|
|
135
137
|
}
|
|
138
|
+
|
|
139
|
+
// ---------------------------------------------------------------------------
|
|
140
|
+
// /crawl — async multi-page scrape (start + status/poll)
|
|
141
|
+
// Docs: https://developers.cloudflare.com/browser-rendering/rest-api/crawl-endpoint/
|
|
142
|
+
// ---------------------------------------------------------------------------
|
|
143
|
+
|
|
144
|
+
export type CrawlSource = "all" | "sitemaps" | "links";
|
|
145
|
+
export type CrawlFormat = "html" | "markdown" | "json";
|
|
146
|
+
|
|
147
|
+
export interface CrawlStartParams {
|
|
148
|
+
url: string;
|
|
149
|
+
limit?: number;
|
|
150
|
+
depth?: number;
|
|
151
|
+
formats?: CrawlFormat[];
|
|
152
|
+
render?: boolean;
|
|
153
|
+
source?: CrawlSource;
|
|
154
|
+
includePatterns?: string[];
|
|
155
|
+
excludePatterns?: string[];
|
|
156
|
+
includeExternalLinks?: boolean;
|
|
157
|
+
includeSubdomains?: boolean;
|
|
158
|
+
/** Content Signals purpose declarations. */
|
|
159
|
+
crawlPurposes?: Array<"search" | "ai-input" | "ai-train">;
|
|
160
|
+
contentUse?: "reference" | "full";
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
export interface CrawlStatusQuery {
|
|
164
|
+
jobId: string;
|
|
165
|
+
/** Max records to return (omit for a full page of results). */
|
|
166
|
+
limit?: number;
|
|
167
|
+
cursor?: string | number;
|
|
168
|
+
/** Filter records: queued|completed|disallowed|skipped|errored|cancelled */
|
|
169
|
+
status?: string;
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
export type CrawlJobStatus =
|
|
173
|
+
| "running"
|
|
174
|
+
| "completed"
|
|
175
|
+
| "errored"
|
|
176
|
+
| "cancelled_due_to_timeout"
|
|
177
|
+
| "cancelled_due_to_limits"
|
|
178
|
+
| "cancelled_by_user"
|
|
179
|
+
| string;
|
|
180
|
+
|
|
181
|
+
export interface CrawlRecord {
|
|
182
|
+
url: string;
|
|
183
|
+
status: string;
|
|
184
|
+
markdown?: string;
|
|
185
|
+
html?: string;
|
|
186
|
+
json?: unknown;
|
|
187
|
+
metadata?: Record<string, unknown>;
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
export interface CrawlJobResult {
|
|
191
|
+
id: string;
|
|
192
|
+
status: CrawlJobStatus;
|
|
193
|
+
browserSecondsUsed?: number;
|
|
194
|
+
total?: number;
|
|
195
|
+
finished?: number;
|
|
196
|
+
records?: CrawlRecord[];
|
|
197
|
+
cursor?: string | number;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
export type CrawlApiResult =
|
|
201
|
+
| { ok: true; jobId: string; job?: CrawlJobResult }
|
|
202
|
+
| { ok: false; error: string };
|
|
203
|
+
|
|
204
|
+
const DEFAULT_CRAWL_FORMATS: CrawlFormat[] = ["markdown"];
|
|
205
|
+
const CRAWL_POLL_INTERVAL_MS = 5_000;
|
|
206
|
+
const CRAWL_POLL_MAX_ATTEMPTS = 60; // 5 min
|
|
207
|
+
const SMALL_CRAWL_LIMIT = 20;
|
|
208
|
+
|
|
209
|
+
function crawlEndpoint(config: BrowserRunConfig, jobId?: string): string {
|
|
210
|
+
const base = (config.apiBase ?? DEFAULT_API_BASE).replace(/\/$/, "");
|
|
211
|
+
const root = `${base}/accounts/${encodeURIComponent(config.accountId)}/browser-rendering/crawl`;
|
|
212
|
+
return jobId ? `${root}/${encodeURIComponent(jobId)}` : root;
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
function crawlAuthHeaders(config: BrowserRunConfig): Record<string, string> {
|
|
216
|
+
return {
|
|
217
|
+
Authorization: `Bearer ${config.apiToken}`,
|
|
218
|
+
"Content-Type": "application/json",
|
|
219
|
+
"user-agent":
|
|
220
|
+
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36",
|
|
221
|
+
};
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
async function readCfJson(res: Response): Promise<{ success?: boolean; result?: unknown; errors?: unknown }> {
|
|
225
|
+
const text = await res.text();
|
|
226
|
+
try {
|
|
227
|
+
return JSON.parse(text) as { success?: boolean; result?: unknown; errors?: unknown };
|
|
228
|
+
} catch {
|
|
229
|
+
throw new Error(`non-JSON response (HTTP ${res.status}): ${text.slice(0, 200)}`);
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
/** Start a crawl job; returns the job id immediately. */
|
|
234
|
+
export async function crawlStart(config: BrowserRunConfig, params: CrawlStartParams): Promise<CrawlApiResult> {
|
|
235
|
+
try {
|
|
236
|
+
const url = assertSafeUrl(params.url);
|
|
237
|
+
const body: Record<string, unknown> = {
|
|
238
|
+
url,
|
|
239
|
+
formats: params.formats?.length ? params.formats : DEFAULT_CRAWL_FORMATS,
|
|
240
|
+
};
|
|
241
|
+
if (params.limit !== undefined) body.limit = params.limit;
|
|
242
|
+
if (params.depth !== undefined) body.depth = params.depth;
|
|
243
|
+
if (params.render !== undefined) body.render = params.render;
|
|
244
|
+
if (params.source !== undefined) body.source = params.source;
|
|
245
|
+
if (params.crawlPurposes !== undefined) body.crawlPurposes = params.crawlPurposes;
|
|
246
|
+
if (params.contentUse !== undefined) body.contentUse = params.contentUse;
|
|
247
|
+
const options: Record<string, unknown> = {};
|
|
248
|
+
if (params.includePatterns?.length) options.includePatterns = params.includePatterns;
|
|
249
|
+
if (params.excludePatterns?.length) options.excludePatterns = params.excludePatterns;
|
|
250
|
+
if (params.includeExternalLinks !== undefined) options.includeExternalLinks = params.includeExternalLinks;
|
|
251
|
+
if (params.includeSubdomains !== undefined) options.includeSubdomains = params.includeSubdomains;
|
|
252
|
+
if (Object.keys(options).length) body.options = options;
|
|
253
|
+
|
|
254
|
+
const res = await fetch(crawlEndpoint(config), {
|
|
255
|
+
method: "POST",
|
|
256
|
+
headers: crawlAuthHeaders(config),
|
|
257
|
+
body: JSON.stringify(body),
|
|
258
|
+
});
|
|
259
|
+
const data = await readCfJson(res);
|
|
260
|
+
if (!res.ok || data.success === false) {
|
|
261
|
+
const err = data.errors ? JSON.stringify(data.errors).slice(0, 300) : `HTTP ${res.status}`;
|
|
262
|
+
return { ok: false, error: `crawl start failed: ${err}` };
|
|
263
|
+
}
|
|
264
|
+
const jobId = typeof data.result === "string" ? data.result : (data.result as { id?: string } | undefined)?.id;
|
|
265
|
+
if (!jobId) return { ok: false, error: "crawl start failed: missing job id in response" };
|
|
266
|
+
return { ok: true, jobId };
|
|
267
|
+
} catch (e) {
|
|
268
|
+
return { ok: false, error: e instanceof Error ? e.message : String(e) };
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
/** Fetch crawl job status / records. */
|
|
273
|
+
export async function crawlStatus(config: BrowserRunConfig, query: CrawlStatusQuery): Promise<CrawlApiResult> {
|
|
274
|
+
try {
|
|
275
|
+
if (!query.jobId || !String(query.jobId).trim()) {
|
|
276
|
+
return { ok: false, error: "jobId is required" };
|
|
277
|
+
}
|
|
278
|
+
const u = new URL(crawlEndpoint(config, query.jobId));
|
|
279
|
+
if (query.limit !== undefined) u.searchParams.set("limit", String(query.limit));
|
|
280
|
+
if (query.cursor !== undefined) u.searchParams.set("cursor", String(query.cursor));
|
|
281
|
+
if (query.status !== undefined) u.searchParams.set("status", String(query.status));
|
|
282
|
+
|
|
283
|
+
const res = await fetch(u.toString(), {
|
|
284
|
+
method: "GET",
|
|
285
|
+
headers: crawlAuthHeaders(config),
|
|
286
|
+
});
|
|
287
|
+
const data = await readCfJson(res);
|
|
288
|
+
if (!res.ok || data.success === false) {
|
|
289
|
+
const err = data.errors ? JSON.stringify(data.errors).slice(0, 300) : `HTTP ${res.status}`;
|
|
290
|
+
return { ok: false, error: `crawl status failed: ${err}` };
|
|
291
|
+
}
|
|
292
|
+
const job = data.result as CrawlJobResult;
|
|
293
|
+
const jobId = job?.id ?? query.jobId;
|
|
294
|
+
return { ok: true, jobId, job };
|
|
295
|
+
} catch (e) {
|
|
296
|
+
return { ok: false, error: e instanceof Error ? e.message : String(e) };
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
function sleep(ms: number): Promise<void> {
|
|
301
|
+
return new Promise((r) => setTimeout(r, ms));
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
/**
|
|
305
|
+
* Start a crawl and optionally poll until a terminal status.
|
|
306
|
+
* `wait` defaults to true when limit is undefined or ≤20, else false.
|
|
307
|
+
*/
|
|
308
|
+
export async function crawl(
|
|
309
|
+
config: BrowserRunConfig,
|
|
310
|
+
params: CrawlStartParams & { wait?: boolean },
|
|
311
|
+
): Promise<CrawlApiResult> {
|
|
312
|
+
const limit = params.limit ?? 10;
|
|
313
|
+
const wait = params.wait ?? limit <= SMALL_CRAWL_LIMIT;
|
|
314
|
+
const started = await crawlStart(config, params);
|
|
315
|
+
if (!started.ok) return started;
|
|
316
|
+
if (!wait) return started;
|
|
317
|
+
|
|
318
|
+
for (let i = 0; i < CRAWL_POLL_MAX_ATTEMPTS; i++) {
|
|
319
|
+
const light = await crawlStatus(config, { jobId: started.jobId, limit: 1 });
|
|
320
|
+
if (!light.ok) return light;
|
|
321
|
+
const status = light.job?.status;
|
|
322
|
+
if (status && status !== "running") {
|
|
323
|
+
return crawlStatus(config, { jobId: started.jobId });
|
|
324
|
+
}
|
|
325
|
+
await sleep(CRAWL_POLL_INTERVAL_MS);
|
|
326
|
+
}
|
|
327
|
+
return {
|
|
328
|
+
ok: false,
|
|
329
|
+
error: `crawl job ${started.jobId} did not complete within timeout (still running)`,
|
|
330
|
+
};
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
/** Whether wait should default on for a given limit (exported for tests). */
|
|
334
|
+
export function defaultCrawlWait(limit?: number): boolean {
|
|
335
|
+
return (limit ?? 10) <= SMALL_CRAWL_LIMIT;
|
|
336
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -5,6 +5,8 @@
|
|
|
5
5
|
// browse(url, action?) read a page as clean markdown (default)
|
|
6
6
|
// screenshot(url) save a PNG of the page, returns local path
|
|
7
7
|
// pdf(url) save a PDF of the page, returns local path
|
|
8
|
+
// crawl(url, ...) multi-page crawl via /crawl (optional wait/poll)
|
|
9
|
+
// crawl_status(jobId, ...) poll / fetch crawl job results
|
|
8
10
|
//
|
|
9
11
|
// Auth: CLOUDFLARE_API_TOKEN (Browser Rendering:Edit) + CLOUDFLARE_ACCOUNT_ID
|
|
10
12
|
// env vars. No Workers, no proxy — direct v4 REST calls.
|
|
@@ -14,7 +16,17 @@ import { Type } from "typebox";
|
|
|
14
16
|
import { mkdirSync, writeFileSync } from "fs";
|
|
15
17
|
import { tmpdir } from "os";
|
|
16
18
|
import { join } from "path";
|
|
17
|
-
import {
|
|
19
|
+
import {
|
|
20
|
+
browserRunAction,
|
|
21
|
+
crawl,
|
|
22
|
+
crawlStatus,
|
|
23
|
+
defaultCrawlWait,
|
|
24
|
+
loadConfig,
|
|
25
|
+
type Action,
|
|
26
|
+
type CrawlFormat,
|
|
27
|
+
type CrawlJobResult,
|
|
28
|
+
type CrawlRecord,
|
|
29
|
+
} from "./api.js";
|
|
18
30
|
|
|
19
31
|
const urlSchema = Type.Object({
|
|
20
32
|
url: Type.String({ minLength: 1, description: "Public http(s) URL, e.g. https://example.com" }),
|
|
@@ -121,4 +133,141 @@ export default function (pi: ExtensionAPI) {
|
|
|
121
133
|
return toolResult(`pdf saved to ${file}`);
|
|
122
134
|
},
|
|
123
135
|
});
|
|
136
|
+
|
|
137
|
+
const truncateRecord = (rec: CrawlRecord, maxMd = 4_000): CrawlRecord => {
|
|
138
|
+
const out: CrawlRecord = { url: rec.url, status: rec.status };
|
|
139
|
+
if (rec.metadata) out.metadata = rec.metadata;
|
|
140
|
+
if (typeof rec.markdown === "string") {
|
|
141
|
+
out.markdown =
|
|
142
|
+
rec.markdown.length > maxMd ? rec.markdown.slice(0, maxMd) + "\n…(truncated)" : rec.markdown;
|
|
143
|
+
}
|
|
144
|
+
if (typeof rec.html === "string") {
|
|
145
|
+
out.html = rec.html.length > maxMd ? rec.html.slice(0, maxMd) + "\n…(truncated)" : rec.html;
|
|
146
|
+
}
|
|
147
|
+
if (rec.json !== undefined) out.json = rec.json;
|
|
148
|
+
return out;
|
|
149
|
+
};
|
|
150
|
+
|
|
151
|
+
const formatCrawlJob = (job: CrawlJobResult, maxRecords = 50): string => {
|
|
152
|
+
const records = (job.records ?? []).slice(0, maxRecords).map((r) => truncateRecord(r));
|
|
153
|
+
return JSON.stringify(
|
|
154
|
+
{
|
|
155
|
+
id: job.id,
|
|
156
|
+
status: job.status,
|
|
157
|
+
total: job.total,
|
|
158
|
+
finished: job.finished,
|
|
159
|
+
browserSecondsUsed: job.browserSecondsUsed,
|
|
160
|
+
cursor: job.cursor,
|
|
161
|
+
recordCount: records.length,
|
|
162
|
+
records,
|
|
163
|
+
},
|
|
164
|
+
null,
|
|
165
|
+
2,
|
|
166
|
+
);
|
|
167
|
+
};
|
|
168
|
+
|
|
169
|
+
const crawlSchema = Type.Object({
|
|
170
|
+
url: Type.String({ minLength: 1, description: "Starting public http(s) URL to crawl" }),
|
|
171
|
+
limit: Type.Optional(Type.Number({ description: "Max pages to crawl (default 10)" })),
|
|
172
|
+
depth: Type.Optional(Type.Number({ description: "Max link depth from the start URL" })),
|
|
173
|
+
formats: Type.Optional(
|
|
174
|
+
Type.Array(Type.Union([Type.Literal("markdown"), Type.Literal("html"), Type.Literal("json")]), {
|
|
175
|
+
description: "Response formats (default [markdown])",
|
|
176
|
+
}),
|
|
177
|
+
),
|
|
178
|
+
render: Type.Optional(Type.Boolean({ description: "Headless Chrome (true) or fast HTML fetch (false)" })),
|
|
179
|
+
source: Type.Optional(
|
|
180
|
+
Type.Union([Type.Literal("all"), Type.Literal("sitemaps"), Type.Literal("links")], {
|
|
181
|
+
description: "URL discovery source (default all)",
|
|
182
|
+
}),
|
|
183
|
+
),
|
|
184
|
+
includePatterns: Type.Optional(Type.Array(Type.String(), { description: "Only visit matching URL patterns" })),
|
|
185
|
+
excludePatterns: Type.Optional(Type.Array(Type.String(), { description: "Skip matching URL patterns" })),
|
|
186
|
+
wait: Type.Optional(
|
|
187
|
+
Type.Boolean({
|
|
188
|
+
description:
|
|
189
|
+
"Poll until done (default true when limit≤20). If false, returns job id immediately.",
|
|
190
|
+
}),
|
|
191
|
+
),
|
|
192
|
+
});
|
|
193
|
+
|
|
194
|
+
const crawlStatusSchema = Type.Object({
|
|
195
|
+
jobId: Type.String({ minLength: 1, description: "Crawl job id from crawl" }),
|
|
196
|
+
limit: Type.Optional(Type.Number({ description: "Max records to return" })),
|
|
197
|
+
cursor: Type.Optional(Type.Union([Type.String(), Type.Number()], { description: "Pagination cursor" })),
|
|
198
|
+
status: Type.Optional(
|
|
199
|
+
Type.String({ description: "Filter records: queued|completed|disallowed|skipped|errored|cancelled" }),
|
|
200
|
+
),
|
|
201
|
+
});
|
|
202
|
+
|
|
203
|
+
pi.registerTool({
|
|
204
|
+
name: "crawl",
|
|
205
|
+
label: "Crawl",
|
|
206
|
+
description:
|
|
207
|
+
"Crawl a public site starting from a URL via Cloudflare Browser Run /crawl. " +
|
|
208
|
+
"Follows links up to limit/depth and returns markdown (default) per page. " +
|
|
209
|
+
"For small jobs (limit≤20) waits for completion by default; for larger jobs " +
|
|
210
|
+
"returns a job id — then use crawl_status.",
|
|
211
|
+
promptSnippet: "crawl a public site (multi-page markdown)",
|
|
212
|
+
promptGuidelines: [
|
|
213
|
+
"Use crawl when you need several pages from one site; use browse for a single URL.",
|
|
214
|
+
"Prefer small limits (≤20) so wait can finish in one call; use crawl_status for long jobs.",
|
|
215
|
+
],
|
|
216
|
+
parameters: crawlSchema,
|
|
217
|
+
executionMode: "sequential",
|
|
218
|
+
async execute(_toolCallId, params) {
|
|
219
|
+
const cfg = typeof config === "object" && "apiToken" in config ? config : null;
|
|
220
|
+
if (!cfg) return toolResult(`crawl unavailable: ${(config as { error: string }).error}`);
|
|
221
|
+
const limit = typeof params.limit === "number" ? params.limit : undefined;
|
|
222
|
+
const wait = typeof params.wait === "boolean" ? params.wait : defaultCrawlWait(limit);
|
|
223
|
+
const r = await crawl(cfg, {
|
|
224
|
+
url: params.url,
|
|
225
|
+
limit,
|
|
226
|
+
depth: typeof params.depth === "number" ? params.depth : undefined,
|
|
227
|
+
formats: (params.formats as CrawlFormat[] | undefined) ?? ["markdown"],
|
|
228
|
+
render: typeof params.render === "boolean" ? params.render : undefined,
|
|
229
|
+
source: params.source as "all" | "sitemaps" | "links" | undefined,
|
|
230
|
+
includePatterns: params.includePatterns as string[] | undefined,
|
|
231
|
+
excludePatterns: params.excludePatterns as string[] | undefined,
|
|
232
|
+
wait,
|
|
233
|
+
});
|
|
234
|
+
if (!r.ok) return toolResult(`crawl failed: ${"error" in r ? r.error : "unknown"}`);
|
|
235
|
+
if (r.job) return toolResult(formatCrawlJob(r.job));
|
|
236
|
+
return toolResult(
|
|
237
|
+
JSON.stringify(
|
|
238
|
+
{
|
|
239
|
+
jobId: r.jobId,
|
|
240
|
+
status: "running",
|
|
241
|
+
message: "Crawl started. Poll with crawl_status(jobId) until status is completed.",
|
|
242
|
+
},
|
|
243
|
+
null,
|
|
244
|
+
2,
|
|
245
|
+
),
|
|
246
|
+
);
|
|
247
|
+
},
|
|
248
|
+
});
|
|
249
|
+
|
|
250
|
+
pi.registerTool({
|
|
251
|
+
name: "crawl_status",
|
|
252
|
+
label: "Crawl status",
|
|
253
|
+
description:
|
|
254
|
+
"Check status or fetch results of a Cloudflare Browser Run crawl job by id. " +
|
|
255
|
+
"Supports limit/cursor pagination and status filters on records.",
|
|
256
|
+
promptSnippet: "poll crawl job status / results",
|
|
257
|
+
parameters: crawlStatusSchema,
|
|
258
|
+
executionMode: "sequential",
|
|
259
|
+
async execute(_toolCallId, params) {
|
|
260
|
+
const cfg = typeof config === "object" && "apiToken" in config ? config : null;
|
|
261
|
+
if (!cfg) return toolResult(`crawl_status unavailable: ${(config as { error: string }).error}`);
|
|
262
|
+
const r = await crawlStatus(cfg, {
|
|
263
|
+
jobId: params.jobId,
|
|
264
|
+
limit: typeof params.limit === "number" ? params.limit : undefined,
|
|
265
|
+
cursor: params.cursor as string | number | undefined,
|
|
266
|
+
status: typeof params.status === "string" ? params.status : undefined,
|
|
267
|
+
});
|
|
268
|
+
if (!r.ok) return toolResult(`crawl_status failed: ${"error" in r ? r.error : "unknown"}`);
|
|
269
|
+
if (r.job) return toolResult(formatCrawlJob(r.job));
|
|
270
|
+
return toolResult(JSON.stringify({ jobId: r.jobId }, null, 2));
|
|
271
|
+
},
|
|
272
|
+
});
|
|
124
273
|
}
|