@zenrows/mcp 2.1.2 → 2.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -7
- package/dist/auth/claim-hint.d.ts +16 -0
- package/dist/auth/claim-hint.js +71 -0
- package/dist/auth/ensure-key.d.ts +59 -0
- package/dist/auth/ensure-key.js +184 -0
- package/dist/batch-api.d.ts +89 -0
- package/dist/batch-api.js +182 -0
- package/dist/http.js +4 -4
- package/dist/index.js +30 -4
- package/dist/server.js +51 -15
- package/dist/tools/account.d.ts +18 -0
- package/dist/tools/account.js +98 -0
- package/dist/tools/batch.d.ts +2 -0
- package/dist/tools/batch.js +245 -0
- package/dist/tools/browser.js +194 -47
- package/dist/tools/extract.d.ts +40 -0
- package/dist/tools/extract.js +232 -0
- package/package.json +6 -3
package/dist/http.js
CHANGED
|
@@ -57,7 +57,7 @@ function mcpServerCard() {
|
|
|
57
57
|
$schema: "https://static.modelcontextprotocol.io/schemas/v1/server-card.schema.json",
|
|
58
58
|
name: "io.zenrows/mcp",
|
|
59
59
|
version: pkg.version,
|
|
60
|
-
description: "
|
|
60
|
+
description: "Zenrows MCP — scrape and extract from protected sites via Fetch (anti-bot bypass, JS rendering, proxies).",
|
|
61
61
|
websiteUrl: "https://www.zenrows.com/mcp",
|
|
62
62
|
remotes: [
|
|
63
63
|
{
|
|
@@ -71,9 +71,9 @@ function mcpServerCard() {
|
|
|
71
71
|
// Older SEP-1649-shaped fields some scanners still expect.
|
|
72
72
|
protocolVersion: "2025-06-18",
|
|
73
73
|
serverInfo: {
|
|
74
|
-
name: "
|
|
74
|
+
name: "Zenrows",
|
|
75
75
|
version: pkg.version,
|
|
76
|
-
description: "
|
|
76
|
+
description: "Zenrows Fetch API via MCP",
|
|
77
77
|
homepage: "https://www.zenrows.com/mcp",
|
|
78
78
|
},
|
|
79
79
|
transport: {
|
|
@@ -138,7 +138,7 @@ app.all("/mcp", async (c) => {
|
|
|
138
138
|
"WWW-Authenticate": `Bearer realm="${AUTH_SERVER}", resource_metadata="${MCP_SERVER}/.well-known/oauth-protected-resource"`,
|
|
139
139
|
// CloudFront strips WWW-Authenticate — add Link header as RFC 8615 fallback
|
|
140
140
|
// so MCP clients can still discover the OAuth server
|
|
141
|
-
|
|
141
|
+
Link: `<${MCP_SERVER}/.well-known/oauth-protected-resource>; rel="oauth-protected-resource"`,
|
|
142
142
|
});
|
|
143
143
|
}
|
|
144
144
|
const transport = new WebStandardStreamableHTTPServerTransport({
|
package/dist/index.js
CHANGED
|
@@ -1,12 +1,38 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
+
import { createRequire } from "module";
|
|
2
3
|
import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
|
|
4
|
+
import { AuthError, ensureApiKey, getZenrowsDir, resolveApiKey } from "./auth/ensure-key.js";
|
|
3
5
|
import { createServer } from "./server.js";
|
|
4
|
-
const
|
|
5
|
-
|
|
6
|
-
|
|
6
|
+
const require = createRequire(import.meta.url);
|
|
7
|
+
const pkg = require("../package.json");
|
|
8
|
+
let apiKey;
|
|
9
|
+
try {
|
|
10
|
+
const existing = resolveApiKey();
|
|
11
|
+
if (existing.key) {
|
|
12
|
+
process.stderr.write(`Using existing API key from ${existing.source} (secrets dir: ${getZenrowsDir()})\n`);
|
|
13
|
+
}
|
|
14
|
+
else {
|
|
15
|
+
const signup = process.env.ZENROWS_AGENT_SIGNUP_URL?.trim() || "https://app.zenrows.com/api/agent/signup (default prod)";
|
|
16
|
+
process.stderr.write(`No API key — will auto-signup via: ${signup}\n`);
|
|
17
|
+
}
|
|
18
|
+
const resolved = await ensureApiKey({
|
|
19
|
+
userAgent: `zenrows/mcp ${pkg.version}`,
|
|
20
|
+
onProvision: (acct) => {
|
|
21
|
+
process.stderr.write(`Created a Zenrows Free plan account.\n` +
|
|
22
|
+
`Claim it anytime (keeps your usage): ${acct.claimUrl}\n` +
|
|
23
|
+
`Key stored in ${getZenrowsDir()}/secrets.json\n`);
|
|
24
|
+
},
|
|
25
|
+
});
|
|
26
|
+
apiKey = resolved.apiKey;
|
|
27
|
+
}
|
|
28
|
+
catch (err) {
|
|
29
|
+
const msg = err instanceof AuthError
|
|
30
|
+
? `Error: ${err.message}\n`
|
|
31
|
+
: `Error: ${err instanceof Error ? err.message : String(err)}\n`;
|
|
32
|
+
process.stderr.write(msg);
|
|
7
33
|
process.exit(1);
|
|
8
34
|
}
|
|
9
35
|
const server = createServer(apiKey);
|
|
10
36
|
const transport = new StdioServerTransport();
|
|
11
37
|
await server.connect(transport);
|
|
12
|
-
process.stderr.write(
|
|
38
|
+
process.stderr.write(`Zenrows MCP server running on stdio (secrets dir: ${getZenrowsDir()})\n`);
|
package/dist/server.js
CHANGED
|
@@ -1,7 +1,12 @@
|
|
|
1
1
|
import { createRequire } from "module";
|
|
2
2
|
import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
|
|
3
3
|
import { z } from "zod";
|
|
4
|
+
import { appendClaimHint } from "./auth/claim-hint.js";
|
|
5
|
+
import { getZenrowsDir, readAccount } from "./auth/ensure-key.js";
|
|
6
|
+
import { registerAccountTools } from "./tools/account.js";
|
|
7
|
+
import { registerBatchTools } from "./tools/batch.js";
|
|
4
8
|
import { registerBrowserTools } from "./tools/browser.js";
|
|
9
|
+
import { registerExtractTool } from "./tools/extract.js";
|
|
5
10
|
const require = createRequire(import.meta.url);
|
|
6
11
|
const pkg = require("../package.json");
|
|
7
12
|
const ZENROWS_API_URL = "https://api.zenrows.com/v1/";
|
|
@@ -23,24 +28,22 @@ export function createServer(apiKey, clientName) {
|
|
|
23
28
|
readOnlyHint: true,
|
|
24
29
|
destructiveHint: false,
|
|
25
30
|
},
|
|
26
|
-
description: `Scrape any webpage and return its content using Zenrows.
|
|
31
|
+
description: `Scrape any webpage and return its content using Zenrows (Fetch).
|
|
27
32
|
|
|
28
|
-
Use
|
|
29
|
-
|
|
33
|
+
Use for full-page content (markdown/HTML/PDF/screenshot). For structured JSON
|
|
34
|
+
fields (products, articles, listings), prefer the extract tool when it fits —
|
|
35
|
+
it returns parsed fields instead of a full page body.
|
|
30
36
|
|
|
31
37
|
When to enable options:
|
|
32
38
|
- js_render: page uses React/Vue/Angular, loads content dynamically, or content
|
|
33
39
|
appears missing on the first attempt
|
|
34
40
|
- premium_proxy: site returns 403/blocked errors even with js_render enabled
|
|
35
41
|
- wait_for: specific content loads after initial render (requires js_render)
|
|
36
|
-
- css_extractor: you only need specific elements, not the whole page
|
|
37
|
-
- autoparse: structured data pages like products or articles
|
|
38
42
|
|
|
39
43
|
Examples:
|
|
40
44
|
Basic: { url: "https://example.com" }
|
|
41
45
|
Dynamic: { url: "https://spa.com", js_render: true }
|
|
42
|
-
Protected:{ url: "https://protected.com", js_render: true, premium_proxy: true }
|
|
43
|
-
Extract: { url: "https://shop.com", css_extractor: '{"title":"h1","price":".price"}' }`,
|
|
46
|
+
Protected:{ url: "https://protected.com", js_render: true, premium_proxy: true }`,
|
|
44
47
|
inputSchema: {
|
|
45
48
|
url: z.string().url().describe("The webpage URL to scrape"),
|
|
46
49
|
js_render: z
|
|
@@ -157,11 +160,7 @@ Examples:
|
|
|
157
160
|
// 'html' is the Zenrows default (no param); all other values are passed through.
|
|
158
161
|
const isScreenshot = params.screenshot || params.screenshot_fullpage || params.screenshot_selector;
|
|
159
162
|
const effectiveType = params.response_type ?? DEFAULT_RESPONSE_TYPE;
|
|
160
|
-
if (!params.autoparse &&
|
|
161
|
-
!params.css_extractor &&
|
|
162
|
-
!params.outputs &&
|
|
163
|
-
!isScreenshot &&
|
|
164
|
-
effectiveType !== "html") {
|
|
163
|
+
if (!params.autoparse && !params.css_extractor && !params.outputs && !isScreenshot && effectiveType !== "html") {
|
|
165
164
|
searchParams.set("response_type", effectiveType);
|
|
166
165
|
}
|
|
167
166
|
let response;
|
|
@@ -187,8 +186,12 @@ Examples:
|
|
|
187
186
|
}
|
|
188
187
|
if (!response.ok) {
|
|
189
188
|
const body = await response.text();
|
|
189
|
+
const text = appendClaimHint(`Zenrows error ${response.status}: ${body}`, {
|
|
190
|
+
status: response.status,
|
|
191
|
+
body,
|
|
192
|
+
});
|
|
190
193
|
return {
|
|
191
|
-
content: [{ type: "text", text
|
|
194
|
+
content: [{ type: "text", text }],
|
|
192
195
|
isError: true,
|
|
193
196
|
};
|
|
194
197
|
}
|
|
@@ -232,7 +235,7 @@ Examples:
|
|
|
232
235
|
}));
|
|
233
236
|
server.registerPrompt("extract_structured_data", {
|
|
234
237
|
title: "Extract Structured Data",
|
|
235
|
-
description: "
|
|
238
|
+
description: "Extract specific structured data from a webpage using CSS selectors.",
|
|
236
239
|
argsSchema: {
|
|
237
240
|
url: z.string().url().describe("The webpage URL to extract data from"),
|
|
238
241
|
fields: z
|
|
@@ -245,7 +248,7 @@ Examples:
|
|
|
245
248
|
role: "user",
|
|
246
249
|
content: {
|
|
247
250
|
type: "text",
|
|
248
|
-
text: `
|
|
251
|
+
text: `Use the Zenrows MCP extract tool on ${url} with mode=css and css_extractor set to ${fields}. Return the extracted data as a clean JSON object.`,
|
|
249
252
|
},
|
|
250
253
|
},
|
|
251
254
|
],
|
|
@@ -267,7 +270,40 @@ Examples:
|
|
|
267
270
|
},
|
|
268
271
|
],
|
|
269
272
|
}));
|
|
273
|
+
registerExtractTool(server, apiKey, getClientName);
|
|
274
|
+
registerBatchTools(server, apiKey);
|
|
275
|
+
registerAccountTools(server, apiKey);
|
|
270
276
|
const BROWSER_URL = process.env.ZENROWS_BROWSER_URL ?? "https://mcp.zenrows.com";
|
|
271
277
|
registerBrowserTools(server, apiKey, BROWSER_URL, getClientName);
|
|
278
|
+
// Always expose account resource; handler re-reads disk so ZENROWS_HOME is visible.
|
|
279
|
+
server.registerResource("zenrows-account", "zenrows://account", {
|
|
280
|
+
description: "Local Zenrows agent account metadata (claim URL for unclaimed Free plans). Re-reads ~/.zenrows or $ZENROWS_HOME on each read.",
|
|
281
|
+
mimeType: "application/json",
|
|
282
|
+
}, async () => {
|
|
283
|
+
const acct = readAccount();
|
|
284
|
+
const home = getZenrowsDir();
|
|
285
|
+
const body = acct
|
|
286
|
+
? {
|
|
287
|
+
accountId: acct.accountId,
|
|
288
|
+
unclaimed: acct.unclaimed,
|
|
289
|
+
claimUrl: acct.claimUrl,
|
|
290
|
+
createdAt: acct.createdAt,
|
|
291
|
+
zenrowsHome: home,
|
|
292
|
+
}
|
|
293
|
+
: {
|
|
294
|
+
unclaimed: false,
|
|
295
|
+
message: "No local agent account file (key from env or missing).",
|
|
296
|
+
zenrowsHome: home,
|
|
297
|
+
};
|
|
298
|
+
return {
|
|
299
|
+
contents: [
|
|
300
|
+
{
|
|
301
|
+
uri: "zenrows://account",
|
|
302
|
+
mimeType: "application/json",
|
|
303
|
+
text: JSON.stringify(body, null, 2),
|
|
304
|
+
},
|
|
305
|
+
],
|
|
306
|
+
};
|
|
307
|
+
});
|
|
272
308
|
return server;
|
|
273
309
|
}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
|
|
2
|
+
type TextContent = {
|
|
3
|
+
type: "text";
|
|
4
|
+
text: string;
|
|
5
|
+
};
|
|
6
|
+
export type AccountOpts = {
|
|
7
|
+
fetchImpl?: typeof fetch;
|
|
8
|
+
};
|
|
9
|
+
/**
|
|
10
|
+
* Read the plan's usage. Split out from the handler so it can be exercised without an
|
|
11
|
+
* MCP server or a live account — the endpoint is the one thing here we cannot try
|
|
12
|
+
* against production from a test.
|
|
13
|
+
*/
|
|
14
|
+
export declare function runAccountUsage(apiKey: string, opts?: AccountOpts): Promise<{
|
|
15
|
+
content: TextContent[];
|
|
16
|
+
}>;
|
|
17
|
+
export declare function registerAccountTools(server: McpServer, apiKey: string, opts?: AccountOpts): void;
|
|
18
|
+
export {};
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
import { createRequire } from "module";
|
|
2
|
+
import { appendClaimHint } from "../auth/claim-hint.js";
|
|
3
|
+
const require = createRequire(import.meta.url);
|
|
4
|
+
const pkg = require("../../package.json");
|
|
5
|
+
/**
|
|
6
|
+
* Let an agent read its own allowance before it runs out of it.
|
|
7
|
+
*
|
|
8
|
+
* Until now nothing on this server could answer "how many credits do I have left?".
|
|
9
|
+
* Responses carry `X-Request-Cost` and `X-Request-Credits` — what a call *cost*, after
|
|
10
|
+
* the fact — but no counterpart to `Concurrency-Limit` / `Concurrency-Remaining`, so an
|
|
11
|
+
* agent could total its own spend and still not know the ceiling. It found out by
|
|
12
|
+
* hitting a 402 telling it to buy a subscription (ACT-1581, ACT-1577).
|
|
13
|
+
*
|
|
14
|
+
* `/v1/subscriptions/self/details` has always had the answer. It does not count against
|
|
15
|
+
* concurrency, which is what makes it safe to call before a batch or on a retry.
|
|
16
|
+
*/
|
|
17
|
+
const SUBSCRIPTION_DETAILS_URL = "https://api.zenrows.com/v1/subscriptions/self/details";
|
|
18
|
+
function err(text, opts = {}) {
|
|
19
|
+
return {
|
|
20
|
+
content: [
|
|
21
|
+
{
|
|
22
|
+
type: "text",
|
|
23
|
+
text: appendClaimHint(text, {
|
|
24
|
+
status: opts.status,
|
|
25
|
+
body: opts.body ?? text,
|
|
26
|
+
message: text,
|
|
27
|
+
}),
|
|
28
|
+
},
|
|
29
|
+
],
|
|
30
|
+
isError: true,
|
|
31
|
+
};
|
|
32
|
+
}
|
|
33
|
+
function json(data) {
|
|
34
|
+
return { content: [{ type: "text", text: JSON.stringify(data) }] };
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* Read the plan's usage. Split out from the handler so it can be exercised without an
|
|
38
|
+
* MCP server or a live account — the endpoint is the one thing here we cannot try
|
|
39
|
+
* against production from a test.
|
|
40
|
+
*/
|
|
41
|
+
export async function runAccountUsage(apiKey, opts = {}) {
|
|
42
|
+
const doFetch = opts.fetchImpl ?? fetch;
|
|
43
|
+
let res;
|
|
44
|
+
try {
|
|
45
|
+
res = await doFetch(SUBSCRIPTION_DETAILS_URL, {
|
|
46
|
+
headers: {
|
|
47
|
+
"X-API-Key": apiKey,
|
|
48
|
+
"User-Agent": `zenrows-mcp/${pkg.version}`,
|
|
49
|
+
},
|
|
50
|
+
});
|
|
51
|
+
}
|
|
52
|
+
catch (e) {
|
|
53
|
+
return err(`Could not reach the Zenrows subscription endpoint: ${e.message}`);
|
|
54
|
+
}
|
|
55
|
+
const body = await res.text();
|
|
56
|
+
if (!res.ok) {
|
|
57
|
+
return err(`Zenrows returned ${res.status} for the subscription details endpoint.\n${body}`, {
|
|
58
|
+
status: res.status,
|
|
59
|
+
body,
|
|
60
|
+
});
|
|
61
|
+
}
|
|
62
|
+
// Passed through verbatim. The response shape is not part of any documented contract,
|
|
63
|
+
// so reshaping it here would mean inventing field names that could drift away from
|
|
64
|
+
// what the API actually sends.
|
|
65
|
+
try {
|
|
66
|
+
return json(JSON.parse(body));
|
|
67
|
+
}
|
|
68
|
+
catch {
|
|
69
|
+
return json({ raw: body });
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
export function registerAccountTools(server, apiKey, opts = {}) {
|
|
73
|
+
server.registerTool("account_usage", {
|
|
74
|
+
annotations: {
|
|
75
|
+
title: "Check Credit Usage",
|
|
76
|
+
readOnlyHint: true,
|
|
77
|
+
destructiveHint: false,
|
|
78
|
+
openWorldHint: false,
|
|
79
|
+
},
|
|
80
|
+
description: `Read the current plan's credit allowance and how much of it is spent.
|
|
81
|
+
|
|
82
|
+
Call this BEFORE a large batch, and after any 402 / AUTH004, to find out whether the
|
|
83
|
+
account is out of credits and when the allowance renews. It is free and does not consume
|
|
84
|
+
a concurrency slot, so it is safe to poll between runs.
|
|
85
|
+
|
|
86
|
+
Credit costs per request: 1 basic, 5 js_render, 10 premium_proxy, 25 both. On a small
|
|
87
|
+
plan a few hundred protected requests can exhaust a month, so check before fanning out.
|
|
88
|
+
|
|
89
|
+
AUTH004 ("usage exceeded") means this allowance is spent. It renews at the end of the
|
|
90
|
+
billing period, so it is not a permanent block: never retry-loop against it. If the
|
|
91
|
+
human does not want to wait for the renewal, relay the way to continue now: add a
|
|
92
|
+
credit pack at https://app.zenrows.com/billing?topup=open (opens the purchase
|
|
93
|
+
directly) or upgrade at https://app.zenrows.com/plans. Prices are per plan; quote them
|
|
94
|
+
only from this tool's response, never from memory.
|
|
95
|
+
AUTH006 is the concurrency limit, which is a different thing entirely.`,
|
|
96
|
+
inputSchema: {},
|
|
97
|
+
}, async () => runAccountUsage(apiKey, opts));
|
|
98
|
+
}
|
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
import { createRequire } from "module";
|
|
2
|
+
import { z } from "zod";
|
|
3
|
+
import { appendClaimHint } from "../auth/claim-hint.js";
|
|
4
|
+
import { BatchError, createJob, getJob, listResults, stopJob, waitForJob } from "../batch-api.js";
|
|
5
|
+
const require = createRequire(import.meta.url);
|
|
6
|
+
const pkg = require("../../package.json");
|
|
7
|
+
function err(data, hint) {
|
|
8
|
+
const raw = typeof data === "string" ? data : JSON.stringify(data);
|
|
9
|
+
return {
|
|
10
|
+
content: [
|
|
11
|
+
{
|
|
12
|
+
type: "text",
|
|
13
|
+
text: appendClaimHint(raw, {
|
|
14
|
+
status: hint?.status,
|
|
15
|
+
code: hint?.code,
|
|
16
|
+
message: hint?.message ?? raw,
|
|
17
|
+
body: raw,
|
|
18
|
+
}),
|
|
19
|
+
},
|
|
20
|
+
],
|
|
21
|
+
isError: true,
|
|
22
|
+
};
|
|
23
|
+
}
|
|
24
|
+
function json(data) {
|
|
25
|
+
return { content: [{ type: "text", text: JSON.stringify(data) }] };
|
|
26
|
+
}
|
|
27
|
+
function batchErr(e) {
|
|
28
|
+
if (e instanceof BatchError) {
|
|
29
|
+
return err(e.toJSON(), { status: e.status, code: e.code, message: e.message });
|
|
30
|
+
}
|
|
31
|
+
return err({
|
|
32
|
+
code: "BATCH_FAILED",
|
|
33
|
+
message: e instanceof Error ? e.message : String(e),
|
|
34
|
+
});
|
|
35
|
+
}
|
|
36
|
+
function normalizeParams(obj) {
|
|
37
|
+
const out = {};
|
|
38
|
+
for (const [k, v] of Object.entries(obj)) {
|
|
39
|
+
if (v === undefined || v === null)
|
|
40
|
+
continue;
|
|
41
|
+
if (typeof v === "string")
|
|
42
|
+
out[k] = v;
|
|
43
|
+
else if (typeof v === "boolean" || typeof v === "number")
|
|
44
|
+
out[k] = String(v);
|
|
45
|
+
else
|
|
46
|
+
out[k] = JSON.stringify(v);
|
|
47
|
+
}
|
|
48
|
+
return out;
|
|
49
|
+
}
|
|
50
|
+
const taskSchema = z.object({
|
|
51
|
+
url: z.string().url().describe("Target URL for this task"),
|
|
52
|
+
external_id: z.string().optional().describe("Optional stable id echoed back on results"),
|
|
53
|
+
metadata: z.unknown().optional().describe("Opaque per-task metadata carried through to results"),
|
|
54
|
+
zenrows_params: z
|
|
55
|
+
.record(z.union([z.string(), z.number(), z.boolean()]))
|
|
56
|
+
.optional()
|
|
57
|
+
.describe("Per-task Zenrows scrape params (js_render, premium_proxy, extract, autoparse, …)"),
|
|
58
|
+
});
|
|
59
|
+
export function registerBatchTools(server, apiKey) {
|
|
60
|
+
const ua = `zenrows/mcp ${pkg.version}`;
|
|
61
|
+
const call = { apiKey, userAgent: ua };
|
|
62
|
+
server.registerTool("batch_create", {
|
|
63
|
+
annotations: { title: "Create Batch Job", readOnlyHint: false, destructiveHint: false },
|
|
64
|
+
description: `Submit a cloud Batch job that fans out many URLs asynchronously (Zenrows Batch API beta).
|
|
65
|
+
|
|
66
|
+
NOT the same as browser_batch — this hits https://async.api.zenrows.com/v1 with X-API-Key.
|
|
67
|
+
Use for large URL lists; prefer scrape/extract for one-off pages.
|
|
68
|
+
|
|
69
|
+
Returns job_id + latest_run.status/stats. Poll with batch_status / batch_wait, then batch_results.
|
|
70
|
+
If you get BATCH_ACCESS_DENIED, the account lacks Batch beta access.`,
|
|
71
|
+
inputSchema: {
|
|
72
|
+
tasks: z
|
|
73
|
+
.array(taskSchema)
|
|
74
|
+
.optional()
|
|
75
|
+
.describe("List of tasks (each needs a url). Prefer this over urls when you need per-task params."),
|
|
76
|
+
urls: z
|
|
77
|
+
.array(z.string().url())
|
|
78
|
+
.optional()
|
|
79
|
+
.describe("Shorthand: list of URLs (converted to tasks). Ignored when tasks is provided."),
|
|
80
|
+
js_render: z.boolean().optional().describe("Job-level js_render for all tasks"),
|
|
81
|
+
premium_proxy: z.boolean().optional().describe("Job-level premium_proxy for all tasks"),
|
|
82
|
+
proxy_country: z
|
|
83
|
+
.string()
|
|
84
|
+
.optional()
|
|
85
|
+
.describe("Job-level ISO country code (requires premium_proxy or mode=auto)"),
|
|
86
|
+
response_type: z.enum(["markdown", "plaintext", "html", "pdf"]).optional().describe("Job-level response_type"),
|
|
87
|
+
zenrows_params: z
|
|
88
|
+
.record(z.union([z.string(), z.number(), z.boolean()]))
|
|
89
|
+
.optional()
|
|
90
|
+
.describe("Additional job-level zenrows_params merged with the flags above"),
|
|
91
|
+
wait: z.boolean().optional().describe("If true, poll until the job reaches a terminal state before returning"),
|
|
92
|
+
wait_timeout_ms: z
|
|
93
|
+
.number()
|
|
94
|
+
.int()
|
|
95
|
+
.min(1000)
|
|
96
|
+
.max(3_600_000)
|
|
97
|
+
.optional()
|
|
98
|
+
.describe("Max wait time when wait=true (default 600000)"),
|
|
99
|
+
},
|
|
100
|
+
}, async (params) => {
|
|
101
|
+
const tasksIn = params.tasks && params.tasks.length > 0 ? params.tasks : (params.urls ?? []).map((url) => ({ url }));
|
|
102
|
+
if (!tasksIn.length) {
|
|
103
|
+
return err({
|
|
104
|
+
code: "INVALID_USAGE",
|
|
105
|
+
message: "Provide tasks (preferred) or urls with at least one URL.",
|
|
106
|
+
});
|
|
107
|
+
}
|
|
108
|
+
const jobParams = { ...(params.zenrows_params ?? {}) };
|
|
109
|
+
if (params.js_render)
|
|
110
|
+
jobParams.js_render = true;
|
|
111
|
+
if (params.premium_proxy)
|
|
112
|
+
jobParams.premium_proxy = true;
|
|
113
|
+
if (params.proxy_country)
|
|
114
|
+
jobParams.proxy_country = params.proxy_country.toLowerCase();
|
|
115
|
+
if (params.response_type)
|
|
116
|
+
jobParams.response_type = params.response_type;
|
|
117
|
+
const body = {
|
|
118
|
+
type: "regular",
|
|
119
|
+
status: "closed",
|
|
120
|
+
tasks: tasksIn.map((t) => {
|
|
121
|
+
const task = { url: t.url };
|
|
122
|
+
if (t.external_id)
|
|
123
|
+
task.external_id = t.external_id;
|
|
124
|
+
if (t.metadata !== undefined)
|
|
125
|
+
task.metadata = t.metadata;
|
|
126
|
+
if (t.zenrows_params)
|
|
127
|
+
task.zenrows_params = normalizeParams(t.zenrows_params);
|
|
128
|
+
return task;
|
|
129
|
+
}),
|
|
130
|
+
...(Object.keys(jobParams).length ? { zenrows_params: normalizeParams(jobParams) } : {}),
|
|
131
|
+
};
|
|
132
|
+
try {
|
|
133
|
+
const job = await createJob(body, call);
|
|
134
|
+
const finished = params.wait === true
|
|
135
|
+
? await waitForJob(job.job_id, {
|
|
136
|
+
...call,
|
|
137
|
+
pollTimeoutMs: params.wait_timeout_ms ?? 600_000,
|
|
138
|
+
})
|
|
139
|
+
: job;
|
|
140
|
+
const run = finished.latest_run ?? {};
|
|
141
|
+
return json({
|
|
142
|
+
ok: true,
|
|
143
|
+
job_id: finished.job_id,
|
|
144
|
+
status: run.status,
|
|
145
|
+
stats: run.stats,
|
|
146
|
+
job: finished,
|
|
147
|
+
});
|
|
148
|
+
}
|
|
149
|
+
catch (e) {
|
|
150
|
+
return batchErr(e);
|
|
151
|
+
}
|
|
152
|
+
});
|
|
153
|
+
server.registerTool("batch_status", {
|
|
154
|
+
annotations: { title: "Batch Job Status", readOnlyHint: true, destructiveHint: false },
|
|
155
|
+
description: "Get status and stats for a Zenrows Batch job (latest_run.status + latest_run.stats).",
|
|
156
|
+
inputSchema: {
|
|
157
|
+
job_id: z.string().describe("Batch job id returned by batch_create"),
|
|
158
|
+
},
|
|
159
|
+
}, async ({ job_id }) => {
|
|
160
|
+
try {
|
|
161
|
+
const job = await getJob(job_id, call);
|
|
162
|
+
const run = job.latest_run ?? {};
|
|
163
|
+
return json({
|
|
164
|
+
ok: true,
|
|
165
|
+
job_id: job.job_id,
|
|
166
|
+
status: run.status,
|
|
167
|
+
stats: run.stats,
|
|
168
|
+
job,
|
|
169
|
+
});
|
|
170
|
+
}
|
|
171
|
+
catch (e) {
|
|
172
|
+
return batchErr(e);
|
|
173
|
+
}
|
|
174
|
+
});
|
|
175
|
+
server.registerTool("batch_results", {
|
|
176
|
+
annotations: { title: "Batch Job Results", readOnlyHint: true, destructiveHint: false },
|
|
177
|
+
description: `List result rows for a Batch job (cursor-paginated server-side; returns the full list).
|
|
178
|
+
|
|
179
|
+
Each row may include task_id, external_id, status, and a short-lived result_url for the body.
|
|
180
|
+
Download result_url soon — presigned links expire.`,
|
|
181
|
+
inputSchema: {
|
|
182
|
+
job_id: z.string().describe("Batch job id"),
|
|
183
|
+
status: z.enum(["successful", "failed", "all"]).optional().describe("Filter results by status (default: all)"),
|
|
184
|
+
},
|
|
185
|
+
}, async ({ job_id, status }) => {
|
|
186
|
+
try {
|
|
187
|
+
const results = await listResults(job_id, { ...call, status });
|
|
188
|
+
return json({ ok: true, job_id, count: results.length, results });
|
|
189
|
+
}
|
|
190
|
+
catch (e) {
|
|
191
|
+
return batchErr(e);
|
|
192
|
+
}
|
|
193
|
+
});
|
|
194
|
+
server.registerTool("batch_cancel", {
|
|
195
|
+
annotations: { title: "Cancel Batch Job", readOnlyHint: false, destructiveHint: true },
|
|
196
|
+
description: "Stop an in-flight Batch job run (POST /jobs/:id/stop).",
|
|
197
|
+
inputSchema: {
|
|
198
|
+
job_id: z.string().describe("Batch job id to stop"),
|
|
199
|
+
},
|
|
200
|
+
}, async ({ job_id }) => {
|
|
201
|
+
try {
|
|
202
|
+
const job = await stopJob(job_id, call);
|
|
203
|
+
const run = job.latest_run ?? {};
|
|
204
|
+
return json({
|
|
205
|
+
ok: true,
|
|
206
|
+
job_id: job.job_id,
|
|
207
|
+
status: run.status,
|
|
208
|
+
stats: run.stats,
|
|
209
|
+
job,
|
|
210
|
+
});
|
|
211
|
+
}
|
|
212
|
+
catch (e) {
|
|
213
|
+
return batchErr(e);
|
|
214
|
+
}
|
|
215
|
+
});
|
|
216
|
+
server.registerTool("batch_wait", {
|
|
217
|
+
annotations: { title: "Wait for Batch Job", readOnlyHint: true, destructiveHint: false },
|
|
218
|
+
description: "Poll batch_status until the job reaches a terminal state (completed, stopped, or deleted).",
|
|
219
|
+
inputSchema: {
|
|
220
|
+
job_id: z.string().describe("Batch job id"),
|
|
221
|
+
timeout_ms: z
|
|
222
|
+
.number()
|
|
223
|
+
.int()
|
|
224
|
+
.min(1000)
|
|
225
|
+
.max(3_600_000)
|
|
226
|
+
.optional()
|
|
227
|
+
.describe("Max wait time in ms (default 600000)"),
|
|
228
|
+
},
|
|
229
|
+
}, async ({ job_id, timeout_ms }) => {
|
|
230
|
+
try {
|
|
231
|
+
const job = await waitForJob(job_id, { ...call, pollTimeoutMs: timeout_ms ?? 600_000 });
|
|
232
|
+
const run = job.latest_run ?? {};
|
|
233
|
+
return json({
|
|
234
|
+
ok: true,
|
|
235
|
+
job_id: job.job_id,
|
|
236
|
+
status: run.status,
|
|
237
|
+
stats: run.stats,
|
|
238
|
+
job,
|
|
239
|
+
});
|
|
240
|
+
}
|
|
241
|
+
catch (e) {
|
|
242
|
+
return batchErr(e);
|
|
243
|
+
}
|
|
244
|
+
});
|
|
245
|
+
}
|