@zenrows/mcp 2.1.2 → 2.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -7
- package/dist/auth/claim-hint.d.ts +16 -0
- package/dist/auth/claim-hint.js +71 -0
- package/dist/auth/ensure-key.d.ts +59 -0
- package/dist/auth/ensure-key.js +184 -0
- package/dist/batch-api.d.ts +89 -0
- package/dist/batch-api.js +182 -0
- package/dist/http.js +4 -4
- package/dist/index.js +30 -4
- package/dist/server.js +51 -15
- package/dist/tools/account.d.ts +18 -0
- package/dist/tools/account.js +98 -0
- package/dist/tools/batch.d.ts +2 -0
- package/dist/tools/batch.js +245 -0
- package/dist/tools/browser.js +194 -47
- package/dist/tools/extract.d.ts +40 -0
- package/dist/tools/extract.js +232 -0
- package/package.json +6 -3
|
@@ -0,0 +1,232 @@
|
|
|
1
|
+
import { createRequire } from "module";
|
|
2
|
+
import { z } from "zod";
|
|
3
|
+
import { appendClaimHint } from "../auth/claim-hint.js";
|
|
4
|
+
const require = createRequire(import.meta.url);
|
|
5
|
+
const pkg = require("../../package.json");
|
|
6
|
+
const ZENROWS_API_URL = "https://api.zenrows.com/v1/";
|
|
7
|
+
function err(text) {
|
|
8
|
+
return {
|
|
9
|
+
content: [{ type: "text", text: appendClaimHint(text, { body: text, message: text }) }],
|
|
10
|
+
isError: true,
|
|
11
|
+
};
|
|
12
|
+
}
|
|
13
|
+
function json(data) {
|
|
14
|
+
return { content: [{ type: "text", text: JSON.stringify(data) }] };
|
|
15
|
+
}
|
|
16
|
+
export function zrErrorCode(body) {
|
|
17
|
+
try {
|
|
18
|
+
const j = JSON.parse(body);
|
|
19
|
+
if (typeof j.code === "string")
|
|
20
|
+
return j.code;
|
|
21
|
+
const m = typeof j.error === "string" ? j.error.match(/\((AUTH\d+)\)/) : null;
|
|
22
|
+
return m?.[1];
|
|
23
|
+
}
|
|
24
|
+
catch {
|
|
25
|
+
return undefined;
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
function isEmptyData(data) {
|
|
29
|
+
if (data === null || data === undefined)
|
|
30
|
+
return true;
|
|
31
|
+
if (Array.isArray(data))
|
|
32
|
+
return data.length === 0;
|
|
33
|
+
if (typeof data === "object")
|
|
34
|
+
return Object.keys(data).length === 0;
|
|
35
|
+
if (typeof data === "string")
|
|
36
|
+
return data.trim() === "";
|
|
37
|
+
return false;
|
|
38
|
+
}
|
|
39
|
+
export function buildExtractParams(apiKey, url, mode, opts) {
|
|
40
|
+
const sp = new URLSearchParams({ apikey: apiKey, url });
|
|
41
|
+
if (mode === "auto")
|
|
42
|
+
sp.set("extract", "auto");
|
|
43
|
+
if (mode === "autoparse")
|
|
44
|
+
sp.set("autoparse", "true");
|
|
45
|
+
if (mode === "css" && opts.css_extractor)
|
|
46
|
+
sp.set("css_extractor", opts.css_extractor);
|
|
47
|
+
if (opts.mode_auto)
|
|
48
|
+
sp.set("mode", "auto");
|
|
49
|
+
if (opts.js_render)
|
|
50
|
+
sp.set("js_render", "true");
|
|
51
|
+
if (opts.premium_proxy)
|
|
52
|
+
sp.set("premium_proxy", "true");
|
|
53
|
+
if (opts.proxy_country)
|
|
54
|
+
sp.set("proxy_country", opts.proxy_country.toUpperCase());
|
|
55
|
+
if (opts.wait_for)
|
|
56
|
+
sp.set("wait_for", opts.wait_for);
|
|
57
|
+
if (opts.wait != null)
|
|
58
|
+
sp.set("wait", String(opts.wait));
|
|
59
|
+
return sp;
|
|
60
|
+
}
|
|
61
|
+
async function callZenrows(apiKey, searchParams, getClientName, fetchImpl) {
|
|
62
|
+
let response;
|
|
63
|
+
try {
|
|
64
|
+
response = await fetchImpl(`${ZENROWS_API_URL}?${searchParams}`, {
|
|
65
|
+
headers: {
|
|
66
|
+
"User-Agent": `zenrows/mcp ${pkg.version}`,
|
|
67
|
+
...(getClientName() ? { "x-mcp-client-name": getClientName() } : {}),
|
|
68
|
+
"x-mcp-tool": "extract",
|
|
69
|
+
},
|
|
70
|
+
});
|
|
71
|
+
}
|
|
72
|
+
catch (e) {
|
|
73
|
+
return {
|
|
74
|
+
ok: false,
|
|
75
|
+
status: 0,
|
|
76
|
+
body: `Network error contacting Zenrows: ${e instanceof Error ? e.message : String(e)}`,
|
|
77
|
+
};
|
|
78
|
+
}
|
|
79
|
+
return { ok: response.ok, status: response.status, body: await response.text() };
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* Core extract logic (testable). AUTH010 on mode=auto retries once with autoparse
|
|
83
|
+
* unless fallback_autoparse is false — same behavior as the CLI extract adapter.
|
|
84
|
+
*/
|
|
85
|
+
export async function runExtract(apiKey, params, options = {}) {
|
|
86
|
+
const getClientName = options.getClientName ?? (() => undefined);
|
|
87
|
+
const fetchImpl = options.fetchImpl ?? fetch;
|
|
88
|
+
const mode = params.mode ?? "auto";
|
|
89
|
+
if (mode === "css" && !params.css_extractor) {
|
|
90
|
+
return {
|
|
91
|
+
ok: false,
|
|
92
|
+
errorText: JSON.stringify({
|
|
93
|
+
code: "INVALID_USAGE",
|
|
94
|
+
message: "mode=css requires css_extractor JSON selector map.",
|
|
95
|
+
}),
|
|
96
|
+
};
|
|
97
|
+
}
|
|
98
|
+
const opts = {
|
|
99
|
+
css_extractor: params.css_extractor,
|
|
100
|
+
js_render: params.js_render,
|
|
101
|
+
premium_proxy: params.premium_proxy,
|
|
102
|
+
proxy_country: params.proxy_country,
|
|
103
|
+
mode_auto: params.mode_auto,
|
|
104
|
+
wait_for: params.wait_for,
|
|
105
|
+
wait: params.wait,
|
|
106
|
+
};
|
|
107
|
+
let usedMode = mode;
|
|
108
|
+
let result = await callZenrows(apiKey, buildExtractParams(apiKey, params.url, mode, opts), getClientName, fetchImpl);
|
|
109
|
+
let fellBackToAutoparse = false;
|
|
110
|
+
if (!result.ok &&
|
|
111
|
+
mode === "auto" &&
|
|
112
|
+
params.fallback_autoparse !== false &&
|
|
113
|
+
result.status === 402 &&
|
|
114
|
+
zrErrorCode(result.body) === "AUTH010") {
|
|
115
|
+
usedMode = "autoparse";
|
|
116
|
+
fellBackToAutoparse = true;
|
|
117
|
+
result = await callZenrows(apiKey, buildExtractParams(apiKey, params.url, "autoparse", opts), getClientName, fetchImpl);
|
|
118
|
+
}
|
|
119
|
+
if (!result.ok) {
|
|
120
|
+
return {
|
|
121
|
+
ok: false,
|
|
122
|
+
errorText: result.status === 0
|
|
123
|
+
? result.body
|
|
124
|
+
: JSON.stringify({
|
|
125
|
+
code: zrErrorCode(result.body) ?? "EXTRACT_FAILED",
|
|
126
|
+
message: `Zenrows error ${result.status}`,
|
|
127
|
+
detail: result.body.slice(0, 500),
|
|
128
|
+
mode: usedMode,
|
|
129
|
+
}),
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
let parsed;
|
|
133
|
+
try {
|
|
134
|
+
parsed = JSON.parse(result.body);
|
|
135
|
+
}
|
|
136
|
+
catch {
|
|
137
|
+
return {
|
|
138
|
+
ok: true,
|
|
139
|
+
mode: usedMode,
|
|
140
|
+
fellBackToAutoparse,
|
|
141
|
+
empty: true,
|
|
142
|
+
data: null,
|
|
143
|
+
raw: result.body,
|
|
144
|
+
};
|
|
145
|
+
}
|
|
146
|
+
let data = parsed;
|
|
147
|
+
let html;
|
|
148
|
+
if (usedMode === "auto" && parsed && typeof parsed === "object" && !Array.isArray(parsed)) {
|
|
149
|
+
const envelope = parsed;
|
|
150
|
+
if ("parsed" in envelope) {
|
|
151
|
+
data = envelope.parsed;
|
|
152
|
+
html = typeof envelope.html === "string" ? envelope.html : undefined;
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
return {
|
|
156
|
+
ok: true,
|
|
157
|
+
mode: usedMode,
|
|
158
|
+
fellBackToAutoparse,
|
|
159
|
+
empty: isEmptyData(data),
|
|
160
|
+
data,
|
|
161
|
+
...(html !== undefined ? { html } : {}),
|
|
162
|
+
};
|
|
163
|
+
}
|
|
164
|
+
export function registerExtractTool(server, apiKey, getClientName) {
|
|
165
|
+
server.registerTool("extract", {
|
|
166
|
+
annotations: {
|
|
167
|
+
title: "Extract Structured Data",
|
|
168
|
+
readOnlyHint: true,
|
|
169
|
+
destructiveHint: false,
|
|
170
|
+
},
|
|
171
|
+
description: `Extract structured data from a webpage via Zenrows.
|
|
172
|
+
|
|
173
|
+
Prefer this over scrape when you need JSON fields (products, articles, listings)
|
|
174
|
+
rather than a full page body.
|
|
175
|
+
Modes:
|
|
176
|
+
- auto (default): extract=auto — site-tailored Extract (open beta; currently free,
|
|
177
|
+
billing may apply later; may fall back to autoparse if the domain is not enabled)
|
|
178
|
+
- autoparse: general-purpose structured JSON on any domain
|
|
179
|
+
- css: css_extractor with an explicit selector map
|
|
180
|
+
|
|
181
|
+
Stealth: js_render, premium_proxy, proxy_country, or mode_auto (Adaptive Stealth Mode).
|
|
182
|
+
For full-page markdown/HTML/screenshots, use scrape instead.`,
|
|
183
|
+
inputSchema: {
|
|
184
|
+
url: z.string().url().describe("The webpage URL to extract from"),
|
|
185
|
+
mode: z
|
|
186
|
+
.enum(["auto", "autoparse", "css"])
|
|
187
|
+
.optional()
|
|
188
|
+
.default("auto")
|
|
189
|
+
.describe("Extraction mode: auto (extract=auto, default), autoparse, or css (requires css_extractor)"),
|
|
190
|
+
css_extractor: z
|
|
191
|
+
.string()
|
|
192
|
+
.optional()
|
|
193
|
+
.describe('Required when mode=css. JSON map of field→selector, e.g. \'{"title":"h1","price":".price"}\''),
|
|
194
|
+
js_render: z.boolean().optional().describe("Enable headless JS rendering (SPAs / dynamic content)"),
|
|
195
|
+
premium_proxy: z
|
|
196
|
+
.boolean()
|
|
197
|
+
.optional()
|
|
198
|
+
.describe("Use premium residential proxies (anti-bot). Higher credit cost."),
|
|
199
|
+
proxy_country: z
|
|
200
|
+
.string()
|
|
201
|
+
.optional()
|
|
202
|
+
.describe("ISO 3166-1 alpha-2 country code. Requires premium_proxy or mode_auto."),
|
|
203
|
+
mode_auto: z.boolean().optional().describe("Enable Adaptive Stealth Mode (mode=auto) for tougher sites"),
|
|
204
|
+
wait_for: z.string().optional().describe("CSS selector to wait for before extracting. Requires js_render."),
|
|
205
|
+
wait: z
|
|
206
|
+
.number()
|
|
207
|
+
.int()
|
|
208
|
+
.min(0)
|
|
209
|
+
.max(30000)
|
|
210
|
+
.optional()
|
|
211
|
+
.describe("Milliseconds to wait after load. Requires js_render."),
|
|
212
|
+
fallback_autoparse: z
|
|
213
|
+
.boolean()
|
|
214
|
+
.optional()
|
|
215
|
+
.default(true)
|
|
216
|
+
.describe("When mode=auto and the domain is not in Extract open beta (AUTH010), retry once with autoparse (default true)"),
|
|
217
|
+
},
|
|
218
|
+
}, async (params) => {
|
|
219
|
+
const outcome = await runExtract(apiKey, params, { getClientName });
|
|
220
|
+
if (!outcome.ok)
|
|
221
|
+
return err(outcome.errorText);
|
|
222
|
+
return json({
|
|
223
|
+
ok: true,
|
|
224
|
+
mode: outcome.mode,
|
|
225
|
+
fellBackToAutoparse: outcome.fellBackToAutoparse,
|
|
226
|
+
empty: outcome.empty,
|
|
227
|
+
data: outcome.data,
|
|
228
|
+
...(outcome.html !== undefined ? { html: outcome.html } : {}),
|
|
229
|
+
...(outcome.raw !== undefined ? { raw: outcome.raw } : {}),
|
|
230
|
+
});
|
|
231
|
+
});
|
|
232
|
+
}
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@zenrows/mcp",
|
|
3
|
-
"version": "2.
|
|
4
|
-
"description": "Zenrows MCP server — Fetch and Browser Sessions for AI coding assistants",
|
|
3
|
+
"version": "2.2.2",
|
|
4
|
+
"description": "Zenrows MCP server — Fetch, Extract, Batch, and Browser Sessions for AI coding assistants",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
7
7
|
"zenrows-mcp": "dist/index.js"
|
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
],
|
|
12
12
|
"scripts": {
|
|
13
13
|
"build": "tsc && chmod +x dist/index.js",
|
|
14
|
+
"check:server-json": "node scripts/sync-server-json-version.mjs --check",
|
|
14
15
|
"clean": "rm -rf dist",
|
|
15
16
|
"dev": "node --env-file=.env --import tsx src/index.ts",
|
|
16
17
|
"dev:http": "node --env-file=.env --import tsx src/http.ts",
|
|
@@ -20,8 +21,10 @@
|
|
|
20
21
|
"lint": "eslint src/**/*.ts",
|
|
21
22
|
"lint:fix": "eslint src/**/*.ts --fix",
|
|
22
23
|
"prepare": "npm run build",
|
|
23
|
-
"prepublishOnly": "npm run clean && npm run build && npm run typecheck && npm run lint",
|
|
24
|
+
"prepublishOnly": "npm run clean && npm run build && npm run typecheck && npm run lint && npm test && npm run check:server-json",
|
|
24
25
|
"publish-beta": "npm publish --tag beta",
|
|
26
|
+
"sync:server-json": "node scripts/sync-server-json-version.mjs",
|
|
27
|
+
"test": "node --import tsx --test tests/*.test.ts",
|
|
25
28
|
"typecheck": "tsc --noEmit"
|
|
26
29
|
},
|
|
27
30
|
"dependencies": {
|