scrapeless-mcp-server 0.6.0 → 0.6.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +29 -28
- package/build/config.js +4 -1
- package/build/index.js +1 -1
- package/build/tools/ai_scraper/aiScraper.js +634 -0
- package/build/tools/ai_scraper/api.js +27 -0
- package/build/tools/crawl/api.js +2 -2
- package/build/tools/index.js +1 -1
- package/package.json +5 -3
package/README.md
CHANGED
|
@@ -10,6 +10,7 @@ Built on the open MCP standard, Scrapeless MCP Server seamlessly connects models
|
|
|
10
10
|
- **Browser automation** for page-level navigation and interaction
|
|
11
11
|
- **Scrape** dynamic, JS-heavy sites—export as HTML, Markdown, or screenshots
|
|
12
12
|
- **Crawl** entire websites by following links and capture each page in multiple formats
|
|
13
|
+
- **AI Scraper** Create an AI Scraper task for ChatGPT, Gemini, Perplexity, Copilot, Google AI Mode, Google AI Overview, Grok, or Alexa
|
|
13
14
|
|
|
14
15
|
Whether you're building an AI research assistant, a coding copilot, or autonomous web agents, this server provides the dynamic context and real-world data your workflows need—**without getting blocked**.
|
|
15
16
|
|
|
@@ -72,7 +73,7 @@ Scrapeless MCP Server supports both **Stdio** and **Streamable HTTP** transport
|
|
|
72
73
|
"command": "npx",
|
|
73
74
|
"args": ["-y", "scrapeless-mcp-server"],
|
|
74
75
|
"env": {
|
|
75
|
-
"
|
|
76
|
+
"SCRAPELESS_API_KEY": "YOUR_SCRAPELESS_KEY"
|
|
76
77
|
}
|
|
77
78
|
}
|
|
78
79
|
}
|
|
@@ -129,32 +130,32 @@ Customize browser session behavior with optional parameters. These can be set vi
|
|
|
129
130
|
|
|
130
131
|
## Supported MCP Tools
|
|
131
132
|
|
|
132
|
-
| Name | Description
|
|
133
|
-
|
|
134
|
-
| google_search | Universal information search engine.
|
|
135
|
-
| google_trends | Get trending search data from Google Trends.
|
|
136
|
-
| browser_create | Create or reuse a cloud browser session using Scrapeless.
|
|
137
|
-
| browser_close | Closes the current session by disconnecting the cloud browser.
|
|
138
|
-
| browser_goto | Navigate browser to a specified URL.
|
|
139
|
-
| browser_go_back | Go back one step in browser history.
|
|
140
|
-
| browser_go_forward | Go forward one step in browser history.
|
|
141
|
-
| browser_click | Click a specific element on the page.
|
|
142
|
-
| browser_type | Type text into a specified input field.
|
|
143
|
-
| browser_press_key | Simulate a key press.
|
|
144
|
-
| browser_wait_for | Wait for a specific page element to appear.
|
|
145
|
-
| browser_wait | Pause execution for a fixed duration.
|
|
146
|
-
| browser_screenshot | Capture a screenshot of the current page.
|
|
147
|
-
| browser_get_html | Get the full HTML of the current page.
|
|
148
|
-
| browser_get_text | Get all visible text from the current page.
|
|
149
|
-
| browser_scroll | Scroll to the bottom of the page.
|
|
150
|
-
| browser_scroll_to | Scroll a specific element into view.
|
|
151
|
-
| scrape_html | Scrape a URL and return its full HTML content.
|
|
152
|
-
| scrape_markdown | Scrape a URL and return its content as Markdown.
|
|
153
|
-
| scrape_screenshot | Capture a high-quality screenshot of any webpage.
|
|
154
|
-
| crawl_start | Start an asynchronous crawl job from a base URL and return its job id.
|
|
155
|
-
| crawl_cancel | Cancel an in-progress crawl job by its id.
|
|
156
|
-
| crawl_result | Poll a crawl job by its id until it completes and return the crawled data.
|
|
157
|
-
|
|
|
133
|
+
| Name | Description |
|
|
134
|
+
|--------------------|-------------------------------------------------------------------------------------------------------------------------|
|
|
135
|
+
| google_search | Universal information search engine. |
|
|
136
|
+
| google_trends | Get trending search data from Google Trends. |
|
|
137
|
+
| browser_create | Create or reuse a cloud browser session using Scrapeless. |
|
|
138
|
+
| browser_close | Closes the current session by disconnecting the cloud browser. |
|
|
139
|
+
| browser_goto | Navigate browser to a specified URL. |
|
|
140
|
+
| browser_go_back | Go back one step in browser history. |
|
|
141
|
+
| browser_go_forward | Go forward one step in browser history. |
|
|
142
|
+
| browser_click | Click a specific element on the page. |
|
|
143
|
+
| browser_type | Type text into a specified input field. |
|
|
144
|
+
| browser_press_key | Simulate a key press. |
|
|
145
|
+
| browser_wait_for | Wait for a specific page element to appear. |
|
|
146
|
+
| browser_wait | Pause execution for a fixed duration. |
|
|
147
|
+
| browser_screenshot | Capture a screenshot of the current page. |
|
|
148
|
+
| browser_get_html | Get the full HTML of the current page. |
|
|
149
|
+
| browser_get_text | Get all visible text from the current page. |
|
|
150
|
+
| browser_scroll | Scroll to the bottom of the page. |
|
|
151
|
+
| browser_scroll_to | Scroll a specific element into view. |
|
|
152
|
+
| scrape_html | Scrape a URL and return its full HTML content. |
|
|
153
|
+
| scrape_markdown | Scrape a URL and return its content as Markdown. |
|
|
154
|
+
| scrape_screenshot | Capture a high-quality screenshot of any webpage. |
|
|
155
|
+
| crawl_start | Start an asynchronous crawl job from a base URL and return its job id. |
|
|
156
|
+
| crawl_cancel | Cancel an in-progress crawl job by its id. |
|
|
157
|
+
| crawl_result | Poll a crawl job by its id until it completes and return the crawled data. |
|
|
158
|
+
| ai_scraper | Create an AI Scraper task for ChatGPT, Gemini, Perplexity, Copilot, Google AI Mode, Google AI Overview, Grok, or Alexa. |
|
|
158
159
|
|
|
159
160
|
## Security Best Practices
|
|
160
161
|
|
|
@@ -164,7 +165,7 @@ When using Scrapeless MCP Server with LLMs (like ChatGPT, Claude, or Cursor), it
|
|
|
164
165
|
|
|
165
166
|
- **Never pass raw scraped content directly into LLM prompts.** Raw HTML, JavaScript, or user-generated text may contain hidden injection payloads.
|
|
166
167
|
- **Sanitize and validate all extracted content.** Strip or escape potentially harmful tags and scripts before using content in downstream logic or AI models.
|
|
167
|
-
- **Prefer structured extraction
|
|
168
|
+
- **Prefer structured extraction to free-form text.** Use tools like `scrape_html`, `scrape_markdown`, or targeted `browser_get_text` with known-safe selectors to extract only the content you trust.
|
|
168
169
|
- **Apply domain or selector whitelisting** when scraping dynamically generated pages, to restrict data flow to known and trusted sources.
|
|
169
170
|
- **Log and monitor all outbound requests** made via browser or scraping tools, especially if you're handling sensitive data, tokens, or internal network access.
|
|
170
171
|
|
package/build/config.js
CHANGED
|
@@ -1,5 +1,8 @@
|
|
|
1
1
|
import { getParamValue } from "@chatmcp/sdk/utils/index.js";
|
|
2
|
-
export const API_KEY = process.env.
|
|
2
|
+
export const API_KEY = process.env.SCRAPELESS_API_KEY?.trim() ||
|
|
3
|
+
getParamValue("SCRAPELESS_API_KEY") ||
|
|
4
|
+
process.env.SCRAPELESS_KEY?.trim() ||
|
|
5
|
+
getParamValue("SCRAPELESS_KEY");
|
|
3
6
|
export const BASE_URL = process.env.SCRAPELESS_BASE_URL?.trim() || "https://api.scrapeless.com";
|
|
4
7
|
export const API_KEY_NAME = "x-api-token";
|
|
5
8
|
export const ServerMode = getParamValue("mode") || "stdio";
|
package/build/index.js
CHANGED
|
@@ -0,0 +1,634 @@
|
|
|
1
|
+
import axios from "axios";
|
|
2
|
+
import z from "zod";
|
|
3
|
+
import { BASE_URL } from "../../config.js";
|
|
4
|
+
import { defineTool } from "../utils.js";
|
|
5
|
+
import { getAiScraperApi, AI_SCRAPER_REQUEST_ENDPOINT, AI_SCRAPER_RESULT_ENDPOINT, } from "./api.js";
|
|
6
|
+
const POLL_INTERVAL_MS = 5000;
|
|
7
|
+
const DEFAULT_TIMEOUT_SECONDS = 180;
|
|
8
|
+
const MIN_TIMEOUT_SECONDS = 60;
|
|
9
|
+
const MAX_TIMEOUT_SECONDS = 600;
|
|
10
|
+
const ACTOR_OPTIONS = [
|
|
11
|
+
"scraper.chatgpt",
|
|
12
|
+
"scraper.gemini",
|
|
13
|
+
"scraper.perplexity",
|
|
14
|
+
"scraper.copilot",
|
|
15
|
+
"scraper.aimode",
|
|
16
|
+
"scraper.overview",
|
|
17
|
+
"scraper.grok",
|
|
18
|
+
"scraper.alexa",
|
|
19
|
+
];
|
|
20
|
+
const MODE_OPTIONS = [
|
|
21
|
+
"search",
|
|
22
|
+
"smart",
|
|
23
|
+
"chat",
|
|
24
|
+
"reasoning",
|
|
25
|
+
"study",
|
|
26
|
+
"MODEL_MODE_FAST",
|
|
27
|
+
"MODEL_MODE_EXPERT",
|
|
28
|
+
"MODEL_MODE_AUTO",
|
|
29
|
+
];
|
|
30
|
+
const ACTOR_CONFIG = {
|
|
31
|
+
"scraper.chatgpt": {
|
|
32
|
+
displayName: "ChatGPT",
|
|
33
|
+
supportsWebSearch: true,
|
|
34
|
+
supportsShopping: true,
|
|
35
|
+
},
|
|
36
|
+
"scraper.gemini": {
|
|
37
|
+
displayName: "Gemini",
|
|
38
|
+
},
|
|
39
|
+
"scraper.perplexity": {
|
|
40
|
+
displayName: "Perplexity",
|
|
41
|
+
supportsWebSearch: true,
|
|
42
|
+
},
|
|
43
|
+
"scraper.copilot": {
|
|
44
|
+
displayName: "Microsoft Copilot",
|
|
45
|
+
supportsMode: true,
|
|
46
|
+
requiresMode: true,
|
|
47
|
+
},
|
|
48
|
+
"scraper.aimode": {
|
|
49
|
+
displayName: "Google AI Mode",
|
|
50
|
+
supportsShopping: true,
|
|
51
|
+
supportsGoogleLocation: true,
|
|
52
|
+
},
|
|
53
|
+
"scraper.overview": {
|
|
54
|
+
displayName: "Google AI Overview",
|
|
55
|
+
supportsShopping: true,
|
|
56
|
+
supportsGoogleLocation: true,
|
|
57
|
+
},
|
|
58
|
+
"scraper.grok": {
|
|
59
|
+
displayName: "Grok",
|
|
60
|
+
supportsMode: true,
|
|
61
|
+
requiresMode: true,
|
|
62
|
+
},
|
|
63
|
+
"scraper.alexa": {
|
|
64
|
+
displayName: "Alexa",
|
|
65
|
+
},
|
|
66
|
+
};
|
|
67
|
+
const RESULT_FIELDS_BY_ACTOR = {
|
|
68
|
+
"scraper.chatgpt": [
|
|
69
|
+
"prompt",
|
|
70
|
+
"result_text",
|
|
71
|
+
"model",
|
|
72
|
+
"web_search",
|
|
73
|
+
"links",
|
|
74
|
+
"search_result",
|
|
75
|
+
"content_references",
|
|
76
|
+
"products",
|
|
77
|
+
"ads",
|
|
78
|
+
"map",
|
|
79
|
+
"search_model_queries",
|
|
80
|
+
],
|
|
81
|
+
"scraper.gemini": [
|
|
82
|
+
"result_text",
|
|
83
|
+
"prompt",
|
|
84
|
+
"citations",
|
|
85
|
+
"related_queries",
|
|
86
|
+
],
|
|
87
|
+
"scraper.perplexity": [
|
|
88
|
+
"prompt",
|
|
89
|
+
"result_text",
|
|
90
|
+
"related_prompt",
|
|
91
|
+
"web_results",
|
|
92
|
+
"media_items",
|
|
93
|
+
],
|
|
94
|
+
"scraper.copilot": [
|
|
95
|
+
"result_text",
|
|
96
|
+
"prompt",
|
|
97
|
+
"mode",
|
|
98
|
+
"links",
|
|
99
|
+
"citations",
|
|
100
|
+
],
|
|
101
|
+
"scraper.aimode": [
|
|
102
|
+
"result_text",
|
|
103
|
+
"result_md",
|
|
104
|
+
"result_html",
|
|
105
|
+
"raw_url",
|
|
106
|
+
"citations",
|
|
107
|
+
"search_result",
|
|
108
|
+
"products",
|
|
109
|
+
],
|
|
110
|
+
"scraper.overview": [
|
|
111
|
+
"content",
|
|
112
|
+
"rawtext",
|
|
113
|
+
"metadata",
|
|
114
|
+
"is_overview_shopping",
|
|
115
|
+
"products",
|
|
116
|
+
"source",
|
|
117
|
+
"web_source",
|
|
118
|
+
"ads",
|
|
119
|
+
],
|
|
120
|
+
"scraper.grok": [
|
|
121
|
+
"conversation",
|
|
122
|
+
"create_time",
|
|
123
|
+
"follow_up_suggestions",
|
|
124
|
+
"full_response",
|
|
125
|
+
"tool_usages",
|
|
126
|
+
"user_model",
|
|
127
|
+
"user_query",
|
|
128
|
+
"web_search_results",
|
|
129
|
+
"footnotes",
|
|
130
|
+
"x_search_results",
|
|
131
|
+
],
|
|
132
|
+
"scraper.alexa": [
|
|
133
|
+
"user_text",
|
|
134
|
+
"md_text",
|
|
135
|
+
"raw_text",
|
|
136
|
+
"completed",
|
|
137
|
+
"answer_fragment_uri",
|
|
138
|
+
"answer_revision",
|
|
139
|
+
"dialog_request_id",
|
|
140
|
+
"endpoint_id",
|
|
141
|
+
"fragment_count",
|
|
142
|
+
"conversation",
|
|
143
|
+
"directives",
|
|
144
|
+
"references",
|
|
145
|
+
"sources",
|
|
146
|
+
"suggestions",
|
|
147
|
+
"products",
|
|
148
|
+
],
|
|
149
|
+
};
|
|
150
|
+
const aiScraperSchema = z.object({
|
|
151
|
+
prompt: z
|
|
152
|
+
.string()
|
|
153
|
+
.min(1)
|
|
154
|
+
.describe("Question or prompt to send to the selected AI Scraper actor."),
|
|
155
|
+
actor: z
|
|
156
|
+
.enum(ACTOR_OPTIONS)
|
|
157
|
+
.describe("Scrapeless actor to run. Choose one explicitly: scraper.chatgpt, scraper.gemini, scraper.perplexity, scraper.copilot, scraper.aimode, scraper.overview, scraper.grok, or scraper.alexa."),
|
|
158
|
+
country: z
|
|
159
|
+
.string()
|
|
160
|
+
.optional()
|
|
161
|
+
.default("US")
|
|
162
|
+
.describe("Country or region code used by the actor. Defaults to US."),
|
|
163
|
+
web_search: z
|
|
164
|
+
.boolean()
|
|
165
|
+
.optional()
|
|
166
|
+
.describe("Enable web search for scraper.chatgpt or scraper.perplexity when explicitly provided."),
|
|
167
|
+
shopping: z
|
|
168
|
+
.boolean()
|
|
169
|
+
.optional()
|
|
170
|
+
.describe("Fetch shopping/product data for scraper.chatgpt, scraper.aimode, or scraper.overview when explicitly provided."),
|
|
171
|
+
mode: z
|
|
172
|
+
.enum(MODE_OPTIONS)
|
|
173
|
+
.optional()
|
|
174
|
+
.describe("Required for scraper.copilot or scraper.grok. Copilot: search, smart, chat, reasoning, study. Grok: MODEL_MODE_FAST, MODEL_MODE_EXPERT, MODEL_MODE_AUTO."),
|
|
175
|
+
location: z
|
|
176
|
+
.string()
|
|
177
|
+
.optional()
|
|
178
|
+
.describe("Google canonical location for Google AI Mode or Google AI Overview. Mutually exclusive with uule."),
|
|
179
|
+
uule: z
|
|
180
|
+
.string()
|
|
181
|
+
.optional()
|
|
182
|
+
.describe("Pre-encoded Google UULE for Google AI Mode or Google AI Overview. Mutually exclusive with location."),
|
|
183
|
+
webhook: z
|
|
184
|
+
.object({
|
|
185
|
+
url: z
|
|
186
|
+
.string()
|
|
187
|
+
.url()
|
|
188
|
+
.describe("Webhook URL Scrapeless calls when the task completes."),
|
|
189
|
+
})
|
|
190
|
+
.optional()
|
|
191
|
+
.describe("Optional webhook object passed to the Scrapeless task request."),
|
|
192
|
+
timeout_seconds: z
|
|
193
|
+
.number()
|
|
194
|
+
.int()
|
|
195
|
+
.min(MIN_TIMEOUT_SECONDS)
|
|
196
|
+
.max(MAX_TIMEOUT_SECONDS)
|
|
197
|
+
.optional()
|
|
198
|
+
.default(DEFAULT_TIMEOUT_SECONDS)
|
|
199
|
+
.describe("Maximum polling time in seconds. Defaults to 180. Minimum is 60 and maximum is 600."),
|
|
200
|
+
});
|
|
201
|
+
const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
|
|
202
|
+
function textResponse(data) {
|
|
203
|
+
return {
|
|
204
|
+
content: [
|
|
205
|
+
{
|
|
206
|
+
type: "text",
|
|
207
|
+
text: `Response:\n\n${JSON.stringify(data)}`,
|
|
208
|
+
},
|
|
209
|
+
],
|
|
210
|
+
};
|
|
211
|
+
}
|
|
212
|
+
function normalizeCountry(country) {
|
|
213
|
+
const normalized = (country ?? "US").trim().toUpperCase();
|
|
214
|
+
return normalized || "US";
|
|
215
|
+
}
|
|
216
|
+
function validateActorParams(params) {
|
|
217
|
+
const config = ACTOR_CONFIG[params.actor];
|
|
218
|
+
if (params.location && params.uule) {
|
|
219
|
+
return {
|
|
220
|
+
code: "invalid_input",
|
|
221
|
+
message: "location and uule are mutually exclusive.",
|
|
222
|
+
};
|
|
223
|
+
}
|
|
224
|
+
if (params.web_search !== undefined && !config.supportsWebSearch) {
|
|
225
|
+
return {
|
|
226
|
+
code: "invalid_input",
|
|
227
|
+
message: `web_search is not a supported option for ${params.actor}.`,
|
|
228
|
+
};
|
|
229
|
+
}
|
|
230
|
+
if (params.shopping !== undefined && !config.supportsShopping) {
|
|
231
|
+
return {
|
|
232
|
+
code: "invalid_input",
|
|
233
|
+
message: `shopping is not a supported top-level option for ${params.actor}.`,
|
|
234
|
+
};
|
|
235
|
+
}
|
|
236
|
+
if ((params.location || params.uule) && !config.supportsGoogleLocation) {
|
|
237
|
+
return {
|
|
238
|
+
code: "invalid_input",
|
|
239
|
+
message: `location and uule are only supported for scraper.aimode and scraper.overview.`,
|
|
240
|
+
};
|
|
241
|
+
}
|
|
242
|
+
if (params.mode !== undefined && !config.supportsMode) {
|
|
243
|
+
return {
|
|
244
|
+
code: "invalid_input",
|
|
245
|
+
message: `mode is only supported for scraper.copilot and scraper.grok.`,
|
|
246
|
+
};
|
|
247
|
+
}
|
|
248
|
+
if (params.actor === "scraper.copilot" && params.mode !== undefined) {
|
|
249
|
+
const allowedModes = ["search", "smart", "chat", "reasoning", "study"];
|
|
250
|
+
if (!allowedModes.includes(params.mode)) {
|
|
251
|
+
return {
|
|
252
|
+
code: "invalid_input",
|
|
253
|
+
message: `mode for scraper.copilot must be one of: ${allowedModes.join(", ")}.`,
|
|
254
|
+
};
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
if (params.actor === "scraper.grok" && params.mode !== undefined) {
|
|
258
|
+
const allowedModes = [
|
|
259
|
+
"MODEL_MODE_FAST",
|
|
260
|
+
"MODEL_MODE_EXPERT",
|
|
261
|
+
"MODEL_MODE_AUTO",
|
|
262
|
+
];
|
|
263
|
+
if (!allowedModes.includes(params.mode)) {
|
|
264
|
+
return {
|
|
265
|
+
code: "invalid_input",
|
|
266
|
+
message: `mode for scraper.grok must be one of: ${allowedModes.join(", ")}.`,
|
|
267
|
+
};
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
if (config.requiresMode &&
|
|
271
|
+
params.mode === undefined) {
|
|
272
|
+
return {
|
|
273
|
+
code: "invalid_input",
|
|
274
|
+
message: `mode is required for ${params.actor}.`,
|
|
275
|
+
};
|
|
276
|
+
}
|
|
277
|
+
return undefined;
|
|
278
|
+
}
|
|
279
|
+
function buildActorInput(params, country) {
|
|
280
|
+
const actorInput = {
|
|
281
|
+
prompt: params.prompt,
|
|
282
|
+
country,
|
|
283
|
+
};
|
|
284
|
+
if (params.web_search !== undefined) {
|
|
285
|
+
actorInput.web_search = params.web_search;
|
|
286
|
+
}
|
|
287
|
+
if (params.shopping !== undefined) {
|
|
288
|
+
actorInput.shopping = params.shopping;
|
|
289
|
+
}
|
|
290
|
+
if (params.mode !== undefined) {
|
|
291
|
+
actorInput.mode = params.mode;
|
|
292
|
+
}
|
|
293
|
+
if (params.location)
|
|
294
|
+
actorInput.location = params.location;
|
|
295
|
+
if (params.uule)
|
|
296
|
+
actorInput.uule = params.uule;
|
|
297
|
+
return actorInput;
|
|
298
|
+
}
|
|
299
|
+
function isRecord(value) {
|
|
300
|
+
return Boolean(value) && typeof value === "object" && !Array.isArray(value);
|
|
301
|
+
}
|
|
302
|
+
function getStringField(source, field) {
|
|
303
|
+
if (!isRecord(source))
|
|
304
|
+
return undefined;
|
|
305
|
+
const value = source[field];
|
|
306
|
+
return typeof value === "string" && value.length > 0 ? value : undefined;
|
|
307
|
+
}
|
|
308
|
+
function extractTaskId(createResponse) {
|
|
309
|
+
return getStringField(createResponse, "task_id");
|
|
310
|
+
}
|
|
311
|
+
function extractRequestId(createResponse) {
|
|
312
|
+
return getStringField(createResponse, "request_id");
|
|
313
|
+
}
|
|
314
|
+
function extractTaskStatus(resultResponse) {
|
|
315
|
+
return getStringField(resultResponse, "status")?.toLowerCase();
|
|
316
|
+
}
|
|
317
|
+
function extractTaskMessage(resultResponse) {
|
|
318
|
+
return getStringField(resultResponse, "message");
|
|
319
|
+
}
|
|
320
|
+
function extractTaskResult(resultResponse) {
|
|
321
|
+
return isRecord(resultResponse) ? resultResponse.task_result : undefined;
|
|
322
|
+
}
|
|
323
|
+
function pickActorResult(actor, result) {
|
|
324
|
+
if (!isRecord(result))
|
|
325
|
+
return result;
|
|
326
|
+
return RESULT_FIELDS_BY_ACTOR[actor].reduce((actorResult, field) => {
|
|
327
|
+
if (field in result)
|
|
328
|
+
actorResult[field] = result[field];
|
|
329
|
+
return actorResult;
|
|
330
|
+
}, {});
|
|
331
|
+
}
|
|
332
|
+
function formatPollHttpError(statusCode, responseData) {
|
|
333
|
+
return {
|
|
334
|
+
status_code: statusCode,
|
|
335
|
+
code: "api_error",
|
|
336
|
+
message: getStringField(responseData, "message") ||
|
|
337
|
+
getStringField(responseData, "error") ||
|
|
338
|
+
`AI scraper result request failed with HTTP ${statusCode}.`,
|
|
339
|
+
response: responseData,
|
|
340
|
+
};
|
|
341
|
+
}
|
|
342
|
+
function formatTaskResultError(code, message, response) {
|
|
343
|
+
return {
|
|
344
|
+
code,
|
|
345
|
+
message,
|
|
346
|
+
response,
|
|
347
|
+
};
|
|
348
|
+
}
|
|
349
|
+
async function pollTask(api, taskId, timeoutSeconds) {
|
|
350
|
+
const startedAtMs = Date.now();
|
|
351
|
+
const startedAt = new Date(startedAtMs).toISOString();
|
|
352
|
+
const deadline = startedAtMs + timeoutSeconds * 1000;
|
|
353
|
+
let latestResult;
|
|
354
|
+
let latestStatus = "processing";
|
|
355
|
+
let latestHttpStatus;
|
|
356
|
+
while (true) {
|
|
357
|
+
const response = await api.get(`${AI_SCRAPER_RESULT_ENDPOINT}/${encodeURIComponent(taskId)}`, {
|
|
358
|
+
validateStatus: () => true,
|
|
359
|
+
});
|
|
360
|
+
latestResult = response.data;
|
|
361
|
+
latestHttpStatus = response.status;
|
|
362
|
+
if (response.status === 429) {
|
|
363
|
+
latestStatus = "rate_limited_retrying";
|
|
364
|
+
}
|
|
365
|
+
else if (response.status < 200 || response.status >= 300) {
|
|
366
|
+
const finishedAtMs = Date.now();
|
|
367
|
+
return {
|
|
368
|
+
taskId,
|
|
369
|
+
status: `http_${response.status}`,
|
|
370
|
+
terminal: true,
|
|
371
|
+
ok: false,
|
|
372
|
+
timedOut: false,
|
|
373
|
+
startedAt,
|
|
374
|
+
finishedAt: new Date(finishedAtMs).toISOString(),
|
|
375
|
+
elapsedMs: finishedAtMs - startedAtMs,
|
|
376
|
+
latestResult,
|
|
377
|
+
httpStatus: response.status,
|
|
378
|
+
error: formatPollHttpError(response.status, response.data),
|
|
379
|
+
};
|
|
380
|
+
}
|
|
381
|
+
else {
|
|
382
|
+
const taskStatus = extractTaskStatus(response.data);
|
|
383
|
+
if (taskStatus === "success") {
|
|
384
|
+
const taskResult = extractTaskResult(response.data);
|
|
385
|
+
const finishedAtMs = Date.now();
|
|
386
|
+
if (!isRecord(taskResult)) {
|
|
387
|
+
return {
|
|
388
|
+
taskId,
|
|
389
|
+
status: "invalid_response",
|
|
390
|
+
terminal: true,
|
|
391
|
+
ok: false,
|
|
392
|
+
timedOut: false,
|
|
393
|
+
startedAt,
|
|
394
|
+
finishedAt: new Date(finishedAtMs).toISOString(),
|
|
395
|
+
elapsedMs: finishedAtMs - startedAtMs,
|
|
396
|
+
latestResult,
|
|
397
|
+
httpStatus: response.status,
|
|
398
|
+
error: formatTaskResultError("invalid_response", "Scrapeless result response status is success but task_result is missing or is not an object.", response.data),
|
|
399
|
+
};
|
|
400
|
+
}
|
|
401
|
+
return {
|
|
402
|
+
taskId,
|
|
403
|
+
status: taskStatus,
|
|
404
|
+
terminal: true,
|
|
405
|
+
ok: true,
|
|
406
|
+
timedOut: false,
|
|
407
|
+
startedAt,
|
|
408
|
+
finishedAt: new Date(finishedAtMs).toISOString(),
|
|
409
|
+
elapsedMs: finishedAtMs - startedAtMs,
|
|
410
|
+
latestResult,
|
|
411
|
+
taskResult,
|
|
412
|
+
httpStatus: response.status,
|
|
413
|
+
};
|
|
414
|
+
}
|
|
415
|
+
if (taskStatus === "failed") {
|
|
416
|
+
const finishedAtMs = Date.now();
|
|
417
|
+
return {
|
|
418
|
+
taskId,
|
|
419
|
+
status: taskStatus,
|
|
420
|
+
terminal: true,
|
|
421
|
+
ok: false,
|
|
422
|
+
timedOut: false,
|
|
423
|
+
startedAt,
|
|
424
|
+
finishedAt: new Date(finishedAtMs).toISOString(),
|
|
425
|
+
elapsedMs: finishedAtMs - startedAtMs,
|
|
426
|
+
latestResult,
|
|
427
|
+
httpStatus: response.status,
|
|
428
|
+
error: formatTaskResultError("task_failed", extractTaskMessage(response.data) || "AI scraper task failed.", response.data),
|
|
429
|
+
};
|
|
430
|
+
}
|
|
431
|
+
if (taskStatus === "pending" || taskStatus === "running") {
|
|
432
|
+
latestStatus = taskStatus;
|
|
433
|
+
}
|
|
434
|
+
else {
|
|
435
|
+
const finishedAtMs = Date.now();
|
|
436
|
+
return {
|
|
437
|
+
taskId,
|
|
438
|
+
status: "invalid_response",
|
|
439
|
+
terminal: true,
|
|
440
|
+
ok: false,
|
|
441
|
+
timedOut: false,
|
|
442
|
+
startedAt,
|
|
443
|
+
finishedAt: new Date(finishedAtMs).toISOString(),
|
|
444
|
+
elapsedMs: finishedAtMs - startedAtMs,
|
|
445
|
+
latestResult,
|
|
446
|
+
httpStatus: response.status,
|
|
447
|
+
error: formatTaskResultError("invalid_response", "Scrapeless result response did not include a valid status: pending, running, success, or failed.", response.data),
|
|
448
|
+
};
|
|
449
|
+
}
|
|
450
|
+
}
|
|
451
|
+
const remainingMs = deadline - Date.now();
|
|
452
|
+
if (remainingMs <= 0) {
|
|
453
|
+
const finishedAtMs = Date.now();
|
|
454
|
+
return {
|
|
455
|
+
taskId,
|
|
456
|
+
status: latestStatus,
|
|
457
|
+
terminal: false,
|
|
458
|
+
ok: false,
|
|
459
|
+
timedOut: true,
|
|
460
|
+
startedAt,
|
|
461
|
+
finishedAt: new Date(finishedAtMs).toISOString(),
|
|
462
|
+
elapsedMs: finishedAtMs - startedAtMs,
|
|
463
|
+
latestResult,
|
|
464
|
+
httpStatus: latestHttpStatus,
|
|
465
|
+
};
|
|
466
|
+
}
|
|
467
|
+
await sleep(Math.min(POLL_INTERVAL_MS, remainingMs));
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
function timeoutHelp(taskId) {
|
|
471
|
+
return {
|
|
472
|
+
message: "The AI scraper task did not finish before the timeout. Fetch the result manually with the Scrapeless API.",
|
|
473
|
+
manual_result_request: {
|
|
474
|
+
method: "GET",
|
|
475
|
+
url: `${BASE_URL}${AI_SCRAPER_RESULT_ENDPOINT}/${taskId}`,
|
|
476
|
+
headers: {
|
|
477
|
+
"x-api-token": "YOUR_SCRAPELESS_KEY",
|
|
478
|
+
},
|
|
479
|
+
},
|
|
480
|
+
};
|
|
481
|
+
}
|
|
482
|
+
function formatApiError(error) {
|
|
483
|
+
if (axios.isAxiosError(error)) {
|
|
484
|
+
const statusCode = error.response?.status;
|
|
485
|
+
const responseData = error.response?.data;
|
|
486
|
+
return {
|
|
487
|
+
status_code: statusCode,
|
|
488
|
+
code: statusCode === 429 ? "rate_limited" : "api_error",
|
|
489
|
+
message: getStringField(responseData, "message") ||
|
|
490
|
+
getStringField(responseData, "error") ||
|
|
491
|
+
error.message,
|
|
492
|
+
response: responseData,
|
|
493
|
+
};
|
|
494
|
+
}
|
|
495
|
+
return {
|
|
496
|
+
code: "unknown_error",
|
|
497
|
+
message: error.message,
|
|
498
|
+
};
|
|
499
|
+
}
|
|
500
|
+
export const aiScraper = defineTool({
|
|
501
|
+
name: "ai_scraper",
|
|
502
|
+
description: `Create an AI Scraper task for an explicit Scrapeless actor, then poll every 5 seconds until the answer is ready or the timeout is reached.
|
|
503
|
+
Supports ChatGPT, Gemini, Perplexity, Copilot, Google AI Mode, Google AI Overview, Grok, and Alexa.
|
|
504
|
+
Defaults to a 3 minute timeout. The timeout can be set from 60 to 600 seconds.
|
|
505
|
+
On timeout, returns the task_id and instructions for manually fetching the result.`,
|
|
506
|
+
inputSchema: aiScraperSchema.shape,
|
|
507
|
+
handle: async (rawParams, client, headers) => {
|
|
508
|
+
const params = aiScraperSchema.parse(rawParams);
|
|
509
|
+
const country = normalizeCountry(params.country);
|
|
510
|
+
const actorConfig = ACTOR_CONFIG[params.actor];
|
|
511
|
+
const startedAtMs = Date.now();
|
|
512
|
+
const startedAt = new Date(startedAtMs).toISOString();
|
|
513
|
+
const validationError = validateActorParams(params);
|
|
514
|
+
if (validationError) {
|
|
515
|
+
return textResponse({
|
|
516
|
+
status: "failed",
|
|
517
|
+
actor: params.actor,
|
|
518
|
+
actor_display_name: actorConfig.displayName,
|
|
519
|
+
error: validationError,
|
|
520
|
+
});
|
|
521
|
+
}
|
|
522
|
+
const actorInput = buildActorInput(params, country);
|
|
523
|
+
const taskRequest = {
|
|
524
|
+
actor: params.actor,
|
|
525
|
+
input: actorInput,
|
|
526
|
+
};
|
|
527
|
+
if (params.webhook) {
|
|
528
|
+
taskRequest.webhook = params.webhook;
|
|
529
|
+
}
|
|
530
|
+
let taskId;
|
|
531
|
+
let requestId;
|
|
532
|
+
let createResponseBody;
|
|
533
|
+
try {
|
|
534
|
+
const api = getAiScraperApi(client, headers);
|
|
535
|
+
const createResponse = await api.post(AI_SCRAPER_REQUEST_ENDPOINT, taskRequest);
|
|
536
|
+
createResponseBody = createResponse.data;
|
|
537
|
+
taskId = extractTaskId(createResponseBody);
|
|
538
|
+
requestId = extractRequestId(createResponseBody);
|
|
539
|
+
if (!taskId) {
|
|
540
|
+
return textResponse({
|
|
541
|
+
status: "failed",
|
|
542
|
+
actor: params.actor,
|
|
543
|
+
actor_display_name: actorConfig.displayName,
|
|
544
|
+
request_id: requestId,
|
|
545
|
+
error: {
|
|
546
|
+
code: "missing_task_id",
|
|
547
|
+
message: "Scrapeless did not return a task id from /api/v2/scraper/request.",
|
|
548
|
+
},
|
|
549
|
+
create_response: createResponseBody,
|
|
550
|
+
});
|
|
551
|
+
}
|
|
552
|
+
const pollResult = await pollTask(api, taskId, params.timeout_seconds);
|
|
553
|
+
const finishedAtMs = Date.now();
|
|
554
|
+
if (pollResult.timedOut) {
|
|
555
|
+
return textResponse({
|
|
556
|
+
status: "timeout",
|
|
557
|
+
task_id: taskId,
|
|
558
|
+
request_id: requestId,
|
|
559
|
+
actor: params.actor,
|
|
560
|
+
actor_display_name: actorConfig.displayName,
|
|
561
|
+
timeout_seconds: params.timeout_seconds,
|
|
562
|
+
poll_interval_seconds: POLL_INTERVAL_MS / 1000,
|
|
563
|
+
execution_time: {
|
|
564
|
+
started_at: startedAt,
|
|
565
|
+
finished_at: new Date(finishedAtMs).toISOString(),
|
|
566
|
+
elapsed_ms: finishedAtMs - startedAtMs,
|
|
567
|
+
},
|
|
568
|
+
latest_status: pollResult.status,
|
|
569
|
+
latest_http_status: pollResult.httpStatus,
|
|
570
|
+
latest_result: pollResult.latestResult,
|
|
571
|
+
help: timeoutHelp(taskId),
|
|
572
|
+
create_response: createResponseBody,
|
|
573
|
+
});
|
|
574
|
+
}
|
|
575
|
+
if (!pollResult.ok) {
|
|
576
|
+
return textResponse({
|
|
577
|
+
status: "failed",
|
|
578
|
+
task_id: taskId,
|
|
579
|
+
request_id: requestId,
|
|
580
|
+
actor: params.actor,
|
|
581
|
+
actor_display_name: actorConfig.displayName,
|
|
582
|
+
timeout_seconds: params.timeout_seconds,
|
|
583
|
+
poll_interval_seconds: POLL_INTERVAL_MS / 1000,
|
|
584
|
+
execution_time: {
|
|
585
|
+
started_at: startedAt,
|
|
586
|
+
finished_at: new Date(finishedAtMs).toISOString(),
|
|
587
|
+
elapsed_ms: finishedAtMs - startedAtMs,
|
|
588
|
+
},
|
|
589
|
+
task_status: pollResult.status,
|
|
590
|
+
result_http_status: pollResult.httpStatus,
|
|
591
|
+
error: pollResult.error,
|
|
592
|
+
latest_result: pollResult.latestResult,
|
|
593
|
+
create_response: createResponseBody,
|
|
594
|
+
});
|
|
595
|
+
}
|
|
596
|
+
return textResponse({
|
|
597
|
+
status: "completed",
|
|
598
|
+
task_id: taskId,
|
|
599
|
+
request_id: requestId,
|
|
600
|
+
actor: params.actor,
|
|
601
|
+
actor_display_name: actorConfig.displayName,
|
|
602
|
+
timeout_seconds: params.timeout_seconds,
|
|
603
|
+
poll_interval_seconds: POLL_INTERVAL_MS / 1000,
|
|
604
|
+
execution_time: {
|
|
605
|
+
started_at: startedAt,
|
|
606
|
+
finished_at: new Date(finishedAtMs).toISOString(),
|
|
607
|
+
elapsed_ms: finishedAtMs - startedAtMs,
|
|
608
|
+
},
|
|
609
|
+
task_status: pollResult.status,
|
|
610
|
+
result_http_status: pollResult.httpStatus,
|
|
611
|
+
result: pickActorResult(params.actor, pollResult.taskResult),
|
|
612
|
+
create_response: createResponseBody,
|
|
613
|
+
});
|
|
614
|
+
}
|
|
615
|
+
catch (error) {
|
|
616
|
+
const finishedAtMs = Date.now();
|
|
617
|
+
return textResponse({
|
|
618
|
+
status: "failed",
|
|
619
|
+
task_id: taskId,
|
|
620
|
+
request_id: requestId,
|
|
621
|
+
actor: params.actor,
|
|
622
|
+
actor_display_name: actorConfig.displayName,
|
|
623
|
+
execution_time: {
|
|
624
|
+
started_at: startedAt,
|
|
625
|
+
finished_at: new Date(finishedAtMs).toISOString(),
|
|
626
|
+
elapsed_ms: finishedAtMs - startedAtMs,
|
|
627
|
+
},
|
|
628
|
+
error: formatApiError(error),
|
|
629
|
+
help: taskId ? timeoutHelp(taskId) : undefined,
|
|
630
|
+
create_response: createResponseBody,
|
|
631
|
+
});
|
|
632
|
+
}
|
|
633
|
+
},
|
|
634
|
+
});
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
import axios from "axios";
|
|
2
|
+
import { API_KEY, BASE_URL, API_KEY_NAME } from "../../config.js";
|
|
3
|
+
/**
|
|
4
|
+
* Build an axios instance targeting the Scrapeless v2 scraper task API.
|
|
5
|
+
*
|
|
6
|
+
* The `x-api-token` is resolved from the per-request client when available
|
|
7
|
+
* (so multi-tenant HTTP mode keeps using the caller's key) and falls back to
|
|
8
|
+
* the `SCRAPELESS_API_KEY` (or legacy `SCRAPELESS_KEY`) environment variable.
|
|
9
|
+
*/
|
|
10
|
+
export function getAiScraperApi(client, headers) {
|
|
11
|
+
// Priority: explicit request header (HTTP multi-tenant) -> key carried by the
|
|
12
|
+
// per-request client -> SCRAPELESS_API_KEY / SCRAPELESS_KEY env (stdio / fallback).
|
|
13
|
+
const apiKey = headers?.[API_KEY_NAME] ||
|
|
14
|
+
client?.scraping?.apiKey ||
|
|
15
|
+
API_KEY ||
|
|
16
|
+
"";
|
|
17
|
+
return axios.create({
|
|
18
|
+
baseURL: BASE_URL,
|
|
19
|
+
headers: {
|
|
20
|
+
"Content-Type": "application/json",
|
|
21
|
+
[API_KEY_NAME]: apiKey,
|
|
22
|
+
},
|
|
23
|
+
timeout: 30000,
|
|
24
|
+
});
|
|
25
|
+
}
|
|
26
|
+
export const AI_SCRAPER_REQUEST_ENDPOINT = "/api/v2/scraper/request";
|
|
27
|
+
export const AI_SCRAPER_RESULT_ENDPOINT = "/api/v2/scraper/result";
|
package/build/tools/crawl/api.js
CHANGED
|
@@ -5,11 +5,11 @@ import { API_KEY, BASE_URL, API_KEY_NAME } from "../../config.js";
|
|
|
5
5
|
*
|
|
6
6
|
* The `x-api-token` is resolved from the per-request client when available
|
|
7
7
|
* (so multi-tenant HTTP mode keeps using the caller's key) and falls back to
|
|
8
|
-
* the `SCRAPELESS_KEY` environment variable.
|
|
8
|
+
* the `SCRAPELESS_API_KEY` (or legacy `SCRAPELESS_KEY`) environment variable.
|
|
9
9
|
*/
|
|
10
10
|
export function getCrawlApi(client, headers) {
|
|
11
11
|
// Priority: explicit request header (HTTP multi-tenant) -> key carried by the
|
|
12
|
-
// per-request client -> SCRAPELESS_KEY env (stdio / fallback).
|
|
12
|
+
// per-request client -> SCRAPELESS_API_KEY / SCRAPELESS_KEY env (stdio / fallback).
|
|
13
13
|
const apiKey = headers?.[API_KEY_NAME] ||
|
|
14
14
|
client?.scrapingCrawl?.crawl?.apiKey ||
|
|
15
15
|
API_KEY ||
|
package/build/tools/index.js
CHANGED
|
@@ -6,4 +6,4 @@ export * from "./universal/scrapeScreenshot.js";
|
|
|
6
6
|
export * from "./crawl/crawlStart.js";
|
|
7
7
|
export * from "./crawl/crawlCancel.js";
|
|
8
8
|
export * from "./crawl/crawlResult.js";
|
|
9
|
-
export * from "./
|
|
9
|
+
export * from "./ai_scraper/aiScraper.js";
|
package/package.json
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "scrapeless-mcp-server",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.3",
|
|
4
|
+
"mcpName": "io.github.scrapeless-ai/scrapeless-mcp-server",
|
|
4
5
|
"main": "index.js",
|
|
5
6
|
"type": "module",
|
|
6
7
|
"bin": {
|
|
@@ -19,17 +20,18 @@
|
|
|
19
20
|
"mcp",
|
|
20
21
|
"google",
|
|
21
22
|
"google search",
|
|
23
|
+
"browser",
|
|
22
24
|
"serpapi",
|
|
23
25
|
"cursor",
|
|
24
26
|
"claude",
|
|
25
27
|
"ai",
|
|
26
28
|
"model-context-protocol",
|
|
27
29
|
"crawl",
|
|
28
|
-
"
|
|
30
|
+
"ai scraper"
|
|
29
31
|
],
|
|
30
32
|
"author": "",
|
|
31
33
|
"license": "ISC",
|
|
32
|
-
"description": "
|
|
34
|
+
"description": "Web search, browser automation, scraping, crawling and CAPTCHA solving for AI agents.",
|
|
33
35
|
"dependencies": {
|
|
34
36
|
"@chatmcp/sdk": "^1.0.5",
|
|
35
37
|
"@modelcontextprotocol/sdk": "^1.8.0",
|