scrapeless-mcp-server 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,138 @@
1
+ import { defineTool, wrapMcpResponse } from "../utils.js";
2
+ import { getCrawlApi, CRAWL_ENDPOINT } from "./api.js";
3
+ import z from "zod";
4
+ const formatEnum = z.enum([
5
+ "markdown",
6
+ "html",
7
+ "rawHtml",
8
+ "links",
9
+ "screenshot",
10
+ "screenshot@fullPage",
11
+ "json",
12
+ ]);
13
+ const scrapeOptionsSchema = z
14
+ .object({
15
+ formats: z
16
+ .array(formatEnum)
17
+ .optional()
18
+ .describe("Formats to include in the output. Defaults to ['markdown']."),
19
+ onlyMainContent: z
20
+ .boolean()
21
+ .optional()
22
+ .describe("Only return the main content of the page excluding headers, navs, footers, etc."),
23
+ includeTags: z
24
+ .array(z.string())
25
+ .optional()
26
+ .describe("Tags to include in the output."),
27
+ excludeTags: z
28
+ .array(z.string())
29
+ .optional()
30
+ .describe("Tags to exclude from the output."),
31
+ headers: z
32
+ .record(z.any())
33
+ .optional()
34
+ .describe("Headers to send with the request. Can be used to send cookies, user-agent, etc."),
35
+ waitFor: z
36
+ .number()
37
+ .optional()
38
+ .describe("Delay in milliseconds before fetching the content, allowing the page sufficient time to load."),
39
+ timeout: z
40
+ .number()
41
+ .optional()
42
+ .describe("Timeout in milliseconds for the request."),
43
+ })
44
+ .describe("Options that control how each page is scraped.");
45
+ const browserOptionsSchema = z
46
+ .object({
47
+ sessionName: z
48
+ .string()
49
+ .optional()
50
+ .describe("A name for your session to facilitate searching and viewing in the historical session list."),
51
+ sessionTTL: z
52
+ .string()
53
+ .optional()
54
+ .describe("Session duration in seconds. Defaults to 180s, customizable between 60s and 900s."),
55
+ sessionRecording: z
56
+ .string()
57
+ .optional()
58
+ .describe("Whether to enable session recording. Defaults to false."),
59
+ proxyCountry: z
60
+ .string()
61
+ .optional()
62
+ .describe("Target country/region for the proxy as a country code (e.g. US, GB, ANY)."),
63
+ proxyURL: z
64
+ .string()
65
+ .optional()
66
+ .describe("Custom proxy URL, e.g. http://user:pass@ip:port. If set, all other proxy_* parameters are ignored."),
67
+ fingerprint: z
68
+ .string()
69
+ .optional()
70
+ .describe("Custom browser fingerprint configuration."),
71
+ })
72
+ .describe("Options that control the underlying scraping browser session.");
73
+ export const crawlStart = defineTool({
74
+ name: "crawl_start",
75
+ description: `Start an asynchronous crawl job. Crawls a website starting from a base URL, following links according to the provided options, and captures page content in various formats (markdown, html, links, screenshot, etc.).
76
+ Returns a job id that can be used with crawl_result to fetch results and crawl_cancel to cancel the job.
77
+ Only 'url' is required; all other parameters are optional.`,
78
+ inputSchema: {
79
+ url: z.string().url().describe("The base URL to start crawling from."),
80
+ limit: z
81
+ .number()
82
+ .optional()
83
+ .describe("Maximum number of pages to crawl. Default limit is 10000."),
84
+ excludePaths: z
85
+ .array(z.string())
86
+ .optional()
87
+ .describe("URL pathname regex patterns that exclude matching URLs from the crawl."),
88
+ includePaths: z
89
+ .array(z.string())
90
+ .optional()
91
+ .describe("URL pathname regex patterns that include matching URLs in the crawl."),
92
+ maxDepth: z
93
+ .number()
94
+ .optional()
95
+ .describe("Maximum depth to crawl relative to the base URL (max number of slashes in the pathname)."),
96
+ maxDiscoveryDepth: z
97
+ .number()
98
+ .optional()
99
+ .describe("Maximum depth to crawl based on discovery order."),
100
+ ignoreSitemap: z
101
+ .boolean()
102
+ .optional()
103
+ .describe("Ignore the website sitemap when crawling."),
104
+ ignoreQueryParameters: z
105
+ .boolean()
106
+ .optional()
107
+ .describe("Do not re-scrape the same path with different (or none) query parameters."),
108
+ deduplicateSimilarURLs: z
109
+ .boolean()
110
+ .optional()
111
+ .describe("Controls whether similar URLs should be deduplicated."),
112
+ regexOnFullURL: z
113
+ .boolean()
114
+ .optional()
115
+ .describe("Controls whether the include/exclude regex should be applied to the full URL."),
116
+ allowBackwardLinks: z
117
+ .boolean()
118
+ .optional()
119
+ .describe("Allow the crawler to follow links that are not part of the URL hierarchy you specify."),
120
+ allowExternalLinks: z
121
+ .boolean()
122
+ .optional()
123
+ .describe("Allow the crawler to follow links to external websites."),
124
+ delay: z
125
+ .number()
126
+ .optional()
127
+ .describe("Delay in seconds between scrapes. This helps respect website rate limits."),
128
+ scrapeOptions: scrapeOptionsSchema.optional(),
129
+ browserOptions: browserOptionsSchema.optional(),
130
+ },
131
+ handle: async (params, client, headers) => {
132
+ return wrapMcpResponse(async () => {
133
+ const api = getCrawlApi(client, headers);
134
+ const { data } = await api.post(CRAWL_ENDPOINT, params);
135
+ return data;
136
+ });
137
+ },
138
+ });
@@ -0,0 +1,37 @@
1
+ import { defineTool, wrapMcpResponse } from "../utils.js";
2
+ import z from "zod";
3
+ export const googleSearch = defineTool({
4
+ name: "google_search",
5
+ description: "Universal Information Search Engine.Retrieves any data information; Explanatory queries (why, how).Comparative analysis requests",
6
+ inputSchema: {
7
+ q: z
8
+ .string()
9
+ .describe("Parameter defines the query you want to search. You can use anything that you would use in a regular Google search. e.g. inurl:, site:, intitle:. We also support advanced search query parameters such as as_dt and as_eq.")
10
+ .default("Top news headlines"),
11
+ hl: z
12
+ .string()
13
+ .describe("Parameter defines the language to use for the Google search. It's a two-letter language code. (e.g., en for English, es for Spanish, or fr for French).")
14
+ .default("en"),
15
+ gl: z
16
+ .string()
17
+ .describe("Parameter defines the country to use for the Google search. It's a two-letter country code. (e.g., us for the United States, uk for United Kingdom, or fr for France).")
18
+ .default("us"),
19
+ },
20
+ handle: async (params, client) => {
21
+ return wrapMcpResponse(async () => {
22
+ const data = await client.deepserp.scrape({
23
+ actor: "scraper.google.search",
24
+ input: params,
25
+ });
26
+ return (data?.organic_results?.map((i) => ({
27
+ position: i.position,
28
+ title: i.title,
29
+ link: i.link,
30
+ redirect_link: i.redirect_link,
31
+ snippet: i.snippet,
32
+ snippet_highlighted_words: i.snippet_highlighted_words,
33
+ source: i.source,
34
+ })) ?? []);
35
+ });
36
+ },
37
+ });
@@ -0,0 +1,208 @@
1
+ import { defineTool, wrapMcpResponse } from "../utils.js";
2
+ import z from "zod";
3
+ export const googleTrends = defineTool({
4
+ name: 'google_trends',
5
+ description: `Get trending search data from Google Trends.
6
+ Restrictions: Activated for queries about trends, popularity, or interest over time.
7
+ Valid: Find the search interest for "AI" over the last year.
8
+ Invalid: A general question like "What is AI?" (use google_search).`,
9
+ inputSchema: {
10
+ q: z
11
+ .string()
12
+ .describe('Parameter defines the query or queries you want to search. You can use anything that you would use in a regular Google Trends search. The maximum number of queries per search is 5 (this only applies to interest_over_time and compared_breakdown_by_region data_type, other types of data will only accept 1 query per search).')
13
+ .default('Mercedes-Benz,BMW X5'),
14
+ data_type: z
15
+ .union([
16
+ z.literal('autocomplete').describe('Auto complete'),
17
+ z.literal('interest_over_time').describe('Interest over time'),
18
+ z.literal('compared_breakdown_by_region').describe('Compared breakdown by region'),
19
+ z.literal('interest_by_subregion').describe('Interest by region'),
20
+ z.literal('related_queries').describe('Related queries'),
21
+ z.literal('related_topics').describe('Related topics')
22
+ ])
23
+ .describe('The supported types are: autocomplete,interest_over_time,compared_breakdown_by_region,interest_by_subregion,related_queries,related_topics.')
24
+ .default('interest_over_time'),
25
+ date: z
26
+ .string()
27
+ .describe('The supported dates are: now 1-H, now 4-H, now 1-d, now 7-d, today 1-m, today 3-m, today 12-m, today 5-y, all.\n\nYou can also pass custom values:\n\nDates from 2004 to present: yyyy-mm-dd yyyy-mm-dd (e.g. 2021-10-15 2022-05-25)\nDates with hours within a week range: yyyy-mm-ddThh yyyy-mm-ddThh (e.g. 2022-05-19T10 2022-05-24T22). Hours will be calculated depending on the tz (time zone) parameter.')
28
+ .default('today 1-m'),
29
+ hl: z
30
+ .string()
31
+ .describe("Parameter defines the language to use for the Google Trends search. It's a two-letter language code. (e.g., en for English, es for Spanish, or fr for French).")
32
+ .default('en'),
33
+ tz: z.string().describe('time zone offset. default is 420.').default('420'),
34
+ geo: z
35
+ .union([
36
+ z.literal(''),
37
+ z.literal('AR'),
38
+ z.literal('AU'),
39
+ z.literal('AT'),
40
+ z.literal('BE'),
41
+ z.literal('BR'),
42
+ z.literal('CA'),
43
+ z.literal('CL'),
44
+ z.literal('CO'),
45
+ z.literal('CZ'),
46
+ z.literal('DK'),
47
+ z.literal('EG'),
48
+ z.literal('FI'),
49
+ z.literal('FR'),
50
+ z.literal('DE'),
51
+ z.literal('GR'),
52
+ z.literal('HK'),
53
+ z.literal('HU'),
54
+ z.literal('IN'),
55
+ z.literal('ID'),
56
+ z.literal('IE'),
57
+ z.literal('IL'),
58
+ z.literal('IT'),
59
+ z.literal('JP'),
60
+ z.literal('KE'),
61
+ z.literal('MY'),
62
+ z.literal('MX'),
63
+ z.literal('NL'),
64
+ z.literal('NZ'),
65
+ z.literal('NG'),
66
+ z.literal('NO'),
67
+ z.literal('PE'),
68
+ z.literal('PH'),
69
+ z.literal('PL'),
70
+ z.literal('PT'),
71
+ z.literal('RO'),
72
+ z.literal('RU'),
73
+ z.literal('SA'),
74
+ z.literal('SG'),
75
+ z.literal('ZA'),
76
+ z.literal('KR'),
77
+ z.literal('ES'),
78
+ z.literal('SE'),
79
+ z.literal('CH'),
80
+ z.literal('TW'),
81
+ z.literal('TH'),
82
+ z.literal('TR'),
83
+ z.literal('UA'),
84
+ z.literal('GB'),
85
+ z.literal('US'),
86
+ z.literal('VN')
87
+ ])
88
+ .describe('Parameter defines the location from where you want the search to originate. It defaults to Worldwide (activated when the value of geo parameter is not set or empty).')
89
+ .optional(),
90
+ cat: z
91
+ .union([
92
+ z.literal('0').describe('All categories'),
93
+ z.literal('3').describe('Arts & Entertainment'),
94
+ z.literal('5').describe('Computers & Electronics'),
95
+ z.literal('7').describe('Finance'),
96
+ z.literal('8').describe('Games'),
97
+ z.literal('11').describe('Home & Garden'),
98
+ z.literal('12').describe('Business & Industrial'),
99
+ z.literal('13').describe('Internet & Telecom'),
100
+ z.literal('14').describe('People & Society'),
101
+ z.literal('16').describe('News'),
102
+ z.literal('18').describe('Shopping'),
103
+ z.literal('19').describe('Law & Government'),
104
+ z.literal('20').describe('Sports'),
105
+ z.literal('22').describe('Books & Literature'),
106
+ z.literal('23').describe('Performing Arts'),
107
+ z.literal('24').describe('Visual Art & Design'),
108
+ z.literal('25').describe('Advertising & Marketing'),
109
+ z.literal('28').describe('Office Services'),
110
+ z.literal('29').describe('Real Estate'),
111
+ z.literal('30').describe('Computer Hardware'),
112
+ z.literal('31').describe('Programming'),
113
+ z.literal('32').describe('Software'),
114
+ z.literal('33').describe('Offbeat'),
115
+ z.literal('34').describe('Movies'),
116
+ z.literal('35').describe('Music & Audio'),
117
+ z.literal('36').describe('TV & Video'),
118
+ z.literal('37').describe('Banking'),
119
+ z.literal('38').describe('Insurance'),
120
+ z.literal('39').describe('Card Games'),
121
+ z.literal('41').describe('Computer & Video Games'),
122
+ z.literal('42').describe('Jazz'),
123
+ z.literal('43').describe('Online Goodies'),
124
+ z.literal('44').describe('Beauty & Fitness'),
125
+ z.literal('45').describe('Health'),
126
+ z.literal('46').describe('Agriculture & Forestry'),
127
+ z.literal('47').describe('Autos & Vehicles'),
128
+ z.literal('48').describe('Construction & Maintenance'),
129
+ z.literal('49').describe('Manufacturing'),
130
+ z.literal('50').describe('Transportation & Logistics'),
131
+ z.literal('53').describe('Web Hosting & Domain Registration'),
132
+ z.literal('54').describe('Social Issues & Advocacy'),
133
+ z.literal('55').describe('Dating & Personals'),
134
+ z.literal('56').describe('Ethnic & Identity Groups'),
135
+ z.literal('57').describe('Charity & Philanthropy'),
136
+ z.literal('58').describe('Parenting'),
137
+ z.literal('59').describe('Religion & Belief'),
138
+ z.literal('60').describe('Jobs'),
139
+ z.literal('61').describe('Classifieds'),
140
+ z.literal('63').describe('Weather'),
141
+ z.literal('64').describe('Antiques & Collectibles'),
142
+ z.literal('65').describe('Hobbies & Leisure'),
143
+ z.literal('66').describe('Pets & Animals'),
144
+ z.literal('67').describe('Travel'),
145
+ z.literal('68').describe('Apparel'),
146
+ z.literal('69').describe('Consumer Resources'),
147
+ z.literal('70').describe('Gifts & Special Event Items'),
148
+ z.literal('71').describe('Food & Drink'),
149
+ z.literal('73').describe('Mass Merchants & Department Stores'),
150
+ z.literal('74').describe('Education'),
151
+ z.literal('75').describe('Legal'),
152
+ z.literal('76').describe('Government'),
153
+ z.literal('77').describe('Enterprise Technology'),
154
+ z.literal('78').describe('Consumer Electronics'),
155
+ z.literal('82').describe('Environmental Issues'),
156
+ z.literal('83').describe('Marketing Services'),
157
+ z.literal('84').describe('Search Engine Optimization & Marketing'),
158
+ z.literal('89').describe('Vehicle Parts & Accessories'),
159
+ z.literal('91').describe('Stereo Systems & Components'),
160
+ z.literal('93').describe('Skin & Nail Care'),
161
+ z.literal('94').describe('Fitness'),
162
+ z.literal('95').describe('Office Supplies'),
163
+ z.literal('96').describe('Real Estate Agencies'),
164
+ z.literal('97').describe('Consumer Advocacy & Protection'),
165
+ z.literal('98').describe('Fashion Designers & Collections'),
166
+ z.literal('99').describe('Gifts'),
167
+ z.literal('100').describe('Cards & Greetings'),
168
+ z.literal('101').describe('Spirituality'),
169
+ z.literal('102').describe('Personals'),
170
+ z.literal('104').describe('ISPs'),
171
+ z.literal('105').describe('Online Games'),
172
+ z.literal('107').describe('Investing'),
173
+ z.literal('108').describe('Language Resources'),
174
+ z.literal('112').describe('Broadcast & Network News'),
175
+ z.literal('113').describe('Gay-Lesbian-Bisexual-Transgender'),
176
+ z.literal('115').describe('Baby Care & Hygiene'),
177
+ z.literal('118').describe('Water Sports'),
178
+ z.literal('119').describe('Wildlife'),
179
+ z.literal('120').describe('Cookware & Diningware'),
180
+ z.literal('121').describe('Grocery & Food Retailers'),
181
+ z.literal('122').describe('Cooking & Recipes'),
182
+ z.literal('123').describe('Tobacco Products'),
183
+ z.literal('124').describe('Clothing Accessories'),
184
+ z.literal('137').describe('Homemaking & Interior Decor'),
185
+ z.literal('138').describe('Vehicle Maintenance'),
186
+ z.literal('143').describe('Face & Body Care'),
187
+ z.literal('144').describe('Unwanted Body & Facial Hair Removal'),
188
+ z.literal('145').describe('Spas & Beauty Services'),
189
+ z.literal('146').describe('Hair Care'),
190
+ z.literal('147').describe('Cosmetology & Beauty Professionals'),
191
+ z.literal('148').describe('Off-Road Vehicles'),
192
+ z.literal('154').describe('Kids & Teens'),
193
+ z.literal('157').describe('Human Resources'),
194
+ z.literal('158').describe('Home Improvement'),
195
+ z.literal('166').describe('Public Safety'),
196
+ z.literal('168').describe('Emergency Services'),
197
+ z.literal('170').describe('Vehicle Licensing & Registration')
198
+ ])
199
+ .describe('Parameter is used to define a search category. The default value is set to 0 ("All categories").')
200
+ .default('0')
201
+ },
202
+ handle: async (params, client) => {
203
+ return wrapMcpResponse(() => client.deepserp.scrape({
204
+ actor: 'scraper.google.trends',
205
+ input: params
206
+ }));
207
+ }
208
+ });
@@ -0,0 +1,9 @@
1
+ export * from "./deepserp/googleSearch.js";
2
+ export * from "./deepserp/googleTrends.js";
3
+ export * from "./universal/scrapeHtml.js";
4
+ export * from "./universal/scrapeMarkdown.js";
5
+ export * from "./universal/scrapeScreenshot.js";
6
+ export * from "./crawl/crawlStart.js";
7
+ export * from "./crawl/crawlCancel.js";
8
+ export * from "./crawl/crawlResult.js";
9
+ export * from "./llm_chat_scraper/llmChatScraper.js";
@@ -0,0 +1,27 @@
1
+ import axios from "axios";
2
+ import { API_KEY, BASE_URL, API_KEY_NAME } from "../../config.js";
3
+ /**
4
+ * Build an axios instance targeting the Scrapeless v2 scraper task API.
5
+ *
6
+ * The `x-api-token` is resolved from the per-request client when available
7
+ * (so multi-tenant HTTP mode keeps using the caller's key) and falls back to
8
+ * the `SCRAPELESS_KEY` environment variable.
9
+ */
10
+ export function getLlmChatScraperApi(client, headers) {
11
+ // Priority: explicit request header (HTTP multi-tenant) -> key carried by the
12
+ // per-request client -> SCRAPELESS_KEY env (stdio / fallback).
13
+ const apiKey = headers?.[API_KEY_NAME] ||
14
+ client?.scraping?.apiKey ||
15
+ API_KEY ||
16
+ "";
17
+ return axios.create({
18
+ baseURL: BASE_URL,
19
+ headers: {
20
+ "Content-Type": "application/json",
21
+ [API_KEY_NAME]: apiKey,
22
+ },
23
+ timeout: 30000,
24
+ });
25
+ }
26
+ export const LLM_CHAT_SCRAPER_REQUEST_ENDPOINT = "/api/v2/scraper/request";
27
+ export const LLM_CHAT_SCRAPER_RESULT_ENDPOINT = "/api/v2/scraper/result";