scrapercity 1.0.4 → 1.0.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bin/mcp.mjs CHANGED
@@ -1,18 +1,22 @@
1
1
  #!/usr/bin/env node
2
2
  // bin/mcp.mjs - ScraperCity MCP Server (stdio transport)
3
- // Connect from Claude Code / Cursor / any MCP client
3
+ // Connect from Claude Code / Cursor / any MCP client.
4
+ //
5
+ // The SCRAPER tools are generated from scraperConfigs.ts (see lib/catalog.generated.mjs)
6
+ // and dispatched generically: the tool's args are POSTed straight to its /api/v1/scrape
7
+ // endpoint. Nothing per-scraper is hand-written here — add a scraper to the config and it
8
+ // shows up automatically on the next publish. Only the non-scraper utility/DB tools below
9
+ // are hand-defined.
4
10
  import { Server } from '@modelcontextprotocol/sdk/server/index.js'
5
11
  import { StdioServerTransport } from '@modelcontextprotocol/sdk/server/stdio.js'
6
12
  import { ListToolsRequestSchema, CallToolRequestSchema } from '@modelcontextprotocol/sdk/types.js'
7
13
  import * as sc from '../lib/client.mjs'
14
+ import { TOOLS as SCRAPER_TOOLS, ENDPOINT_BY_TOOL } from '../lib/catalog.generated.mjs'
8
15
 
9
- const server = new Server(
10
- { name: 'scrapercity', version: '1.0.0' },
11
- { capabilities: { tools: {} } }
12
- )
16
+ const server = new Server({ name: 'scrapercity', version: '1.0.0' }, { capabilities: { tools: {} } })
13
17
 
14
- // ── Tool definitions ──────────────────────────────────────────
15
- const TOOLS = [
18
+ // ── Non-scraper tools (utility + database), hand-defined ──────
19
+ const UTILITY_TOOLS = [
16
20
  {
17
21
  name: 'check_wallet',
18
22
  description: 'Check account balance, plan, and billing info. Call this FIRST to verify credits before running any scrape.',
@@ -21,425 +25,88 @@ const TOOLS = [
21
25
  {
22
26
  name: 'list_runs',
23
27
  description: 'List recent scraper runs with status, lead counts, and costs.',
24
- inputSchema: {
25
- type: 'object',
26
- properties: {
27
- hours: { type: 'number', description: 'How many hours back to look (default 24, max 168)', default: 24 },
28
- limit: { type: 'number', description: 'Max runs to return (default 20, max 100)', default: 20 }
29
- }
30
- }
31
- },
32
- {
33
- name: 'scrape_apollo',
34
- description: 'Scrape leads from Apollo.io using a search URL. IMPORTANT: Apollo delivery takes 11-48+ hours. Use webhooks (configure at app.scrapercity.com/dashboard/webhooks) instead of polling. Returns a runId to check status later.',
35
- inputSchema: {
36
- type: 'object',
37
- properties: {
38
- url: { type: 'string', description: 'Full Apollo.io search URL (from the People search page)' },
39
- count: { type: 'number', description: 'Number of leads to pull (min 500, max 50000)', default: 1000 },
40
- fileName: { type: 'string', description: 'Name for this export', default: '' }
41
- },
42
- required: ['url']
43
- }
44
- },
45
- {
46
- name: 'scrape_maps',
47
- description: 'Scrape businesses from Google Maps. Returns businesses with names, addresses, phone numbers, websites, emails, ratings. Typically completes in 5-30 minutes.',
48
- inputSchema: {
49
- type: 'object',
50
- properties: {
51
- query: { type: 'string', description: 'Search keyword, e.g. "plumbers", "dentists", "restaurants"' },
52
- location: { type: 'string', description: 'City/area, e.g. "Denver, CO", "London, UK"' },
53
- limit: { type: 'number', description: 'Max places to scrape (default 500)', default: 500 }
54
- },
55
- required: ['query', 'location']
56
- }
57
- },
58
- {
59
- name: 'validate_emails',
60
- description: 'Validate a list of email addresses. Returns deliverability status, catch-all detection, MX records, and company info. Typically completes in 1-10 minutes.',
61
- inputSchema: {
62
- type: 'object',
63
- properties: {
64
- emails: { type: 'array', items: { type: 'string' }, description: 'Email addresses to validate' }
65
- },
66
- required: ['emails']
67
- }
68
- },
69
- {
70
- name: 'find_emails',
71
- description: 'Find business email addresses given a person name and company. Provide first/last name + domain or company name.',
72
- inputSchema: {
73
- type: 'object',
74
- properties: {
75
- contacts: {
76
- type: 'array',
77
- items: {
78
- type: 'object',
79
- properties: {
80
- first_name: { type: 'string' },
81
- last_name: { type: 'string' },
82
- full_name: { type: 'string', description: 'Alternative to first/last' },
83
- domain: { type: 'string', description: 'Company domain, e.g. "acme.com"' },
84
- company_name: { type: 'string', description: 'Alternative to domain' }
85
- }
86
- },
87
- description: 'Contacts to find emails for. Each needs a name AND a company/domain.'
88
- },
89
- autoValidateEmails: { type: 'boolean', description: 'Auto-validate found emails', default: false },
90
- autoFindMobiles: { type: 'boolean', description: 'Auto-find mobile numbers', default: false }
91
- },
92
- required: ['contacts']
93
- }
94
- },
95
- {
96
- name: 'find_mobiles',
97
- description: 'Find mobile phone numbers from LinkedIn URLs or work emails.',
98
- inputSchema: {
99
- type: 'object',
100
- properties: {
101
- inputs: { type: 'array', items: { type: 'string' }, description: 'LinkedIn profile URLs or work email addresses' }
102
- },
103
- required: ['inputs']
104
- }
105
- },
106
- {
107
- name: 'find_people',
108
- description: 'Skip trace / people finder. Look up personal info by name, email, phone, or address.',
109
- inputSchema: {
110
- type: 'object',
111
- properties: {
112
- name: { type: 'array', items: { type: 'string' }, description: 'Full names to search' },
113
- email: { type: 'array', items: { type: 'string' }, description: 'Email addresses to search' },
114
- phone_number: { type: 'array', items: { type: 'string' }, description: 'Phone numbers to search' },
115
- street_citystatezip: { type: 'array', items: { type: 'string' }, description: 'Addresses (format: "123 Main St, City, ST 12345")' },
116
- max_results: { type: 'number', description: 'Results per search (default 1)', default: 1 }
117
- }
118
- }
119
- },
120
- {
121
- name: 'scrape_store_leads',
122
- description: 'Get ecommerce store data (Shopify, WooCommerce, etc). Returns domains, emails, phones, social profiles, revenue estimates. Instant results from cached database.',
123
- inputSchema: {
124
- type: 'object',
125
- properties: {
126
- platform: { type: 'string', description: 'e.g. "shopify", "woocommerce", "bigcommerce"', default: 'shopify' },
127
- countryCode: { type: 'string', description: 'e.g. "US", "GB", "CA"' },
128
- category: { type: 'string', description: 'Store category filter' },
129
- city: { type: 'string', description: 'City filter' },
130
- technologies: { type: 'string', description: 'Technology filter' },
131
- apps: { type: 'string', description: 'Installed app filter' },
132
- emails: { type: 'boolean', description: 'Only stores with emails', default: false },
133
- phones: { type: 'boolean', description: 'Only stores with phones', default: false },
134
- instagram: { type: 'boolean', description: 'Only stores with Instagram' },
135
- facebook: { type: 'boolean', description: 'Only stores with Facebook' },
136
- totalLeads: { type: 'number', description: 'Number of leads to pull', default: 1000 }
137
- }
138
- }
139
- },
140
- {
141
- name: 'scrape_builtwith',
142
- description: 'Find all websites using a specific technology. Returns domains with contact info. $4.99 per search.',
143
- inputSchema: {
144
- type: 'object',
145
- properties: {
146
- technology: { type: 'string', description: 'Technology name, e.g. "Shopify", "Stripe", "HubSpot"' },
147
- fileName: { type: 'string', description: 'Export name' }
148
- },
149
- required: ['technology']
150
- }
151
- },
152
- {
153
- name: 'search_criminal_records',
154
- description: 'Search criminal records by name. $1 per search, only charged if records found.',
155
- inputSchema: {
156
- type: 'object',
157
- properties: {
158
- name: { type: 'string', description: 'Full name to search' },
159
- state: { type: 'string', description: 'US state code, e.g. "CA", "TX" (optional, searches all if omitted)' },
160
- dob: { type: 'string', description: 'Date of birth "MM/DD/YYYY" (optional, improves accuracy)' }
161
- },
162
- required: ['name']
163
- }
164
- },
165
- {
166
- name: 'scrape_airbnb',
167
- description: 'Scrape Airbnb host emails by city or listing URL. Returns host contact info including email addresses.',
168
- inputSchema: {
169
- type: 'object',
170
- properties: {
171
- mode: { type: 'string', enum: ['city', 'single', 'bulk'], description: 'Search mode', default: 'city' },
172
- city: { type: 'array', items: { type: 'string' }, description: 'Cities to search (for city mode), e.g. ["Miami, FL"]' },
173
- listingUrl: { type: 'string', description: 'Single Airbnb listing URL (for single mode)' },
174
- bulkListings: { type: 'array', items: { type: 'string' }, description: 'Multiple listing URLs (for bulk mode)' },
175
- maxResults: { type: 'number', description: 'Max results for billing cap', default: 100 },
176
- checkin: { type: 'string', description: 'Check-in date YYYY-MM-DD' },
177
- checkout: { type: 'string', description: 'Check-out date YYYY-MM-DD' },
178
- maxPages: { type: 'number', description: 'Max pages to crawl per city', default: 1 },
179
- onlyUniqueEmails: { type: 'boolean', description: 'Deduplicate by email', default: false },
180
- onlyProHosts: { type: 'boolean', description: 'Only professional hosts', default: false }
181
- }
182
- }
183
- },
184
- {
185
- name: 'scrape_youtube_email',
186
- description: 'Find business emails for YouTube channels. Pass channel handles (@ChannelName) or URLs.',
187
- inputSchema: {
188
- type: 'object',
189
- properties: {
190
- channels: { type: 'array', items: { type: 'string' }, description: 'YouTube channel handles or URLs, e.g. ["@MrBeast", "https://youtube.com/@Channel"]' }
191
- },
192
- required: ['channels']
193
- }
194
- },
195
- {
196
- name: 'scrape_website_finder',
197
- description: 'Find contact info (emails, phones, social links) from a list of website domains. Optionally filter by job title.',
198
- inputSchema: {
199
- type: 'object',
200
- properties: {
201
- domains: { type: 'array', items: { type: 'string' }, description: 'Website domains, e.g. ["acme.com", "example.com"]' },
202
- jobTitle: { type: 'string', description: 'Filter contacts by job title, e.g. "CEO"' }
203
- },
204
- required: ['domains']
205
- }
206
- },
207
- {
208
- name: 'scrape_yelp',
209
- description: 'Scrape business listings from Yelp. Search by keyword + location, or provide direct Yelp URLs.',
210
- inputSchema: {
211
- type: 'object',
212
- properties: {
213
- searchTerms: { type: 'array', items: { type: 'string' }, description: 'Search keywords, e.g. ["plumbers", "dentists"]' },
214
- locations: { type: 'array', items: { type: 'string' }, description: 'Locations, e.g. ["Denver, CO"]' },
215
- directUrls: { type: 'array', items: { type: 'string' }, description: 'Direct Yelp search/business URLs' },
216
- searchLimit: { type: 'number', description: 'Max results per search', default: 10 }
217
- }
218
- }
219
- },
220
- {
221
- name: 'scrape_angi',
222
- description: 'Scrape service provider listings from Angi (Angie\'s List). Search by keyword and zip codes.',
223
- inputSchema: {
224
- type: 'object',
225
- properties: {
226
- keyword: { type: 'string', description: 'Service type, e.g. "plumbers", "electricians"' },
227
- zipCodes: { type: 'array', items: { type: 'string' }, description: 'ZIP codes to search, e.g. ["80202", "80203"]' },
228
- maxItems: { type: 'number', description: 'Max results', default: 100 }
229
- },
230
- required: ['keyword']
231
- }
232
- },
233
- {
234
- name: 'scrape_zillow_agents',
235
- description: 'Scrape real estate agent listings from Zillow by location.',
236
- inputSchema: {
237
- type: 'object',
238
- properties: {
239
- location: { type: 'string', description: 'City or area, e.g. "Denver, CO"' },
240
- specialty: { type: 'string', description: 'Agent specialty, e.g. "buyer", "seller"' },
241
- language: { type: 'string', description: 'Language filter' },
242
- searchLimit: { type: 'number', description: 'Max results', default: 10 }
243
- },
244
- required: ['location']
245
- }
246
- },
247
- {
248
- name: 'scrape_bizbuysell',
249
- description: 'Scrape business-for-sale listings from BizBuySell. Provide search result URLs.',
250
- inputSchema: {
251
- type: 'object',
252
- properties: {
253
- startUrls: { type: 'array', items: { type: 'string' }, description: 'BizBuySell search URLs' },
254
- maxItems: { type: 'number', description: 'Max listings to scrape', default: 100 }
255
- },
256
- required: ['startUrls']
257
- }
258
- },
259
- {
260
- name: 'scrape_crexi',
261
- description: 'Scrape commercial real estate listings from Crexi. Provide search result URLs.',
262
- inputSchema: {
263
- type: 'object',
264
- properties: {
265
- startUrls: { type: 'array', items: { type: 'string' }, description: 'Crexi search URLs' }
266
- },
267
- required: ['startUrls']
268
- }
269
- },
270
- {
271
- name: 'scrape_property_lookup',
272
- description: 'Look up property data by address. Optionally include owner contact information. $0.15 per address.',
273
- inputSchema: {
274
- type: 'object',
275
- properties: {
276
- addresses: { type: 'array', items: { type: 'string' }, description: 'Full addresses, e.g. ["123 Main St, Denver, CO 80202"]' },
277
- includeOwnerContact: { type: 'boolean', description: 'Include property owner contact info', default: false }
278
- },
279
- required: ['addresses']
280
- }
28
+ inputSchema: { type: 'object', properties: {
29
+ hours: { type: 'number', description: 'How many hours back to look (default 24, max 168)', default: 24 },
30
+ limit: { type: 'number', description: 'Max runs to return (default 20, max 100)', default: 20 }
31
+ } }
281
32
  },
282
33
  {
283
34
  name: 'check_run_status',
284
35
  description: 'Check the status of a scraper run. Returns status (RUNNING/SUCCEEDED/FAILED/CANCELLED), lead count, and download URL when complete.',
285
- inputSchema: {
286
- type: 'object',
287
- properties: {
288
- runId: { type: 'string', description: 'The run ID returned when starting a scrape' }
289
- },
290
- required: ['runId']
291
- }
36
+ inputSchema: { type: 'object', properties: { runId: { type: 'string', description: 'The run ID returned when starting a scrape' } }, required: ['runId'] }
292
37
  },
293
38
  {
294
39
  name: 'download_results',
295
40
  description: 'Download the CSV results of a completed scraper run. Only works when status is SUCCEEDED.',
296
- inputSchema: {
297
- type: 'object',
298
- properties: {
299
- runId: { type: 'string', description: 'The run ID to download results for' },
300
- outputPath: { type: 'string', description: 'File path to save CSV (default: {runId}.csv)' }
301
- },
302
- required: ['runId']
303
- }
41
+ inputSchema: { type: 'object', properties: {
42
+ runId: { type: 'string', description: 'The run ID to download results for' },
43
+ outputPath: { type: 'string', description: 'File path to save CSV (default: {runId}.csv)' }
44
+ }, required: ['runId'] }
304
45
  },
305
46
  {
306
47
  name: 'cancel_run',
307
48
  description: 'Cancel a running scraper job.',
308
- inputSchema: {
309
- type: 'object',
310
- properties: {
311
- runId: { type: 'string', description: 'The run ID to cancel' }
312
- },
313
- required: ['runId']
314
- }
49
+ inputSchema: { type: 'object', properties: { runId: { type: 'string', description: 'The run ID to cancel' } }, required: ['runId'] }
315
50
  },
316
51
  {
317
52
  name: 'query_lead_database',
318
- description: 'Query the B2B lead database directly (requires $649/mo plan). Returns contacts with names, emails, phones, titles, companies. 100 per request, paginate with page param.',
319
- inputSchema: {
320
- type: 'object',
321
- properties: {
322
- title: { type: 'string', description: 'Job title filter' },
323
- industry: { type: 'string', description: 'Company industry' },
324
- country: { type: 'string', description: 'Person country' },
325
- state: { type: 'string', description: 'Person state' },
326
- city: { type: 'string', description: 'Person city' },
327
- companyName: { type: 'string', description: 'Company name' },
328
- companyDomain: { type: 'string', description: 'Company domain' },
329
- seniority: { type: 'string', description: 'e.g. "vp", "director", "c_suite"' },
330
- department: { type: 'string', description: 'e.g. "sales", "engineering"' },
331
- hasEmail: { type: 'boolean', description: 'Only contacts with email' },
332
- hasPhone: { type: 'boolean', description: 'Only contacts with phone' },
333
- page: { type: 'number', description: 'Page number (default 1)', default: 1 },
334
- limit: { type: 'number', description: 'Results per page (max 100)', default: 50 }
335
- }
336
- }
53
+ description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies. 100 per request, paginate with page param. Included with the $149/mo plan.',
54
+ inputSchema: { type: 'object', properties: {
55
+ title: { type: 'string', description: 'Job title filter' },
56
+ industry: { type: 'string', description: 'Company industry' },
57
+ country: { type: 'string', description: 'Person country' },
58
+ state: { type: 'string', description: 'Person state' },
59
+ city: { type: 'string', description: 'Person city' },
60
+ companyName: { type: 'string', description: 'Company name' },
61
+ companyDomain: { type: 'string', description: 'Company domain' },
62
+ seniority: { type: 'string', description: 'e.g. "vp", "director", "c_suite"' },
63
+ department: { type: 'string', description: 'e.g. "sales", "engineering"' },
64
+ hasEmail: { type: 'boolean', description: 'Only contacts with email' },
65
+ hasPhone: { type: 'boolean', description: 'Only contacts with phone' },
66
+ page: { type: 'number', description: 'Page number (default 1)', default: 1 },
67
+ limit: { type: 'number', description: 'Results per page (max 100)', default: 50 }
68
+ } }
337
69
  }
338
70
  ]
339
71
 
340
- // ── Register tool list handler ────────────────────────────────
341
- server.setRequestHandler(ListToolsRequestSchema, async () => {
342
- return { tools: TOOLS }
343
- })
72
+ const TOOLS = [...UTILITY_TOOLS, ...SCRAPER_TOOLS]
73
+
74
+ server.setRequestHandler(ListToolsRequestSchema, async () => ({ tools: TOOLS }))
344
75
 
345
- // ── Tool execution handler ────────────────────────────────────
346
76
  server.setRequestHandler(CallToolRequestSchema, async (request) => {
347
- const { name, arguments: args } = request.params
77
+ const { name, arguments: args = {} } = request.params
348
78
  try {
349
79
  let result
350
- switch (name) {
351
- case 'check_wallet':
352
- result = await sc.wallet()
353
- break
354
- case 'list_runs':
355
- result = await sc.runs(args?.hours || 24, args?.limit || 20)
356
- break
357
- case 'scrape_apollo':
358
- result = await sc.apollo(args.url, args.count || 1000, args.fileName || '')
359
- result._note = 'Apollo takes 11-48+ hours. Configure webhook at app.scrapercity.com/dashboard/webhooks instead of polling.'
360
- break
361
- case 'scrape_maps':
362
- result = await sc.maps(args.query, args.location, args.limit || 500)
363
- break
364
- case 'validate_emails':
365
- result = await sc.emailValidate(args.emails)
366
- break
367
- case 'find_emails':
368
- result = await sc.emailFind(args.contacts, {
369
- validate: args.autoValidateEmails || false,
370
- mobiles: args.autoFindMobiles || false
371
- })
372
- break
373
- case 'find_mobiles':
374
- result = await sc.mobileFinder(args.inputs)
375
- break
376
- case 'find_people':
377
- result = await sc.peopleFinder(args)
378
- break
379
- case 'scrape_store_leads':
380
- result = await sc.storeLeads(args)
381
- break
382
- case 'scrape_builtwith':
383
- result = await sc.builtwith(args.technology, args.fileName)
384
- break
385
- case 'search_criminal_records':
386
- result = await sc.criminal(args.name, args.state, args.dob)
387
- break
388
- case 'scrape_airbnb':
389
- result = await sc.airbnb(args)
390
- break
391
- case 'scrape_youtube_email':
392
- result = await sc.youtubeEmail(args.channels)
393
- break
394
- case 'scrape_website_finder':
395
- result = await sc.websiteFinder(args.domains, args.jobTitle)
396
- break
397
- case 'scrape_yelp':
398
- result = await sc.yelp(args)
399
- break
400
- case 'scrape_angi':
401
- result = await sc.angi(args.keyword, args.zipCodes || [], args.maxItems || 100)
402
- break
403
- case 'scrape_zillow_agents':
404
- result = await sc.zillowAgents(args.location, { specialty: args.specialty, language: args.language, searchLimit: args.searchLimit || 10 })
405
- break
406
- case 'scrape_bizbuysell':
407
- result = await sc.bizbuysell(args.startUrls, args.maxItems || 100)
408
- break
409
- case 'scrape_crexi':
410
- result = await sc.crexi(args.startUrls)
411
- break
412
- case 'scrape_property_lookup':
413
- result = await sc.propertyLookup(args.addresses, args.includeOwnerContact || false)
414
- break
415
- case 'check_run_status':
416
- result = await sc.status(args.runId)
417
- break
418
- case 'download_results': {
419
- const dl = await sc.download(args.runId, args.outputPath)
420
- result = { success: true, path: dl.path, sizeKB: Math.round(dl.bytes / 1024) }
421
- break
80
+ // Generated scraper tools: POST args straight to the canonical endpoint.
81
+ if (ENDPOINT_BY_TOOL[name]) {
82
+ result = await sc.postEndpoint(ENDPOINT_BY_TOOL[name], args)
83
+ } else {
84
+ switch (name) {
85
+ case 'check_wallet': result = await sc.wallet(); break
86
+ case 'list_runs': result = await sc.runs(args.hours || 24, args.limit || 20); break
87
+ case 'check_run_status': result = await sc.status(args.runId); break
88
+ case 'download_results': {
89
+ const dl = await sc.download(args.runId, args.outputPath)
90
+ result = { success: true, path: dl.path, sizeKB: Math.round(dl.bytes / 1024) }
91
+ break
92
+ }
93
+ case 'cancel_run': result = await sc.cancel(args.runId); break
94
+ case 'query_lead_database': {
95
+ const params = { ...args }
96
+ if (params.hasEmail) params.hasEmail = 'true'
97
+ if (params.hasPhone) params.hasPhone = 'true'
98
+ result = await sc.dbLeads(params)
99
+ break
100
+ }
101
+ default:
102
+ return { content: [{ type: 'text', text: `Unknown tool: ${name}` }], isError: true }
422
103
  }
423
- case 'cancel_run':
424
- result = await sc.cancel(args.runId)
425
- break
426
- case 'query_lead_database': {
427
- const params = { ...args }
428
- if (params.hasEmail) { params.hasEmail = 'true'; }
429
- if (params.hasPhone) { params.hasPhone = 'true'; }
430
- result = await sc.dbLeads(params)
431
- break
432
- }
433
- default:
434
- return { content: [{ type: 'text', text: `Unknown tool: ${name}` }], isError: true }
435
104
  }
436
-
437
105
  return { content: [{ type: 'text', text: JSON.stringify(result, null, 2) }] }
438
106
  } catch (e) {
439
107
  return { content: [{ type: 'text', text: `Error: ${e.message}` }], isError: true }
440
108
  }
441
109
  })
442
110
 
443
- // ── Start ─────────────────────────────────────────────────────
444
111
  const transport = new StdioServerTransport()
445
112
  await server.connect(transport)