scrapercity 1.0.14 → 1.0.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENT_DOCS.txt +16 -5
- package/SKILL.md +2 -2
- package/bin/cli.mjs +37 -3
- package/bin/mcp.mjs +36 -2
- package/lib/catalog.generated.mjs +1 -1
- package/lib/client.mjs +8 -1
- package/package.json +1 -1
package/AGENT_DOCS.txt
CHANGED
|
@@ -189,8 +189,11 @@ GET /api/v1/database/leads?title=CTO&country=United States&hasEmail=true&page=1&
|
|
|
189
189
|
Company filters: keywords (company tags, whole-word, any match), revenueMin, revenueMax (USD),
|
|
190
190
|
companyCountry, companyState, companyCity (company HQ, not the person)
|
|
191
191
|
Exclusions: notTitle, notKeywords, notIndustry
|
|
192
|
-
|
|
193
|
-
The response adds
|
|
192
|
+
url=<people-search URL>: use that search's filters (other params override).
|
|
193
|
+
The response adds search_translation: { applied, approximated, not_supported }.
|
|
194
|
+
A URL with no filters the database can use returns 400.
|
|
195
|
+
POST /api/v1/database/leads with a JSON body takes the same parameters (arrays as JSON
|
|
196
|
+
arrays). Use it for a long url or long lists: a query string over ~16KB is refused.
|
|
194
197
|
Country: full name ("United States"); US, USA, UK, UAE also work.
|
|
195
198
|
Array params use repeated keys: ?seniority=vp&seniority=director
|
|
196
199
|
(keywords / notKeywords also accept commas: ?keywords=saas,fintech)
|
|
@@ -202,16 +205,24 @@ GET /api/v1/database/leads?title=CTO&country=United States&hasEmail=true&page=1&
|
|
|
202
205
|
Every lead returned is saved to your account. Only NEW leads count toward the
|
|
203
206
|
100k/day limit; fetching a lead you already have again is free.
|
|
204
207
|
Returns: { data: [...leads], pagination: { page, limit, total, totalPages, has_more, next_after }, rate_limit,
|
|
205
|
-
|
|
208
|
+
search_translation? }
|
|
206
209
|
Lead fields include company_industry, company_size, employees_count, company_website,
|
|
207
210
|
company_annual_revenue, company_city, company_state, company_country, keywords.
|
|
208
211
|
409 with filters_not_applied_yet: the company filters listed can't be used yet (company
|
|
209
212
|
data still being prepared right after launch). Nothing is returned or counted; retry later
|
|
210
213
|
or drop those filters.
|
|
214
|
+
503: the search took too long to run. Narrow it (fewer titles or keywords, add a location
|
|
215
|
+
or company size) and retry.
|
|
216
|
+
|
|
217
|
+
GET /api/v1/database/local-businesses?category=Dentist&country=US&hasContact=true&limit=100
|
|
218
|
+
Local businesses ($149/mo plan and up, 100k new businesses/day), same pattern as above.
|
|
219
|
+
Filters: category (repeat for several), country (two-letter code), state, city, postalCode,
|
|
220
|
+
query (business name and description), minRating, maxRating, minReviews, maxReviews,
|
|
221
|
+
hasEmail, hasPhone, hasWebsite, hasContact (only businesses with a named contact person).
|
|
222
|
+
Optional: excludeDelivered=true, after=<id> (as above).
|
|
211
223
|
|
|
212
|
-
GET /api/v1/database/local-businesses?...
|
|
213
224
|
GET /api/v1/database/ecommerce?...
|
|
214
|
-
Same pattern (including excludeDelivered and after), different
|
|
225
|
+
Same pattern (including excludeDelivered and after), different dataset.
|
|
215
226
|
|
|
216
227
|
────────────────────────────────────────
|
|
217
228
|
ERROR CODES
|
package/SKILL.md
CHANGED
|
@@ -78,8 +78,8 @@ Apollo scrapes take **up to 4 days** to deliver. Do NOT poll in a loop.
|
|
|
78
78
|
| POST | `/api/v1/scrape/cancel/{runId}` | Cancel running job |
|
|
79
79
|
| GET | `/api/v1/scrape/logs/{runId}` | Run logs |
|
|
80
80
|
| GET | `/api/v1/apollo-status` | Apollo service health: `{status: "url-based" or "legacy", message, timestamp}` |
|
|
81
|
-
| GET | `/api/v1/database/leads?title=CTO&country=United%20States&hasEmail=true&limit=100` | Lead DB ($149/mo plan and up, 100k new leads/day). Optional: `excludeDelivered=true` (only leads you don't have yet), `after=0` then `pagination.next_after` (cursor paging). Company filters: `keywords`, `revenueMin`/`revenueMax`, `companyCountry`/`companyState`/`companyCity`. Exclusions: `notTitle`, `notKeywords`, `notIndustry`. `
|
|
82
|
-
| GET | `/api/v1/database/local-businesses
|
|
81
|
+
| GET | `/api/v1/database/leads?title=CTO&country=United%20States&hasEmail=true&limit=100` | Lead DB ($149/mo plan and up, 100k new leads/day). Optional: `excludeDelivered=true` (only leads you don't have yet), `after=0` then `pagination.next_after` (cursor paging). Company filters: `keywords`, `revenueMin`/`revenueMax`, `companyCountry`/`companyState`/`companyCity`. Exclusions: `notTitle`, `notKeywords`, `notIndustry`. `url=<people-search URL>` searches with that URL's filters. POST with a JSON body takes the same parameters (use it for a long `url`) |
|
|
82
|
+
| GET | `/api/v1/database/local-businesses?category=Dentist&country=US&hasContact=true&limit=100` | Local biz DB ($149/mo plan and up, 100k new businesses/day). Filters: `category`, `country`, `state`, `city`, `postalCode`, `query`, `minRating`/`maxRating`, `minReviews`/`maxReviews`, `hasEmail`, `hasPhone`, `hasWebsite`, `hasContact` (only businesses with a named contact person). Same optional `excludeDelivered` / `after` |
|
|
83
83
|
| GET | `/api/v1/database/ecommerce?...` | Ecommerce DB ($149/mo plan and up). Same optional `excludeDelivered` / `after` |
|
|
84
84
|
|
|
85
85
|
## Status Values
|
package/bin/cli.mjs
CHANGED
|
@@ -376,7 +376,7 @@ async function main() {
|
|
|
376
376
|
'--seniority': 'seniority', '--department': 'department',
|
|
377
377
|
'--page': 'page', '--limit': 'limit',
|
|
378
378
|
'--min-employees': 'minEmployees', '--max-employees': 'maxEmployees',
|
|
379
|
-
'--after': 'after', '--apollo-url': 'apolloUrl',
|
|
379
|
+
'--after': 'after', '--url': 'url', '--apollo-url': 'apolloUrl',
|
|
380
380
|
'--keywords': 'keywords', '--revenue-min': 'revenueMin', '--revenue-max': 'revenueMax',
|
|
381
381
|
'--company-country': 'companyCountry', '--company-state': 'companyState',
|
|
382
382
|
'--company-city': 'companyCity', '--not-title': 'notTitle',
|
|
@@ -395,7 +395,35 @@ async function main() {
|
|
|
395
395
|
const r = await sc.dbLeads(params)
|
|
396
396
|
console.log(`${r.pagination?.total || '?'} total leads, page ${r.pagination?.page || 1} of ${r.pagination?.totalPages || '?'}`)
|
|
397
397
|
if (r.pagination?.next_after) console.log(`Next page: --after ${r.pagination.next_after}`)
|
|
398
|
-
|
|
398
|
+
const tr = r.search_translation || r.apollo_translation
|
|
399
|
+
if (tr?.not_supported?.length) console.log(`Not supported from the search URL: ${tr.not_supported.join(', ')}`)
|
|
400
|
+
console.log(json(r.data?.slice(0, 3) || r))
|
|
401
|
+
if (r.data?.length > 3) console.log(`... and ${r.data.length - 3} more`)
|
|
402
|
+
break
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
// ── Database: Local businesses ($149/mo plan and up) ──
|
|
406
|
+
case 'db-local': {
|
|
407
|
+
const params = {}
|
|
408
|
+
const keyFor = { '--category': 'category', '--country': 'country', '--state': 'state', '--city': 'city',
|
|
409
|
+
'--postal-code': 'postalCode', '--query': 'query', '--min-rating': 'minRating', '--max-rating': 'maxRating',
|
|
410
|
+
'--min-reviews': 'minReviews', '--max-reviews': 'maxReviews', '--page': 'page', '--limit': 'limit',
|
|
411
|
+
'--after': 'after' }
|
|
412
|
+
// List filters take commas: --category "Dentist,Orthodontist" sends both.
|
|
413
|
+
const lists = new Set(['--category', '--country'])
|
|
414
|
+
for (const f of Object.keys(keyFor)) {
|
|
415
|
+
const v = flag(f)
|
|
416
|
+
if (v === undefined) continue
|
|
417
|
+
params[keyFor[f]] = lists.has(f) ? String(v).split(',').map(s => s.trim()).filter(Boolean) : v
|
|
418
|
+
}
|
|
419
|
+
if (flagBool('--has-email')) params.hasEmail = 'true'
|
|
420
|
+
if (flagBool('--has-phone')) params.hasPhone = 'true'
|
|
421
|
+
if (flagBool('--has-website')) params.hasWebsite = 'true'
|
|
422
|
+
if (flagBool('--has-contact')) params.hasContact = 'true'
|
|
423
|
+
if (flagBool('--exclude-delivered')) params.excludeDelivered = 'true'
|
|
424
|
+
const r = await sc.dbLocalBusinesses(params)
|
|
425
|
+
console.log(`${r.pagination?.total || '?'} total businesses, page ${r.pagination?.page || 1} of ${r.pagination?.totalPages || '?'}`)
|
|
426
|
+
if (r.pagination?.next_after) console.log(`Next page: --after ${r.pagination.next_after}`)
|
|
399
427
|
console.log(json(r.data?.slice(0, 3) || r))
|
|
400
428
|
if (r.data?.length > 3) console.log(`... and ${r.data.length - 3} more`)
|
|
401
429
|
break
|
|
@@ -457,9 +485,15 @@ ScraperCity CLI - B2B lead generation from your terminal
|
|
|
457
485
|
--min-employees --max-employees --has-email --has-phone
|
|
458
486
|
--keywords --revenue-min --revenue-max --company-country --company-state --company-city
|
|
459
487
|
--not-title --not-keywords --not-industry
|
|
460
|
-
--
|
|
488
|
+
--url "<people-search URL>" Search with that URL's filters
|
|
461
489
|
--exclude-delivered Only leads you don't already have
|
|
462
490
|
--after <id> Cursor paging (start with 0, then use the printed next id)
|
|
491
|
+
scrapercity db-local [filters] Query local business database
|
|
492
|
+
--category --country --state --city --postal-code --query
|
|
493
|
+
--min-rating --max-rating --min-reviews --max-reviews
|
|
494
|
+
--has-email --has-phone --has-website --has-contact (only businesses with a named contact person)
|
|
495
|
+
--exclude-delivered Only businesses you don't already have
|
|
496
|
+
--after <id> Cursor paging (start with 0, then use the printed next id)
|
|
463
497
|
|
|
464
498
|
Env: SCRAPERCITY_API_KEY=... or scrapercity login
|
|
465
499
|
Docs: https://scrapercity.com/agents
|
package/bin/mcp.mjs
CHANGED
|
@@ -50,9 +50,9 @@ const UTILITY_TOOLS = [
|
|
|
50
50
|
},
|
|
51
51
|
{
|
|
52
52
|
name: 'query_lead_database',
|
|
53
|
-
description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies (plus company website, revenue, HQ location and keywords). Up to 100 per request. Included with the $149/mo plan (100,000 new leads a day; leads you already have do not count again). For a daily pull of only new leads set excludeDelivered=true. For big pulls page with after: send after="0" first, then the pagination.next_after from each response until it is null. Have
|
|
53
|
+
description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies (plus company website, revenue, HQ location and keywords). Up to 100 per request. Included with the $149/mo plan (100,000 new leads a day; leads you already have do not count again). For a daily pull of only new leads set excludeDelivered=true. For big pulls page with after: send after="0" first, then the pagination.next_after from each response until it is null. Have a saved people-search URL? Pass it as url to search with its filters right away (search_translation in the response says what was applied).',
|
|
54
54
|
inputSchema: { type: 'object', properties: {
|
|
55
|
-
|
|
55
|
+
url: { type: 'string', description: 'A people-search URL (the address of a people search with its filters). Its filters become the search; any other filter you pass overrides the matching one.' },
|
|
56
56
|
title: { type: 'string', description: 'Job title filter, comma separated for several (partial match, e.g. "CEO, Founder")' },
|
|
57
57
|
industry: { type: 'array', items: { type: 'string' }, description: 'Company industries (any match)' },
|
|
58
58
|
country: { type: 'array', items: { type: 'string' }, description: 'Person countries, full names like "United States" (US, USA, UK, UAE also work)' },
|
|
@@ -81,6 +81,30 @@ const UTILITY_TOOLS = [
|
|
|
81
81
|
excludeDelivered: { type: 'boolean', description: 'Skip leads this account already has (from the API or unlocked in the dashboard). With this on, keep page at 1 or use after.' },
|
|
82
82
|
after: { type: 'string', description: 'Cursor paging: return leads after this lead id. Use "0" to start, then the pagination.next_after from each response.' }
|
|
83
83
|
} }
|
|
84
|
+
},
|
|
85
|
+
{
|
|
86
|
+
name: 'query_local_business_database',
|
|
87
|
+
description: 'Query the local business database directly. Returns businesses with phone numbers, emails, websites, addresses and ratings, many with a named contact person. Up to 100 per request. Included with the $149/mo plan (100,000 new businesses a day; businesses you already have do not count again). Set hasContact=true for only businesses with a named contact person. For a daily pull of only new businesses set excludeDelivered=true. For big pulls page with after: send after="0" first, then the pagination.next_after from each response until it is null.',
|
|
88
|
+
inputSchema: { type: 'object', properties: {
|
|
89
|
+
category: { type: 'array', items: { type: 'string' }, description: 'Business categories, e.g. ["Dentist"]. A word also finds categories that contain it ("restaurant" finds "Mexican restaurant")' },
|
|
90
|
+
country: { type: 'array', items: { type: 'string' }, description: 'Two-letter country codes, e.g. ["US"]' },
|
|
91
|
+
state: { type: 'string', description: 'State or region, comma separated for several (US abbreviations work, e.g. "TX")' },
|
|
92
|
+
city: { type: 'string', description: 'City (partial match)' },
|
|
93
|
+
postalCode: { type: 'string', description: 'Postal or zip code' },
|
|
94
|
+
query: { type: 'string', description: 'Search business names and descriptions' },
|
|
95
|
+
minRating: { type: 'number', description: 'Minimum Google rating (1-5)' },
|
|
96
|
+
maxRating: { type: 'number', description: 'Maximum Google rating (1-5)' },
|
|
97
|
+
minReviews: { type: 'number', description: 'Minimum number of reviews' },
|
|
98
|
+
maxReviews: { type: 'number', description: 'Maximum number of reviews' },
|
|
99
|
+
hasEmail: { type: 'boolean', description: 'Only businesses with an email address' },
|
|
100
|
+
hasPhone: { type: 'boolean', description: 'Only businesses with a phone number' },
|
|
101
|
+
hasWebsite: { type: 'boolean', description: 'Only businesses with a website' },
|
|
102
|
+
hasContact: { type: 'boolean', description: 'Only businesses with a named contact person' },
|
|
103
|
+
page: { type: 'number', description: 'Page number (default 1)', default: 1 },
|
|
104
|
+
limit: { type: 'number', description: 'Results per page (max 100)', default: 50 },
|
|
105
|
+
excludeDelivered: { type: 'boolean', description: 'Skip businesses this account already has (from the API or unlocked in the dashboard). With this on, keep page at 1 or use after.' },
|
|
106
|
+
after: { type: 'string', description: 'Cursor paging: return businesses after this id. Use "0" to start, then the pagination.next_after from each response.' }
|
|
107
|
+
} }
|
|
84
108
|
}
|
|
85
109
|
]
|
|
86
110
|
|
|
@@ -116,6 +140,16 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
116
140
|
result = await sc.dbLeads(params)
|
|
117
141
|
break
|
|
118
142
|
}
|
|
143
|
+
case 'query_local_business_database': {
|
|
144
|
+
const params = { ...args }
|
|
145
|
+
for (const k of ['hasEmail', 'hasPhone', 'hasWebsite', 'hasContact', 'excludeDelivered']) {
|
|
146
|
+
if (params[k]) params[k] = 'true'
|
|
147
|
+
else delete params[k]
|
|
148
|
+
}
|
|
149
|
+
if (params.after !== undefined && params.after !== null) params.after = String(params.after)
|
|
150
|
+
result = await sc.dbLocalBusinesses(params)
|
|
151
|
+
break
|
|
152
|
+
}
|
|
119
153
|
default:
|
|
120
154
|
return { content: [{ type: 'text', text: `Unknown tool: ${name}` }], isError: true }
|
|
121
155
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
// AUTO-GENERATED by scripts/generate.mjs — DO NOT EDIT BY HAND.
|
|
2
2
|
// Source of truth: scraperConfigs.ts (+ src/config/scrapers.ts). Regenerate: npm run generate
|
|
3
|
-
// Generated: 2026-10-
|
|
3
|
+
// Generated: 2026-10-08T22:24:15.688Z
|
|
4
4
|
export const TOOLS = [
|
|
5
5
|
{
|
|
6
6
|
"name": "scrape_apollo",
|
package/lib/client.mjs
CHANGED
|
@@ -166,14 +166,21 @@ export const scrape = (slug, body) =>
|
|
|
166
166
|
export const postEndpoint = (endpoint, body) => post(endpoint, body)
|
|
167
167
|
|
|
168
168
|
// ── Databases ($649 plan) ─────────────────────────────────────
|
|
169
|
+
// A long request (a big Apollo URL, a long title or keyword list) goes as a POST with a JSON body:
|
|
170
|
+
// a query string past ~16KB is refused by the server before the search runs.
|
|
171
|
+
const LONG_QUERY = 6000
|
|
169
172
|
export const dbLeads = (params) => {
|
|
170
173
|
const qs = new URLSearchParams()
|
|
174
|
+
const body = {}
|
|
171
175
|
for (const [k, v] of Object.entries(params)) {
|
|
172
176
|
if (v === undefined || v === null || v === '') continue
|
|
173
177
|
if (Array.isArray(v)) v.forEach(x => qs.append(k, x))
|
|
174
178
|
else qs.append(k, String(v))
|
|
179
|
+
body[k] = v
|
|
175
180
|
}
|
|
176
|
-
|
|
181
|
+
const qstr = qs.toString()
|
|
182
|
+
if (qstr.length > LONG_QUERY) return post(ENDPOINTS['database-leads'], body)
|
|
183
|
+
return get(`${ENDPOINTS['database-leads']}?${qstr}`)
|
|
177
184
|
}
|
|
178
185
|
|
|
179
186
|
export const dbLocalBusinesses = (params) => {
|