scrapercity 1.0.12 → 1.0.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENT_DOCS.txt +31 -14
- package/README.md +1 -1
- package/SKILL.md +4 -5
- package/bin/cli.mjs +30 -12
- package/bin/mcp.mjs +28 -10
- package/lib/catalog.generated.mjs +24 -22
- package/package.json +1 -1
package/AGENT_DOCS.txt
CHANGED
|
@@ -40,17 +40,13 @@ SCRAPE ENDPOINTS (POST /api/v1/scrape/{slug})
|
|
|
40
40
|
|
|
41
41
|
apollo
|
|
42
42
|
Body: { url: "https://app.apollo.io/#/people?...", count: 1000, fileName: "My Export" }
|
|
43
|
-
Returns: { runId, message }
|
|
43
|
+
Returns: { runId, message, lead_database? }
|
|
44
|
+
lead_database (when there are matches): { matches, url, api } - leads with the same
|
|
45
|
+
filters already in the Lead Database (ones you don't have), a dashboard link, and the
|
|
46
|
+
/api/v1/database/leads call that returns them now.
|
|
44
47
|
Cost: $0.0039/lead | Min 500, max 50000
|
|
45
48
|
DELIVERY: UP TO 4 DAYS. Use webhook, not polling.
|
|
46
|
-
|
|
47
|
-
apollo-filters
|
|
48
|
-
Body: { seniorityLevel, functionDept, companyIndustry, personCountry, personState,
|
|
49
|
-
companyCountry, companyState, companySize, personTitles:[], companyDomains:[],
|
|
50
|
-
companyKeywords:[], personCities:[], companyCities:[], hasPhone, count, fileName }
|
|
51
|
-
At least one filter required.
|
|
52
|
-
Returns: { runId, message }
|
|
53
|
-
Cost: $0.0039/lead | DELIVERY: UP TO 4 DAYS.
|
|
49
|
+
To filter by title/location/industry without a URL, use the Lead Database (below).
|
|
54
50
|
|
|
55
51
|
maps
|
|
56
52
|
Body: { searchStringsArray: ["plumbers"], locationQuery: "Denver, CO", maxCrawledPlacesPerSearch: 500 }
|
|
@@ -180,21 +176,42 @@ GET /api/v1/runs?hours=24&limit=50
|
|
|
180
176
|
Returns: { runs: [{ run_id, status, handled, url, file_name, created_at }] }
|
|
181
177
|
|
|
182
178
|
GET /api/v1/apollo-status
|
|
183
|
-
Returns: {
|
|
179
|
+
Returns: { status: "url-based" | "legacy", message, timestamp } - "url-based" means the
|
|
180
|
+
Apollo URL endpoint is available
|
|
184
181
|
|
|
185
182
|
────────────────────────────────────────
|
|
186
|
-
DATABASES (GET, $
|
|
183
|
+
DATABASES (GET, $149/mo plan and up, 100k new records/day)
|
|
187
184
|
────────────────────────────────────────
|
|
188
185
|
|
|
189
186
|
GET /api/v1/database/leads?title=CTO&country=United States&hasEmail=true&page=1&limit=100
|
|
190
187
|
Filters: title, industry, country, state, city, companyName, companyDomain,
|
|
191
|
-
|
|
188
|
+
seniority, department, hasEmail, hasPhone, minEmployees, maxEmployees, socialUrl
|
|
189
|
+
Company filters: keywords (company tags, whole-word, any match), revenueMin, revenueMax (USD),
|
|
190
|
+
companyCountry, companyState, companyCity (company HQ, not the person)
|
|
191
|
+
Exclusions: notTitle, notKeywords, notIndustry
|
|
192
|
+
apolloUrl=<Apollo people-search URL>: use that search's filters (other params override).
|
|
193
|
+
The response adds apollo_translation: { applied, approximated, not_supported }.
|
|
194
|
+
Country: full name ("United States"); US, USA, UK, UAE also work.
|
|
192
195
|
Array params use repeated keys: ?seniority=vp&seniority=director
|
|
193
|
-
|
|
196
|
+
(keywords / notKeywords also accept commas: ?keywords=saas,fintech)
|
|
197
|
+
Optional:
|
|
198
|
+
excludeDelivered=true skip leads this account already has (from the API or the
|
|
199
|
+
dashboard). For daily pulls. Keep page=1 (or use after).
|
|
200
|
+
after=<id> cursor paging in id order. Send after=0 first, then
|
|
201
|
+
pagination.next_after from each response until it is null.
|
|
202
|
+
Every lead returned is saved to your account. Only NEW leads count toward the
|
|
203
|
+
100k/day limit; fetching a lead you already have again is free.
|
|
204
|
+
Returns: { data: [...leads], pagination: { page, limit, total, totalPages, has_more, next_after }, rate_limit,
|
|
205
|
+
apollo_translation? }
|
|
206
|
+
Lead fields include company_industry, company_size, employees_count, company_website,
|
|
207
|
+
company_annual_revenue, company_city, company_state, company_country, keywords.
|
|
208
|
+
409 with filters_not_applied_yet: the company filters listed can't be used yet (company
|
|
209
|
+
data still being prepared right after launch). Nothing is returned or counted; retry later
|
|
210
|
+
or drop those filters.
|
|
194
211
|
|
|
195
212
|
GET /api/v1/database/local-businesses?...
|
|
196
213
|
GET /api/v1/database/ecommerce?...
|
|
197
|
-
Same pattern, different datasets.
|
|
214
|
+
Same pattern (including excludeDelivered and after), different datasets.
|
|
198
215
|
|
|
199
216
|
────────────────────────────────────────
|
|
200
217
|
ERROR CODES
|
package/README.md
CHANGED
|
@@ -104,7 +104,7 @@ curl -O https://app.scrapercity.com/api/downloads/RUN_ID \
|
|
|
104
104
|
| **Store Leads** | Shopify/WooCommerce stores with contacts | $0.0039/lead |
|
|
105
105
|
| **BuiltWith** | All sites using a technology | $4.99/search |
|
|
106
106
|
| **Criminal Records** | Background check by name | $1.00 if found |
|
|
107
|
-
| **Lead Database** | 3M+ B2B contacts, instant query ($
|
|
107
|
+
| **Lead Database** | 3M+ B2B contacts, instant query ($149/mo plan and up) | Included |
|
|
108
108
|
|
|
109
109
|
## How It Works
|
|
110
110
|
|
package/SKILL.md
CHANGED
|
@@ -45,7 +45,6 @@ Auth: `Authorization: Bearer $SCRAPERCITY_API_KEY` on all requests.
|
|
|
45
45
|
| Slug | Input | Cost | Speed |
|
|
46
46
|
|------|-------|------|-------|
|
|
47
47
|
| `apollo` | `{url, count, fileName}` | $0.0039/lead | ~4 DAYS (use webhook) |
|
|
48
|
-
| `apollo-filters` | `{seniorityLevel, functionDept, companyIndustry, personCountry, personState, companySize, personTitles[], companyDomains[], count}` | $0.0039/lead | ~4 DAYS |
|
|
49
48
|
| `maps` | `{searchStringsArray:["query"], locationQuery, maxCrawledPlacesPerSearch}` | $0.01/place | 5-30 min |
|
|
50
49
|
| `email-validator` | `{emails:["a@b.com"]}` | $0.0036/email | 1-10 min |
|
|
51
50
|
| `email-finder` | `{contacts:[{first_name,last_name,domain}], autoValidateEmails, autoFindMobiles}` | $0.05/contact | 1-10 min |
|
|
@@ -78,10 +77,10 @@ Apollo scrapes take **up to 4 days** to deliver. Do NOT poll in a loop.
|
|
|
78
77
|
| GET | `/api/v1/runs?hours=24&limit=50` | Recent runs |
|
|
79
78
|
| POST | `/api/v1/scrape/cancel/{runId}` | Cancel running job |
|
|
80
79
|
| GET | `/api/v1/scrape/logs/{runId}` | Run logs |
|
|
81
|
-
| GET | `/api/v1/apollo-status` | Apollo service health |
|
|
82
|
-
| GET | `/api/v1/database/leads?title=CTO&country=
|
|
83
|
-
| GET | `/api/v1/database/local-businesses?...` | Local biz DB ($
|
|
84
|
-
| GET | `/api/v1/database/ecommerce?...` | Ecommerce DB ($
|
|
80
|
+
| GET | `/api/v1/apollo-status` | Apollo service health: `{status: "url-based" or "legacy", message, timestamp}` |
|
|
81
|
+
| GET | `/api/v1/database/leads?title=CTO&country=United%20States&hasEmail=true&limit=100` | Lead DB ($149/mo plan and up, 100k new leads/day). Optional: `excludeDelivered=true` (only leads you don't have yet), `after=0` then `pagination.next_after` (cursor paging). Company filters: `keywords`, `revenueMin`/`revenueMax`, `companyCountry`/`companyState`/`companyCity`. Exclusions: `notTitle`, `notKeywords`, `notIndustry`. `apolloUrl=<Apollo people-search URL>` searches with that URL's filters |
|
|
82
|
+
| GET | `/api/v1/database/local-businesses?...` | Local biz DB ($149/mo plan and up). Same optional `excludeDelivered` / `after` |
|
|
83
|
+
| GET | `/api/v1/database/ecommerce?...` | Ecommerce DB ($149/mo plan and up). Same optional `excludeDelivered` / `after` |
|
|
85
84
|
|
|
86
85
|
## Status Values
|
|
87
86
|
`RUNNING` → `SUCCEEDED` or `FAILED` or `CANCELLED`
|
package/bin/cli.mjs
CHANGED
|
@@ -5,7 +5,9 @@ import fs from 'fs'
|
|
|
5
5
|
import readline from 'readline'
|
|
6
6
|
|
|
7
7
|
const [,, cmd, ...args] = process.argv
|
|
8
|
-
|
|
8
|
+
// Read "--name value" and remove both from args. (It used to splice first and then read
|
|
9
|
+
// args[i], which by then was the NEXT flag, so every value flag got the wrong value.)
|
|
10
|
+
const flag = (name) => { const i = args.indexOf(name); if (i === -1) return undefined; const v = args[i + 1]; args.splice(i, 2); return v ?? true }
|
|
9
11
|
const flagBool = (name) => { const i = args.indexOf(name); if (i !== -1) { args.splice(i, 1); return true } return false }
|
|
10
12
|
const json = (d) => JSON.stringify(d, null, 2)
|
|
11
13
|
|
|
@@ -365,26 +367,35 @@ async function main() {
|
|
|
365
367
|
break
|
|
366
368
|
}
|
|
367
369
|
|
|
368
|
-
// ── Database: Leads ($
|
|
370
|
+
// ── Database: Leads ($149/mo plan and up) ─────────────
|
|
369
371
|
case 'db-leads': {
|
|
370
372
|
const params = {}
|
|
371
|
-
|
|
372
|
-
'--company', '--domain', '--company-size', '--seniority',
|
|
373
|
-
'--department', '--page', '--limit', '--min-employees', '--max-employees']) {
|
|
374
|
-
const v = flag(f)
|
|
375
|
-
if (v === undefined) continue
|
|
376
|
-
const key = { '--title': 'title', '--industry': 'industry', '--country': 'country',
|
|
373
|
+
const keyFor = { '--title': 'title', '--industry': 'industry', '--country': 'country',
|
|
377
374
|
'--state': 'state', '--city': 'city', '--company': 'companyName',
|
|
378
|
-
'--domain': 'companyDomain',
|
|
375
|
+
'--domain': 'companyDomain',
|
|
379
376
|
'--seniority': 'seniority', '--department': 'department',
|
|
380
377
|
'--page': 'page', '--limit': 'limit',
|
|
381
|
-
'--min-employees': 'minEmployees', '--max-employees': 'maxEmployees'
|
|
382
|
-
|
|
378
|
+
'--min-employees': 'minEmployees', '--max-employees': 'maxEmployees',
|
|
379
|
+
'--after': 'after', '--apollo-url': 'apolloUrl',
|
|
380
|
+
'--keywords': 'keywords', '--revenue-min': 'revenueMin', '--revenue-max': 'revenueMax',
|
|
381
|
+
'--company-country': 'companyCountry', '--company-state': 'companyState',
|
|
382
|
+
'--company-city': 'companyCity', '--not-title': 'notTitle',
|
|
383
|
+
'--not-keywords': 'notKeywords', '--not-industry': 'notIndustry' }
|
|
384
|
+
// List filters take commas: --country "United States,Canada" sends both.
|
|
385
|
+
const lists = new Set(['--industry', '--country', '--state', '--seniority', '--department',
|
|
386
|
+
'--company-country', '--company-state', '--not-industry'])
|
|
387
|
+
for (const f of Object.keys(keyFor)) {
|
|
388
|
+
const v = flag(f)
|
|
389
|
+
if (v === undefined) continue
|
|
390
|
+
params[keyFor[f]] = lists.has(f) ? v.split(',').map(s => s.trim()).filter(Boolean) : v
|
|
383
391
|
}
|
|
384
392
|
if (flagBool('--has-email')) params.hasEmail = 'true'
|
|
385
393
|
if (flagBool('--has-phone')) params.hasPhone = 'true'
|
|
394
|
+
if (flagBool('--exclude-delivered')) params.excludeDelivered = 'true'
|
|
386
395
|
const r = await sc.dbLeads(params)
|
|
387
396
|
console.log(`${r.pagination?.total || '?'} total leads, page ${r.pagination?.page || 1} of ${r.pagination?.totalPages || '?'}`)
|
|
397
|
+
if (r.pagination?.next_after) console.log(`Next page: --after ${r.pagination.next_after}`)
|
|
398
|
+
if (r.apollo_translation?.not_supported?.length) console.log(`Not supported from the Apollo URL: ${r.apollo_translation.not_supported.join(', ')}`)
|
|
388
399
|
console.log(json(r.data?.slice(0, 3) || r))
|
|
389
400
|
if (r.data?.length > 3) console.log(`... and ${r.data.length - 3} more`)
|
|
390
401
|
break
|
|
@@ -440,8 +451,15 @@ ScraperCity CLI - B2B lead generation from your terminal
|
|
|
440
451
|
scrapercity logs <runId> View run logs
|
|
441
452
|
scrapercity runs [--hours 24] List recent runs
|
|
442
453
|
|
|
443
|
-
Database ($
|
|
454
|
+
Database ($149/mo plan and up):
|
|
444
455
|
scrapercity db-leads [filters] Query lead database
|
|
456
|
+
--title --industry --country --state --city --company --domain --seniority --department
|
|
457
|
+
--min-employees --max-employees --has-email --has-phone
|
|
458
|
+
--keywords --revenue-min --revenue-max --company-country --company-state --company-city
|
|
459
|
+
--not-title --not-keywords --not-industry
|
|
460
|
+
--apollo-url "<Apollo people-search URL>" Search with that URL's filters
|
|
461
|
+
--exclude-delivered Only leads you don't already have
|
|
462
|
+
--after <id> Cursor paging (start with 0, then use the printed next id)
|
|
445
463
|
|
|
446
464
|
Env: SCRAPERCITY_API_KEY=... or scrapercity login
|
|
447
465
|
Docs: https://scrapercity.com/agents
|
package/bin/mcp.mjs
CHANGED
|
@@ -50,21 +50,36 @@ const UTILITY_TOOLS = [
|
|
|
50
50
|
},
|
|
51
51
|
{
|
|
52
52
|
name: 'query_lead_database',
|
|
53
|
-
description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies
|
|
53
|
+
description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies (plus company website, revenue, HQ location and keywords). Up to 100 per request. Included with the $149/mo plan (100,000 new leads a day; leads you already have do not count again). For a daily pull of only new leads set excludeDelivered=true. For big pulls page with after: send after="0" first, then the pagination.next_after from each response until it is null. Have an Apollo people-search URL? Pass it as apolloUrl to search with its filters right away (apollo_translation in the response says what was applied).',
|
|
54
54
|
inputSchema: { type: 'object', properties: {
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
55
|
+
apolloUrl: { type: 'string', description: 'An Apollo people-search URL (app.apollo.io/#/people?...). Its filters become the search; any other filter you pass overrides the matching one.' },
|
|
56
|
+
title: { type: 'string', description: 'Job title filter, comma separated for several (partial match, e.g. "CEO, Founder")' },
|
|
57
|
+
industry: { type: 'array', items: { type: 'string' }, description: 'Company industries (any match)' },
|
|
58
|
+
country: { type: 'array', items: { type: 'string' }, description: 'Person countries, full names like "United States" (US, USA, UK, UAE also work)' },
|
|
59
|
+
state: { type: 'array', items: { type: 'string' }, description: 'Person states' },
|
|
60
|
+
city: { type: 'string', description: 'Person city, comma separated for several' },
|
|
60
61
|
companyName: { type: 'string', description: 'Company name' },
|
|
61
|
-
companyDomain: { type: 'string', description: 'Company domain' },
|
|
62
|
-
seniority: { type: 'string', description: 'e.g. "vp", "director", "c_suite"' },
|
|
63
|
-
department: { type: 'string', description: 'e.g. "sales", "engineering"' },
|
|
62
|
+
companyDomain: { type: 'string', description: 'Company domain(s), comma separated' },
|
|
63
|
+
seniority: { type: 'array', items: { type: 'string' }, description: 'e.g. ["vp", "director", "c_suite"]' },
|
|
64
|
+
department: { type: 'array', items: { type: 'string' }, description: 'e.g. ["sales", "engineering"]' },
|
|
65
|
+
minEmployees: { type: 'number', description: 'Minimum company employee count' },
|
|
66
|
+
maxEmployees: { type: 'number', description: 'Maximum company employee count' },
|
|
67
|
+
keywords: { type: 'array', items: { type: 'string' }, description: 'Company keywords/tags, any match (e.g. ["saas", "fintech"])' },
|
|
68
|
+
revenueMin: { type: 'number', description: 'Minimum company annual revenue, USD' },
|
|
69
|
+
revenueMax: { type: 'number', description: 'Maximum company annual revenue, USD' },
|
|
70
|
+
companyCountry: { type: 'array', items: { type: 'string' }, description: 'Company HQ countries (not where the person is)' },
|
|
71
|
+
companyState: { type: 'array', items: { type: 'string' }, description: 'Company HQ states' },
|
|
72
|
+
companyCity: { type: 'string', description: 'Company HQ city' },
|
|
73
|
+
notTitle: { type: 'string', description: 'Exclude titles containing any of these (comma separated)' },
|
|
74
|
+
notKeywords: { type: 'array', items: { type: 'string' }, description: 'Exclude companies with any of these keywords' },
|
|
75
|
+
notIndustry: { type: 'array', items: { type: 'string' }, description: 'Exclude these industries' },
|
|
76
|
+
socialUrl: { type: 'string', description: 'Match a social profile URL' },
|
|
64
77
|
hasEmail: { type: 'boolean', description: 'Only contacts with email' },
|
|
65
78
|
hasPhone: { type: 'boolean', description: 'Only contacts with phone' },
|
|
66
79
|
page: { type: 'number', description: 'Page number (default 1)', default: 1 },
|
|
67
|
-
limit: { type: 'number', description: 'Results per page (max 100)', default: 50 }
|
|
80
|
+
limit: { type: 'number', description: 'Results per page (max 100)', default: 50 },
|
|
81
|
+
excludeDelivered: { type: 'boolean', description: 'Skip leads this account already has (from the API or unlocked in the dashboard). With this on, keep page at 1 or use after.' },
|
|
82
|
+
after: { type: 'string', description: 'Cursor paging: return leads after this lead id. Use "0" to start, then the pagination.next_after from each response.' }
|
|
68
83
|
} }
|
|
69
84
|
}
|
|
70
85
|
]
|
|
@@ -95,6 +110,9 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
95
110
|
const params = { ...args }
|
|
96
111
|
if (params.hasEmail) params.hasEmail = 'true'
|
|
97
112
|
if (params.hasPhone) params.hasPhone = 'true'
|
|
113
|
+
if (params.excludeDelivered) params.excludeDelivered = 'true'
|
|
114
|
+
else delete params.excludeDelivered
|
|
115
|
+
if (params.after !== undefined && params.after !== null) params.after = String(params.after)
|
|
98
116
|
result = await sc.dbLeads(params)
|
|
99
117
|
break
|
|
100
118
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
// AUTO-GENERATED by scripts/generate.mjs — DO NOT EDIT BY HAND.
|
|
2
2
|
// Source of truth: scraperConfigs.ts (+ src/config/scrapers.ts). Regenerate: npm run generate
|
|
3
|
-
// Generated: 2026-
|
|
3
|
+
// Generated: 2026-10-01T23:44:06.142Z
|
|
4
4
|
export const TOOLS = [
|
|
5
5
|
{
|
|
6
6
|
"name": "scrape_apollo",
|
|
@@ -559,29 +559,31 @@ export const TOOLS = [
|
|
|
559
559
|
"contacts": {
|
|
560
560
|
"type": "array",
|
|
561
561
|
"items": {
|
|
562
|
-
"type": "object"
|
|
562
|
+
"type": "object",
|
|
563
|
+
"properties": {
|
|
564
|
+
"first_name": {
|
|
565
|
+
"type": "string",
|
|
566
|
+
"description": "Person's first name (required if full_name not provided)"
|
|
567
|
+
},
|
|
568
|
+
"last_name": {
|
|
569
|
+
"type": "string",
|
|
570
|
+
"description": "Person's last name"
|
|
571
|
+
},
|
|
572
|
+
"full_name": {
|
|
573
|
+
"type": "string",
|
|
574
|
+
"description": "Full name (alternative to first_name + last_name)"
|
|
575
|
+
},
|
|
576
|
+
"domain": {
|
|
577
|
+
"type": "string",
|
|
578
|
+
"description": "Company domain (preferred over company_name)"
|
|
579
|
+
},
|
|
580
|
+
"company_name": {
|
|
581
|
+
"type": "string",
|
|
582
|
+
"description": "Company name (use if domain not available)"
|
|
583
|
+
}
|
|
584
|
+
}
|
|
563
585
|
},
|
|
564
586
|
"description": "Array of contact objects to look up"
|
|
565
|
-
},
|
|
566
|
-
"contacts[].first_name": {
|
|
567
|
-
"type": "string",
|
|
568
|
-
"description": "Person's first name (required if full_name not provided)"
|
|
569
|
-
},
|
|
570
|
-
"contacts[].last_name": {
|
|
571
|
-
"type": "string",
|
|
572
|
-
"description": "Person's last name"
|
|
573
|
-
},
|
|
574
|
-
"contacts[].full_name": {
|
|
575
|
-
"type": "string",
|
|
576
|
-
"description": "Full name (alternative to first_name + last_name)"
|
|
577
|
-
},
|
|
578
|
-
"contacts[].domain": {
|
|
579
|
-
"type": "string",
|
|
580
|
-
"description": "Company domain (preferred over company_name)"
|
|
581
|
-
},
|
|
582
|
-
"contacts[].company_name": {
|
|
583
|
-
"type": "string",
|
|
584
|
-
"description": "Company name (use if domain not available)"
|
|
585
587
|
}
|
|
586
588
|
},
|
|
587
589
|
"required": [
|