scrapercity 1.0.13 → 1.0.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENT_DOCS.txt +26 -11
- package/SKILL.md +2 -3
- package/bin/cli.mjs +24 -10
- package/bin/mcp.mjs +22 -9
- package/lib/catalog.generated.mjs +1 -1
- package/lib/client.mjs +8 -1
- package/package.json +1 -1
package/AGENT_DOCS.txt
CHANGED
|
@@ -40,17 +40,13 @@ SCRAPE ENDPOINTS (POST /api/v1/scrape/{slug})
|
|
|
40
40
|
|
|
41
41
|
apollo
|
|
42
42
|
Body: { url: "https://app.apollo.io/#/people?...", count: 1000, fileName: "My Export" }
|
|
43
|
-
Returns: { runId, message }
|
|
43
|
+
Returns: { runId, message, lead_database? }
|
|
44
|
+
lead_database (when there are matches): { matches, url, api } - leads with the same
|
|
45
|
+
filters already in the Lead Database (ones you don't have), a dashboard link, and the
|
|
46
|
+
/api/v1/database/leads call that returns them now.
|
|
44
47
|
Cost: $0.0039/lead | Min 500, max 50000
|
|
45
48
|
DELIVERY: UP TO 4 DAYS. Use webhook, not polling.
|
|
46
|
-
|
|
47
|
-
apollo-filters
|
|
48
|
-
Body: { seniorityLevel, functionDept, companyIndustry, personCountry, personState,
|
|
49
|
-
companyCountry, companyState, companySize, personTitles:[], companyDomains:[],
|
|
50
|
-
companyKeywords:[], personCities:[], companyCities:[], hasPhone, count, fileName }
|
|
51
|
-
At least one filter required.
|
|
52
|
-
Returns: { runId, message }
|
|
53
|
-
Cost: $0.0039/lead | DELIVERY: UP TO 4 DAYS.
|
|
49
|
+
To filter by title/location/industry without a URL, use the Lead Database (below).
|
|
54
50
|
|
|
55
51
|
maps
|
|
56
52
|
Body: { searchStringsArray: ["plumbers"], locationQuery: "Denver, CO", maxCrawledPlacesPerSearch: 500 }
|
|
@@ -180,7 +176,8 @@ GET /api/v1/runs?hours=24&limit=50
|
|
|
180
176
|
Returns: { runs: [{ run_id, status, handled, url, file_name, created_at }] }
|
|
181
177
|
|
|
182
178
|
GET /api/v1/apollo-status
|
|
183
|
-
Returns: {
|
|
179
|
+
Returns: { status: "url-based" | "legacy", message, timestamp } - "url-based" means the
|
|
180
|
+
Apollo URL endpoint is available
|
|
184
181
|
|
|
185
182
|
────────────────────────────────────────
|
|
186
183
|
DATABASES (GET, $149/mo plan and up, 100k new records/day)
|
|
@@ -189,7 +186,17 @@ DATABASES (GET, $149/mo plan and up, 100k new records/day)
|
|
|
189
186
|
GET /api/v1/database/leads?title=CTO&country=United States&hasEmail=true&page=1&limit=100
|
|
190
187
|
Filters: title, industry, country, state, city, companyName, companyDomain,
|
|
191
188
|
seniority, department, hasEmail, hasPhone, minEmployees, maxEmployees, socialUrl
|
|
189
|
+
Company filters: keywords (company tags, whole-word, any match), revenueMin, revenueMax (USD),
|
|
190
|
+
companyCountry, companyState, companyCity (company HQ, not the person)
|
|
191
|
+
Exclusions: notTitle, notKeywords, notIndustry
|
|
192
|
+
url=<people-search URL>: use that search's filters (other params override).
|
|
193
|
+
The response adds search_translation: { applied, approximated, not_supported }.
|
|
194
|
+
A URL with no filters the database can use returns 400.
|
|
195
|
+
POST /api/v1/database/leads with a JSON body takes the same parameters (arrays as JSON
|
|
196
|
+
arrays). Use it for a long url or long lists: a query string over ~16KB is refused.
|
|
197
|
+
Country: full name ("United States"); US, USA, UK, UAE also work.
|
|
192
198
|
Array params use repeated keys: ?seniority=vp&seniority=director
|
|
199
|
+
(keywords / notKeywords also accept commas: ?keywords=saas,fintech)
|
|
193
200
|
Optional:
|
|
194
201
|
excludeDelivered=true skip leads this account already has (from the API or the
|
|
195
202
|
dashboard). For daily pulls. Keep page=1 (or use after).
|
|
@@ -197,7 +204,15 @@ GET /api/v1/database/leads?title=CTO&country=United States&hasEmail=true&page=1&
|
|
|
197
204
|
pagination.next_after from each response until it is null.
|
|
198
205
|
Every lead returned is saved to your account. Only NEW leads count toward the
|
|
199
206
|
100k/day limit; fetching a lead you already have again is free.
|
|
200
|
-
Returns: { data: [...leads], pagination: { page, limit, total, totalPages, has_more, next_after }, rate_limit
|
|
207
|
+
Returns: { data: [...leads], pagination: { page, limit, total, totalPages, has_more, next_after }, rate_limit,
|
|
208
|
+
search_translation? }
|
|
209
|
+
Lead fields include company_industry, company_size, employees_count, company_website,
|
|
210
|
+
company_annual_revenue, company_city, company_state, company_country, keywords.
|
|
211
|
+
409 with filters_not_applied_yet: the company filters listed can't be used yet (company
|
|
212
|
+
data still being prepared right after launch). Nothing is returned or counted; retry later
|
|
213
|
+
or drop those filters.
|
|
214
|
+
503: the search took too long to run. Narrow it (fewer titles or keywords, add a location
|
|
215
|
+
or company size) and retry.
|
|
201
216
|
|
|
202
217
|
GET /api/v1/database/local-businesses?...
|
|
203
218
|
GET /api/v1/database/ecommerce?...
|
package/SKILL.md
CHANGED
|
@@ -45,7 +45,6 @@ Auth: `Authorization: Bearer $SCRAPERCITY_API_KEY` on all requests.
|
|
|
45
45
|
| Slug | Input | Cost | Speed |
|
|
46
46
|
|------|-------|------|-------|
|
|
47
47
|
| `apollo` | `{url, count, fileName}` | $0.0039/lead | ~4 DAYS (use webhook) |
|
|
48
|
-
| `apollo-filters` | `{seniorityLevel, functionDept, companyIndustry, personCountry, personState, companySize, personTitles[], companyDomains[], count}` | $0.0039/lead | ~4 DAYS |
|
|
49
48
|
| `maps` | `{searchStringsArray:["query"], locationQuery, maxCrawledPlacesPerSearch}` | $0.01/place | 5-30 min |
|
|
50
49
|
| `email-validator` | `{emails:["a@b.com"]}` | $0.0036/email | 1-10 min |
|
|
51
50
|
| `email-finder` | `{contacts:[{first_name,last_name,domain}], autoValidateEmails, autoFindMobiles}` | $0.05/contact | 1-10 min |
|
|
@@ -78,8 +77,8 @@ Apollo scrapes take **up to 4 days** to deliver. Do NOT poll in a loop.
|
|
|
78
77
|
| GET | `/api/v1/runs?hours=24&limit=50` | Recent runs |
|
|
79
78
|
| POST | `/api/v1/scrape/cancel/{runId}` | Cancel running job |
|
|
80
79
|
| GET | `/api/v1/scrape/logs/{runId}` | Run logs |
|
|
81
|
-
| GET | `/api/v1/apollo-status` | Apollo service health |
|
|
82
|
-
| GET | `/api/v1/database/leads?title=CTO&country=United%20States&hasEmail=true&limit=100` | Lead DB ($149/mo plan and up, 100k new leads/day). Optional: `excludeDelivered=true` (only leads you don't have yet), `after=0` then `pagination.next_after` (cursor paging) |
|
|
80
|
+
| GET | `/api/v1/apollo-status` | Apollo service health: `{status: "url-based" or "legacy", message, timestamp}` |
|
|
81
|
+
| GET | `/api/v1/database/leads?title=CTO&country=United%20States&hasEmail=true&limit=100` | Lead DB ($149/mo plan and up, 100k new leads/day). Optional: `excludeDelivered=true` (only leads you don't have yet), `after=0` then `pagination.next_after` (cursor paging). Company filters: `keywords`, `revenueMin`/`revenueMax`, `companyCountry`/`companyState`/`companyCity`. Exclusions: `notTitle`, `notKeywords`, `notIndustry`. `url=<people-search URL>` searches with that URL's filters. POST with a JSON body takes the same parameters (use it for a long `url`) |
|
|
83
82
|
| GET | `/api/v1/database/local-businesses?...` | Local biz DB ($149/mo plan and up). Same optional `excludeDelivered` / `after` |
|
|
84
83
|
| GET | `/api/v1/database/ecommerce?...` | Ecommerce DB ($149/mo plan and up). Same optional `excludeDelivered` / `after` |
|
|
85
84
|
|
package/bin/cli.mjs
CHANGED
|
@@ -5,7 +5,9 @@ import fs from 'fs'
|
|
|
5
5
|
import readline from 'readline'
|
|
6
6
|
|
|
7
7
|
const [,, cmd, ...args] = process.argv
|
|
8
|
-
|
|
8
|
+
// Read "--name value" and remove both from args. (It used to splice first and then read
|
|
9
|
+
// args[i], which by then was the NEXT flag, so every value flag got the wrong value.)
|
|
10
|
+
const flag = (name) => { const i = args.indexOf(name); if (i === -1) return undefined; const v = args[i + 1]; args.splice(i, 2); return v ?? true }
|
|
9
11
|
const flagBool = (name) => { const i = args.indexOf(name); if (i !== -1) { args.splice(i, 1); return true } return false }
|
|
10
12
|
const json = (d) => JSON.stringify(d, null, 2)
|
|
11
13
|
|
|
@@ -368,19 +370,24 @@ async function main() {
|
|
|
368
370
|
// ── Database: Leads ($149/mo plan and up) ─────────────
|
|
369
371
|
case 'db-leads': {
|
|
370
372
|
const params = {}
|
|
371
|
-
|
|
372
|
-
'--company', '--domain', '--company-size', '--seniority',
|
|
373
|
-
'--department', '--page', '--limit', '--min-employees', '--max-employees', '--after']) {
|
|
374
|
-
const v = flag(f)
|
|
375
|
-
if (v === undefined) continue
|
|
376
|
-
const key = { '--title': 'title', '--industry': 'industry', '--country': 'country',
|
|
373
|
+
const keyFor = { '--title': 'title', '--industry': 'industry', '--country': 'country',
|
|
377
374
|
'--state': 'state', '--city': 'city', '--company': 'companyName',
|
|
378
|
-
'--domain': 'companyDomain',
|
|
375
|
+
'--domain': 'companyDomain',
|
|
379
376
|
'--seniority': 'seniority', '--department': 'department',
|
|
380
377
|
'--page': 'page', '--limit': 'limit',
|
|
381
378
|
'--min-employees': 'minEmployees', '--max-employees': 'maxEmployees',
|
|
382
|
-
'--after': 'after'
|
|
383
|
-
|
|
379
|
+
'--after': 'after', '--url': 'url', '--apollo-url': 'apolloUrl',
|
|
380
|
+
'--keywords': 'keywords', '--revenue-min': 'revenueMin', '--revenue-max': 'revenueMax',
|
|
381
|
+
'--company-country': 'companyCountry', '--company-state': 'companyState',
|
|
382
|
+
'--company-city': 'companyCity', '--not-title': 'notTitle',
|
|
383
|
+
'--not-keywords': 'notKeywords', '--not-industry': 'notIndustry' }
|
|
384
|
+
// List filters take commas: --country "United States,Canada" sends both.
|
|
385
|
+
const lists = new Set(['--industry', '--country', '--state', '--seniority', '--department',
|
|
386
|
+
'--company-country', '--company-state', '--not-industry'])
|
|
387
|
+
for (const f of Object.keys(keyFor)) {
|
|
388
|
+
const v = flag(f)
|
|
389
|
+
if (v === undefined) continue
|
|
390
|
+
params[keyFor[f]] = lists.has(f) ? v.split(',').map(s => s.trim()).filter(Boolean) : v
|
|
384
391
|
}
|
|
385
392
|
if (flagBool('--has-email')) params.hasEmail = 'true'
|
|
386
393
|
if (flagBool('--has-phone')) params.hasPhone = 'true'
|
|
@@ -388,6 +395,8 @@ async function main() {
|
|
|
388
395
|
const r = await sc.dbLeads(params)
|
|
389
396
|
console.log(`${r.pagination?.total || '?'} total leads, page ${r.pagination?.page || 1} of ${r.pagination?.totalPages || '?'}`)
|
|
390
397
|
if (r.pagination?.next_after) console.log(`Next page: --after ${r.pagination.next_after}`)
|
|
398
|
+
const tr = r.search_translation || r.apollo_translation
|
|
399
|
+
if (tr?.not_supported?.length) console.log(`Not supported from the search URL: ${tr.not_supported.join(', ')}`)
|
|
391
400
|
console.log(json(r.data?.slice(0, 3) || r))
|
|
392
401
|
if (r.data?.length > 3) console.log(`... and ${r.data.length - 3} more`)
|
|
393
402
|
break
|
|
@@ -445,6 +454,11 @@ ScraperCity CLI - B2B lead generation from your terminal
|
|
|
445
454
|
|
|
446
455
|
Database ($149/mo plan and up):
|
|
447
456
|
scrapercity db-leads [filters] Query lead database
|
|
457
|
+
--title --industry --country --state --city --company --domain --seniority --department
|
|
458
|
+
--min-employees --max-employees --has-email --has-phone
|
|
459
|
+
--keywords --revenue-min --revenue-max --company-country --company-state --company-city
|
|
460
|
+
--not-title --not-keywords --not-industry
|
|
461
|
+
--url "<people-search URL>" Search with that URL's filters
|
|
448
462
|
--exclude-delivered Only leads you don't already have
|
|
449
463
|
--after <id> Cursor paging (start with 0, then use the printed next id)
|
|
450
464
|
|
package/bin/mcp.mjs
CHANGED
|
@@ -50,17 +50,30 @@ const UTILITY_TOOLS = [
|
|
|
50
50
|
},
|
|
51
51
|
{
|
|
52
52
|
name: 'query_lead_database',
|
|
53
|
-
description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies. Up to 100 per request. Included with the $149/mo plan (100,000 new leads a day; leads you already have do not count again). For a daily pull of only new leads set excludeDelivered=true. For big pulls page with after: send after="0" first, then the pagination.next_after from each response until it is null.',
|
|
53
|
+
description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies (plus company website, revenue, HQ location and keywords). Up to 100 per request. Included with the $149/mo plan (100,000 new leads a day; leads you already have do not count again). For a daily pull of only new leads set excludeDelivered=true. For big pulls page with after: send after="0" first, then the pagination.next_after from each response until it is null. Have a saved people-search URL? Pass it as url to search with its filters right away (search_translation in the response says what was applied).',
|
|
54
54
|
inputSchema: { type: 'object', properties: {
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
55
|
+
url: { type: 'string', description: 'A people-search URL (the address of a people search with its filters). Its filters become the search; any other filter you pass overrides the matching one.' },
|
|
56
|
+
title: { type: 'string', description: 'Job title filter, comma separated for several (partial match, e.g. "CEO, Founder")' },
|
|
57
|
+
industry: { type: 'array', items: { type: 'string' }, description: 'Company industries (any match)' },
|
|
58
|
+
country: { type: 'array', items: { type: 'string' }, description: 'Person countries, full names like "United States" (US, USA, UK, UAE also work)' },
|
|
59
|
+
state: { type: 'array', items: { type: 'string' }, description: 'Person states' },
|
|
60
|
+
city: { type: 'string', description: 'Person city, comma separated for several' },
|
|
60
61
|
companyName: { type: 'string', description: 'Company name' },
|
|
61
|
-
companyDomain: { type: 'string', description: 'Company domain' },
|
|
62
|
-
seniority: { type: 'string', description: 'e.g. "vp", "director", "c_suite"' },
|
|
63
|
-
department: { type: 'string', description: 'e.g. "sales", "engineering"' },
|
|
62
|
+
companyDomain: { type: 'string', description: 'Company domain(s), comma separated' },
|
|
63
|
+
seniority: { type: 'array', items: { type: 'string' }, description: 'e.g. ["vp", "director", "c_suite"]' },
|
|
64
|
+
department: { type: 'array', items: { type: 'string' }, description: 'e.g. ["sales", "engineering"]' },
|
|
65
|
+
minEmployees: { type: 'number', description: 'Minimum company employee count' },
|
|
66
|
+
maxEmployees: { type: 'number', description: 'Maximum company employee count' },
|
|
67
|
+
keywords: { type: 'array', items: { type: 'string' }, description: 'Company keywords/tags, any match (e.g. ["saas", "fintech"])' },
|
|
68
|
+
revenueMin: { type: 'number', description: 'Minimum company annual revenue, USD' },
|
|
69
|
+
revenueMax: { type: 'number', description: 'Maximum company annual revenue, USD' },
|
|
70
|
+
companyCountry: { type: 'array', items: { type: 'string' }, description: 'Company HQ countries (not where the person is)' },
|
|
71
|
+
companyState: { type: 'array', items: { type: 'string' }, description: 'Company HQ states' },
|
|
72
|
+
companyCity: { type: 'string', description: 'Company HQ city' },
|
|
73
|
+
notTitle: { type: 'string', description: 'Exclude titles containing any of these (comma separated)' },
|
|
74
|
+
notKeywords: { type: 'array', items: { type: 'string' }, description: 'Exclude companies with any of these keywords' },
|
|
75
|
+
notIndustry: { type: 'array', items: { type: 'string' }, description: 'Exclude these industries' },
|
|
76
|
+
socialUrl: { type: 'string', description: 'Match a social profile URL' },
|
|
64
77
|
hasEmail: { type: 'boolean', description: 'Only contacts with email' },
|
|
65
78
|
hasPhone: { type: 'boolean', description: 'Only contacts with phone' },
|
|
66
79
|
page: { type: 'number', description: 'Page number (default 1)', default: 1 },
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
// AUTO-GENERATED by scripts/generate.mjs — DO NOT EDIT BY HAND.
|
|
2
2
|
// Source of truth: scraperConfigs.ts (+ src/config/scrapers.ts). Regenerate: npm run generate
|
|
3
|
-
// Generated: 2026-
|
|
3
|
+
// Generated: 2026-10-02T14:29:37.617Z
|
|
4
4
|
export const TOOLS = [
|
|
5
5
|
{
|
|
6
6
|
"name": "scrape_apollo",
|
package/lib/client.mjs
CHANGED
|
@@ -166,14 +166,21 @@ export const scrape = (slug, body) =>
|
|
|
166
166
|
export const postEndpoint = (endpoint, body) => post(endpoint, body)
|
|
167
167
|
|
|
168
168
|
// ── Databases ($649 plan) ─────────────────────────────────────
|
|
169
|
+
// A long request (a big Apollo URL, a long title or keyword list) goes as a POST with a JSON body:
|
|
170
|
+
// a query string past ~16KB is refused by the server before the search runs.
|
|
171
|
+
const LONG_QUERY = 6000
|
|
169
172
|
export const dbLeads = (params) => {
|
|
170
173
|
const qs = new URLSearchParams()
|
|
174
|
+
const body = {}
|
|
171
175
|
for (const [k, v] of Object.entries(params)) {
|
|
172
176
|
if (v === undefined || v === null || v === '') continue
|
|
173
177
|
if (Array.isArray(v)) v.forEach(x => qs.append(k, x))
|
|
174
178
|
else qs.append(k, String(v))
|
|
179
|
+
body[k] = v
|
|
175
180
|
}
|
|
176
|
-
|
|
181
|
+
const qstr = qs.toString()
|
|
182
|
+
if (qstr.length > LONG_QUERY) return post(ENDPOINTS['database-leads'], body)
|
|
183
|
+
return get(`${ENDPOINTS['database-leads']}?${qstr}`)
|
|
177
184
|
}
|
|
178
185
|
|
|
179
186
|
export const dbLocalBusinesses = (params) => {
|