scrapercity 1.0.13 → 1.0.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENT_DOCS.txt +21 -11
- package/SKILL.md +2 -3
- package/bin/cli.mjs +23 -10
- package/bin/mcp.mjs +22 -9
- package/lib/catalog.generated.mjs +1 -1
- package/package.json +1 -1
package/AGENT_DOCS.txt
CHANGED
|
@@ -40,17 +40,13 @@ SCRAPE ENDPOINTS (POST /api/v1/scrape/{slug})
|
|
|
40
40
|
|
|
41
41
|
apollo
|
|
42
42
|
Body: { url: "https://app.apollo.io/#/people?...", count: 1000, fileName: "My Export" }
|
|
43
|
-
Returns: { runId, message }
|
|
43
|
+
Returns: { runId, message, lead_database? }
|
|
44
|
+
lead_database (when there are matches): { matches, url, api } - leads with the same
|
|
45
|
+
filters already in the Lead Database (ones you don't have), a dashboard link, and the
|
|
46
|
+
/api/v1/database/leads call that returns them now.
|
|
44
47
|
Cost: $0.0039/lead | Min 500, max 50000
|
|
45
48
|
DELIVERY: UP TO 4 DAYS. Use webhook, not polling.
|
|
46
|
-
|
|
47
|
-
apollo-filters
|
|
48
|
-
Body: { seniorityLevel, functionDept, companyIndustry, personCountry, personState,
|
|
49
|
-
companyCountry, companyState, companySize, personTitles:[], companyDomains:[],
|
|
50
|
-
companyKeywords:[], personCities:[], companyCities:[], hasPhone, count, fileName }
|
|
51
|
-
At least one filter required.
|
|
52
|
-
Returns: { runId, message }
|
|
53
|
-
Cost: $0.0039/lead | DELIVERY: UP TO 4 DAYS.
|
|
49
|
+
To filter by title/location/industry without a URL, use the Lead Database (below).
|
|
54
50
|
|
|
55
51
|
maps
|
|
56
52
|
Body: { searchStringsArray: ["plumbers"], locationQuery: "Denver, CO", maxCrawledPlacesPerSearch: 500 }
|
|
@@ -180,7 +176,8 @@ GET /api/v1/runs?hours=24&limit=50
|
|
|
180
176
|
Returns: { runs: [{ run_id, status, handled, url, file_name, created_at }] }
|
|
181
177
|
|
|
182
178
|
GET /api/v1/apollo-status
|
|
183
|
-
Returns: {
|
|
179
|
+
Returns: { status: "url-based" | "legacy", message, timestamp } - "url-based" means the
|
|
180
|
+
Apollo URL endpoint is available
|
|
184
181
|
|
|
185
182
|
────────────────────────────────────────
|
|
186
183
|
DATABASES (GET, $149/mo plan and up, 100k new records/day)
|
|
@@ -189,7 +186,14 @@ DATABASES (GET, $149/mo plan and up, 100k new records/day)
|
|
|
189
186
|
GET /api/v1/database/leads?title=CTO&country=United States&hasEmail=true&page=1&limit=100
|
|
190
187
|
Filters: title, industry, country, state, city, companyName, companyDomain,
|
|
191
188
|
seniority, department, hasEmail, hasPhone, minEmployees, maxEmployees, socialUrl
|
|
189
|
+
Company filters: keywords (company tags, whole-word, any match), revenueMin, revenueMax (USD),
|
|
190
|
+
companyCountry, companyState, companyCity (company HQ, not the person)
|
|
191
|
+
Exclusions: notTitle, notKeywords, notIndustry
|
|
192
|
+
apolloUrl=<Apollo people-search URL>: use that search's filters (other params override).
|
|
193
|
+
The response adds apollo_translation: { applied, approximated, not_supported }.
|
|
194
|
+
Country: full name ("United States"); US, USA, UK, UAE also work.
|
|
192
195
|
Array params use repeated keys: ?seniority=vp&seniority=director
|
|
196
|
+
(keywords / notKeywords also accept commas: ?keywords=saas,fintech)
|
|
193
197
|
Optional:
|
|
194
198
|
excludeDelivered=true skip leads this account already has (from the API or the
|
|
195
199
|
dashboard). For daily pulls. Keep page=1 (or use after).
|
|
@@ -197,7 +201,13 @@ GET /api/v1/database/leads?title=CTO&country=United States&hasEmail=true&page=1&
|
|
|
197
201
|
pagination.next_after from each response until it is null.
|
|
198
202
|
Every lead returned is saved to your account. Only NEW leads count toward the
|
|
199
203
|
100k/day limit; fetching a lead you already have again is free.
|
|
200
|
-
Returns: { data: [...leads], pagination: { page, limit, total, totalPages, has_more, next_after }, rate_limit
|
|
204
|
+
Returns: { data: [...leads], pagination: { page, limit, total, totalPages, has_more, next_after }, rate_limit,
|
|
205
|
+
apollo_translation? }
|
|
206
|
+
Lead fields include company_industry, company_size, employees_count, company_website,
|
|
207
|
+
company_annual_revenue, company_city, company_state, company_country, keywords.
|
|
208
|
+
409 with filters_not_applied_yet: the company filters listed can't be used yet (company
|
|
209
|
+
data still being prepared right after launch). Nothing is returned or counted; retry later
|
|
210
|
+
or drop those filters.
|
|
201
211
|
|
|
202
212
|
GET /api/v1/database/local-businesses?...
|
|
203
213
|
GET /api/v1/database/ecommerce?...
|
package/SKILL.md
CHANGED
|
@@ -45,7 +45,6 @@ Auth: `Authorization: Bearer $SCRAPERCITY_API_KEY` on all requests.
|
|
|
45
45
|
| Slug | Input | Cost | Speed |
|
|
46
46
|
|------|-------|------|-------|
|
|
47
47
|
| `apollo` | `{url, count, fileName}` | $0.0039/lead | ~4 DAYS (use webhook) |
|
|
48
|
-
| `apollo-filters` | `{seniorityLevel, functionDept, companyIndustry, personCountry, personState, companySize, personTitles[], companyDomains[], count}` | $0.0039/lead | ~4 DAYS |
|
|
49
48
|
| `maps` | `{searchStringsArray:["query"], locationQuery, maxCrawledPlacesPerSearch}` | $0.01/place | 5-30 min |
|
|
50
49
|
| `email-validator` | `{emails:["a@b.com"]}` | $0.0036/email | 1-10 min |
|
|
51
50
|
| `email-finder` | `{contacts:[{first_name,last_name,domain}], autoValidateEmails, autoFindMobiles}` | $0.05/contact | 1-10 min |
|
|
@@ -78,8 +77,8 @@ Apollo scrapes take **up to 4 days** to deliver. Do NOT poll in a loop.
|
|
|
78
77
|
| GET | `/api/v1/runs?hours=24&limit=50` | Recent runs |
|
|
79
78
|
| POST | `/api/v1/scrape/cancel/{runId}` | Cancel running job |
|
|
80
79
|
| GET | `/api/v1/scrape/logs/{runId}` | Run logs |
|
|
81
|
-
| GET | `/api/v1/apollo-status` | Apollo service health |
|
|
82
|
-
| GET | `/api/v1/database/leads?title=CTO&country=United%20States&hasEmail=true&limit=100` | Lead DB ($149/mo plan and up, 100k new leads/day). Optional: `excludeDelivered=true` (only leads you don't have yet), `after=0` then `pagination.next_after` (cursor paging) |
|
|
80
|
+
| GET | `/api/v1/apollo-status` | Apollo service health: `{status: "url-based" or "legacy", message, timestamp}` |
|
|
81
|
+
| GET | `/api/v1/database/leads?title=CTO&country=United%20States&hasEmail=true&limit=100` | Lead DB ($149/mo plan and up, 100k new leads/day). Optional: `excludeDelivered=true` (only leads you don't have yet), `after=0` then `pagination.next_after` (cursor paging). Company filters: `keywords`, `revenueMin`/`revenueMax`, `companyCountry`/`companyState`/`companyCity`. Exclusions: `notTitle`, `notKeywords`, `notIndustry`. `apolloUrl=<Apollo people-search URL>` searches with that URL's filters |
|
|
83
82
|
| GET | `/api/v1/database/local-businesses?...` | Local biz DB ($149/mo plan and up). Same optional `excludeDelivered` / `after` |
|
|
84
83
|
| GET | `/api/v1/database/ecommerce?...` | Ecommerce DB ($149/mo plan and up). Same optional `excludeDelivered` / `after` |
|
|
85
84
|
|
package/bin/cli.mjs
CHANGED
|
@@ -5,7 +5,9 @@ import fs from 'fs'
|
|
|
5
5
|
import readline from 'readline'
|
|
6
6
|
|
|
7
7
|
const [,, cmd, ...args] = process.argv
|
|
8
|
-
|
|
8
|
+
// Read "--name value" and remove both from args. (It used to splice first and then read
|
|
9
|
+
// args[i], which by then was the NEXT flag, so every value flag got the wrong value.)
|
|
10
|
+
const flag = (name) => { const i = args.indexOf(name); if (i === -1) return undefined; const v = args[i + 1]; args.splice(i, 2); return v ?? true }
|
|
9
11
|
const flagBool = (name) => { const i = args.indexOf(name); if (i !== -1) { args.splice(i, 1); return true } return false }
|
|
10
12
|
const json = (d) => JSON.stringify(d, null, 2)
|
|
11
13
|
|
|
@@ -368,19 +370,24 @@ async function main() {
|
|
|
368
370
|
// ── Database: Leads ($149/mo plan and up) ─────────────
|
|
369
371
|
case 'db-leads': {
|
|
370
372
|
const params = {}
|
|
371
|
-
|
|
372
|
-
'--company', '--domain', '--company-size', '--seniority',
|
|
373
|
-
'--department', '--page', '--limit', '--min-employees', '--max-employees', '--after']) {
|
|
374
|
-
const v = flag(f)
|
|
375
|
-
if (v === undefined) continue
|
|
376
|
-
const key = { '--title': 'title', '--industry': 'industry', '--country': 'country',
|
|
373
|
+
const keyFor = { '--title': 'title', '--industry': 'industry', '--country': 'country',
|
|
377
374
|
'--state': 'state', '--city': 'city', '--company': 'companyName',
|
|
378
|
-
'--domain': 'companyDomain',
|
|
375
|
+
'--domain': 'companyDomain',
|
|
379
376
|
'--seniority': 'seniority', '--department': 'department',
|
|
380
377
|
'--page': 'page', '--limit': 'limit',
|
|
381
378
|
'--min-employees': 'minEmployees', '--max-employees': 'maxEmployees',
|
|
382
|
-
'--after': 'after'
|
|
383
|
-
|
|
379
|
+
'--after': 'after', '--apollo-url': 'apolloUrl',
|
|
380
|
+
'--keywords': 'keywords', '--revenue-min': 'revenueMin', '--revenue-max': 'revenueMax',
|
|
381
|
+
'--company-country': 'companyCountry', '--company-state': 'companyState',
|
|
382
|
+
'--company-city': 'companyCity', '--not-title': 'notTitle',
|
|
383
|
+
'--not-keywords': 'notKeywords', '--not-industry': 'notIndustry' }
|
|
384
|
+
// List filters take commas: --country "United States,Canada" sends both.
|
|
385
|
+
const lists = new Set(['--industry', '--country', '--state', '--seniority', '--department',
|
|
386
|
+
'--company-country', '--company-state', '--not-industry'])
|
|
387
|
+
for (const f of Object.keys(keyFor)) {
|
|
388
|
+
const v = flag(f)
|
|
389
|
+
if (v === undefined) continue
|
|
390
|
+
params[keyFor[f]] = lists.has(f) ? v.split(',').map(s => s.trim()).filter(Boolean) : v
|
|
384
391
|
}
|
|
385
392
|
if (flagBool('--has-email')) params.hasEmail = 'true'
|
|
386
393
|
if (flagBool('--has-phone')) params.hasPhone = 'true'
|
|
@@ -388,6 +395,7 @@ async function main() {
|
|
|
388
395
|
const r = await sc.dbLeads(params)
|
|
389
396
|
console.log(`${r.pagination?.total || '?'} total leads, page ${r.pagination?.page || 1} of ${r.pagination?.totalPages || '?'}`)
|
|
390
397
|
if (r.pagination?.next_after) console.log(`Next page: --after ${r.pagination.next_after}`)
|
|
398
|
+
if (r.apollo_translation?.not_supported?.length) console.log(`Not supported from the Apollo URL: ${r.apollo_translation.not_supported.join(', ')}`)
|
|
391
399
|
console.log(json(r.data?.slice(0, 3) || r))
|
|
392
400
|
if (r.data?.length > 3) console.log(`... and ${r.data.length - 3} more`)
|
|
393
401
|
break
|
|
@@ -445,6 +453,11 @@ ScraperCity CLI - B2B lead generation from your terminal
|
|
|
445
453
|
|
|
446
454
|
Database ($149/mo plan and up):
|
|
447
455
|
scrapercity db-leads [filters] Query lead database
|
|
456
|
+
--title --industry --country --state --city --company --domain --seniority --department
|
|
457
|
+
--min-employees --max-employees --has-email --has-phone
|
|
458
|
+
--keywords --revenue-min --revenue-max --company-country --company-state --company-city
|
|
459
|
+
--not-title --not-keywords --not-industry
|
|
460
|
+
--apollo-url "<Apollo people-search URL>" Search with that URL's filters
|
|
448
461
|
--exclude-delivered Only leads you don't already have
|
|
449
462
|
--after <id> Cursor paging (start with 0, then use the printed next id)
|
|
450
463
|
|
package/bin/mcp.mjs
CHANGED
|
@@ -50,17 +50,30 @@ const UTILITY_TOOLS = [
|
|
|
50
50
|
},
|
|
51
51
|
{
|
|
52
52
|
name: 'query_lead_database',
|
|
53
|
-
description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies. Up to 100 per request. Included with the $149/mo plan (100,000 new leads a day; leads you already have do not count again). For a daily pull of only new leads set excludeDelivered=true. For big pulls page with after: send after="0" first, then the pagination.next_after from each response until it is null.',
|
|
53
|
+
description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies (plus company website, revenue, HQ location and keywords). Up to 100 per request. Included with the $149/mo plan (100,000 new leads a day; leads you already have do not count again). For a daily pull of only new leads set excludeDelivered=true. For big pulls page with after: send after="0" first, then the pagination.next_after from each response until it is null. Have an Apollo people-search URL? Pass it as apolloUrl to search with its filters right away (apollo_translation in the response says what was applied).',
|
|
54
54
|
inputSchema: { type: 'object', properties: {
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
55
|
+
apolloUrl: { type: 'string', description: 'An Apollo people-search URL (app.apollo.io/#/people?...). Its filters become the search; any other filter you pass overrides the matching one.' },
|
|
56
|
+
title: { type: 'string', description: 'Job title filter, comma separated for several (partial match, e.g. "CEO, Founder")' },
|
|
57
|
+
industry: { type: 'array', items: { type: 'string' }, description: 'Company industries (any match)' },
|
|
58
|
+
country: { type: 'array', items: { type: 'string' }, description: 'Person countries, full names like "United States" (US, USA, UK, UAE also work)' },
|
|
59
|
+
state: { type: 'array', items: { type: 'string' }, description: 'Person states' },
|
|
60
|
+
city: { type: 'string', description: 'Person city, comma separated for several' },
|
|
60
61
|
companyName: { type: 'string', description: 'Company name' },
|
|
61
|
-
companyDomain: { type: 'string', description: 'Company domain' },
|
|
62
|
-
seniority: { type: 'string', description: 'e.g. "vp", "director", "c_suite"' },
|
|
63
|
-
department: { type: 'string', description: 'e.g. "sales", "engineering"' },
|
|
62
|
+
companyDomain: { type: 'string', description: 'Company domain(s), comma separated' },
|
|
63
|
+
seniority: { type: 'array', items: { type: 'string' }, description: 'e.g. ["vp", "director", "c_suite"]' },
|
|
64
|
+
department: { type: 'array', items: { type: 'string' }, description: 'e.g. ["sales", "engineering"]' },
|
|
65
|
+
minEmployees: { type: 'number', description: 'Minimum company employee count' },
|
|
66
|
+
maxEmployees: { type: 'number', description: 'Maximum company employee count' },
|
|
67
|
+
keywords: { type: 'array', items: { type: 'string' }, description: 'Company keywords/tags, any match (e.g. ["saas", "fintech"])' },
|
|
68
|
+
revenueMin: { type: 'number', description: 'Minimum company annual revenue, USD' },
|
|
69
|
+
revenueMax: { type: 'number', description: 'Maximum company annual revenue, USD' },
|
|
70
|
+
companyCountry: { type: 'array', items: { type: 'string' }, description: 'Company HQ countries (not where the person is)' },
|
|
71
|
+
companyState: { type: 'array', items: { type: 'string' }, description: 'Company HQ states' },
|
|
72
|
+
companyCity: { type: 'string', description: 'Company HQ city' },
|
|
73
|
+
notTitle: { type: 'string', description: 'Exclude titles containing any of these (comma separated)' },
|
|
74
|
+
notKeywords: { type: 'array', items: { type: 'string' }, description: 'Exclude companies with any of these keywords' },
|
|
75
|
+
notIndustry: { type: 'array', items: { type: 'string' }, description: 'Exclude these industries' },
|
|
76
|
+
socialUrl: { type: 'string', description: 'Match a social profile URL' },
|
|
64
77
|
hasEmail: { type: 'boolean', description: 'Only contacts with email' },
|
|
65
78
|
hasPhone: { type: 'boolean', description: 'Only contacts with phone' },
|
|
66
79
|
page: { type: 'number', description: 'Page number (default 1)', default: 1 },
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
// AUTO-GENERATED by scripts/generate.mjs — DO NOT EDIT BY HAND.
|
|
2
2
|
// Source of truth: scraperConfigs.ts (+ src/config/scrapers.ts). Regenerate: npm run generate
|
|
3
|
-
// Generated: 2026-
|
|
3
|
+
// Generated: 2026-10-01T23:44:06.142Z
|
|
4
4
|
export const TOOLS = [
|
|
5
5
|
{
|
|
6
6
|
"name": "scrape_apollo",
|