scrapercity 1.0.12 → 1.0.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/AGENT_DOCS.txt CHANGED
@@ -40,17 +40,13 @@ SCRAPE ENDPOINTS (POST /api/v1/scrape/{slug})
40
40
 
41
41
  apollo
42
42
  Body: { url: "https://app.apollo.io/#/people?...", count: 1000, fileName: "My Export" }
43
- Returns: { runId, message }
43
+ Returns: { runId, message, lead_database? }
44
+ lead_database (when there are matches): { matches, url, api } - leads with the same
45
+ filters already in the Lead Database (ones you don't have), a dashboard link, and the
46
+ /api/v1/database/leads call that returns them now.
44
47
  Cost: $0.0039/lead | Min 500, max 50000
45
48
  DELIVERY: UP TO 4 DAYS. Use webhook, not polling.
46
-
47
- apollo-filters
48
- Body: { seniorityLevel, functionDept, companyIndustry, personCountry, personState,
49
- companyCountry, companyState, companySize, personTitles:[], companyDomains:[],
50
- companyKeywords:[], personCities:[], companyCities:[], hasPhone, count, fileName }
51
- At least one filter required.
52
- Returns: { runId, message }
53
- Cost: $0.0039/lead | DELIVERY: UP TO 4 DAYS.
49
+ To filter by title/location/industry without a URL, use the Lead Database (below).
54
50
 
55
51
  maps
56
52
  Body: { searchStringsArray: ["plumbers"], locationQuery: "Denver, CO", maxCrawledPlacesPerSearch: 500 }
@@ -180,21 +176,42 @@ GET /api/v1/runs?hours=24&limit=50
180
176
  Returns: { runs: [{ run_id, status, handled, url, file_name, created_at }] }
181
177
 
182
178
  GET /api/v1/apollo-status
183
- Returns: { useApify, message } - check if Apollo URL endpoint is available
179
+ Returns: { status: "url-based" | "legacy", message, timestamp } - "url-based" means the
180
+ Apollo URL endpoint is available
184
181
 
185
182
  ────────────────────────────────────────
186
- DATABASES (GET, $649 plan only, 100k/day)
183
+ DATABASES (GET, $149/mo plan and up, 100k new records/day)
187
184
  ────────────────────────────────────────
188
185
 
189
186
  GET /api/v1/database/leads?title=CTO&country=United States&hasEmail=true&page=1&limit=100
190
187
  Filters: title, industry, country, state, city, companyName, companyDomain,
191
- companySize, seniority, department, hasEmail, hasPhone, minEmployees, maxEmployees
188
+ seniority, department, hasEmail, hasPhone, minEmployees, maxEmployees, socialUrl
189
+ Company filters: keywords (company tags, whole-word, any match), revenueMin, revenueMax (USD),
190
+ companyCountry, companyState, companyCity (company HQ, not the person)
191
+ Exclusions: notTitle, notKeywords, notIndustry
192
+ apolloUrl=<Apollo people-search URL>: use that search's filters (other params override).
193
+ The response adds apollo_translation: { applied, approximated, not_supported }.
194
+ Country: full name ("United States"); US, USA, UK, UAE also work.
192
195
  Array params use repeated keys: ?seniority=vp&seniority=director
193
- Returns: { data: [...leads], pagination: { page, limit, total, totalPages } }
196
+ (keywords / notKeywords also accept commas: ?keywords=saas,fintech)
197
+ Optional:
198
+ excludeDelivered=true skip leads this account already has (from the API or the
199
+ dashboard). For daily pulls. Keep page=1 (or use after).
200
+ after=<id> cursor paging in id order. Send after=0 first, then
201
+ pagination.next_after from each response until it is null.
202
+ Every lead returned is saved to your account. Only NEW leads count toward the
203
+ 100k/day limit; fetching a lead you already have again is free.
204
+ Returns: { data: [...leads], pagination: { page, limit, total, totalPages, has_more, next_after }, rate_limit,
205
+ apollo_translation? }
206
+ Lead fields include company_industry, company_size, employees_count, company_website,
207
+ company_annual_revenue, company_city, company_state, company_country, keywords.
208
+ 409 with filters_not_applied_yet: the company filters listed can't be used yet (company
209
+ data still being prepared right after launch). Nothing is returned or counted; retry later
210
+ or drop those filters.
194
211
 
195
212
  GET /api/v1/database/local-businesses?...
196
213
  GET /api/v1/database/ecommerce?...
197
- Same pattern, different datasets.
214
+ Same pattern (including excludeDelivered and after), different datasets.
198
215
 
199
216
  ────────────────────────────────────────
200
217
  ERROR CODES
package/README.md CHANGED
@@ -104,7 +104,7 @@ curl -O https://app.scrapercity.com/api/downloads/RUN_ID \
104
104
  | **Store Leads** | Shopify/WooCommerce stores with contacts | $0.0039/lead |
105
105
  | **BuiltWith** | All sites using a technology | $4.99/search |
106
106
  | **Criminal Records** | Background check by name | $1.00 if found |
107
- | **Lead Database** | 3M+ B2B contacts, instant query ($649 plan) | Included |
107
+ | **Lead Database** | 3M+ B2B contacts, instant query ($149/mo plan and up) | Included |
108
108
 
109
109
  ## How It Works
110
110
 
package/SKILL.md CHANGED
@@ -45,7 +45,6 @@ Auth: `Authorization: Bearer $SCRAPERCITY_API_KEY` on all requests.
45
45
  | Slug | Input | Cost | Speed |
46
46
  |------|-------|------|-------|
47
47
  | `apollo` | `{url, count, fileName}` | $0.0039/lead | ~4 DAYS (use webhook) |
48
- | `apollo-filters` | `{seniorityLevel, functionDept, companyIndustry, personCountry, personState, companySize, personTitles[], companyDomains[], count}` | $0.0039/lead | ~4 DAYS |
49
48
  | `maps` | `{searchStringsArray:["query"], locationQuery, maxCrawledPlacesPerSearch}` | $0.01/place | 5-30 min |
50
49
  | `email-validator` | `{emails:["a@b.com"]}` | $0.0036/email | 1-10 min |
51
50
  | `email-finder` | `{contacts:[{first_name,last_name,domain}], autoValidateEmails, autoFindMobiles}` | $0.05/contact | 1-10 min |
@@ -78,10 +77,10 @@ Apollo scrapes take **up to 4 days** to deliver. Do NOT poll in a loop.
78
77
  | GET | `/api/v1/runs?hours=24&limit=50` | Recent runs |
79
78
  | POST | `/api/v1/scrape/cancel/{runId}` | Cancel running job |
80
79
  | GET | `/api/v1/scrape/logs/{runId}` | Run logs |
81
- | GET | `/api/v1/apollo-status` | Apollo service health |
82
- | GET | `/api/v1/database/leads?title=CTO&country=US&hasEmail=true&page=1&limit=100` | Lead DB ($649 plan, 100k/day) |
83
- | GET | `/api/v1/database/local-businesses?...` | Local biz DB ($649 plan) |
84
- | GET | `/api/v1/database/ecommerce?...` | Ecommerce DB ($649 plan) |
80
+ | GET | `/api/v1/apollo-status` | Apollo service health: `{status: "url-based" or "legacy", message, timestamp}` |
81
+ | GET | `/api/v1/database/leads?title=CTO&country=United%20States&hasEmail=true&limit=100` | Lead DB ($149/mo plan and up, 100k new leads/day). Optional: `excludeDelivered=true` (only leads you don't have yet), `after=0` then `pagination.next_after` (cursor paging). Company filters: `keywords`, `revenueMin`/`revenueMax`, `companyCountry`/`companyState`/`companyCity`. Exclusions: `notTitle`, `notKeywords`, `notIndustry`. `apolloUrl=<Apollo people-search URL>` searches with that URL's filters |
82
+ | GET | `/api/v1/database/local-businesses?...` | Local biz DB ($149/mo plan and up). Same optional `excludeDelivered` / `after` |
83
+ | GET | `/api/v1/database/ecommerce?...` | Ecommerce DB ($149/mo plan and up). Same optional `excludeDelivered` / `after` |
85
84
 
86
85
  ## Status Values
87
86
  `RUNNING` → `SUCCEEDED` or `FAILED` or `CANCELLED`
package/bin/cli.mjs CHANGED
@@ -5,7 +5,9 @@ import fs from 'fs'
5
5
  import readline from 'readline'
6
6
 
7
7
  const [,, cmd, ...args] = process.argv
8
- const flag = (name) => { const i = args.indexOf(name); return i !== -1 ? (args.splice(i, 2), args[i] || true) : undefined }
8
+ // Read "--name value" and remove both from args. (It used to splice first and then read
9
+ // args[i], which by then was the NEXT flag, so every value flag got the wrong value.)
10
+ const flag = (name) => { const i = args.indexOf(name); if (i === -1) return undefined; const v = args[i + 1]; args.splice(i, 2); return v ?? true }
9
11
  const flagBool = (name) => { const i = args.indexOf(name); if (i !== -1) { args.splice(i, 1); return true } return false }
10
12
  const json = (d) => JSON.stringify(d, null, 2)
11
13
 
@@ -365,26 +367,35 @@ async function main() {
365
367
  break
366
368
  }
367
369
 
368
- // ── Database: Leads ($649) ────────────────────────────
370
+ // ── Database: Leads ($149/mo plan and up) ─────────────
369
371
  case 'db-leads': {
370
372
  const params = {}
371
- for (const f of ['--title', '--industry', '--country', '--state', '--city',
372
- '--company', '--domain', '--company-size', '--seniority',
373
- '--department', '--page', '--limit', '--min-employees', '--max-employees']) {
374
- const v = flag(f)
375
- if (v === undefined) continue
376
- const key = { '--title': 'title', '--industry': 'industry', '--country': 'country',
373
+ const keyFor = { '--title': 'title', '--industry': 'industry', '--country': 'country',
377
374
  '--state': 'state', '--city': 'city', '--company': 'companyName',
378
- '--domain': 'companyDomain', '--company-size': 'companySize',
375
+ '--domain': 'companyDomain',
379
376
  '--seniority': 'seniority', '--department': 'department',
380
377
  '--page': 'page', '--limit': 'limit',
381
- '--min-employees': 'minEmployees', '--max-employees': 'maxEmployees' }[f]
382
- params[key] = v
378
+ '--min-employees': 'minEmployees', '--max-employees': 'maxEmployees',
379
+ '--after': 'after', '--apollo-url': 'apolloUrl',
380
+ '--keywords': 'keywords', '--revenue-min': 'revenueMin', '--revenue-max': 'revenueMax',
381
+ '--company-country': 'companyCountry', '--company-state': 'companyState',
382
+ '--company-city': 'companyCity', '--not-title': 'notTitle',
383
+ '--not-keywords': 'notKeywords', '--not-industry': 'notIndustry' }
384
+ // List filters take commas: --country "United States,Canada" sends both.
385
+ const lists = new Set(['--industry', '--country', '--state', '--seniority', '--department',
386
+ '--company-country', '--company-state', '--not-industry'])
387
+ for (const f of Object.keys(keyFor)) {
388
+ const v = flag(f)
389
+ if (v === undefined) continue
390
+ params[keyFor[f]] = lists.has(f) ? v.split(',').map(s => s.trim()).filter(Boolean) : v
383
391
  }
384
392
  if (flagBool('--has-email')) params.hasEmail = 'true'
385
393
  if (flagBool('--has-phone')) params.hasPhone = 'true'
394
+ if (flagBool('--exclude-delivered')) params.excludeDelivered = 'true'
386
395
  const r = await sc.dbLeads(params)
387
396
  console.log(`${r.pagination?.total || '?'} total leads, page ${r.pagination?.page || 1} of ${r.pagination?.totalPages || '?'}`)
397
+ if (r.pagination?.next_after) console.log(`Next page: --after ${r.pagination.next_after}`)
398
+ if (r.apollo_translation?.not_supported?.length) console.log(`Not supported from the Apollo URL: ${r.apollo_translation.not_supported.join(', ')}`)
388
399
  console.log(json(r.data?.slice(0, 3) || r))
389
400
  if (r.data?.length > 3) console.log(`... and ${r.data.length - 3} more`)
390
401
  break
@@ -440,8 +451,15 @@ ScraperCity CLI - B2B lead generation from your terminal
440
451
  scrapercity logs <runId> View run logs
441
452
  scrapercity runs [--hours 24] List recent runs
442
453
 
443
- Database ($649 plan):
454
+ Database ($149/mo plan and up):
444
455
  scrapercity db-leads [filters] Query lead database
456
+ --title --industry --country --state --city --company --domain --seniority --department
457
+ --min-employees --max-employees --has-email --has-phone
458
+ --keywords --revenue-min --revenue-max --company-country --company-state --company-city
459
+ --not-title --not-keywords --not-industry
460
+ --apollo-url "<Apollo people-search URL>" Search with that URL's filters
461
+ --exclude-delivered Only leads you don't already have
462
+ --after <id> Cursor paging (start with 0, then use the printed next id)
445
463
 
446
464
  Env: SCRAPERCITY_API_KEY=... or scrapercity login
447
465
  Docs: https://scrapercity.com/agents
package/bin/mcp.mjs CHANGED
@@ -50,21 +50,36 @@ const UTILITY_TOOLS = [
50
50
  },
51
51
  {
52
52
  name: 'query_lead_database',
53
- description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies. 100 per request, paginate with page param. Included with the $149/mo plan.',
53
+ description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies (plus company website, revenue, HQ location and keywords). Up to 100 per request. Included with the $149/mo plan (100,000 new leads a day; leads you already have do not count again). For a daily pull of only new leads set excludeDelivered=true. For big pulls page with after: send after="0" first, then the pagination.next_after from each response until it is null. Have an Apollo people-search URL? Pass it as apolloUrl to search with its filters right away (apollo_translation in the response says what was applied).',
54
54
  inputSchema: { type: 'object', properties: {
55
- title: { type: 'string', description: 'Job title filter' },
56
- industry: { type: 'string', description: 'Company industry' },
57
- country: { type: 'string', description: 'Person country' },
58
- state: { type: 'string', description: 'Person state' },
59
- city: { type: 'string', description: 'Person city' },
55
+ apolloUrl: { type: 'string', description: 'An Apollo people-search URL (app.apollo.io/#/people?...). Its filters become the search; any other filter you pass overrides the matching one.' },
56
+ title: { type: 'string', description: 'Job title filter, comma separated for several (partial match, e.g. "CEO, Founder")' },
57
+ industry: { type: 'array', items: { type: 'string' }, description: 'Company industries (any match)' },
58
+ country: { type: 'array', items: { type: 'string' }, description: 'Person countries, full names like "United States" (US, USA, UK, UAE also work)' },
59
+ state: { type: 'array', items: { type: 'string' }, description: 'Person states' },
60
+ city: { type: 'string', description: 'Person city, comma separated for several' },
60
61
  companyName: { type: 'string', description: 'Company name' },
61
- companyDomain: { type: 'string', description: 'Company domain' },
62
- seniority: { type: 'string', description: 'e.g. "vp", "director", "c_suite"' },
63
- department: { type: 'string', description: 'e.g. "sales", "engineering"' },
62
+ companyDomain: { type: 'string', description: 'Company domain(s), comma separated' },
63
+ seniority: { type: 'array', items: { type: 'string' }, description: 'e.g. ["vp", "director", "c_suite"]' },
64
+ department: { type: 'array', items: { type: 'string' }, description: 'e.g. ["sales", "engineering"]' },
65
+ minEmployees: { type: 'number', description: 'Minimum company employee count' },
66
+ maxEmployees: { type: 'number', description: 'Maximum company employee count' },
67
+ keywords: { type: 'array', items: { type: 'string' }, description: 'Company keywords/tags, any match (e.g. ["saas", "fintech"])' },
68
+ revenueMin: { type: 'number', description: 'Minimum company annual revenue, USD' },
69
+ revenueMax: { type: 'number', description: 'Maximum company annual revenue, USD' },
70
+ companyCountry: { type: 'array', items: { type: 'string' }, description: 'Company HQ countries (not where the person is)' },
71
+ companyState: { type: 'array', items: { type: 'string' }, description: 'Company HQ states' },
72
+ companyCity: { type: 'string', description: 'Company HQ city' },
73
+ notTitle: { type: 'string', description: 'Exclude titles containing any of these (comma separated)' },
74
+ notKeywords: { type: 'array', items: { type: 'string' }, description: 'Exclude companies with any of these keywords' },
75
+ notIndustry: { type: 'array', items: { type: 'string' }, description: 'Exclude these industries' },
76
+ socialUrl: { type: 'string', description: 'Match a social profile URL' },
64
77
  hasEmail: { type: 'boolean', description: 'Only contacts with email' },
65
78
  hasPhone: { type: 'boolean', description: 'Only contacts with phone' },
66
79
  page: { type: 'number', description: 'Page number (default 1)', default: 1 },
67
- limit: { type: 'number', description: 'Results per page (max 100)', default: 50 }
80
+ limit: { type: 'number', description: 'Results per page (max 100)', default: 50 },
81
+ excludeDelivered: { type: 'boolean', description: 'Skip leads this account already has (from the API or unlocked in the dashboard). With this on, keep page at 1 or use after.' },
82
+ after: { type: 'string', description: 'Cursor paging: return leads after this lead id. Use "0" to start, then the pagination.next_after from each response.' }
68
83
  } }
69
84
  }
70
85
  ]
@@ -95,6 +110,9 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
95
110
  const params = { ...args }
96
111
  if (params.hasEmail) params.hasEmail = 'true'
97
112
  if (params.hasPhone) params.hasPhone = 'true'
113
+ if (params.excludeDelivered) params.excludeDelivered = 'true'
114
+ else delete params.excludeDelivered
115
+ if (params.after !== undefined && params.after !== null) params.after = String(params.after)
98
116
  result = await sc.dbLeads(params)
99
117
  break
100
118
  }
@@ -1,6 +1,6 @@
1
1
  // AUTO-GENERATED by scripts/generate.mjs — DO NOT EDIT BY HAND.
2
2
  // Source of truth: scraperConfigs.ts (+ src/config/scrapers.ts). Regenerate: npm run generate
3
- // Generated: 2026-09-25T02:44:56.288Z
3
+ // Generated: 2026-10-01T23:44:06.142Z
4
4
  export const TOOLS = [
5
5
  {
6
6
  "name": "scrape_apollo",
@@ -559,29 +559,31 @@ export const TOOLS = [
559
559
  "contacts": {
560
560
  "type": "array",
561
561
  "items": {
562
- "type": "object"
562
+ "type": "object",
563
+ "properties": {
564
+ "first_name": {
565
+ "type": "string",
566
+ "description": "Person's first name (required if full_name not provided)"
567
+ },
568
+ "last_name": {
569
+ "type": "string",
570
+ "description": "Person's last name"
571
+ },
572
+ "full_name": {
573
+ "type": "string",
574
+ "description": "Full name (alternative to first_name + last_name)"
575
+ },
576
+ "domain": {
577
+ "type": "string",
578
+ "description": "Company domain (preferred over company_name)"
579
+ },
580
+ "company_name": {
581
+ "type": "string",
582
+ "description": "Company name (use if domain not available)"
583
+ }
584
+ }
563
585
  },
564
586
  "description": "Array of contact objects to look up"
565
- },
566
- "contacts[].first_name": {
567
- "type": "string",
568
- "description": "Person's first name (required if full_name not provided)"
569
- },
570
- "contacts[].last_name": {
571
- "type": "string",
572
- "description": "Person's last name"
573
- },
574
- "contacts[].full_name": {
575
- "type": "string",
576
- "description": "Full name (alternative to first_name + last_name)"
577
- },
578
- "contacts[].domain": {
579
- "type": "string",
580
- "description": "Company domain (preferred over company_name)"
581
- },
582
- "contacts[].company_name": {
583
- "type": "string",
584
- "description": "Company name (use if domain not available)"
585
587
  }
586
588
  },
587
589
  "required": [
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "scrapercity",
3
- "version": "1.0.12",
3
+ "version": "1.0.14",
4
4
  "description": "ScraperCity CLI & MCP Server - B2B lead generation for AI agents",
5
5
  "type": "module",
6
6
  "bin": {