scrapercity 1.0.13 → 1.0.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/AGENT_DOCS.txt CHANGED
@@ -40,17 +40,13 @@ SCRAPE ENDPOINTS (POST /api/v1/scrape/{slug})
40
40
 
41
41
  apollo
42
42
  Body: { url: "https://app.apollo.io/#/people?...", count: 1000, fileName: "My Export" }
43
- Returns: { runId, message }
43
+ Returns: { runId, message, lead_database? }
44
+ lead_database (when there are matches): { matches, url, api } - leads with the same
45
+ filters already in the Lead Database (ones you don't have), a dashboard link, and the
46
+ /api/v1/database/leads call that returns them now.
44
47
  Cost: $0.0039/lead | Min 500, max 50000
45
48
  DELIVERY: UP TO 4 DAYS. Use webhook, not polling.
46
-
47
- apollo-filters
48
- Body: { seniorityLevel, functionDept, companyIndustry, personCountry, personState,
49
- companyCountry, companyState, companySize, personTitles:[], companyDomains:[],
50
- companyKeywords:[], personCities:[], companyCities:[], hasPhone, count, fileName }
51
- At least one filter required.
52
- Returns: { runId, message }
53
- Cost: $0.0039/lead | DELIVERY: UP TO 4 DAYS.
49
+ To filter by title/location/industry without a URL, use the Lead Database (below).
54
50
 
55
51
  maps
56
52
  Body: { searchStringsArray: ["plumbers"], locationQuery: "Denver, CO", maxCrawledPlacesPerSearch: 500 }
@@ -180,7 +176,8 @@ GET /api/v1/runs?hours=24&limit=50
180
176
  Returns: { runs: [{ run_id, status, handled, url, file_name, created_at }] }
181
177
 
182
178
  GET /api/v1/apollo-status
183
- Returns: { useApify, message } - check if Apollo URL endpoint is available
179
+ Returns: { status: "url-based" | "legacy", message, timestamp } - "url-based" means the
180
+ Apollo URL endpoint is available
184
181
 
185
182
  ────────────────────────────────────────
186
183
  DATABASES (GET, $149/mo plan and up, 100k new records/day)
@@ -189,7 +186,17 @@ DATABASES (GET, $149/mo plan and up, 100k new records/day)
189
186
  GET /api/v1/database/leads?title=CTO&country=United States&hasEmail=true&page=1&limit=100
190
187
  Filters: title, industry, country, state, city, companyName, companyDomain,
191
188
  seniority, department, hasEmail, hasPhone, minEmployees, maxEmployees, socialUrl
189
+ Company filters: keywords (company tags, whole-word, any match), revenueMin, revenueMax (USD),
190
+ companyCountry, companyState, companyCity (company HQ, not the person)
191
+ Exclusions: notTitle, notKeywords, notIndustry
192
+ url=<people-search URL>: use that search's filters (other params override).
193
+ The response adds search_translation: { applied, approximated, not_supported }.
194
+ A URL with no filters the database can use returns 400.
195
+ POST /api/v1/database/leads with a JSON body takes the same parameters (arrays as JSON
196
+ arrays). Use it for a long url or long lists: a query string over ~16KB is refused.
197
+ Country: full name ("United States"); US, USA, UK, UAE also work.
192
198
  Array params use repeated keys: ?seniority=vp&seniority=director
199
+ (keywords / notKeywords also accept commas: ?keywords=saas,fintech)
193
200
  Optional:
194
201
  excludeDelivered=true skip leads this account already has (from the API or the
195
202
  dashboard). For daily pulls. Keep page=1 (or use after).
@@ -197,7 +204,15 @@ GET /api/v1/database/leads?title=CTO&country=United States&hasEmail=true&page=1&
197
204
  pagination.next_after from each response until it is null.
198
205
  Every lead returned is saved to your account. Only NEW leads count toward the
199
206
  100k/day limit; fetching a lead you already have again is free.
200
- Returns: { data: [...leads], pagination: { page, limit, total, totalPages, has_more, next_after }, rate_limit }
207
+ Returns: { data: [...leads], pagination: { page, limit, total, totalPages, has_more, next_after }, rate_limit,
208
+ search_translation? }
209
+ Lead fields include company_industry, company_size, employees_count, company_website,
210
+ company_annual_revenue, company_city, company_state, company_country, keywords.
211
+ 409 with filters_not_applied_yet: the company filters listed can't be used yet (company
212
+ data still being prepared right after launch). Nothing is returned or counted; retry later
213
+ or drop those filters.
214
+ 503: the search took too long to run. Narrow it (fewer titles or keywords, add a location
215
+ or company size) and retry.
201
216
 
202
217
  GET /api/v1/database/local-businesses?...
203
218
  GET /api/v1/database/ecommerce?...
package/SKILL.md CHANGED
@@ -45,7 +45,6 @@ Auth: `Authorization: Bearer $SCRAPERCITY_API_KEY` on all requests.
45
45
  | Slug | Input | Cost | Speed |
46
46
  |------|-------|------|-------|
47
47
  | `apollo` | `{url, count, fileName}` | $0.0039/lead | ~4 DAYS (use webhook) |
48
- | `apollo-filters` | `{seniorityLevel, functionDept, companyIndustry, personCountry, personState, companySize, personTitles[], companyDomains[], count}` | $0.0039/lead | ~4 DAYS |
49
48
  | `maps` | `{searchStringsArray:["query"], locationQuery, maxCrawledPlacesPerSearch}` | $0.01/place | 5-30 min |
50
49
  | `email-validator` | `{emails:["a@b.com"]}` | $0.0036/email | 1-10 min |
51
50
  | `email-finder` | `{contacts:[{first_name,last_name,domain}], autoValidateEmails, autoFindMobiles}` | $0.05/contact | 1-10 min |
@@ -78,8 +77,8 @@ Apollo scrapes take **up to 4 days** to deliver. Do NOT poll in a loop.
78
77
  | GET | `/api/v1/runs?hours=24&limit=50` | Recent runs |
79
78
  | POST | `/api/v1/scrape/cancel/{runId}` | Cancel running job |
80
79
  | GET | `/api/v1/scrape/logs/{runId}` | Run logs |
81
- | GET | `/api/v1/apollo-status` | Apollo service health |
82
- | GET | `/api/v1/database/leads?title=CTO&country=United%20States&hasEmail=true&limit=100` | Lead DB ($149/mo plan and up, 100k new leads/day). Optional: `excludeDelivered=true` (only leads you don't have yet), `after=0` then `pagination.next_after` (cursor paging) |
80
+ | GET | `/api/v1/apollo-status` | Apollo service health: `{status: "url-based" or "legacy", message, timestamp}` |
81
+ | GET | `/api/v1/database/leads?title=CTO&country=United%20States&hasEmail=true&limit=100` | Lead DB ($149/mo plan and up, 100k new leads/day). Optional: `excludeDelivered=true` (only leads you don't have yet), `after=0` then `pagination.next_after` (cursor paging). Company filters: `keywords`, `revenueMin`/`revenueMax`, `companyCountry`/`companyState`/`companyCity`. Exclusions: `notTitle`, `notKeywords`, `notIndustry`. `url=<people-search URL>` searches with that URL's filters. POST with a JSON body takes the same parameters (use it for a long `url`) |
83
82
  | GET | `/api/v1/database/local-businesses?...` | Local biz DB ($149/mo plan and up). Same optional `excludeDelivered` / `after` |
84
83
  | GET | `/api/v1/database/ecommerce?...` | Ecommerce DB ($149/mo plan and up). Same optional `excludeDelivered` / `after` |
85
84
 
package/bin/cli.mjs CHANGED
@@ -5,7 +5,9 @@ import fs from 'fs'
5
5
  import readline from 'readline'
6
6
 
7
7
  const [,, cmd, ...args] = process.argv
8
- const flag = (name) => { const i = args.indexOf(name); return i !== -1 ? (args.splice(i, 2), args[i] || true) : undefined }
8
+ // Read "--name value" and remove both from args. (It used to splice first and then read
9
+ // args[i], which by then was the NEXT flag, so every value flag got the wrong value.)
10
+ const flag = (name) => { const i = args.indexOf(name); if (i === -1) return undefined; const v = args[i + 1]; args.splice(i, 2); return v ?? true }
9
11
  const flagBool = (name) => { const i = args.indexOf(name); if (i !== -1) { args.splice(i, 1); return true } return false }
10
12
  const json = (d) => JSON.stringify(d, null, 2)
11
13
 
@@ -368,19 +370,24 @@ async function main() {
368
370
  // ── Database: Leads ($149/mo plan and up) ─────────────
369
371
  case 'db-leads': {
370
372
  const params = {}
371
- for (const f of ['--title', '--industry', '--country', '--state', '--city',
372
- '--company', '--domain', '--company-size', '--seniority',
373
- '--department', '--page', '--limit', '--min-employees', '--max-employees', '--after']) {
374
- const v = flag(f)
375
- if (v === undefined) continue
376
- const key = { '--title': 'title', '--industry': 'industry', '--country': 'country',
373
+ const keyFor = { '--title': 'title', '--industry': 'industry', '--country': 'country',
377
374
  '--state': 'state', '--city': 'city', '--company': 'companyName',
378
- '--domain': 'companyDomain', '--company-size': 'companySize',
375
+ '--domain': 'companyDomain',
379
376
  '--seniority': 'seniority', '--department': 'department',
380
377
  '--page': 'page', '--limit': 'limit',
381
378
  '--min-employees': 'minEmployees', '--max-employees': 'maxEmployees',
382
- '--after': 'after' }[f]
383
- params[key] = v
379
+ '--after': 'after', '--url': 'url', '--apollo-url': 'apolloUrl',
380
+ '--keywords': 'keywords', '--revenue-min': 'revenueMin', '--revenue-max': 'revenueMax',
381
+ '--company-country': 'companyCountry', '--company-state': 'companyState',
382
+ '--company-city': 'companyCity', '--not-title': 'notTitle',
383
+ '--not-keywords': 'notKeywords', '--not-industry': 'notIndustry' }
384
+ // List filters take commas: --country "United States,Canada" sends both.
385
+ const lists = new Set(['--industry', '--country', '--state', '--seniority', '--department',
386
+ '--company-country', '--company-state', '--not-industry'])
387
+ for (const f of Object.keys(keyFor)) {
388
+ const v = flag(f)
389
+ if (v === undefined) continue
390
+ params[keyFor[f]] = lists.has(f) ? v.split(',').map(s => s.trim()).filter(Boolean) : v
384
391
  }
385
392
  if (flagBool('--has-email')) params.hasEmail = 'true'
386
393
  if (flagBool('--has-phone')) params.hasPhone = 'true'
@@ -388,6 +395,8 @@ async function main() {
388
395
  const r = await sc.dbLeads(params)
389
396
  console.log(`${r.pagination?.total || '?'} total leads, page ${r.pagination?.page || 1} of ${r.pagination?.totalPages || '?'}`)
390
397
  if (r.pagination?.next_after) console.log(`Next page: --after ${r.pagination.next_after}`)
398
+ const tr = r.search_translation || r.apollo_translation
399
+ if (tr?.not_supported?.length) console.log(`Not supported from the search URL: ${tr.not_supported.join(', ')}`)
391
400
  console.log(json(r.data?.slice(0, 3) || r))
392
401
  if (r.data?.length > 3) console.log(`... and ${r.data.length - 3} more`)
393
402
  break
@@ -445,6 +454,11 @@ ScraperCity CLI - B2B lead generation from your terminal
445
454
 
446
455
  Database ($149/mo plan and up):
447
456
  scrapercity db-leads [filters] Query lead database
457
+ --title --industry --country --state --city --company --domain --seniority --department
458
+ --min-employees --max-employees --has-email --has-phone
459
+ --keywords --revenue-min --revenue-max --company-country --company-state --company-city
460
+ --not-title --not-keywords --not-industry
461
+ --url "<people-search URL>" Search with that URL's filters
448
462
  --exclude-delivered Only leads you don't already have
449
463
  --after <id> Cursor paging (start with 0, then use the printed next id)
450
464
 
package/bin/mcp.mjs CHANGED
@@ -50,17 +50,30 @@ const UTILITY_TOOLS = [
50
50
  },
51
51
  {
52
52
  name: 'query_lead_database',
53
- description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies. Up to 100 per request. Included with the $149/mo plan (100,000 new leads a day; leads you already have do not count again). For a daily pull of only new leads set excludeDelivered=true. For big pulls page with after: send after="0" first, then the pagination.next_after from each response until it is null.',
53
+ description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies (plus company website, revenue, HQ location and keywords). Up to 100 per request. Included with the $149/mo plan (100,000 new leads a day; leads you already have do not count again). For a daily pull of only new leads set excludeDelivered=true. For big pulls page with after: send after="0" first, then the pagination.next_after from each response until it is null. Have a saved people-search URL? Pass it as url to search with its filters right away (search_translation in the response says what was applied).',
54
54
  inputSchema: { type: 'object', properties: {
55
- title: { type: 'string', description: 'Job title filter' },
56
- industry: { type: 'string', description: 'Company industry' },
57
- country: { type: 'string', description: 'Person country' },
58
- state: { type: 'string', description: 'Person state' },
59
- city: { type: 'string', description: 'Person city' },
55
+ url: { type: 'string', description: 'A people-search URL (the address of a people search with its filters). Its filters become the search; any other filter you pass overrides the matching one.' },
56
+ title: { type: 'string', description: 'Job title filter, comma separated for several (partial match, e.g. "CEO, Founder")' },
57
+ industry: { type: 'array', items: { type: 'string' }, description: 'Company industries (any match)' },
58
+ country: { type: 'array', items: { type: 'string' }, description: 'Person countries, full names like "United States" (US, USA, UK, UAE also work)' },
59
+ state: { type: 'array', items: { type: 'string' }, description: 'Person states' },
60
+ city: { type: 'string', description: 'Person city, comma separated for several' },
60
61
  companyName: { type: 'string', description: 'Company name' },
61
- companyDomain: { type: 'string', description: 'Company domain' },
62
- seniority: { type: 'string', description: 'e.g. "vp", "director", "c_suite"' },
63
- department: { type: 'string', description: 'e.g. "sales", "engineering"' },
62
+ companyDomain: { type: 'string', description: 'Company domain(s), comma separated' },
63
+ seniority: { type: 'array', items: { type: 'string' }, description: 'e.g. ["vp", "director", "c_suite"]' },
64
+ department: { type: 'array', items: { type: 'string' }, description: 'e.g. ["sales", "engineering"]' },
65
+ minEmployees: { type: 'number', description: 'Minimum company employee count' },
66
+ maxEmployees: { type: 'number', description: 'Maximum company employee count' },
67
+ keywords: { type: 'array', items: { type: 'string' }, description: 'Company keywords/tags, any match (e.g. ["saas", "fintech"])' },
68
+ revenueMin: { type: 'number', description: 'Minimum company annual revenue, USD' },
69
+ revenueMax: { type: 'number', description: 'Maximum company annual revenue, USD' },
70
+ companyCountry: { type: 'array', items: { type: 'string' }, description: 'Company HQ countries (not where the person is)' },
71
+ companyState: { type: 'array', items: { type: 'string' }, description: 'Company HQ states' },
72
+ companyCity: { type: 'string', description: 'Company HQ city' },
73
+ notTitle: { type: 'string', description: 'Exclude titles containing any of these (comma separated)' },
74
+ notKeywords: { type: 'array', items: { type: 'string' }, description: 'Exclude companies with any of these keywords' },
75
+ notIndustry: { type: 'array', items: { type: 'string' }, description: 'Exclude these industries' },
76
+ socialUrl: { type: 'string', description: 'Match a social profile URL' },
64
77
  hasEmail: { type: 'boolean', description: 'Only contacts with email' },
65
78
  hasPhone: { type: 'boolean', description: 'Only contacts with phone' },
66
79
  page: { type: 'number', description: 'Page number (default 1)', default: 1 },
@@ -1,6 +1,6 @@
1
1
  // AUTO-GENERATED by scripts/generate.mjs — DO NOT EDIT BY HAND.
2
2
  // Source of truth: scraperConfigs.ts (+ src/config/scrapers.ts). Regenerate: npm run generate
3
- // Generated: 2026-09-28T13:03:17.780Z
3
+ // Generated: 2026-10-02T14:29:37.617Z
4
4
  export const TOOLS = [
5
5
  {
6
6
  "name": "scrape_apollo",
package/lib/client.mjs CHANGED
@@ -166,14 +166,21 @@ export const scrape = (slug, body) =>
166
166
  export const postEndpoint = (endpoint, body) => post(endpoint, body)
167
167
 
168
168
  // ── Databases ($649 plan) ─────────────────────────────────────
169
+ // A long request (a big Apollo URL, a long title or keyword list) goes as a POST with a JSON body:
170
+ // a query string past ~16KB is refused by the server before the search runs.
171
+ const LONG_QUERY = 6000
169
172
  export const dbLeads = (params) => {
170
173
  const qs = new URLSearchParams()
174
+ const body = {}
171
175
  for (const [k, v] of Object.entries(params)) {
172
176
  if (v === undefined || v === null || v === '') continue
173
177
  if (Array.isArray(v)) v.forEach(x => qs.append(k, x))
174
178
  else qs.append(k, String(v))
179
+ body[k] = v
175
180
  }
176
- return get(`${ENDPOINTS['database-leads']}?${qs}`)
181
+ const qstr = qs.toString()
182
+ if (qstr.length > LONG_QUERY) return post(ENDPOINTS['database-leads'], body)
183
+ return get(`${ENDPOINTS['database-leads']}?${qstr}`)
177
184
  }
178
185
 
179
186
  export const dbLocalBusinesses = (params) => {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "scrapercity",
3
- "version": "1.0.13",
3
+ "version": "1.0.15",
4
4
  "description": "ScraperCity CLI & MCP Server - B2B lead generation for AI agents",
5
5
  "type": "module",
6
6
  "bin": {