scrapercity 1.0.14 → 1.0.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENT_DOCS.txt +8 -3
- package/SKILL.md +1 -1
- package/bin/cli.mjs +4 -3
- package/bin/mcp.mjs +2 -2
- package/lib/catalog.generated.mjs +1 -1
- package/lib/client.mjs +8 -1
- package/package.json +1 -1
package/AGENT_DOCS.txt
CHANGED
|
@@ -189,8 +189,11 @@ GET /api/v1/database/leads?title=CTO&country=United States&hasEmail=true&page=1&
|
|
|
189
189
|
Company filters: keywords (company tags, whole-word, any match), revenueMin, revenueMax (USD),
|
|
190
190
|
companyCountry, companyState, companyCity (company HQ, not the person)
|
|
191
191
|
Exclusions: notTitle, notKeywords, notIndustry
|
|
192
|
-
|
|
193
|
-
The response adds
|
|
192
|
+
url=<people-search URL>: use that search's filters (other params override).
|
|
193
|
+
The response adds search_translation: { applied, approximated, not_supported }.
|
|
194
|
+
A URL with no filters the database can use returns 400.
|
|
195
|
+
POST /api/v1/database/leads with a JSON body takes the same parameters (arrays as JSON
|
|
196
|
+
arrays). Use it for a long url or long lists: a query string over ~16KB is refused.
|
|
194
197
|
Country: full name ("United States"); US, USA, UK, UAE also work.
|
|
195
198
|
Array params use repeated keys: ?seniority=vp&seniority=director
|
|
196
199
|
(keywords / notKeywords also accept commas: ?keywords=saas,fintech)
|
|
@@ -202,12 +205,14 @@ GET /api/v1/database/leads?title=CTO&country=United States&hasEmail=true&page=1&
|
|
|
202
205
|
Every lead returned is saved to your account. Only NEW leads count toward the
|
|
203
206
|
100k/day limit; fetching a lead you already have again is free.
|
|
204
207
|
Returns: { data: [...leads], pagination: { page, limit, total, totalPages, has_more, next_after }, rate_limit,
|
|
205
|
-
|
|
208
|
+
search_translation? }
|
|
206
209
|
Lead fields include company_industry, company_size, employees_count, company_website,
|
|
207
210
|
company_annual_revenue, company_city, company_state, company_country, keywords.
|
|
208
211
|
409 with filters_not_applied_yet: the company filters listed can't be used yet (company
|
|
209
212
|
data still being prepared right after launch). Nothing is returned or counted; retry later
|
|
210
213
|
or drop those filters.
|
|
214
|
+
503: the search took too long to run. Narrow it (fewer titles or keywords, add a location
|
|
215
|
+
or company size) and retry.
|
|
211
216
|
|
|
212
217
|
GET /api/v1/database/local-businesses?...
|
|
213
218
|
GET /api/v1/database/ecommerce?...
|
package/SKILL.md
CHANGED
|
@@ -78,7 +78,7 @@ Apollo scrapes take **up to 4 days** to deliver. Do NOT poll in a loop.
|
|
|
78
78
|
| POST | `/api/v1/scrape/cancel/{runId}` | Cancel running job |
|
|
79
79
|
| GET | `/api/v1/scrape/logs/{runId}` | Run logs |
|
|
80
80
|
| GET | `/api/v1/apollo-status` | Apollo service health: `{status: "url-based" or "legacy", message, timestamp}` |
|
|
81
|
-
| GET | `/api/v1/database/leads?title=CTO&country=United%20States&hasEmail=true&limit=100` | Lead DB ($149/mo plan and up, 100k new leads/day). Optional: `excludeDelivered=true` (only leads you don't have yet), `after=0` then `pagination.next_after` (cursor paging). Company filters: `keywords`, `revenueMin`/`revenueMax`, `companyCountry`/`companyState`/`companyCity`. Exclusions: `notTitle`, `notKeywords`, `notIndustry`. `
|
|
81
|
+
| GET | `/api/v1/database/leads?title=CTO&country=United%20States&hasEmail=true&limit=100` | Lead DB ($149/mo plan and up, 100k new leads/day). Optional: `excludeDelivered=true` (only leads you don't have yet), `after=0` then `pagination.next_after` (cursor paging). Company filters: `keywords`, `revenueMin`/`revenueMax`, `companyCountry`/`companyState`/`companyCity`. Exclusions: `notTitle`, `notKeywords`, `notIndustry`. `url=<people-search URL>` searches with that URL's filters. POST with a JSON body takes the same parameters (use it for a long `url`) |
|
|
82
82
|
| GET | `/api/v1/database/local-businesses?...` | Local biz DB ($149/mo plan and up). Same optional `excludeDelivered` / `after` |
|
|
83
83
|
| GET | `/api/v1/database/ecommerce?...` | Ecommerce DB ($149/mo plan and up). Same optional `excludeDelivered` / `after` |
|
|
84
84
|
|
package/bin/cli.mjs
CHANGED
|
@@ -376,7 +376,7 @@ async function main() {
|
|
|
376
376
|
'--seniority': 'seniority', '--department': 'department',
|
|
377
377
|
'--page': 'page', '--limit': 'limit',
|
|
378
378
|
'--min-employees': 'minEmployees', '--max-employees': 'maxEmployees',
|
|
379
|
-
'--after': 'after', '--apollo-url': 'apolloUrl',
|
|
379
|
+
'--after': 'after', '--url': 'url', '--apollo-url': 'apolloUrl',
|
|
380
380
|
'--keywords': 'keywords', '--revenue-min': 'revenueMin', '--revenue-max': 'revenueMax',
|
|
381
381
|
'--company-country': 'companyCountry', '--company-state': 'companyState',
|
|
382
382
|
'--company-city': 'companyCity', '--not-title': 'notTitle',
|
|
@@ -395,7 +395,8 @@ async function main() {
|
|
|
395
395
|
const r = await sc.dbLeads(params)
|
|
396
396
|
console.log(`${r.pagination?.total || '?'} total leads, page ${r.pagination?.page || 1} of ${r.pagination?.totalPages || '?'}`)
|
|
397
397
|
if (r.pagination?.next_after) console.log(`Next page: --after ${r.pagination.next_after}`)
|
|
398
|
-
|
|
398
|
+
const tr = r.search_translation || r.apollo_translation
|
|
399
|
+
if (tr?.not_supported?.length) console.log(`Not supported from the search URL: ${tr.not_supported.join(', ')}`)
|
|
399
400
|
console.log(json(r.data?.slice(0, 3) || r))
|
|
400
401
|
if (r.data?.length > 3) console.log(`... and ${r.data.length - 3} more`)
|
|
401
402
|
break
|
|
@@ -457,7 +458,7 @@ ScraperCity CLI - B2B lead generation from your terminal
|
|
|
457
458
|
--min-employees --max-employees --has-email --has-phone
|
|
458
459
|
--keywords --revenue-min --revenue-max --company-country --company-state --company-city
|
|
459
460
|
--not-title --not-keywords --not-industry
|
|
460
|
-
--
|
|
461
|
+
--url "<people-search URL>" Search with that URL's filters
|
|
461
462
|
--exclude-delivered Only leads you don't already have
|
|
462
463
|
--after <id> Cursor paging (start with 0, then use the printed next id)
|
|
463
464
|
|
package/bin/mcp.mjs
CHANGED
|
@@ -50,9 +50,9 @@ const UTILITY_TOOLS = [
|
|
|
50
50
|
},
|
|
51
51
|
{
|
|
52
52
|
name: 'query_lead_database',
|
|
53
|
-
description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies (plus company website, revenue, HQ location and keywords). Up to 100 per request. Included with the $149/mo plan (100,000 new leads a day; leads you already have do not count again). For a daily pull of only new leads set excludeDelivered=true. For big pulls page with after: send after="0" first, then the pagination.next_after from each response until it is null. Have
|
|
53
|
+
description: 'Query the B2B lead database directly. Returns contacts with names, emails, phones, titles, companies (plus company website, revenue, HQ location and keywords). Up to 100 per request. Included with the $149/mo plan (100,000 new leads a day; leads you already have do not count again). For a daily pull of only new leads set excludeDelivered=true. For big pulls page with after: send after="0" first, then the pagination.next_after from each response until it is null. Have a saved people-search URL? Pass it as url to search with its filters right away (search_translation in the response says what was applied).',
|
|
54
54
|
inputSchema: { type: 'object', properties: {
|
|
55
|
-
|
|
55
|
+
url: { type: 'string', description: 'A people-search URL (the address of a people search with its filters). Its filters become the search; any other filter you pass overrides the matching one.' },
|
|
56
56
|
title: { type: 'string', description: 'Job title filter, comma separated for several (partial match, e.g. "CEO, Founder")' },
|
|
57
57
|
industry: { type: 'array', items: { type: 'string' }, description: 'Company industries (any match)' },
|
|
58
58
|
country: { type: 'array', items: { type: 'string' }, description: 'Person countries, full names like "United States" (US, USA, UK, UAE also work)' },
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
// AUTO-GENERATED by scripts/generate.mjs — DO NOT EDIT BY HAND.
|
|
2
2
|
// Source of truth: scraperConfigs.ts (+ src/config/scrapers.ts). Regenerate: npm run generate
|
|
3
|
-
// Generated: 2026-10-
|
|
3
|
+
// Generated: 2026-10-02T14:29:37.617Z
|
|
4
4
|
export const TOOLS = [
|
|
5
5
|
{
|
|
6
6
|
"name": "scrape_apollo",
|
package/lib/client.mjs
CHANGED
|
@@ -166,14 +166,21 @@ export const scrape = (slug, body) =>
|
|
|
166
166
|
export const postEndpoint = (endpoint, body) => post(endpoint, body)
|
|
167
167
|
|
|
168
168
|
// ── Databases ($649 plan) ─────────────────────────────────────
|
|
169
|
+
// A long request (a big Apollo URL, a long title or keyword list) goes as a POST with a JSON body:
|
|
170
|
+
// a query string past ~16KB is refused by the server before the search runs.
|
|
171
|
+
const LONG_QUERY = 6000
|
|
169
172
|
export const dbLeads = (params) => {
|
|
170
173
|
const qs = new URLSearchParams()
|
|
174
|
+
const body = {}
|
|
171
175
|
for (const [k, v] of Object.entries(params)) {
|
|
172
176
|
if (v === undefined || v === null || v === '') continue
|
|
173
177
|
if (Array.isArray(v)) v.forEach(x => qs.append(k, x))
|
|
174
178
|
else qs.append(k, String(v))
|
|
179
|
+
body[k] = v
|
|
175
180
|
}
|
|
176
|
-
|
|
181
|
+
const qstr = qs.toString()
|
|
182
|
+
if (qstr.length > LONG_QUERY) return post(ENDPOINTS['database-leads'], body)
|
|
183
|
+
return get(`${ENDPOINTS['database-leads']}?${qstr}`)
|
|
177
184
|
}
|
|
178
185
|
|
|
179
186
|
export const dbLocalBusinesses = (params) => {
|