scrapercity 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENT_DOCS.txt +205 -0
- package/README.md +128 -0
- package/SKILL.md +98 -0
- package/bin/cli.mjs +518 -0
- package/bin/mcp.mjs +474 -0
- package/lib/client.mjs +215 -0
- package/package.json +30 -0
package/bin/cli.mjs
ADDED
|
@@ -0,0 +1,518 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// bin/cli.mjs - ScraperCity CLI
|
|
3
|
+
import * as sc from '../lib/client.mjs'
|
|
4
|
+
import fs from 'fs'
|
|
5
|
+
import readline from 'readline'
|
|
6
|
+
|
|
7
|
+
const [,, cmd, ...args] = process.argv
|
|
8
|
+
const flag = (name) => { const i = args.indexOf(name); return i !== -1 ? (args.splice(i, 2), args[i] || true) : undefined }
|
|
9
|
+
const flagBool = (name) => { const i = args.indexOf(name); if (i !== -1) { args.splice(i, 1); return true } return false }
|
|
10
|
+
const json = (d) => JSON.stringify(d, null, 2)
|
|
11
|
+
|
|
12
|
+
function die(msg) { console.error(`✗ ${msg}`); process.exit(1) }
|
|
13
|
+
|
|
14
|
+
async function main() {
|
|
15
|
+
try {
|
|
16
|
+
switch (cmd) {
|
|
17
|
+
|
|
18
|
+
// ── Auth ──────────────────────────────────────────────
|
|
19
|
+
case 'login': {
|
|
20
|
+
const key = args[0] || await ask('API key: ')
|
|
21
|
+
sc.saveApiKey(key.trim())
|
|
22
|
+
console.log('✓ Saved to ~/.scrapercityrc')
|
|
23
|
+
break
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
// ── Wallet ────────────────────────────────────────────
|
|
27
|
+
case 'wallet': {
|
|
28
|
+
const w = await sc.wallet()
|
|
29
|
+
console.log(`Plan: ${w.plan.name} ($${w.plan.limit_dollars}/cycle)`)
|
|
30
|
+
console.log(`Balance: $${w.wallet.total_balance_dollars} (subscription: $${w.wallet.balance_dollars} + purchased: $${w.wallet.purchased_credits_dollars})`)
|
|
31
|
+
console.log(`Used: $${w.plan.used_dollars} (${w.plan.usage_percent}%)`)
|
|
32
|
+
if (w.billing.next_billing_date) console.log(`Renews: ${w.billing.next_billing_date}`)
|
|
33
|
+
break
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
// ── Runs ──────────────────────────────────────────────
|
|
37
|
+
case 'runs': {
|
|
38
|
+
const hours = flag('--hours') || 24
|
|
39
|
+
const limit = flag('--limit') || 20
|
|
40
|
+
const data = await sc.runs(hours, limit)
|
|
41
|
+
if (!data.runs?.length) { console.log('No runs found.'); break }
|
|
42
|
+
for (const r of data.runs) {
|
|
43
|
+
const cost = r.handled && r.price_per_lead_micro ? `$${((r.handled * r.price_per_lead_micro) / 1e6).toFixed(2)}` : ''
|
|
44
|
+
console.log(`${r.status.padEnd(10)} ${r.run_id.substring(0, 12).padEnd(14)} ${String(r.handled || 0).padEnd(6)} leads ${cost.padEnd(8)} ${r.file_name || r.url || ''}`)
|
|
45
|
+
}
|
|
46
|
+
break
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
// ── Status ────────────────────────────────────────────
|
|
50
|
+
case 'status': {
|
|
51
|
+
if (!args[0]) die('Usage: scrapercity status <runId>')
|
|
52
|
+
const s = await sc.status(args[0])
|
|
53
|
+
console.log(`Status: ${s.status}`)
|
|
54
|
+
console.log(`Message: ${s.statusMessage}`)
|
|
55
|
+
console.log(`Handled: ${s.handled} / ${s.requested}`)
|
|
56
|
+
if (s.outputUrl) console.log(`Download: scrapercity download ${args[0]}`)
|
|
57
|
+
break
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// ── Download ──────────────────────────────────────────
|
|
61
|
+
case 'download': {
|
|
62
|
+
if (!args[0]) die('Usage: scrapercity download <runId> [--output file.csv]')
|
|
63
|
+
const out = flag('--output') || flag('-o')
|
|
64
|
+
const result = await sc.download(args[0], out)
|
|
65
|
+
console.log(`✓ ${result.path} (${(result.bytes / 1024).toFixed(1)} KB)`)
|
|
66
|
+
break
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
// ── Cancel ────────────────────────────────────────────
|
|
70
|
+
case 'cancel': {
|
|
71
|
+
if (!args[0]) die('Usage: scrapercity cancel <runId>')
|
|
72
|
+
const r = await sc.cancel(args[0])
|
|
73
|
+
console.log(`✓ ${r.message}`)
|
|
74
|
+
break
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
// ── Logs ──────────────────────────────────────────────
|
|
78
|
+
case 'logs': {
|
|
79
|
+
if (!args[0]) die('Usage: scrapercity logs <runId>')
|
|
80
|
+
const r = await sc.logs(args[0])
|
|
81
|
+
console.log(r.log || r.logs || json(r))
|
|
82
|
+
break
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
// ── Apollo (URL-based) ────────────────────────────────
|
|
86
|
+
case 'apollo': {
|
|
87
|
+
const url = args[0]
|
|
88
|
+
if (!url) die('Usage: scrapercity apollo <apollo-url> [--count 1000] [--name "My Export"]')
|
|
89
|
+
const count = flag('--count') || 1000
|
|
90
|
+
const name = flag('--name') || ''
|
|
91
|
+
const r = await sc.apollo(url, +count, name)
|
|
92
|
+
console.log(`✓ Apollo scrape started`)
|
|
93
|
+
console.log(` Run ID: ${r.runId}`)
|
|
94
|
+
console.log(` ${r.message || ''}`)
|
|
95
|
+
console.log(` Note: Apollo takes up to 4 days. Set up a webhook at app.scrapercity.com/dashboard/webhooks to get notified.`)
|
|
96
|
+
console.log(` Check: scrapercity status ${r.runId}`)
|
|
97
|
+
break
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
// ── Apollo Filters ────────────────────────────────────
|
|
101
|
+
case 'apollo-filters': {
|
|
102
|
+
const filters = {}
|
|
103
|
+
for (const f of ['--seniority', '--function', '--industry', '--country', '--state',
|
|
104
|
+
'--company-country', '--company-state', '--company-size', '--count', '--name', '--has-phone']) {
|
|
105
|
+
const v = flag(f)
|
|
106
|
+
if (v === undefined) continue
|
|
107
|
+
const key = {
|
|
108
|
+
'--seniority': 'seniorityLevel', '--function': 'functionDept',
|
|
109
|
+
'--industry': 'companyIndustry', '--country': 'personCountry',
|
|
110
|
+
'--state': 'personState', '--company-country': 'companyCountry',
|
|
111
|
+
'--company-state': 'companyState', '--company-size': 'companySize',
|
|
112
|
+
'--count': 'count', '--name': 'fileName', '--has-phone': 'hasPhone'
|
|
113
|
+
}[f]
|
|
114
|
+
filters[key] = f === '--has-phone' ? true : (f === '--count' ? +v : v)
|
|
115
|
+
}
|
|
116
|
+
// Array flags
|
|
117
|
+
for (const f of ['--titles', '--domains', '--keywords', '--person-cities', '--company-cities']) {
|
|
118
|
+
const v = flag(f)
|
|
119
|
+
if (!v) continue
|
|
120
|
+
const key = { '--titles': 'personTitles', '--domains': 'companyDomains',
|
|
121
|
+
'--keywords': 'companyKeywords', '--person-cities': 'personCities',
|
|
122
|
+
'--company-cities': 'companyCities' }[f]
|
|
123
|
+
filters[key] = v.split(',').map(s => s.trim())
|
|
124
|
+
}
|
|
125
|
+
if (!Object.keys(filters).length) die('At least one filter required. Example: scrapercity apollo-filters --industry "computer software" --country "United States" --count 1000')
|
|
126
|
+
const r = await sc.apolloFilters(filters)
|
|
127
|
+
console.log(`✓ Apollo filter search started`)
|
|
128
|
+
console.log(` Run ID: ${r.runId}`)
|
|
129
|
+
console.log(` ${r.message || ''}`)
|
|
130
|
+
console.log(` Note: Apollo takes up to 4 days. Set up a webhook to get notified.`)
|
|
131
|
+
break
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
// ── Maps ──────────────────────────────────────────────
|
|
135
|
+
case 'maps': {
|
|
136
|
+
const query = flag('--query') || flag('-q')
|
|
137
|
+
const location = flag('--location') || flag('-l')
|
|
138
|
+
const limit = flag('--limit') || 500
|
|
139
|
+
if (!query || !location) die('Usage: scrapercity maps --query "plumbers" --location "Denver, CO" [--limit 500]')
|
|
140
|
+
const r = await sc.maps(query, location, +limit)
|
|
141
|
+
console.log(`✓ Maps scrape started`)
|
|
142
|
+
console.log(` Run ID: ${r.runId}`)
|
|
143
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
144
|
+
break
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
// ── Email Validate ────────────────────────────────────
|
|
148
|
+
case 'email-validate': {
|
|
149
|
+
let emails = args.filter(a => a.includes('@'))
|
|
150
|
+
const file = flag('--file')
|
|
151
|
+
if (file) {
|
|
152
|
+
const content = fs.readFileSync(file, 'utf-8')
|
|
153
|
+
emails = [...emails, ...content.split('\n').map(l => l.trim()).filter(l => l.includes('@'))]
|
|
154
|
+
}
|
|
155
|
+
if (!emails.length) die('Usage: scrapercity email-validate user@example.com ... OR --file emails.txt')
|
|
156
|
+
const r = await sc.emailValidate(emails)
|
|
157
|
+
console.log(`✓ Validating ${r.emailCount || emails.length} emails`)
|
|
158
|
+
console.log(` Run ID: ${r.runId}`)
|
|
159
|
+
console.log(` Est. cost: $${r.estimatedCost?.toFixed(2) || 'N/A'}`)
|
|
160
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
161
|
+
break
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
// ── Email Find ────────────────────────────────────────
|
|
165
|
+
case 'email-find': {
|
|
166
|
+
const first = flag('--first')
|
|
167
|
+
const last = flag('--last')
|
|
168
|
+
const fullName = flag('--full-name')
|
|
169
|
+
const domain = flag('--domain')
|
|
170
|
+
const company = flag('--company')
|
|
171
|
+
const file = flag('--file')
|
|
172
|
+
const validate = flagBool('--validate')
|
|
173
|
+
const mobiles = flagBool('--mobiles')
|
|
174
|
+
|
|
175
|
+
let contacts = []
|
|
176
|
+
if (file) {
|
|
177
|
+
// Expect CSV: first_name,last_name,domain
|
|
178
|
+
const lines = fs.readFileSync(file, 'utf-8').split('\n').filter(Boolean)
|
|
179
|
+
const header = lines[0].toLowerCase()
|
|
180
|
+
if (header.includes('first')) {
|
|
181
|
+
for (const line of lines.slice(1)) {
|
|
182
|
+
const [fn, ln, d, cn] = line.split(',').map(s => s.trim())
|
|
183
|
+
contacts.push({ first_name: fn, last_name: ln, domain: d, company_name: cn || '' })
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
} else if (first || fullName) {
|
|
187
|
+
contacts.push({ first_name: first || '', last_name: last || '', full_name: fullName || '', domain: domain || '', company_name: company || '' })
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
if (!contacts.length) die('Usage: scrapercity email-find --first John --last Doe --domain acme.com OR --file contacts.csv')
|
|
191
|
+
const r = await sc.emailFind(contacts, { validate, mobiles })
|
|
192
|
+
console.log(`✓ Finding emails for ${r.contactCount || contacts.length} contacts`)
|
|
193
|
+
console.log(` Run ID: ${r.runId}`)
|
|
194
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
195
|
+
break
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
// ── Mobile Finder ─────────────────────────────────────
|
|
199
|
+
case 'mobile-find': {
|
|
200
|
+
let inputs = args.filter(a => !a.startsWith('-'))
|
|
201
|
+
const file = flag('--file')
|
|
202
|
+
if (file) {
|
|
203
|
+
inputs = [...inputs, ...fs.readFileSync(file, 'utf-8').split('\n').map(l => l.trim()).filter(Boolean)]
|
|
204
|
+
}
|
|
205
|
+
if (!inputs.length) die('Usage: scrapercity mobile-find linkedin.com/in/johndoe user@company.com OR --file inputs.txt')
|
|
206
|
+
const r = await sc.mobileFinder(inputs)
|
|
207
|
+
console.log(`✓ Finding mobiles for ${r.inputCount || inputs.length} inputs`)
|
|
208
|
+
console.log(` Run ID: ${r.runId}`)
|
|
209
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
210
|
+
break
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
// ── People Finder / Skip Trace ────────────────────────
|
|
214
|
+
case 'people-find': {
|
|
215
|
+
const name = flag('--name')
|
|
216
|
+
const email = flag('--email')
|
|
217
|
+
const phone = flag('--phone')
|
|
218
|
+
const address = flag('--address')
|
|
219
|
+
const maxResults = flag('--max-results') || 1
|
|
220
|
+
const params = { max_results: +maxResults }
|
|
221
|
+
if (name) params.name = name.includes(',') ? name.split(',').map(s => s.trim()) : [name]
|
|
222
|
+
if (email) params.email = email.includes(',') ? email.split(',').map(s => s.trim()) : [email]
|
|
223
|
+
if (phone) params.phone_number = phone.includes(',') ? phone.split(',').map(s => s.trim()) : [phone]
|
|
224
|
+
if (address) params.street_citystatezip = address.includes('|') ? address.split('|').map(s => s.trim()) : [address]
|
|
225
|
+
if (!name && !email && !phone && !address) die('Usage: scrapercity people-find --name "John Doe" [--email a@b.com] [--address "123 Main St, City, ST 12345"]')
|
|
226
|
+
const r = await sc.peopleFinder(params)
|
|
227
|
+
console.log(`✓ People finder started`)
|
|
228
|
+
console.log(` Run ID: ${r.runId}`)
|
|
229
|
+
console.log(` Est. cost: $${r.estimatedCost?.toFixed(2) || 'N/A'}`)
|
|
230
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
231
|
+
break
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
// ── Store Leads ───────────────────────────────────────
|
|
235
|
+
case 'store-leads': {
|
|
236
|
+
const params = {}
|
|
237
|
+
for (const f of ['--platform', '--country', '--category', '--city', '--technologies', '--apps', '--total-leads', '--name']) {
|
|
238
|
+
const v = flag(f)
|
|
239
|
+
if (v === undefined) continue
|
|
240
|
+
const key = { '--platform': 'platform', '--country': 'countryCode', '--category': 'category',
|
|
241
|
+
'--city': 'city', '--technologies': 'technologies', '--apps': 'apps',
|
|
242
|
+
'--total-leads': 'totalLeads', '--name': 'fileName' }[f]
|
|
243
|
+
params[key] = f === '--total-leads' ? +v : v
|
|
244
|
+
}
|
|
245
|
+
for (const f of ['--emails', '--phones', '--instagram', '--facebook', '--tiktok', '--youtube', '--linkedin']) {
|
|
246
|
+
if (flagBool(f)) params[f.slice(2)] = true
|
|
247
|
+
}
|
|
248
|
+
if (!params.totalLeads) params.totalLeads = 1000
|
|
249
|
+
const r = await sc.storeLeads(params)
|
|
250
|
+
console.log(`✓ Store leads started - ${r.totalLeads || 'N/A'} leads`)
|
|
251
|
+
console.log(` Run ID: ${r.runId}`)
|
|
252
|
+
console.log(` Est. cost: $${r.estimatedCost?.toFixed(2) || 'N/A'}`)
|
|
253
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
254
|
+
break
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
// ── BuiltWith ─────────────────────────────────────────
|
|
258
|
+
case 'builtwith': {
|
|
259
|
+
const tech = args.filter(a => !a.startsWith('-')).join(' ') || flag('--tech')
|
|
260
|
+
if (!tech) die('Usage: scrapercity builtwith "Shopify"')
|
|
261
|
+
const name = flag('--name')
|
|
262
|
+
const r = await sc.builtwith(tech, name)
|
|
263
|
+
console.log(`✓ BuiltWith search started for "${r.technology}"`)
|
|
264
|
+
console.log(` Run ID: ${r.runId}`)
|
|
265
|
+
console.log(` Est. cost: $${r.estimatedCost?.toFixed(2) || 'N/A'}`)
|
|
266
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
267
|
+
break
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
// ── Criminal Records ──────────────────────────────────
|
|
271
|
+
case 'criminal': {
|
|
272
|
+
const name = flag('--name') || args.filter(a => !a.startsWith('-')).join(' ')
|
|
273
|
+
const state = flag('--state')
|
|
274
|
+
const dob = flag('--dob')
|
|
275
|
+
if (!name) die('Usage: scrapercity criminal --name "John Smith" [--state CA] [--dob 02/07/1992]')
|
|
276
|
+
const r = await sc.criminal(name, state, dob)
|
|
277
|
+
console.log(`✓ Criminal search queued: "${r.name}"`)
|
|
278
|
+
console.log(` Run ID: ${r.runId}`)
|
|
279
|
+
console.log(` State: ${r.state || 'all'}`)
|
|
280
|
+
console.log(` ${r.note || ''}`)
|
|
281
|
+
break
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
// ── Airbnb Email ──────────────────────────────────────
|
|
285
|
+
case 'airbnb': {
|
|
286
|
+
const city = flag('--city')
|
|
287
|
+
const listingUrl = flag('--url')
|
|
288
|
+
const maxResults = flag('--max-results') || flag('--limit') || 100
|
|
289
|
+
const mode = listingUrl ? 'single' : 'city'
|
|
290
|
+
const params = { mode, maxResults: +maxResults }
|
|
291
|
+
if (city) params.city = city.split(',').map(s => s.trim())
|
|
292
|
+
if (listingUrl) params.listingUrl = listingUrl
|
|
293
|
+
const checkin = flag('--checkin'); if (checkin) params.checkin = checkin
|
|
294
|
+
const checkout = flag('--checkout'); if (checkout) params.checkout = checkout
|
|
295
|
+
if (!city && !listingUrl) die('Usage: scrapercity airbnb --city "Miami, FL" --limit 100 OR --url <airbnb-listing-url>')
|
|
296
|
+
const r = await sc.airbnb(params)
|
|
297
|
+
console.log(`✓ Airbnb scrape started`)
|
|
298
|
+
console.log(` Run ID: ${r.runId}`)
|
|
299
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
300
|
+
break
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
// ── YouTube Email ─────────────────────────────────────
|
|
304
|
+
case 'youtube-email': {
|
|
305
|
+
let channels = args.filter(a => !a.startsWith('-'))
|
|
306
|
+
const file = flag('--file')
|
|
307
|
+
if (file) {
|
|
308
|
+
channels = [...channels, ...fs.readFileSync(file, 'utf-8').split('\n').map(l => l.trim()).filter(Boolean)]
|
|
309
|
+
}
|
|
310
|
+
if (!channels.length) die('Usage: scrapercity youtube-email @ChannelHandle https://youtube.com/@Channel OR --file channels.txt')
|
|
311
|
+
const r = await sc.youtubeEmail(channels)
|
|
312
|
+
console.log(`✓ YouTube email scrape started`)
|
|
313
|
+
console.log(` Run ID: ${r.runId}`)
|
|
314
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
315
|
+
break
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
// ── Website Finder ────────────────────────────────────
|
|
319
|
+
case 'website-finder': {
|
|
320
|
+
let domains = args.filter(a => !a.startsWith('-'))
|
|
321
|
+
const file = flag('--file')
|
|
322
|
+
const jobTitle = flag('--title')
|
|
323
|
+
if (file) {
|
|
324
|
+
domains = [...domains, ...fs.readFileSync(file, 'utf-8').split('\n').map(l => l.trim()).filter(Boolean)]
|
|
325
|
+
}
|
|
326
|
+
if (!domains.length) die('Usage: scrapercity website-finder acme.com example.com OR --file domains.txt [--title "CEO"]')
|
|
327
|
+
const r = await sc.websiteFinder(domains, jobTitle)
|
|
328
|
+
console.log(`✓ Website finder started for ${domains.length} domains`)
|
|
329
|
+
console.log(` Run ID: ${r.runId}`)
|
|
330
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
331
|
+
break
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
// ── Yelp ──────────────────────────────────────────────
|
|
335
|
+
case 'yelp': {
|
|
336
|
+
const searchTerms = flag('--query') || flag('-q')
|
|
337
|
+
const locations = flag('--location') || flag('-l')
|
|
338
|
+
const directUrls = flag('--urls')
|
|
339
|
+
const limit = flag('--limit') || 10
|
|
340
|
+
const params = { searchLimit: +limit }
|
|
341
|
+
if (searchTerms) params.searchTerms = searchTerms.split(',').map(s => s.trim())
|
|
342
|
+
if (locations) params.locations = locations.split(',').map(s => s.trim())
|
|
343
|
+
if (directUrls) params.directUrls = directUrls.split(',').map(s => s.trim())
|
|
344
|
+
if (!searchTerms && !directUrls) die('Usage: scrapercity yelp -q "plumbers" -l "Denver, CO" [--limit 10] OR --urls <yelp-urls>')
|
|
345
|
+
const r = await sc.yelp(params)
|
|
346
|
+
console.log(`✓ Yelp scrape started`)
|
|
347
|
+
console.log(` Run ID: ${r.runId}`)
|
|
348
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
349
|
+
break
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
// ── Angi (Angie's List) ───────────────────────────────
|
|
353
|
+
case 'angi': {
|
|
354
|
+
const keyword = flag('--query') || flag('-q')
|
|
355
|
+
const zipCodes = flag('--zips')
|
|
356
|
+
const maxItems = flag('--limit') || 100
|
|
357
|
+
if (!keyword) die('Usage: scrapercity angi -q "plumbers" --zips "80202,80203" [--limit 100]')
|
|
358
|
+
const zips = zipCodes ? zipCodes.split(',').map(s => s.trim()) : []
|
|
359
|
+
const r = await sc.angi(keyword, zips, +maxItems)
|
|
360
|
+
console.log(`✓ Angi scrape started`)
|
|
361
|
+
console.log(` Run ID: ${r.runId}`)
|
|
362
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
363
|
+
break
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
// ── Zillow Agents ─────────────────────────────────────
|
|
367
|
+
case 'zillow-agents': {
|
|
368
|
+
const location = flag('--location') || flag('-l')
|
|
369
|
+
const specialty = flag('--specialty')
|
|
370
|
+
const language = flag('--language')
|
|
371
|
+
const limit = flag('--limit') || 10
|
|
372
|
+
if (!location) die('Usage: scrapercity zillow-agents -l "Denver, CO" [--specialty "buyer" ] [--limit 10]')
|
|
373
|
+
const r = await sc.zillowAgents(location, { specialty, language, searchLimit: +limit })
|
|
374
|
+
console.log(`✓ Zillow agents scrape started`)
|
|
375
|
+
console.log(` Run ID: ${r.runId}`)
|
|
376
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
377
|
+
break
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
// ── BizBuySell ────────────────────────────────────────
|
|
381
|
+
case 'bizbuysell': {
|
|
382
|
+
const urls = args.filter(a => !a.startsWith('-'))
|
|
383
|
+
const limit = flag('--limit') || 100
|
|
384
|
+
if (!urls.length) die('Usage: scrapercity bizbuysell <bizbuysell-search-url> [--limit 100]')
|
|
385
|
+
const r = await sc.bizbuysell(urls, +limit)
|
|
386
|
+
console.log(`✓ BizBuySell scrape started`)
|
|
387
|
+
console.log(` Run ID: ${r.runId}`)
|
|
388
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
389
|
+
break
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
// ── Crexi ─────────────────────────────────────────────
|
|
393
|
+
case 'crexi': {
|
|
394
|
+
const urls = args.filter(a => !a.startsWith('-'))
|
|
395
|
+
if (!urls.length) die('Usage: scrapercity crexi <crexi-search-url>')
|
|
396
|
+
const r = await sc.crexi(urls)
|
|
397
|
+
console.log(`✓ Crexi scrape started`)
|
|
398
|
+
console.log(` Run ID: ${r.runId}`)
|
|
399
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
400
|
+
break
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
// ── Property Lookup ───────────────────────────────────
|
|
404
|
+
case 'property-lookup': {
|
|
405
|
+
let addresses = args.filter(a => !a.startsWith('-'))
|
|
406
|
+
const file = flag('--file')
|
|
407
|
+
const ownerContact = flagBool('--owner-contact')
|
|
408
|
+
if (file) {
|
|
409
|
+
addresses = [...addresses, ...fs.readFileSync(file, 'utf-8').split('\n').map(l => l.trim()).filter(Boolean)]
|
|
410
|
+
}
|
|
411
|
+
if (!addresses.length) die('Usage: scrapercity property-lookup "123 Main St, Denver, CO 80202" [--owner-contact] OR --file addresses.txt')
|
|
412
|
+
const r = await sc.propertyLookup(addresses, ownerContact)
|
|
413
|
+
console.log(`✓ Property lookup started for ${addresses.length} addresses`)
|
|
414
|
+
console.log(` Run ID: ${r.runId}`)
|
|
415
|
+
console.log(` Poll: scrapercity status ${r.runId}`)
|
|
416
|
+
break
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
// ── Database: Leads ($649) ────────────────────────────
|
|
420
|
+
case 'db-leads': {
|
|
421
|
+
const params = {}
|
|
422
|
+
for (const f of ['--title', '--industry', '--country', '--state', '--city',
|
|
423
|
+
'--company', '--domain', '--company-size', '--seniority',
|
|
424
|
+
'--department', '--page', '--limit', '--min-employees', '--max-employees']) {
|
|
425
|
+
const v = flag(f)
|
|
426
|
+
if (v === undefined) continue
|
|
427
|
+
const key = { '--title': 'title', '--industry': 'industry', '--country': 'country',
|
|
428
|
+
'--state': 'state', '--city': 'city', '--company': 'companyName',
|
|
429
|
+
'--domain': 'companyDomain', '--company-size': 'companySize',
|
|
430
|
+
'--seniority': 'seniority', '--department': 'department',
|
|
431
|
+
'--page': 'page', '--limit': 'limit',
|
|
432
|
+
'--min-employees': 'minEmployees', '--max-employees': 'maxEmployees' }[f]
|
|
433
|
+
params[key] = v
|
|
434
|
+
}
|
|
435
|
+
if (flagBool('--has-email')) params.hasEmail = 'true'
|
|
436
|
+
if (flagBool('--has-phone')) params.hasPhone = 'true'
|
|
437
|
+
const r = await sc.dbLeads(params)
|
|
438
|
+
console.log(`${r.pagination?.total || '?'} total leads, page ${r.pagination?.page || 1} of ${r.pagination?.totalPages || '?'}`)
|
|
439
|
+
console.log(json(r.data?.slice(0, 3) || r))
|
|
440
|
+
if (r.data?.length > 3) console.log(`... and ${r.data.length - 3} more`)
|
|
441
|
+
break
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
// ── Poll (convenience) ────────────────────────────────
|
|
445
|
+
case 'poll': {
|
|
446
|
+
if (!args[0]) die('Usage: scrapercity poll <runId> [--interval 15]')
|
|
447
|
+
const interval = +(flag('--interval') || 15) * 1000
|
|
448
|
+
console.log(`Polling ${args[0]} every ${interval / 1000}s...`)
|
|
449
|
+
const result = await sc.pollUntilDone(args[0], {
|
|
450
|
+
interval,
|
|
451
|
+
onStatus: (s) => process.stdout.write(`\r ${s.status} - ${s.handled}/${s.requested} leads`)
|
|
452
|
+
})
|
|
453
|
+
console.log(`\n✓ Done: ${result.handled} leads`)
|
|
454
|
+
console.log(` Download: scrapercity download ${args[0]}`)
|
|
455
|
+
break
|
|
456
|
+
}
|
|
457
|
+
|
|
458
|
+
// ── Help ──────────────────────────────────────────────
|
|
459
|
+
case 'help': case '--help': case '-h': case undefined: {
|
|
460
|
+
console.log(`
|
|
461
|
+
ScraperCity CLI - B2B lead generation from your terminal
|
|
462
|
+
|
|
463
|
+
Auth:
|
|
464
|
+
scrapercity login [key] Save API key to ~/.scrapercityrc
|
|
465
|
+
scrapercity wallet Check balance and plan
|
|
466
|
+
|
|
467
|
+
Scrapers:
|
|
468
|
+
scrapercity apollo <url> [--count N] Apollo scrape (URL-based, ~4 day delivery)
|
|
469
|
+
scrapercity apollo-filters [filters] Apollo scrape (filter-based)
|
|
470
|
+
scrapercity maps -q <query> -l <loc> Google Maps scrape
|
|
471
|
+
scrapercity email-validate <emails> Validate email addresses
|
|
472
|
+
scrapercity email-find --first X --last Y --domain Z Find business emails
|
|
473
|
+
scrapercity mobile-find <linkedin/email> Find mobile numbers
|
|
474
|
+
scrapercity people-find --name "X" Skip trace / people finder
|
|
475
|
+
scrapercity store-leads [--platform X] Ecommerce store leads
|
|
476
|
+
scrapercity builtwith "Technology" Sites using a technology
|
|
477
|
+
scrapercity criminal --name "X" Criminal records search
|
|
478
|
+
scrapercity airbnb --city "Miami, FL" Airbnb host emails
|
|
479
|
+
scrapercity youtube-email @Channel YouTuber business emails
|
|
480
|
+
scrapercity website-finder acme.com Contact info from domains
|
|
481
|
+
scrapercity yelp -q "query" -l "City" Yelp business scraper
|
|
482
|
+
scrapercity angi -q "query" --zips X Angi (Angie's List) scraper
|
|
483
|
+
scrapercity zillow-agents -l "City" Zillow real estate agents
|
|
484
|
+
scrapercity bizbuysell <url> BizBuySell listings
|
|
485
|
+
scrapercity crexi <url> Crexi commercial real estate
|
|
486
|
+
scrapercity property-lookup "address" Property data + owner contact
|
|
487
|
+
|
|
488
|
+
Status:
|
|
489
|
+
scrapercity status <runId> Check run status
|
|
490
|
+
scrapercity poll <runId> Poll until complete
|
|
491
|
+
scrapercity download <runId> [-o f.csv] Download results CSV
|
|
492
|
+
scrapercity cancel <runId> Cancel a running scrape
|
|
493
|
+
scrapercity logs <runId> View run logs
|
|
494
|
+
scrapercity runs [--hours 24] List recent runs
|
|
495
|
+
|
|
496
|
+
Database ($649 plan):
|
|
497
|
+
scrapercity db-leads [filters] Query lead database
|
|
498
|
+
|
|
499
|
+
Env: SCRAPERCITY_API_KEY=... or scrapercity login
|
|
500
|
+
Docs: https://scrapercity.com/agents
|
|
501
|
+
`)
|
|
502
|
+
break
|
|
503
|
+
}
|
|
504
|
+
|
|
505
|
+
default:
|
|
506
|
+
die(`Unknown command: ${cmd}. Run: scrapercity help`)
|
|
507
|
+
}
|
|
508
|
+
} catch (e) {
|
|
509
|
+
die(e.message)
|
|
510
|
+
}
|
|
511
|
+
}
|
|
512
|
+
|
|
513
|
+
function ask(prompt) {
|
|
514
|
+
const rl = readline.createInterface({ input: process.stdin, output: process.stderr })
|
|
515
|
+
return new Promise(resolve => rl.question(prompt, (a) => { rl.close(); resolve(a) }))
|
|
516
|
+
}
|
|
517
|
+
|
|
518
|
+
main()
|