@datafuel/sdk 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +192 -0
- package/dist/index.cjs +922 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.cts +554 -0
- package/dist/index.d.ts +554 -0
- package/dist/index.js +873 -0
- package/dist/index.js.map +1 -0
- package/package.json +59 -0
- package/src/client.ts +525 -0
- package/src/core.ts +388 -0
- package/src/errors.ts +136 -0
- package/src/index.ts +69 -0
- package/src/models.ts +362 -0
package/dist/index.d.cts
ADDED
|
@@ -0,0 +1,554 @@
|
|
|
1
|
+
/** Request options and response shapes. */
|
|
2
|
+
/** State of a task, job or crawl. Unknown states are treated as not final. */
|
|
3
|
+
type Status = "created" | "pending" | "processing" | "completed" | "completed_with_errors" | "failed" | "cancelled";
|
|
4
|
+
/** Whether a state is final. */
|
|
5
|
+
declare function isDone(status: string | undefined): boolean;
|
|
6
|
+
/**
|
|
7
|
+
* Shape of the scraped content.
|
|
8
|
+
*
|
|
9
|
+
* `html` raw page (API default), `markdown` cleaned text (best for LLMs),
|
|
10
|
+
* `json` schema.org / JSON-LD, `png` / `jpeg` full-page screenshot.
|
|
11
|
+
*/
|
|
12
|
+
type Format = "html" | "markdown" | "json" | "png" | "jpeg";
|
|
13
|
+
/** AI assistant `ask` can query. Engines can be switched off at runtime. */
|
|
14
|
+
type Engine = "openai" | "gemini" | "google_ai_mode" | "perplexity" | "copilot";
|
|
15
|
+
/** Proxy plan. Matched case-insensitively; deployments may define others. */
|
|
16
|
+
type ProxyType = "Basic" | "Premium" | (string & {});
|
|
17
|
+
/** The exit the request leaves from. Every field optional: the account default. */
|
|
18
|
+
interface Proxy {
|
|
19
|
+
type?: ProxyType;
|
|
20
|
+
/** ISO 3166-1 alpha-2, e.g. "US". */
|
|
21
|
+
country?: string;
|
|
22
|
+
city?: string;
|
|
23
|
+
state?: string;
|
|
24
|
+
asn?: string;
|
|
25
|
+
/**
|
|
26
|
+
* Sticky session: the same exit across requests, `ttl` in seconds. Read by
|
|
27
|
+
* `scrape` and `map` only, and sent in attributes rather than the envelope.
|
|
28
|
+
*/
|
|
29
|
+
sessionId?: string;
|
|
30
|
+
ttl?: number;
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* Post-process the page with an LLM, using your own key.
|
|
34
|
+
*
|
|
35
|
+
* Not supported on crawls: the API rejects `result_use_ai` there.
|
|
36
|
+
*/
|
|
37
|
+
interface AI {
|
|
38
|
+
/** What to extract. */
|
|
39
|
+
prompt?: string;
|
|
40
|
+
/** Example JSON object the output must follow. */
|
|
41
|
+
format?: unknown;
|
|
42
|
+
provider?: "openai" | "anthropic" | "google";
|
|
43
|
+
model?: string;
|
|
44
|
+
apiKey?: string;
|
|
45
|
+
}
|
|
46
|
+
/** Page options, shared by `scrape`, jobs and crawls. */
|
|
47
|
+
interface ScrapeOptions {
|
|
48
|
+
format?: Format;
|
|
49
|
+
/**
|
|
50
|
+
* Render the page in a real browser: slower, five times the credits on a
|
|
51
|
+
* Basic proxy. Leave it off unless the page comes back empty without it.
|
|
52
|
+
*/
|
|
53
|
+
jsRendering?: boolean;
|
|
54
|
+
waitFor?: string;
|
|
55
|
+
waitForTimeoutMs?: number;
|
|
56
|
+
jsInstructions?: unknown;
|
|
57
|
+
blockResource?: string;
|
|
58
|
+
/** Markdown only: always render just the `<main>` / `<article>` container. */
|
|
59
|
+
mainContentOnly?: boolean;
|
|
60
|
+
/** Markdown only: `false` drops images and saves tokens. */
|
|
61
|
+
includeImages?: boolean;
|
|
62
|
+
/**
|
|
63
|
+
* Extract only matching elements: field name to CSS selector. Append
|
|
64
|
+
* ` @attr` to read an attribute, e.g. `{ links: "a @href" }`.
|
|
65
|
+
*/
|
|
66
|
+
extract?: string | Record<string, string>;
|
|
67
|
+
/** The same with RE2 patterns on the raw HTML. */
|
|
68
|
+
extractRegex?: string | Record<string, string>;
|
|
69
|
+
/**
|
|
70
|
+
* Built-in extractors, comma separated: images, links, headings,
|
|
71
|
+
* phone_numbers, emails, meta_tags, tables, schema_org, all.
|
|
72
|
+
*/
|
|
73
|
+
template?: string;
|
|
74
|
+
method?: string;
|
|
75
|
+
body?: string;
|
|
76
|
+
contentType?: string;
|
|
77
|
+
headers?: Record<string, string>;
|
|
78
|
+
headerOrder?: string[];
|
|
79
|
+
cookies?: string;
|
|
80
|
+
userAgent?: string;
|
|
81
|
+
/** chrome, firefox, safari, edge. */
|
|
82
|
+
userAgentType?: string;
|
|
83
|
+
ai?: AI;
|
|
84
|
+
}
|
|
85
|
+
/** Options every call accepts. */
|
|
86
|
+
interface CallOptions {
|
|
87
|
+
proxy?: Proxy;
|
|
88
|
+
/** Generated per request when omitted. Max 255 characters. */
|
|
89
|
+
idempotencyKey?: string;
|
|
90
|
+
/** Overrides the client's timeout for this call. */
|
|
91
|
+
timeoutMs?: number;
|
|
92
|
+
signal?: AbortSignal;
|
|
93
|
+
}
|
|
94
|
+
/** The raw `result` object of a task. */
|
|
95
|
+
interface Payload {
|
|
96
|
+
data?: unknown;
|
|
97
|
+
status?: Status;
|
|
98
|
+
error_detail?: string;
|
|
99
|
+
status_code?: number;
|
|
100
|
+
}
|
|
101
|
+
/**
|
|
102
|
+
* One scraped target with its metadata.
|
|
103
|
+
*
|
|
104
|
+
* Read the metadata before the content: a 200 can be an error page,
|
|
105
|
+
* `redirected` means `finalUrl` is not what you asked for, and a 404 still
|
|
106
|
+
* completes and bills.
|
|
107
|
+
*/
|
|
108
|
+
declare class Result {
|
|
109
|
+
readonly id: string | undefined;
|
|
110
|
+
readonly status: Status | undefined;
|
|
111
|
+
/** HTTP status the target answered. */
|
|
112
|
+
readonly statusCode: number | undefined;
|
|
113
|
+
readonly finalUrl: string | undefined;
|
|
114
|
+
readonly redirected: boolean;
|
|
115
|
+
/** Charged for this task, 0 when it failed. */
|
|
116
|
+
readonly creditsUsed: number;
|
|
117
|
+
readonly durationMs: number | undefined;
|
|
118
|
+
readonly blocked: boolean;
|
|
119
|
+
/** Anti-bot vendor recognised when blocked. */
|
|
120
|
+
readonly protection: string | undefined;
|
|
121
|
+
readonly error: string | undefined;
|
|
122
|
+
/** The raw `result` object. Prefer `text`, `data` and `image`. */
|
|
123
|
+
readonly payload: Payload | undefined;
|
|
124
|
+
/** Everything the API sent, including fields this SDK does not know yet. */
|
|
125
|
+
readonly raw: Record<string, unknown>;
|
|
126
|
+
constructor(raw: Record<string, unknown>);
|
|
127
|
+
/** Effective state: the envelope's, else the payload's, else inferred. */
|
|
128
|
+
get state(): string;
|
|
129
|
+
/** Whether the task has not finished yet (job and crawl results list stubs). */
|
|
130
|
+
get pending(): boolean;
|
|
131
|
+
/** Whether the task completed and the target was not blocked. */
|
|
132
|
+
get ok(): boolean;
|
|
133
|
+
/** The structured output: format json, extract, template, AI. */
|
|
134
|
+
get data(): unknown;
|
|
135
|
+
/** Content of an html or markdown result; the raw JSON for structured ones. */
|
|
136
|
+
get text(): string;
|
|
137
|
+
/** Bytes of a png or jpeg screenshot. Throws when the result is not one. */
|
|
138
|
+
get image(): Uint8Array;
|
|
139
|
+
/** `data`, or throw when the task carried none. */
|
|
140
|
+
requireData(): unknown;
|
|
141
|
+
/** Throw TaskFailed, or Blocked, when this task failed. Refunded either way. */
|
|
142
|
+
raiseForStatus(): void;
|
|
143
|
+
}
|
|
144
|
+
/** One crawled page: where it was found, plus the scrape result. */
|
|
145
|
+
declare class CrawlPage extends Result {
|
|
146
|
+
readonly url: string;
|
|
147
|
+
readonly depth: number;
|
|
148
|
+
readonly taskId: string | undefined;
|
|
149
|
+
constructor(raw: Record<string, unknown>);
|
|
150
|
+
}
|
|
151
|
+
/** One discovered URL. */
|
|
152
|
+
interface Link {
|
|
153
|
+
url: string;
|
|
154
|
+
/** "sitemap" or "page". */
|
|
155
|
+
source: string;
|
|
156
|
+
/** Page links only. */
|
|
157
|
+
title?: string;
|
|
158
|
+
/** Sitemap links only. */
|
|
159
|
+
lastmod?: string;
|
|
160
|
+
}
|
|
161
|
+
/**
|
|
162
|
+
* What `map` found.
|
|
163
|
+
*
|
|
164
|
+
* `reason` says why `links` is empty: page_blocked, page_error,
|
|
165
|
+
* page_unreachable, no_links_on_page (try scrape with jsRendering), no_sitemap,
|
|
166
|
+
* sitemap_no_entries, search_no_match.
|
|
167
|
+
*/
|
|
168
|
+
interface SiteMap {
|
|
169
|
+
url: string;
|
|
170
|
+
links: Link[];
|
|
171
|
+
total: number;
|
|
172
|
+
/** `limit` or the time budget cut the list. */
|
|
173
|
+
truncated: boolean;
|
|
174
|
+
sitemaps: string[];
|
|
175
|
+
page_status_code: number;
|
|
176
|
+
reason?: string;
|
|
177
|
+
credits: number;
|
|
178
|
+
/** Envelope of the underlying task. */
|
|
179
|
+
task: Result;
|
|
180
|
+
}
|
|
181
|
+
/** Progress of a crawl. */
|
|
182
|
+
interface CrawlStatus {
|
|
183
|
+
status: Status;
|
|
184
|
+
/** Empty while the crawl still grows: max_pages, max_depth_exhausted, insufficient_credits, cancelled. */
|
|
185
|
+
stop_reason?: string;
|
|
186
|
+
pages: {
|
|
187
|
+
discovered: number;
|
|
188
|
+
enqueued: number;
|
|
189
|
+
done: number;
|
|
190
|
+
failed: number;
|
|
191
|
+
skipped: number;
|
|
192
|
+
};
|
|
193
|
+
depth_reached: number;
|
|
194
|
+
/**
|
|
195
|
+
* Charged at queue time; refunds of failed pages are not subtracted. Sum
|
|
196
|
+
* `creditsUsed` over the pages for the net figure.
|
|
197
|
+
*/
|
|
198
|
+
total_cost: number;
|
|
199
|
+
/** Whether the crawl reached a final state. */
|
|
200
|
+
done: boolean;
|
|
201
|
+
}
|
|
202
|
+
/** One page of crawl results. */
|
|
203
|
+
interface CrawlResultsPage {
|
|
204
|
+
pages: CrawlPage[];
|
|
205
|
+
/** Absent on the last page. */
|
|
206
|
+
nextCursor?: string;
|
|
207
|
+
}
|
|
208
|
+
/** A finished crawl: why it stopped and what it found. */
|
|
209
|
+
interface CrawlResult {
|
|
210
|
+
id: string;
|
|
211
|
+
status: CrawlStatus;
|
|
212
|
+
pages: CrawlPage[];
|
|
213
|
+
}
|
|
214
|
+
/** Progress of a job. */
|
|
215
|
+
interface JobStatus {
|
|
216
|
+
status: Status;
|
|
217
|
+
tasks_count: number;
|
|
218
|
+
tasks_done: number;
|
|
219
|
+
tasks_remaining: number;
|
|
220
|
+
total_cost: number;
|
|
221
|
+
done: boolean;
|
|
222
|
+
}
|
|
223
|
+
/** Every task of a job. A job completes even when some of its tasks failed. */
|
|
224
|
+
interface JobResults {
|
|
225
|
+
id: string;
|
|
226
|
+
tasks_count: number;
|
|
227
|
+
tasks_failed: number;
|
|
228
|
+
tasks_complete: number;
|
|
229
|
+
tasks: Result[];
|
|
230
|
+
}
|
|
231
|
+
/** The account behind the API key. */
|
|
232
|
+
interface Profile {
|
|
233
|
+
email: string;
|
|
234
|
+
username: string;
|
|
235
|
+
current_concurrency: number;
|
|
236
|
+
concurrency_limit: number;
|
|
237
|
+
credit_balance: number;
|
|
238
|
+
monthly_credit_limit: number;
|
|
239
|
+
}
|
|
240
|
+
/** One task type or LLM engine, and whether it accepts new work. */
|
|
241
|
+
interface Capability {
|
|
242
|
+
name: string;
|
|
243
|
+
enabled: boolean;
|
|
244
|
+
/** Operator's note when switched off. */
|
|
245
|
+
reason?: string;
|
|
246
|
+
}
|
|
247
|
+
/** What the API accepts right now. */
|
|
248
|
+
declare class Capabilities {
|
|
249
|
+
readonly modules: Capability[];
|
|
250
|
+
readonly engines: Capability[];
|
|
251
|
+
constructor(raw: Record<string, unknown>);
|
|
252
|
+
/** Whether a task type accepts work. Unknown names report false. */
|
|
253
|
+
moduleEnabled(name: string): boolean;
|
|
254
|
+
/** Whether an LLM engine accepts work. Unknown names report false. */
|
|
255
|
+
engineEnabled(name: Engine | string): boolean;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
/**
|
|
259
|
+
* Request building and answer classification. Pure functions, no I/O.
|
|
260
|
+
*
|
|
261
|
+
* Every decision lives here — which envelope a module needs, where the sticky
|
|
262
|
+
* session keys go, how `ai` expands, whether an answer is retryable — so the
|
|
263
|
+
* client only moves bytes.
|
|
264
|
+
*/
|
|
265
|
+
|
|
266
|
+
declare const DEFAULT_BASE_URL = "https://scraping-api.datafuel.ai/api/v1";
|
|
267
|
+
/** Kept in step with package.json by a test; see test/hardening.test.ts. */
|
|
268
|
+
declare const VERSION = "0.1.0";
|
|
269
|
+
/** Options for the llm_scraping module, which reads its proxy country here. */
|
|
270
|
+
interface AskOptions {
|
|
271
|
+
engine: Engine | string;
|
|
272
|
+
websearch?: boolean;
|
|
273
|
+
followUp?: string;
|
|
274
|
+
country?: string;
|
|
275
|
+
format?: string;
|
|
276
|
+
}
|
|
277
|
+
/** Options for `map`. */
|
|
278
|
+
interface MapOptions {
|
|
279
|
+
/** Keep only links whose URL or title contains this. */
|
|
280
|
+
search?: string;
|
|
281
|
+
/** Map only this sitemap, from a previous `SiteMap.sitemaps`. */
|
|
282
|
+
sitemap?: string;
|
|
283
|
+
/** Default 5000, capped at 10000. */
|
|
284
|
+
limit?: number;
|
|
285
|
+
includeSubdomains?: boolean;
|
|
286
|
+
ignoreSitemap?: boolean;
|
|
287
|
+
sitemapOnly?: boolean;
|
|
288
|
+
userAgent?: string;
|
|
289
|
+
userAgentType?: string;
|
|
290
|
+
}
|
|
291
|
+
/** Options for `crawl`, on top of the page options applied to every page. */
|
|
292
|
+
interface CrawlOptions extends ScrapeOptions {
|
|
293
|
+
/** Default 100, max 10000. The hard budget of the crawl. */
|
|
294
|
+
maxPages?: number;
|
|
295
|
+
/** Default 3, max 10. The start URL is depth 0. */
|
|
296
|
+
maxDepth?: number;
|
|
297
|
+
/** RE2 on path?query; when set only matches are followed. */
|
|
298
|
+
includePaths?: string[];
|
|
299
|
+
/** RE2 on path?query; exclude wins. */
|
|
300
|
+
excludePaths?: string[];
|
|
301
|
+
includeSubdomains?: boolean;
|
|
302
|
+
allowBackwardLinks?: boolean;
|
|
303
|
+
/** Pages in flight, default 5. */
|
|
304
|
+
concurrency?: number;
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
/** The client. It only moves bytes; `core` decides what goes on the wire. */
|
|
308
|
+
|
|
309
|
+
/** How to build the client. */
|
|
310
|
+
interface ClientOptions {
|
|
311
|
+
/** Falls back to `process.env.DATAFUEL_API_KEY`. */
|
|
312
|
+
apiKey?: string;
|
|
313
|
+
baseUrl?: string;
|
|
314
|
+
/** Milliseconds. `null` disables the bound; scrape blocks until the page is ready. */
|
|
315
|
+
timeoutMs?: number | null;
|
|
316
|
+
/** Retries after a network error or a 429/502/503/504 answer. Default 2. */
|
|
317
|
+
maxRetries?: number;
|
|
318
|
+
/** How often `waitJob` and `waitCrawl` poll. Default 2000ms. */
|
|
319
|
+
pollIntervalMs?: number;
|
|
320
|
+
/** Prefixes the SDK User-Agent with your application's. */
|
|
321
|
+
userAgent?: string;
|
|
322
|
+
/** Swap the fetch implementation, e.g. in tests. */
|
|
323
|
+
fetch?: typeof globalThis.fetch;
|
|
324
|
+
}
|
|
325
|
+
/**
|
|
326
|
+
* Client for the DataFuel scraping API.
|
|
327
|
+
*
|
|
328
|
+
* ```ts
|
|
329
|
+
* const df = new DataFuel(); // reads DATAFUEL_API_KEY
|
|
330
|
+
* const markdown = await df.markdown("https://example.com");
|
|
331
|
+
* ```
|
|
332
|
+
*
|
|
333
|
+
* Pick the call by the shape of the work: one URL is {@link scrape}, a site's
|
|
334
|
+
* URL list is {@link map}, many pages from a start URL is {@link crawl}, a list
|
|
335
|
+
* of known URLs is {@link runJob}, a question for an AI engine is {@link ask}.
|
|
336
|
+
*
|
|
337
|
+
* Every write carries an `Idempotency-Key`, generated per request, so a retry
|
|
338
|
+
* attaches to the task already running instead of charging twice.
|
|
339
|
+
*/
|
|
340
|
+
declare class DataFuel {
|
|
341
|
+
readonly baseUrl: string;
|
|
342
|
+
readonly timeoutMs: number | null;
|
|
343
|
+
readonly maxRetries: number;
|
|
344
|
+
readonly pollIntervalMs: number;
|
|
345
|
+
readonly userAgent: string;
|
|
346
|
+
private readonly apiKey;
|
|
347
|
+
private readonly fetchImpl;
|
|
348
|
+
constructor(options?: ClientOptions | string);
|
|
349
|
+
/** Pause between attempts. Overridable so tests need not wait in real time. */
|
|
350
|
+
protected sleep(ms: number): Promise<void>;
|
|
351
|
+
private send;
|
|
352
|
+
private signalFor;
|
|
353
|
+
/**
|
|
354
|
+
* Send a task and wait for its result.
|
|
355
|
+
*
|
|
356
|
+
* On a 202 the API is still working: the identical request goes out again
|
|
357
|
+
* under the identical idempotency key, so it attaches to the running task
|
|
358
|
+
* rather than starting a second one.
|
|
359
|
+
*/
|
|
360
|
+
private runTask;
|
|
361
|
+
/**
|
|
362
|
+
* Fetch one URL and wait for the result.
|
|
363
|
+
*
|
|
364
|
+
* Throws {@link Blocked} when the target refused the request and
|
|
365
|
+
* {@link TaskFailed} otherwise; both carry `.result`.
|
|
366
|
+
*/
|
|
367
|
+
scrape(url: string, options?: ScrapeOptions & CallOptions): Promise<Result>;
|
|
368
|
+
/** The one-liner: the page as LLM-ready markdown, images dropped. */
|
|
369
|
+
markdown(url: string, options?: ScrapeOptions & CallOptions): Promise<string>;
|
|
370
|
+
/**
|
|
371
|
+
* Return a task by id, e.g. one created by a job or a crawl.
|
|
372
|
+
*
|
|
373
|
+
* A task that is still running comes back with `pending` true.
|
|
374
|
+
*/
|
|
375
|
+
getTask(taskId: string, options?: CallOptions): Promise<Result>;
|
|
376
|
+
/** Send a prompt to an AI engine and return its answer. */
|
|
377
|
+
ask(prompt: string, options: AskOptions & CallOptions): Promise<Result>;
|
|
378
|
+
/**
|
|
379
|
+
* List the URLs of a site without scraping them.
|
|
380
|
+
*
|
|
381
|
+
* One credit per call on a Basic proxy, however many links come back. Use it
|
|
382
|
+
* before a crawl to see how big a section is.
|
|
383
|
+
*/
|
|
384
|
+
map(url: string, options?: MapOptions & CallOptions): Promise<SiteMap>;
|
|
385
|
+
/**
|
|
386
|
+
* Queue a crawl and return its id immediately.
|
|
387
|
+
*
|
|
388
|
+
* Follows links from the start URL and scrapes every page. Pages are charged
|
|
389
|
+
* like single scrapes when they are queued; failed and blocked pages are
|
|
390
|
+
* refunded. Unset limits use the API defaults: 100 pages, depth 3, 5 in flight.
|
|
391
|
+
*/
|
|
392
|
+
startCrawl(url: string, options?: CrawlOptions & CallOptions): Promise<string>;
|
|
393
|
+
/** Return the progress of a crawl. */
|
|
394
|
+
getCrawl(crawlId: string, options?: CallOptions): Promise<CrawlStatus>;
|
|
395
|
+
/** One page of results, in discovery order. `limit` unset uses the API default. */
|
|
396
|
+
crawlResults(crawlId: string, options?: CallOptions & {
|
|
397
|
+
cursor?: string;
|
|
398
|
+
limit?: number;
|
|
399
|
+
}): Promise<CrawlResultsPage>;
|
|
400
|
+
/**
|
|
401
|
+
* Walk every page of a crawl, fetching result pages as needed.
|
|
402
|
+
*
|
|
403
|
+
* Readable while the crawl runs; unfinished pages report `pending`.
|
|
404
|
+
*/
|
|
405
|
+
crawlPages(crawlId: string, options?: CallOptions & {
|
|
406
|
+
limit?: number;
|
|
407
|
+
}): AsyncGenerator<CrawlPage>;
|
|
408
|
+
/** Poll until the crawl is done. Without `timeoutMs` it waits indefinitely. */
|
|
409
|
+
waitCrawl(crawlId: string, options?: CallOptions): Promise<CrawlStatus>;
|
|
410
|
+
/**
|
|
411
|
+
* Start a crawl, wait for it, and return every page.
|
|
412
|
+
*
|
|
413
|
+
* Check `stop_reason`: `insufficient_credits` means it ended early. For large
|
|
414
|
+
* crawls prefer {@link startCrawl} + {@link waitCrawl} + {@link crawlPages},
|
|
415
|
+
* which stream instead of holding every page in memory. On {@link WaitTimeout}
|
|
416
|
+
* the error carries the id: the crawl keeps running and billing.
|
|
417
|
+
*/
|
|
418
|
+
crawl(url: string, options?: CrawlOptions & CallOptions): Promise<CrawlResult>;
|
|
419
|
+
/**
|
|
420
|
+
* Queue a batch of known URLs and return the job id.
|
|
421
|
+
*
|
|
422
|
+
* Cheaper and more predictable than a crawl when you already have the URLs.
|
|
423
|
+
* `sequential` runs them one after the other; the default runs them
|
|
424
|
+
* concurrently, bounded by the account's concurrency limit.
|
|
425
|
+
*/
|
|
426
|
+
createJob(urls: string[], options?: ScrapeOptions & CallOptions & {
|
|
427
|
+
sequential?: boolean;
|
|
428
|
+
}): Promise<string>;
|
|
429
|
+
/** Queue a batch of prompts for an AI engine and return the job id. */
|
|
430
|
+
createAskJob(prompts: string[], options: AskOptions & CallOptions & {
|
|
431
|
+
sequential?: boolean;
|
|
432
|
+
}): Promise<string>;
|
|
433
|
+
/** Return the progress of a job. */
|
|
434
|
+
getJob(jobId: string, options?: CallOptions): Promise<JobStatus>;
|
|
435
|
+
/**
|
|
436
|
+
* Return the per-task results of a job.
|
|
437
|
+
*
|
|
438
|
+
* A job completes even when some of its tasks failed: check `task.ok` or call
|
|
439
|
+
* `task.raiseForStatus()` per task.
|
|
440
|
+
*/
|
|
441
|
+
jobResults(jobId: string, options?: CallOptions): Promise<JobResults>;
|
|
442
|
+
/** Poll until the job is done. Without `timeoutMs` it waits indefinitely. */
|
|
443
|
+
waitJob(jobId: string, options?: CallOptions): Promise<JobStatus>;
|
|
444
|
+
/** Create a job, wait for it, and return its results. */
|
|
445
|
+
runJob(urls: string[], options?: ScrapeOptions & CallOptions & {
|
|
446
|
+
sequential?: boolean;
|
|
447
|
+
}): Promise<JobResults>;
|
|
448
|
+
/** Create a prompt job, wait for it, and return its results. */
|
|
449
|
+
runAskJob(prompts: string[], options: AskOptions & CallOptions & {
|
|
450
|
+
sequential?: boolean;
|
|
451
|
+
}): Promise<JobResults>;
|
|
452
|
+
/** Which task types and LLM engines are switched on right now. */
|
|
453
|
+
capabilities(options?: CallOptions): Promise<Capabilities>;
|
|
454
|
+
/** Remaining credits. */
|
|
455
|
+
balance(options?: CallOptions): Promise<number>;
|
|
456
|
+
/** The account behind the API key. */
|
|
457
|
+
me(options?: CallOptions): Promise<Profile>;
|
|
458
|
+
/**
|
|
459
|
+
* Revoke the current key and return the new one.
|
|
460
|
+
*
|
|
461
|
+
* This is the only time the new key is shown. The client keeps using the
|
|
462
|
+
* revoked one: store the new key and build a new client with it.
|
|
463
|
+
*/
|
|
464
|
+
resetApiKey(options?: CallOptions): Promise<string>;
|
|
465
|
+
}
|
|
466
|
+
|
|
467
|
+
/**
|
|
468
|
+
* Errors this package throws.
|
|
469
|
+
*
|
|
470
|
+
* Two families, as in the Go and Python SDKs:
|
|
471
|
+
*
|
|
472
|
+
* - {@link APIError} and its subclasses — the API refused the request.
|
|
473
|
+
* - {@link TaskFailed} / {@link Blocked} — the API accepted the task but the
|
|
474
|
+
* page could not be scraped. Failed tasks are refunded.
|
|
475
|
+
*/
|
|
476
|
+
|
|
477
|
+
/** Base class for every error this package throws. */
|
|
478
|
+
declare class DataFuelError extends Error {
|
|
479
|
+
constructor(message: string, options?: {
|
|
480
|
+
cause?: unknown;
|
|
481
|
+
});
|
|
482
|
+
}
|
|
483
|
+
/** No key was passed and DATAFUEL_API_KEY is empty. Thrown before any request. */
|
|
484
|
+
declare class NoApiKey extends DataFuelError {
|
|
485
|
+
}
|
|
486
|
+
/** The request never got an answer: DNS, connection, abort, read timeout. */
|
|
487
|
+
declare class TransportError extends DataFuelError {
|
|
488
|
+
}
|
|
489
|
+
/** A non-2xx answer from the API itself. */
|
|
490
|
+
declare class APIError extends DataFuelError {
|
|
491
|
+
readonly status: number;
|
|
492
|
+
readonly code: string;
|
|
493
|
+
/** Seconds the API asked us to wait, from Retry-After. 0 when absent. */
|
|
494
|
+
readonly retryAfter: number;
|
|
495
|
+
constructor(status: number, code: string, message: string, retryAfter?: number);
|
|
496
|
+
}
|
|
497
|
+
/** 401: the API key is missing or invalid. */
|
|
498
|
+
declare class Unauthorized extends APIError {
|
|
499
|
+
}
|
|
500
|
+
/** 404: unknown id, or one that belongs to another account. */
|
|
501
|
+
declare class NotFound extends APIError {
|
|
502
|
+
}
|
|
503
|
+
/** 429: the account's request rate or concurrency limit was reached. */
|
|
504
|
+
declare class RateLimited extends APIError {
|
|
505
|
+
}
|
|
506
|
+
/** 402: not enough credits for this task. */
|
|
507
|
+
declare class InsufficientCredits extends APIError {
|
|
508
|
+
}
|
|
509
|
+
/** 400 INVALID_ATTRIBUTES: the attributes do not match the task type. */
|
|
510
|
+
declare class InvalidAttributes extends APIError {
|
|
511
|
+
}
|
|
512
|
+
/** 422: the key was already used for a different request. */
|
|
513
|
+
declare class IdempotencyKeyReused extends APIError {
|
|
514
|
+
}
|
|
515
|
+
/** 503: an operator switched something off. The message carries the reason. */
|
|
516
|
+
declare class Unavailable extends APIError {
|
|
517
|
+
}
|
|
518
|
+
/** 503 MODULE_UNAVAILABLE: this task type is switched off. Nothing was charged. */
|
|
519
|
+
declare class ModuleUnavailable extends Unavailable {
|
|
520
|
+
}
|
|
521
|
+
/** 503 ENGINE_UNAVAILABLE: this LLM engine is switched off. Nothing was charged. */
|
|
522
|
+
declare class EngineUnavailable extends Unavailable {
|
|
523
|
+
}
|
|
524
|
+
/**
|
|
525
|
+
* The task was accepted but could not be completed. Refunded.
|
|
526
|
+
*
|
|
527
|
+
* `result` carries the envelope: `statusCode`, `blocked`, `protection`, `error`.
|
|
528
|
+
*/
|
|
529
|
+
declare class TaskFailed extends DataFuelError {
|
|
530
|
+
readonly result: Result;
|
|
531
|
+
constructor(result: Result);
|
|
532
|
+
/** Anti-bot vendor recognised on the target, when there was one. */
|
|
533
|
+
get protection(): string | undefined;
|
|
534
|
+
}
|
|
535
|
+
/**
|
|
536
|
+
* The target refused or challenged the request (403/429/503, anti-bot wall).
|
|
537
|
+
*
|
|
538
|
+
* Refunded. Retry with `jsRendering: true` or a Premium proxy.
|
|
539
|
+
*/
|
|
540
|
+
declare class Blocked extends TaskFailed {
|
|
541
|
+
}
|
|
542
|
+
/**
|
|
543
|
+
* A wait ran out of time. The work keeps running and billing server-side.
|
|
544
|
+
*
|
|
545
|
+
* `id` picks it back up (`waitCrawl`, `jobResults`, or a re-send under the same
|
|
546
|
+
* idempotency key); `status` is the last one seen, when there was one.
|
|
547
|
+
*/
|
|
548
|
+
declare class WaitTimeout extends DataFuelError {
|
|
549
|
+
readonly id: string | undefined;
|
|
550
|
+
readonly status: unknown;
|
|
551
|
+
constructor(message: string, id?: string, status?: unknown);
|
|
552
|
+
}
|
|
553
|
+
|
|
554
|
+
export { type AI, APIError, type AskOptions, Blocked, type CallOptions, Capabilities, type Capability, type ClientOptions, type CrawlOptions, CrawlPage, type CrawlResult, type CrawlResultsPage, type CrawlStatus, DEFAULT_BASE_URL, DataFuel, DataFuelError, type Engine, EngineUnavailable, type Format, IdempotencyKeyReused, InsufficientCredits, InvalidAttributes, type JobResults, type JobStatus, type Link, type MapOptions, ModuleUnavailable, NoApiKey, NotFound, type Payload, type Profile, type Proxy, type ProxyType, RateLimited, Result, type ScrapeOptions, type SiteMap, type Status, TaskFailed, TransportError, Unauthorized, Unavailable, VERSION, WaitTimeout, isDone };
|