@datafuel/sdk 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +192 -0
- package/dist/index.cjs +922 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.cts +554 -0
- package/dist/index.d.ts +554 -0
- package/dist/index.js +873 -0
- package/dist/index.js.map +1 -0
- package/package.json +59 -0
- package/src/client.ts +525 -0
- package/src/core.ts +388 -0
- package/src/errors.ts +136 -0
- package/src/index.ts +69 -0
- package/src/models.ts +362 -0
package/src/models.ts
ADDED
|
@@ -0,0 +1,362 @@
|
|
|
1
|
+
/** Request options and response shapes. */
|
|
2
|
+
|
|
3
|
+
import { Blocked, DataFuelError, TaskFailed } from "./errors.js";
|
|
4
|
+
|
|
5
|
+
/** State of a task, job or crawl. Unknown states are treated as not final. */
|
|
6
|
+
export type Status =
|
|
7
|
+
| "created"
|
|
8
|
+
| "pending"
|
|
9
|
+
| "processing"
|
|
10
|
+
| "completed"
|
|
11
|
+
| "completed_with_errors"
|
|
12
|
+
| "failed"
|
|
13
|
+
| "cancelled";
|
|
14
|
+
|
|
15
|
+
const TERMINAL = new Set(["completed", "completed_with_errors", "failed", "cancelled"]);
|
|
16
|
+
|
|
17
|
+
/** Whether a state is final. */
|
|
18
|
+
export function isDone(status: string | undefined): boolean {
|
|
19
|
+
return status !== undefined && TERMINAL.has(status);
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* Shape of the scraped content.
|
|
24
|
+
*
|
|
25
|
+
* `html` raw page (API default), `markdown` cleaned text (best for LLMs),
|
|
26
|
+
* `json` schema.org / JSON-LD, `png` / `jpeg` full-page screenshot.
|
|
27
|
+
*/
|
|
28
|
+
export type Format = "html" | "markdown" | "json" | "png" | "jpeg";
|
|
29
|
+
|
|
30
|
+
/** AI assistant `ask` can query. Engines can be switched off at runtime. */
|
|
31
|
+
export type Engine = "openai" | "gemini" | "google_ai_mode" | "perplexity" | "copilot";
|
|
32
|
+
|
|
33
|
+
/** Proxy plan. Matched case-insensitively; deployments may define others. */
|
|
34
|
+
export type ProxyType = "Basic" | "Premium" | (string & {});
|
|
35
|
+
|
|
36
|
+
/** The exit the request leaves from. Every field optional: the account default. */
|
|
37
|
+
export interface Proxy {
|
|
38
|
+
type?: ProxyType;
|
|
39
|
+
/** ISO 3166-1 alpha-2, e.g. "US". */
|
|
40
|
+
country?: string;
|
|
41
|
+
city?: string;
|
|
42
|
+
state?: string;
|
|
43
|
+
asn?: string;
|
|
44
|
+
/**
|
|
45
|
+
* Sticky session: the same exit across requests, `ttl` in seconds. Read by
|
|
46
|
+
* `scrape` and `map` only, and sent in attributes rather than the envelope.
|
|
47
|
+
*/
|
|
48
|
+
sessionId?: string;
|
|
49
|
+
ttl?: number;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Post-process the page with an LLM, using your own key.
|
|
54
|
+
*
|
|
55
|
+
* Not supported on crawls: the API rejects `result_use_ai` there.
|
|
56
|
+
*/
|
|
57
|
+
export interface AI {
|
|
58
|
+
/** What to extract. */
|
|
59
|
+
prompt?: string;
|
|
60
|
+
/** Example JSON object the output must follow. */
|
|
61
|
+
format?: unknown;
|
|
62
|
+
provider?: "openai" | "anthropic" | "google";
|
|
63
|
+
model?: string;
|
|
64
|
+
apiKey?: string;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** Page options, shared by `scrape`, jobs and crawls. */
|
|
68
|
+
export interface ScrapeOptions {
|
|
69
|
+
format?: Format;
|
|
70
|
+
/**
|
|
71
|
+
* Render the page in a real browser: slower, five times the credits on a
|
|
72
|
+
* Basic proxy. Leave it off unless the page comes back empty without it.
|
|
73
|
+
*/
|
|
74
|
+
jsRendering?: boolean;
|
|
75
|
+
waitFor?: string;
|
|
76
|
+
waitForTimeoutMs?: number;
|
|
77
|
+
jsInstructions?: unknown;
|
|
78
|
+
blockResource?: string;
|
|
79
|
+
/** Markdown only: always render just the `<main>` / `<article>` container. */
|
|
80
|
+
mainContentOnly?: boolean;
|
|
81
|
+
/** Markdown only: `false` drops images and saves tokens. */
|
|
82
|
+
includeImages?: boolean;
|
|
83
|
+
/**
|
|
84
|
+
* Extract only matching elements: field name to CSS selector. Append
|
|
85
|
+
* ` @attr` to read an attribute, e.g. `{ links: "a @href" }`.
|
|
86
|
+
*/
|
|
87
|
+
extract?: string | Record<string, string>;
|
|
88
|
+
/** The same with RE2 patterns on the raw HTML. */
|
|
89
|
+
extractRegex?: string | Record<string, string>;
|
|
90
|
+
/**
|
|
91
|
+
* Built-in extractors, comma separated: images, links, headings,
|
|
92
|
+
* phone_numbers, emails, meta_tags, tables, schema_org, all.
|
|
93
|
+
*/
|
|
94
|
+
template?: string;
|
|
95
|
+
method?: string;
|
|
96
|
+
body?: string;
|
|
97
|
+
contentType?: string;
|
|
98
|
+
headers?: Record<string, string>;
|
|
99
|
+
headerOrder?: string[];
|
|
100
|
+
cookies?: string;
|
|
101
|
+
userAgent?: string;
|
|
102
|
+
/** chrome, firefox, safari, edge. */
|
|
103
|
+
userAgentType?: string;
|
|
104
|
+
ai?: AI;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/** Options every call accepts. */
|
|
108
|
+
export interface CallOptions {
|
|
109
|
+
proxy?: Proxy;
|
|
110
|
+
/** Generated per request when omitted. Max 255 characters. */
|
|
111
|
+
idempotencyKey?: string;
|
|
112
|
+
/** Overrides the client's timeout for this call. */
|
|
113
|
+
timeoutMs?: number;
|
|
114
|
+
signal?: AbortSignal;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/** The raw `result` object of a task. */
|
|
118
|
+
export interface Payload {
|
|
119
|
+
data?: unknown;
|
|
120
|
+
status?: Status;
|
|
121
|
+
error_detail?: string;
|
|
122
|
+
status_code?: number;
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* One scraped target with its metadata.
|
|
127
|
+
*
|
|
128
|
+
* Read the metadata before the content: a 200 can be an error page,
|
|
129
|
+
* `redirected` means `finalUrl` is not what you asked for, and a 404 still
|
|
130
|
+
* completes and bills.
|
|
131
|
+
*/
|
|
132
|
+
export class Result {
|
|
133
|
+
readonly id: string | undefined;
|
|
134
|
+
readonly status: Status | undefined;
|
|
135
|
+
/** HTTP status the target answered. */
|
|
136
|
+
readonly statusCode: number | undefined;
|
|
137
|
+
readonly finalUrl: string | undefined;
|
|
138
|
+
readonly redirected: boolean;
|
|
139
|
+
/** Charged for this task, 0 when it failed. */
|
|
140
|
+
readonly creditsUsed: number;
|
|
141
|
+
readonly durationMs: number | undefined;
|
|
142
|
+
readonly blocked: boolean;
|
|
143
|
+
/** Anti-bot vendor recognised when blocked. */
|
|
144
|
+
readonly protection: string | undefined;
|
|
145
|
+
readonly error: string | undefined;
|
|
146
|
+
/** The raw `result` object. Prefer `text`, `data` and `image`. */
|
|
147
|
+
readonly payload: Payload | undefined;
|
|
148
|
+
/** Everything the API sent, including fields this SDK does not know yet. */
|
|
149
|
+
readonly raw: Record<string, unknown>;
|
|
150
|
+
|
|
151
|
+
constructor(raw: Record<string, unknown>) {
|
|
152
|
+
this.raw = raw;
|
|
153
|
+
this.id = str(raw.id);
|
|
154
|
+
this.status = str(raw.status) as Status | undefined;
|
|
155
|
+
this.statusCode = num(raw.status_code);
|
|
156
|
+
this.finalUrl = str(raw.final_url);
|
|
157
|
+
this.redirected = raw.redirected === true;
|
|
158
|
+
this.creditsUsed = num(raw.credits_used) ?? 0;
|
|
159
|
+
this.durationMs = num(raw.duration_ms);
|
|
160
|
+
this.blocked = raw.blocked === true;
|
|
161
|
+
this.protection = str(raw.protection);
|
|
162
|
+
this.error = str(raw.error);
|
|
163
|
+
this.payload =
|
|
164
|
+
raw.result !== null && typeof raw.result === "object" ? (raw.result as Payload) : undefined;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/** Effective state: the envelope's, else the payload's, else inferred. */
|
|
168
|
+
get state(): string {
|
|
169
|
+
if (this.status) return this.status;
|
|
170
|
+
if (this.payload?.status) return this.payload.status;
|
|
171
|
+
if (this.payload?.data !== undefined && this.payload.data !== null) return "completed";
|
|
172
|
+
return "pending";
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/** Whether the task has not finished yet (job and crawl results list stubs). */
|
|
176
|
+
get pending(): boolean {
|
|
177
|
+
return !isDone(this.state);
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/** Whether the task completed and the target was not blocked. */
|
|
181
|
+
get ok(): boolean {
|
|
182
|
+
return this.state === "completed" && !this.blocked;
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/** The structured output: format json, extract, template, AI. */
|
|
186
|
+
get data(): unknown {
|
|
187
|
+
return this.payload?.data;
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/** Content of an html or markdown result; the raw JSON for structured ones. */
|
|
191
|
+
get text(): string {
|
|
192
|
+
const value = this.data;
|
|
193
|
+
if (value === undefined || value === null) return "";
|
|
194
|
+
return typeof value === "string" ? value : JSON.stringify(value);
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
/** Bytes of a png or jpeg screenshot. Throws when the result is not one. */
|
|
198
|
+
get image(): Uint8Array {
|
|
199
|
+
try {
|
|
200
|
+
return Uint8Array.from(atob(this.text), (char) => char.charCodeAt(0));
|
|
201
|
+
} catch (cause) {
|
|
202
|
+
throw new DataFuelError("result is not a png or jpeg screenshot", { cause });
|
|
203
|
+
}
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
/** `data`, or throw when the task carried none. */
|
|
207
|
+
requireData(): unknown {
|
|
208
|
+
if (this.data === undefined || this.data === null) {
|
|
209
|
+
throw new DataFuelError(`result has no data (status ${this.state})`);
|
|
210
|
+
}
|
|
211
|
+
return this.data;
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/** Throw TaskFailed, or Blocked, when this task failed. Refunded either way. */
|
|
215
|
+
raiseForStatus(): void {
|
|
216
|
+
if (this.state === "failed") {
|
|
217
|
+
throw this.blocked ? new Blocked(this) : new TaskFailed(this);
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
/** One crawled page: where it was found, plus the scrape result. */
|
|
223
|
+
export class CrawlPage extends Result {
|
|
224
|
+
readonly url: string;
|
|
225
|
+
readonly depth: number;
|
|
226
|
+
readonly taskId: string | undefined;
|
|
227
|
+
|
|
228
|
+
constructor(raw: Record<string, unknown>) {
|
|
229
|
+
super(raw);
|
|
230
|
+
this.url = str(raw.url) ?? "";
|
|
231
|
+
this.depth = num(raw.depth) ?? 0;
|
|
232
|
+
this.taskId = str(raw.task_id);
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
/** One discovered URL. */
|
|
237
|
+
export interface Link {
|
|
238
|
+
url: string;
|
|
239
|
+
/** "sitemap" or "page". */
|
|
240
|
+
source: string;
|
|
241
|
+
/** Page links only. */
|
|
242
|
+
title?: string;
|
|
243
|
+
/** Sitemap links only. */
|
|
244
|
+
lastmod?: string;
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
/**
|
|
248
|
+
* What `map` found.
|
|
249
|
+
*
|
|
250
|
+
* `reason` says why `links` is empty: page_blocked, page_error,
|
|
251
|
+
* page_unreachable, no_links_on_page (try scrape with jsRendering), no_sitemap,
|
|
252
|
+
* sitemap_no_entries, search_no_match.
|
|
253
|
+
*/
|
|
254
|
+
export interface SiteMap {
|
|
255
|
+
url: string;
|
|
256
|
+
links: Link[];
|
|
257
|
+
total: number;
|
|
258
|
+
/** `limit` or the time budget cut the list. */
|
|
259
|
+
truncated: boolean;
|
|
260
|
+
sitemaps: string[];
|
|
261
|
+
page_status_code: number;
|
|
262
|
+
reason?: string;
|
|
263
|
+
credits: number;
|
|
264
|
+
/** Envelope of the underlying task. */
|
|
265
|
+
task: Result;
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
/** Progress of a crawl. */
|
|
269
|
+
export interface CrawlStatus {
|
|
270
|
+
status: Status;
|
|
271
|
+
/** Empty while the crawl still grows: max_pages, max_depth_exhausted, insufficient_credits, cancelled. */
|
|
272
|
+
stop_reason?: string;
|
|
273
|
+
pages: { discovered: number; enqueued: number; done: number; failed: number; skipped: number };
|
|
274
|
+
depth_reached: number;
|
|
275
|
+
/**
|
|
276
|
+
* Charged at queue time; refunds of failed pages are not subtracted. Sum
|
|
277
|
+
* `creditsUsed` over the pages for the net figure.
|
|
278
|
+
*/
|
|
279
|
+
total_cost: number;
|
|
280
|
+
/** Whether the crawl reached a final state. */
|
|
281
|
+
done: boolean;
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
/** One page of crawl results. */
|
|
285
|
+
export interface CrawlResultsPage {
|
|
286
|
+
pages: CrawlPage[];
|
|
287
|
+
/** Absent on the last page. */
|
|
288
|
+
nextCursor?: string;
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
/** A finished crawl: why it stopped and what it found. */
|
|
292
|
+
export interface CrawlResult {
|
|
293
|
+
id: string;
|
|
294
|
+
status: CrawlStatus;
|
|
295
|
+
pages: CrawlPage[];
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
/** Progress of a job. */
|
|
299
|
+
export interface JobStatus {
|
|
300
|
+
status: Status;
|
|
301
|
+
tasks_count: number;
|
|
302
|
+
tasks_done: number;
|
|
303
|
+
tasks_remaining: number;
|
|
304
|
+
total_cost: number;
|
|
305
|
+
done: boolean;
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
/** Every task of a job. A job completes even when some of its tasks failed. */
|
|
309
|
+
export interface JobResults {
|
|
310
|
+
id: string;
|
|
311
|
+
tasks_count: number;
|
|
312
|
+
tasks_failed: number;
|
|
313
|
+
tasks_complete: number;
|
|
314
|
+
tasks: Result[];
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
/** The account behind the API key. */
|
|
318
|
+
export interface Profile {
|
|
319
|
+
email: string;
|
|
320
|
+
username: string;
|
|
321
|
+
current_concurrency: number;
|
|
322
|
+
concurrency_limit: number;
|
|
323
|
+
credit_balance: number;
|
|
324
|
+
monthly_credit_limit: number;
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
/** One task type or LLM engine, and whether it accepts new work. */
|
|
328
|
+
export interface Capability {
|
|
329
|
+
name: string;
|
|
330
|
+
enabled: boolean;
|
|
331
|
+
/** Operator's note when switched off. */
|
|
332
|
+
reason?: string;
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
/** What the API accepts right now. */
|
|
336
|
+
export class Capabilities {
|
|
337
|
+
readonly modules: Capability[];
|
|
338
|
+
readonly engines: Capability[];
|
|
339
|
+
|
|
340
|
+
constructor(raw: Record<string, unknown>) {
|
|
341
|
+
this.modules = Array.isArray(raw.modules) ? (raw.modules as Capability[]) : [];
|
|
342
|
+
this.engines = Array.isArray(raw.engines) ? (raw.engines as Capability[]) : [];
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
/** Whether a task type accepts work. Unknown names report false. */
|
|
346
|
+
moduleEnabled(name: string): boolean {
|
|
347
|
+
return this.modules.some((item) => item.name === name && item.enabled);
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
/** Whether an LLM engine accepts work. Unknown names report false. */
|
|
351
|
+
engineEnabled(name: Engine | string): boolean {
|
|
352
|
+
return this.engines.some((item) => item.name === name && item.enabled);
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
function str(value: unknown): string | undefined {
|
|
357
|
+
return typeof value === "string" && value !== "" ? value : undefined;
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
function num(value: unknown): number | undefined {
|
|
361
|
+
return typeof value === "number" ? value : undefined;
|
|
362
|
+
}
|