@datafuel/sdk 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/client.ts ADDED
@@ -0,0 +1,525 @@
1
+ /** The client. It only moves bytes; `core` decides what goes on the wire. */
2
+
3
+ import * as core from "./core.js";
4
+ import type { AskOptions, CrawlOptions, MapOptions } from "./core.js";
5
+ import { apiError, DataFuelError, NoApiKey, TransportError, WaitTimeout } from "./errors.js";
6
+ import type {
7
+ CallOptions,
8
+ Capability,
9
+ CrawlResult,
10
+ CrawlResultsPage,
11
+ CrawlStatus,
12
+ JobResults,
13
+ JobStatus,
14
+ Profile,
15
+ ScrapeOptions,
16
+ SiteMap,
17
+ } from "./models.js";
18
+ import { Capabilities, CrawlPage, isDone, Result } from "./models.js";
19
+
20
+ /** How to build the client. */
21
+ export interface ClientOptions {
22
+ /** Falls back to `process.env.DATAFUEL_API_KEY`. */
23
+ apiKey?: string;
24
+ baseUrl?: string;
25
+ /** Milliseconds. `null` disables the bound; scrape blocks until the page is ready. */
26
+ timeoutMs?: number | null;
27
+ /** Retries after a network error or a 429/502/503/504 answer. Default 2. */
28
+ maxRetries?: number;
29
+ /** How often `waitJob` and `waitCrawl` poll. Default 2000ms. */
30
+ pollIntervalMs?: number;
31
+ /** Prefixes the SDK User-Agent with your application's. */
32
+ userAgent?: string;
33
+ /** Swap the fetch implementation, e.g. in tests. */
34
+ fetch?: typeof globalThis.fetch;
35
+ }
36
+
37
+ type Send = CallOptions & { timeoutMs?: number; signal?: AbortSignal };
38
+
39
+ /**
40
+ * Client for the DataFuel scraping API.
41
+ *
42
+ * ```ts
43
+ * const df = new DataFuel(); // reads DATAFUEL_API_KEY
44
+ * const markdown = await df.markdown("https://example.com");
45
+ * ```
46
+ *
47
+ * Pick the call by the shape of the work: one URL is {@link scrape}, a site's
48
+ * URL list is {@link map}, many pages from a start URL is {@link crawl}, a list
49
+ * of known URLs is {@link runJob}, a question for an AI engine is {@link ask}.
50
+ *
51
+ * Every write carries an `Idempotency-Key`, generated per request, so a retry
52
+ * attaches to the task already running instead of charging twice.
53
+ */
54
+ export class DataFuel {
55
+ readonly baseUrl: string;
56
+ readonly timeoutMs: number | null;
57
+ readonly maxRetries: number;
58
+ readonly pollIntervalMs: number;
59
+ readonly userAgent: string;
60
+ private readonly apiKey: string;
61
+ private readonly fetchImpl: typeof globalThis.fetch;
62
+
63
+ constructor(options: ClientOptions | string = {}) {
64
+ const opts: ClientOptions = typeof options === "string" ? { apiKey: options } : options;
65
+ this.apiKey = opts.apiKey ?? envApiKey() ?? "";
66
+ this.baseUrl = (opts.baseUrl ?? core.DEFAULT_BASE_URL).replace(/\/+$/, "");
67
+ this.timeoutMs = opts.timeoutMs === undefined ? core.DEFAULT_TIMEOUT_MS : opts.timeoutMs;
68
+ this.maxRetries = Math.max(opts.maxRetries ?? 2, 0);
69
+ // A non-positive interval would poll in a tight loop against the API.
70
+ this.pollIntervalMs =
71
+ opts.pollIntervalMs !== undefined && opts.pollIntervalMs > 0 ? opts.pollIntervalMs : 2_000;
72
+ const sdk = `datafuel-js/${core.VERSION}`;
73
+ this.userAgent = opts.userAgent ? `${opts.userAgent} ${sdk}` : sdk;
74
+ // Bound: a detached fetch throws "Illegal invocation" in browsers and
75
+ // workers, even though Node tolerates it.
76
+ this.fetchImpl = opts.fetch ?? globalThis.fetch.bind(globalThis);
77
+ }
78
+
79
+ // --- transport ---------------------------------------------------------
80
+
81
+ /** Pause between attempts. Overridable so tests need not wait in real time. */
82
+ protected sleep(ms: number): Promise<void> {
83
+ return new Promise((resolve) => setTimeout(resolve, ms));
84
+ }
85
+
86
+ private async send(request: core.Request, opts: Send = {}): Promise<unknown> {
87
+ if (!this.apiKey) {
88
+ throw new NoApiKey("no API key: pass one to the client or set DATAFUEL_API_KEY");
89
+ }
90
+ const url = new URL(this.baseUrl + request.path);
91
+ for (const [name, value] of Object.entries(request.params ?? {})) {
92
+ url.searchParams.set(name, value);
93
+ }
94
+ const init: RequestInit = {
95
+ method: request.method,
96
+ headers: core.headers(this.apiKey, this.userAgent, request),
97
+ // Redirects stay off: a redirect to another host would carry X-API-Key
98
+ // to it, since fetch does not strip custom headers.
99
+ redirect: "manual",
100
+ };
101
+ if (request.body !== undefined) init.body = JSON.stringify(request.body);
102
+
103
+ for (let attempt = 0; ; attempt++) {
104
+ let error: unknown;
105
+ try {
106
+ const signal = this.signalFor(opts);
107
+ // Detached on purpose: calling `this.fetchImpl(...)` would invoke it as
108
+ // a method of the client, and a fetch bound to another receiver (or
109
+ // none) throws "Illegal invocation".
110
+ const send = this.fetchImpl;
111
+ const response = await send(url, signal ? { ...init, signal } : init);
112
+ return await parse(response);
113
+ } catch (caught) {
114
+ if (isAbort(caught)) {
115
+ throw new TransportError("the request was aborted or timed out", { cause: caught });
116
+ }
117
+ error =
118
+ caught instanceof DataFuelError
119
+ ? caught
120
+ : new TransportError(String(caught), {
121
+ cause: caught,
122
+ });
123
+ }
124
+ const retryAfter =
125
+ error instanceof Object && "retryAfter" in error ? Number(error.retryAfter) : 0;
126
+ if (attempt >= this.maxRetries || !request.retryable || !core.shouldRetry(error)) throw error;
127
+ await this.sleep(core.backoff(attempt, retryAfter));
128
+ }
129
+ }
130
+
131
+ private signalFor(opts: Send): AbortSignal | undefined {
132
+ const budget = opts.timeoutMs ?? this.timeoutMs;
133
+ const timeout =
134
+ budget === null || budget === undefined ? undefined : AbortSignal.timeout(budget);
135
+ if (timeout && opts.signal) return AbortSignal.any([timeout, opts.signal]);
136
+ return timeout ?? opts.signal;
137
+ }
138
+
139
+ /**
140
+ * Send a task and wait for its result.
141
+ *
142
+ * On a 202 the API is still working: the identical request goes out again
143
+ * under the identical idempotency key, so it attaches to the running task
144
+ * rather than starting a second one.
145
+ */
146
+ private async runTask(request: core.Request, opts: Send): Promise<Result> {
147
+ const budget = opts.timeoutMs ?? this.timeoutMs ?? core.DEFAULT_TIMEOUT_MS;
148
+ const deadline = Date.now() + budget;
149
+ let body: unknown;
150
+ for (let attempt = 0; ; attempt++) {
151
+ body = await this.send(request, opts);
152
+ if (!core.stillProcessing(body)) break;
153
+ const delay = core.pollDelay(attempt);
154
+ if (Date.now() + delay >= deadline) {
155
+ throw new WaitTimeout(
156
+ "the task is still processing; re-send under the same idempotency key to pick it up",
157
+ request.idempotencyKey,
158
+ );
159
+ }
160
+ await this.sleep(delay);
161
+ }
162
+ const result = new Result(record(body));
163
+ result.raiseForStatus();
164
+ return result;
165
+ }
166
+
167
+ // --- pages -------------------------------------------------------------
168
+
169
+ /**
170
+ * Fetch one URL and wait for the result.
171
+ *
172
+ * Throws {@link Blocked} when the target refused the request and
173
+ * {@link TaskFailed} otherwise; both carry `.result`.
174
+ */
175
+ async scrape(url: string, options: ScrapeOptions & CallOptions = {}): Promise<Result> {
176
+ const request = core.buildScrape(url, options, options.proxy, core.key(options.idempotencyKey));
177
+ return this.runTask(request, options);
178
+ }
179
+
180
+ /** The one-liner: the page as LLM-ready markdown, images dropped. */
181
+ async markdown(url: string, options: ScrapeOptions & CallOptions = {}): Promise<string> {
182
+ const result = await this.scrape(url, { ...options, format: "markdown", includeImages: false });
183
+ return result.text;
184
+ }
185
+
186
+ /**
187
+ * Return a task by id, e.g. one created by a job or a crawl.
188
+ *
189
+ * A task that is still running comes back with `pending` true.
190
+ */
191
+ async getTask(taskId: string, options: CallOptions = {}): Promise<Result> {
192
+ const body = await this.send(
193
+ new core.Request("GET", `/task/${core.pathSegment(taskId)}`),
194
+ options,
195
+ );
196
+ if (core.stillProcessing(body)) return new Result({ id: taskId, status: "processing" });
197
+ const result = new Result(record(body));
198
+ result.raiseForStatus();
199
+ return result;
200
+ }
201
+
202
+ /** Send a prompt to an AI engine and return its answer. */
203
+ async ask(prompt: string, options: AskOptions & CallOptions): Promise<Result> {
204
+ const request = core.buildAsk(prompt, options, core.key(options.idempotencyKey));
205
+ return this.runTask(request, options);
206
+ }
207
+
208
+ /**
209
+ * List the URLs of a site without scraping them.
210
+ *
211
+ * One credit per call on a Basic proxy, however many links come back. Use it
212
+ * before a crawl to see how big a section is.
213
+ */
214
+ async map(url: string, options: MapOptions & CallOptions = {}): Promise<SiteMap> {
215
+ const request = core.buildMap(url, options, options.proxy, core.key(options.idempotencyKey));
216
+ const task = await this.runTask(request, options);
217
+ const data = record(task.requireData()) as unknown as SiteMap;
218
+ return { ...data, links: data.links ?? [], sitemaps: data.sitemaps ?? [], task };
219
+ }
220
+
221
+ // --- crawl -------------------------------------------------------------
222
+
223
+ /**
224
+ * Queue a crawl and return its id immediately.
225
+ *
226
+ * Follows links from the start URL and scrapes every page. Pages are charged
227
+ * like single scrapes when they are queued; failed and blocked pages are
228
+ * refunded. Unset limits use the API defaults: 100 pages, depth 3, 5 in flight.
229
+ */
230
+ async startCrawl(url: string, options: CrawlOptions & CallOptions = {}): Promise<string> {
231
+ const request = core.buildCrawl(url, options, options.proxy, core.key(options.idempotencyKey));
232
+ return strField(await this.send(request, options), "job_id");
233
+ }
234
+
235
+ /** Return the progress of a crawl. */
236
+ async getCrawl(crawlId: string, options: CallOptions = {}): Promise<CrawlStatus> {
237
+ const body = record(
238
+ await this.send(new core.Request("GET", `/crawl/${core.pathSegment(crawlId)}`), options),
239
+ );
240
+ const raw = body as unknown as Partial<CrawlStatus>;
241
+ return {
242
+ ...raw,
243
+ status: raw.status as CrawlStatus["status"],
244
+ // Counters the caller reads unconditionally must exist even on a thin answer.
245
+ pages: {
246
+ discovered: 0,
247
+ enqueued: 0,
248
+ done: 0,
249
+ failed: 0,
250
+ skipped: 0,
251
+ ...(raw.pages ?? {}),
252
+ },
253
+ depth_reached: raw.depth_reached ?? 0,
254
+ total_cost: raw.total_cost ?? 0,
255
+ done: isDone(raw.status),
256
+ };
257
+ }
258
+
259
+ /** One page of results, in discovery order. `limit` unset uses the API default. */
260
+ async crawlResults(
261
+ crawlId: string,
262
+ options: CallOptions & { cursor?: string; limit?: number } = {},
263
+ ): Promise<CrawlResultsPage> {
264
+ const body = record(
265
+ await this.send(core.crawlResultsRequest(crawlId, options.cursor, options.limit), options),
266
+ );
267
+ const pages = Array.isArray(body.pages) ? body.pages : [];
268
+ const next = typeof body.next_cursor === "string" ? body.next_cursor : undefined;
269
+ return {
270
+ pages: pages.map((page) => new CrawlPage(record(page))),
271
+ ...(next ? { nextCursor: next } : {}),
272
+ };
273
+ }
274
+
275
+ /**
276
+ * Walk every page of a crawl, fetching result pages as needed.
277
+ *
278
+ * Readable while the crawl runs; unfinished pages report `pending`.
279
+ */
280
+ async *crawlPages(
281
+ crawlId: string,
282
+ options: CallOptions & { limit?: number } = {},
283
+ ): AsyncGenerator<CrawlPage> {
284
+ let cursor: string | undefined;
285
+ const seen = new Set<string>();
286
+ for (;;) {
287
+ const batch = await this.crawlResults(crawlId, { ...options, ...(cursor ? { cursor } : {}) });
288
+ yield* batch.pages;
289
+ cursor = batch.nextCursor;
290
+ if (!cursor) return;
291
+ if (seen.has(cursor)) {
292
+ throw new DataFuelError("the API repeated a crawl cursor; stopping to avoid a loop");
293
+ }
294
+ seen.add(cursor);
295
+ }
296
+ }
297
+
298
+ /** Poll until the crawl is done. Without `timeoutMs` it waits indefinitely. */
299
+ async waitCrawl(crawlId: string, options: CallOptions = {}): Promise<CrawlStatus> {
300
+ // The timeout bounds the whole wait, not each poll.
301
+ const { timeoutMs, ...perPoll } = options;
302
+ const deadline = timeoutMs === undefined ? null : Date.now() + timeoutMs;
303
+ for (;;) {
304
+ const status = await this.getCrawl(crawlId, perPoll);
305
+ if (status.done) return status;
306
+ if (deadline !== null && Date.now() + this.pollIntervalMs >= deadline) {
307
+ throw new WaitTimeout(`crawl ${crawlId} is still running`, crawlId, status);
308
+ }
309
+ await this.sleep(this.pollIntervalMs);
310
+ }
311
+ }
312
+
313
+ /**
314
+ * Start a crawl, wait for it, and return every page.
315
+ *
316
+ * Check `stop_reason`: `insufficient_credits` means it ended early. For large
317
+ * crawls prefer {@link startCrawl} + {@link waitCrawl} + {@link crawlPages},
318
+ * which stream instead of holding every page in memory. On {@link WaitTimeout}
319
+ * the error carries the id: the crawl keeps running and billing.
320
+ */
321
+ async crawl(url: string, options: CrawlOptions & CallOptions = {}): Promise<CrawlResult> {
322
+ const id = await this.startCrawl(url, options);
323
+ const status = await this.waitCrawl(id, options);
324
+ const pages: CrawlPage[] = [];
325
+ for await (const page of this.crawlPages(id, options)) pages.push(page);
326
+ return { id, status, pages };
327
+ }
328
+
329
+ // --- jobs --------------------------------------------------------------
330
+
331
+ /**
332
+ * Queue a batch of known URLs and return the job id.
333
+ *
334
+ * Cheaper and more predictable than a crawl when you already have the URLs.
335
+ * `sequential` runs them one after the other; the default runs them
336
+ * concurrently, bounded by the account's concurrency limit.
337
+ */
338
+ async createJob(
339
+ urls: string[],
340
+ options: ScrapeOptions & CallOptions & { sequential?: boolean } = {},
341
+ ): Promise<string> {
342
+ const request = core.buildJob(
343
+ urls,
344
+ options,
345
+ options.proxy,
346
+ options.sequential ?? false,
347
+ core.key(options.idempotencyKey),
348
+ );
349
+ return strField(await this.send(request, options), "id");
350
+ }
351
+
352
+ /** Queue a batch of prompts for an AI engine and return the job id. */
353
+ async createAskJob(
354
+ prompts: string[],
355
+ options: AskOptions & CallOptions & { sequential?: boolean },
356
+ ): Promise<string> {
357
+ const request = core.buildAskJob(
358
+ prompts,
359
+ options,
360
+ options.sequential ?? false,
361
+ core.key(options.idempotencyKey),
362
+ );
363
+ return strField(await this.send(request, options), "id");
364
+ }
365
+
366
+ /** Return the progress of a job. */
367
+ async getJob(jobId: string, options: CallOptions = {}): Promise<JobStatus> {
368
+ const body = record(
369
+ await this.send(new core.Request("GET", `/job/${core.pathSegment(jobId)}`), options),
370
+ );
371
+ const raw = body as unknown as Partial<JobStatus>;
372
+ return {
373
+ status: raw.status as JobStatus["status"],
374
+ tasks_count: raw.tasks_count ?? 0,
375
+ tasks_done: raw.tasks_done ?? 0,
376
+ tasks_remaining: raw.tasks_remaining ?? 0,
377
+ total_cost: raw.total_cost ?? 0,
378
+ done: isDone(raw.status),
379
+ };
380
+ }
381
+
382
+ /**
383
+ * Return the per-task results of a job.
384
+ *
385
+ * A job completes even when some of its tasks failed: check `task.ok` or call
386
+ * `task.raiseForStatus()` per task.
387
+ */
388
+ async jobResults(jobId: string, options: CallOptions = {}): Promise<JobResults> {
389
+ const body = record(
390
+ await this.send(new core.Request("GET", `/job/${core.pathSegment(jobId)}/results`), options),
391
+ );
392
+ const tasks = Array.isArray(body.tasks_result) ? body.tasks_result : [];
393
+ return {
394
+ id: jobId,
395
+ tasks_count: Number(body.tasks_count ?? 0),
396
+ tasks_failed: Number(body.tasks_failed ?? 0),
397
+ tasks_complete: Number(body.tasks_complete ?? 0),
398
+ tasks: tasks.map((task) => new Result(record(task))),
399
+ };
400
+ }
401
+
402
+ /** Poll until the job is done. Without `timeoutMs` it waits indefinitely. */
403
+ async waitJob(jobId: string, options: CallOptions = {}): Promise<JobStatus> {
404
+ // The timeout bounds the whole wait, not each poll.
405
+ const { timeoutMs, ...perPoll } = options;
406
+ const deadline = timeoutMs === undefined ? null : Date.now() + timeoutMs;
407
+ for (;;) {
408
+ const status = await this.getJob(jobId, perPoll);
409
+ if (status.done) return status;
410
+ if (deadline !== null && Date.now() + this.pollIntervalMs >= deadline) {
411
+ throw new WaitTimeout(`job ${jobId} is still running`, jobId, status);
412
+ }
413
+ await this.sleep(this.pollIntervalMs);
414
+ }
415
+ }
416
+
417
+ /** Create a job, wait for it, and return its results. */
418
+ async runJob(
419
+ urls: string[],
420
+ options: ScrapeOptions & CallOptions & { sequential?: boolean } = {},
421
+ ): Promise<JobResults> {
422
+ const id = await this.createJob(urls, options);
423
+ await this.waitJob(id, options);
424
+ return this.jobResults(id, options);
425
+ }
426
+
427
+ /** Create a prompt job, wait for it, and return its results. */
428
+ async runAskJob(
429
+ prompts: string[],
430
+ options: AskOptions & CallOptions & { sequential?: boolean },
431
+ ): Promise<JobResults> {
432
+ const id = await this.createAskJob(prompts, options);
433
+ await this.waitJob(id, options);
434
+ return this.jobResults(id, options);
435
+ }
436
+
437
+ // --- account -----------------------------------------------------------
438
+
439
+ /** Which task types and LLM engines are switched on right now. */
440
+ async capabilities(options: CallOptions = {}): Promise<Capabilities> {
441
+ return new Capabilities(
442
+ record(await this.send(new core.Request("GET", "/capabilities"), options)),
443
+ );
444
+ }
445
+
446
+ /** Remaining credits. */
447
+ async balance(options: CallOptions = {}): Promise<number> {
448
+ const body = await this.send(new core.Request("GET", "/users/@me/balance"), options);
449
+ return intField(body, "balance");
450
+ }
451
+
452
+ /** The account behind the API key. */
453
+ async me(options: CallOptions = {}): Promise<Profile> {
454
+ const body = await this.send(new core.Request("GET", "/users/@me"), options);
455
+ return record(body) as unknown as Profile;
456
+ }
457
+
458
+ /**
459
+ * Revoke the current key and return the new one.
460
+ *
461
+ * This is the only time the new key is shown. The client keeps using the
462
+ * revoked one: store the new key and build a new client with it.
463
+ */
464
+ async resetApiKey(options: CallOptions = {}): Promise<string> {
465
+ const body = await this.send(new core.Request("POST", "/users/@me/api-key/reset"), options);
466
+ return strField(body, "api_key");
467
+ }
468
+ }
469
+
470
+ export type { Capability };
471
+
472
+ /** The API key from the environment, on runtimes that have one. */
473
+ function envApiKey(): string | undefined {
474
+ return typeof process !== "undefined" ? process.env?.DATAFUEL_API_KEY : undefined;
475
+ }
476
+
477
+ async function parse(response: Response): Promise<unknown> {
478
+ const text = await response.text();
479
+ let body: unknown;
480
+ if (text.length > 0) {
481
+ try {
482
+ body = JSON.parse(text);
483
+ } catch {
484
+ body = text;
485
+ }
486
+ }
487
+ if (!response.ok) {
488
+ const header = response.headers.get("Retry-After");
489
+ const retryAfter = header !== null && !Number.isNaN(Number(header)) ? Number(header) : 0;
490
+ throw apiError(response.status, body, retryAfter);
491
+ }
492
+ return body;
493
+ }
494
+
495
+ function isAbort(error: unknown): boolean {
496
+ return error instanceof Error && (error.name === "AbortError" || error.name === "TimeoutError");
497
+ }
498
+
499
+ /** Turn a surprise answer into a DataFuelError rather than a TypeError later. */
500
+ function record(body: unknown): Record<string, unknown> {
501
+ if (body === null || typeof body !== "object" || Array.isArray(body)) {
502
+ throw new DataFuelError(`unexpected answer from the API: ${JSON.stringify(body) ?? "empty"}`);
503
+ }
504
+ return body as Record<string, unknown>;
505
+ }
506
+
507
+ function field(body: unknown, name: string): unknown {
508
+ const value = record(body)[name];
509
+ if (value === undefined || value === null) {
510
+ throw new DataFuelError(`unexpected answer from the API: no ${name} field`);
511
+ }
512
+ return value;
513
+ }
514
+
515
+ function strField(body: unknown, name: string): string {
516
+ return String(field(body, name));
517
+ }
518
+
519
+ function intField(body: unknown, name: string): number {
520
+ const value = Number(field(body, name));
521
+ if (Number.isNaN(value)) {
522
+ throw new DataFuelError(`unexpected answer from the API: ${name} is not a number`);
523
+ }
524
+ return value;
525
+ }