@octocrawl/sdk 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +17 -0
- package/dist/index.cjs +1278 -0
- package/dist/index.js +1240 -0
- package/dist/types/cjs/client.d.ts +362 -0
- package/dist/types/cjs/contracts/access.d.ts +166 -0
- package/dist/types/cjs/contracts/actions.d.ts +191 -0
- package/dist/types/cjs/contracts/api.d.ts +891 -0
- package/dist/types/cjs/contracts/benchmark.d.ts +116 -0
- package/dist/types/cjs/contracts/checkpoint.d.ts +164 -0
- package/dist/types/cjs/contracts/compliance.d.ts +412 -0
- package/dist/types/cjs/contracts/crawl.d.ts +302 -0
- package/dist/types/cjs/contracts/delivery.d.ts +136 -0
- package/dist/types/cjs/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/cjs/contracts/execution.d.ts +197 -0
- package/dist/types/cjs/contracts/extractor.d.ts +379 -0
- package/dist/types/cjs/contracts/file.d.ts +117 -0
- package/dist/types/cjs/contracts/firecrawl.d.ts +258 -0
- package/dist/types/cjs/contracts/groundTruth.d.ts +77 -0
- package/dist/types/cjs/contracts/identityBundle.d.ts +70 -0
- package/dist/types/cjs/contracts/index.d.ts +30 -0
- package/dist/types/cjs/contracts/map.d.ts +180 -0
- package/dist/types/cjs/contracts/monitor.d.ts +217 -0
- package/dist/types/cjs/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/cjs/contracts/policy.d.ts +93 -0
- package/dist/types/cjs/contracts/proxy.d.ts +52 -0
- package/dist/types/cjs/contracts/recipe.d.ts +74 -0
- package/dist/types/cjs/contracts/regexSafety.d.ts +31 -0
- package/dist/types/cjs/contracts/result.d.ts +503 -0
- package/dist/types/cjs/contracts/session.d.ts +51 -0
- package/dist/types/cjs/contracts/ssrf.d.ts +16 -0
- package/dist/types/cjs/contracts/status.d.ts +27 -0
- package/dist/types/cjs/contracts/structured.d.ts +185 -0
- package/dist/types/cjs/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/cjs/contracts/tokens.d.ts +47 -0
- package/dist/types/cjs/index.d.ts +9 -0
- package/dist/types/cjs/package.json +1 -0
- package/dist/types/cjs/version.d.ts +8 -0
- package/dist/types/cjs/watcher.d.ts +151 -0
- package/dist/types/esm/client.d.ts +362 -0
- package/dist/types/esm/contracts/access.d.ts +166 -0
- package/dist/types/esm/contracts/actions.d.ts +191 -0
- package/dist/types/esm/contracts/api.d.ts +891 -0
- package/dist/types/esm/contracts/benchmark.d.ts +116 -0
- package/dist/types/esm/contracts/checkpoint.d.ts +164 -0
- package/dist/types/esm/contracts/compliance.d.ts +412 -0
- package/dist/types/esm/contracts/crawl.d.ts +302 -0
- package/dist/types/esm/contracts/delivery.d.ts +136 -0
- package/dist/types/esm/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/esm/contracts/execution.d.ts +197 -0
- package/dist/types/esm/contracts/extractor.d.ts +379 -0
- package/dist/types/esm/contracts/file.d.ts +117 -0
- package/dist/types/esm/contracts/firecrawl.d.ts +258 -0
- package/dist/types/esm/contracts/groundTruth.d.ts +77 -0
- package/dist/types/esm/contracts/identityBundle.d.ts +70 -0
- package/dist/types/esm/contracts/index.d.ts +30 -0
- package/dist/types/esm/contracts/map.d.ts +180 -0
- package/dist/types/esm/contracts/monitor.d.ts +217 -0
- package/dist/types/esm/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/esm/contracts/policy.d.ts +93 -0
- package/dist/types/esm/contracts/proxy.d.ts +52 -0
- package/dist/types/esm/contracts/recipe.d.ts +74 -0
- package/dist/types/esm/contracts/regexSafety.d.ts +31 -0
- package/dist/types/esm/contracts/result.d.ts +503 -0
- package/dist/types/esm/contracts/session.d.ts +51 -0
- package/dist/types/esm/contracts/ssrf.d.ts +16 -0
- package/dist/types/esm/contracts/status.d.ts +27 -0
- package/dist/types/esm/contracts/structured.d.ts +185 -0
- package/dist/types/esm/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/esm/contracts/tokens.d.ts +47 -0
- package/dist/types/esm/index.d.ts +9 -0
- package/dist/types/esm/version.d.ts +8 -0
- package/dist/types/esm/watcher.d.ts +151 -0
- package/package.json +40 -0
|
@@ -0,0 +1,891 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Native REST contract for scrape + crawl.
|
|
3
|
+
*
|
|
4
|
+
* Request fields match the product CLI. Responses are FetchResult /
|
|
5
|
+
* CrawlReport — no second result enum. Types only.
|
|
6
|
+
*/
|
|
7
|
+
import { type CrawlMode, type RobotsOverride } from './compliance.js';
|
|
8
|
+
import type { CrawlError, CrawlPage, CrawlPageList, CrawlReport, SitemapMode } from './crawl.js';
|
|
9
|
+
import { type FetchOptions } from './execution.js';
|
|
10
|
+
import type { FetchResult, FetchWarning, LadderRunAudit, TraceEvent } from './result.js';
|
|
11
|
+
import type { DocumentExtraction, PageMetadata } from './extractor.js';
|
|
12
|
+
import type { EvidenceRecord } from './evidenceRecord.js';
|
|
13
|
+
import type { ScrapeFormat, StructuredExtractionResult } from './structured.js';
|
|
14
|
+
import type { WebhookPayloadFormat } from './delivery.js';
|
|
15
|
+
export declare const CRAWL_MODES: readonly ["research", "standard", "authed"];
|
|
16
|
+
export type ApiCrawlMode = (typeof CRAWL_MODES)[number];
|
|
17
|
+
/** A scrape's deadline when the request sets no `timeout`, and the largest one it may set. */
|
|
18
|
+
export declare const DEFAULT_SCRAPE_TIMEOUT_MS = 300000;
|
|
19
|
+
export declare const MIN_SCRAPE_TIMEOUT_MS = 1000;
|
|
20
|
+
export declare const MAX_WAIT_FOR_MS = 60000;
|
|
21
|
+
/**
|
|
22
|
+
* Per-page capture options shared by scrape, batch and crawl (for batch and
|
|
23
|
+
* crawl they apply to every page). A robots override is never one of them:
|
|
24
|
+
* it names one URL (`ScrapeRequest.robotsOverride`, `BatchStartRequest.robotsOverrides`).
|
|
25
|
+
* Nor are `includeHtml`, `includeRawHtml`, `includeImages`, `attributes` and
|
|
26
|
+
* `screenshot`: the `html`, `rawHtml`, `images`, `attributes` and `screenshot`
|
|
27
|
+
* formats ask for those.
|
|
28
|
+
*/
|
|
29
|
+
export interface PageOptions extends Omit<FetchOptions, 'robotsOverride' | 'includeHtml' | 'includeRawHtml' | 'includeImages' | 'attributes' | 'screenshot' | 'list'> {
|
|
30
|
+
/**
|
|
31
|
+
* The whole scrape's deadline in milliseconds, 1 000 to 300 000; default
|
|
32
|
+
* 300 000. When it fires the result is `partial` with the best content a
|
|
33
|
+
* rung produced so far, or `failed` with `timeout`, never an error.
|
|
34
|
+
*/
|
|
35
|
+
timeout?: number;
|
|
36
|
+
/**
|
|
37
|
+
* The http lane alone, no browser escalation: a page that needs script
|
|
38
|
+
* execution returns the http lane's own verdict (a shell is
|
|
39
|
+
* `failed`/`empty_unverified`, never rendered), the ladder audit records
|
|
40
|
+
* the rungs it dropped (`ladder_channels_filtered`), and `agentHints` says
|
|
41
|
+
* when the http lane asked for the browser lane. Not a lane option: the
|
|
42
|
+
* engine selects channels by it. Default false. A URL the server binds to
|
|
43
|
+
* the browser lane refuses it with HTTP 400.
|
|
44
|
+
*/
|
|
45
|
+
fastMode?: boolean;
|
|
46
|
+
/**
|
|
47
|
+
* Reuse a stored result of this page fetched at most this many
|
|
48
|
+
* milliseconds ago, under the same options, instead of fetching it; 0 to
|
|
49
|
+
* MAX_CACHE_AGE_MS. Default 0: nothing is looked up and the page is
|
|
50
|
+
* fetched. A reused result says so (`cacheState: "hit"`, `cachedAt`) and
|
|
51
|
+
* carries the original fetch's evidence; a looked-up page that had none
|
|
52
|
+
* says `miss`. Not available in mode `authed`.
|
|
53
|
+
*/
|
|
54
|
+
maxAge?: number;
|
|
55
|
+
/**
|
|
56
|
+
* Reuse only a stored result at least this many milliseconds old; 0 to
|
|
57
|
+
* MAX_CACHE_AGE_MS, at most `maxAge`. Without `maxAge` it looks up a
|
|
58
|
+
* result of any age from this one on.
|
|
59
|
+
*/
|
|
60
|
+
minAge?: number;
|
|
61
|
+
/**
|
|
62
|
+
* Store this page's result for later reuse when it succeeds. Default true,
|
|
63
|
+
* except for a request with custom `headers`, which stores only with
|
|
64
|
+
* `true` (the stored trace keeps their values); mode `authed` never stores.
|
|
65
|
+
*/
|
|
66
|
+
storeInCache?: boolean;
|
|
67
|
+
/**
|
|
68
|
+
* Cache only: answer from a stored result and never fetch; a page with
|
|
69
|
+
* none is `failed` with `cache_miss`. `maxAge` and `minAge` still bound
|
|
70
|
+
* the age when given; `maxAge: 0` contradicts it. A crawl in this mode
|
|
71
|
+
* reads no sitemap, so it needs `sitemap: "skip"`.
|
|
72
|
+
*/
|
|
73
|
+
lockdown?: boolean;
|
|
74
|
+
}
|
|
75
|
+
/** The largest `maxAge` or `minAge` a request may set: ten years in milliseconds. */
|
|
76
|
+
export declare const MAX_CACHE_AGE_MS = 315360000000;
|
|
77
|
+
/** The cache options of a request, as PageOptions names them. */
|
|
78
|
+
export type CacheOptions = Pick<PageOptions, 'maxAge' | 'minAge' | 'storeInCache' | 'lockdown'>;
|
|
79
|
+
/**
|
|
80
|
+
* Whether a request looks a page up in the cache: a `maxAge` above 0, a
|
|
81
|
+
* `minAge` or `lockdown`, unless `maxAge` is 0. Otherwise nothing is looked
|
|
82
|
+
* up, and the result carries no `cacheState`.
|
|
83
|
+
*/
|
|
84
|
+
export declare function cacheLookupRequested(options: CacheOptions): boolean;
|
|
85
|
+
/**
|
|
86
|
+
* What the cache did for a result, read from its trace: `hit` with the
|
|
87
|
+
* reused fetch's time after a `cache_hit` event, `miss` after a
|
|
88
|
+
* `cache_miss` event, nothing when the cache was not asked. The last such
|
|
89
|
+
* event decides.
|
|
90
|
+
*/
|
|
91
|
+
export declare function cacheStateOf(trace: readonly TraceEvent[]): Pick<ScrapeMetadata, 'cacheState' | 'cachedAt'>;
|
|
92
|
+
/** The caveats a scrape response may carry for an agent: what to change about the request, in one sentence each. */
|
|
93
|
+
export type AgentHints = readonly string[];
|
|
94
|
+
/**
|
|
95
|
+
* Who a request is from, for W2L's own records only: the scrape record and
|
|
96
|
+
* the task checkpoint carry both, the crawl and batch status reports them as
|
|
97
|
+
* `attribution`, and nothing sent to the target changes. Each is 1 to 100
|
|
98
|
+
* printable ASCII characters without spaces (`^[\x21-\x7e]+$`).
|
|
99
|
+
*/
|
|
100
|
+
export interface RequestAttribution {
|
|
101
|
+
/** The client that made the request: the SDK sends `js-sdk@<version>`, the MCP server `mcp-<client name>@<client version>`. */
|
|
102
|
+
origin?: string;
|
|
103
|
+
/** The caller's own label for the integration or workflow the request belongs to. */
|
|
104
|
+
integration?: string;
|
|
105
|
+
}
|
|
106
|
+
/**
|
|
107
|
+
* The facts of one scrape call, under Firecrawl's names, on the full and
|
|
108
|
+
* compact scrape responses (`metadata`, beside the page's own declarations)
|
|
109
|
+
* and on `/fc`. Nothing here is guessed: `proxyUsed` is the route the
|
|
110
|
+
* answering lane recorded, `timezone` the time zone the browser lane
|
|
111
|
+
* declares (null on the HTTP lane, where none goes on the wire), and the
|
|
112
|
+
* concurrency pair says whether the per-origin ceiling held an attempt back.
|
|
113
|
+
*/
|
|
114
|
+
export interface ScrapeMetadata {
|
|
115
|
+
/** A UUID minted for this call; `GET /v1/scrapes/:id` returns its record. */
|
|
116
|
+
scrapeId: string;
|
|
117
|
+
/** The URL as requested (`requestedUrl`). */
|
|
118
|
+
sourceURL: string;
|
|
119
|
+
/** The final URL, after redirects (`evidence.finalUrl`). */
|
|
120
|
+
url: string;
|
|
121
|
+
/** The status of the response that answered `url` (`evidence.httpStatus`); null when none did. */
|
|
122
|
+
statusCode: number | null;
|
|
123
|
+
/** That response's `content-type` header (`evidence.contentType`); null when there was none. */
|
|
124
|
+
contentType: string | null;
|
|
125
|
+
/** `operator` when the request went through the server's environment proxy, `user` when the compliance record names the caller's own egress, else null. */
|
|
126
|
+
proxyUsed: 'operator' | 'user' | null;
|
|
127
|
+
/** The IANA time zone the browser lane declares; null for an HTTP-lane result. */
|
|
128
|
+
timezone: string | null;
|
|
129
|
+
/** True when the per-origin concurrency ceiling (`W2L_PER_HOST_CONCURRENCY`) held an attempt of this scrape back. */
|
|
130
|
+
concurrencyLimited: boolean;
|
|
131
|
+
/** Milliseconds the attempts of this scrape waited on that ceiling, cooldown and pacing excluded; 0 when none did. */
|
|
132
|
+
concurrencyQueueDurationMs: number;
|
|
133
|
+
/**
|
|
134
|
+
* Whether the cache answered: `hit` when a stored result was reused (the
|
|
135
|
+
* rest of the response is that fetch's), `miss` when one was looked up
|
|
136
|
+
* and none fit. Absent when nothing was looked up (no `maxAge` above 0,
|
|
137
|
+
* no `minAge`, no `lockdown`): never a guessed `miss`.
|
|
138
|
+
*/
|
|
139
|
+
cacheState?: 'hit' | 'miss';
|
|
140
|
+
/** On a hit, when the reused result was fetched (its `evidenceRecord.fetchedAt`). Absent otherwise. */
|
|
141
|
+
cachedAt?: string;
|
|
142
|
+
}
|
|
143
|
+
/** A scrape response's `metadata`: the page's declarations (all null on a page that was not read as content) and the call's facts. */
|
|
144
|
+
export type ScrapeResponseMetadata = PageMetadata & ScrapeMetadata;
|
|
145
|
+
export interface ScrapeRequest extends PageOptions, RequestAttribution {
|
|
146
|
+
url: string;
|
|
147
|
+
mode?: ApiCrawlMode;
|
|
148
|
+
allowlistedDomains?: readonly string[];
|
|
149
|
+
formats?: readonly ScrapeFormat[];
|
|
150
|
+
/** Include outbound links. Kept separate from content formats. */
|
|
151
|
+
includeLinks?: boolean;
|
|
152
|
+
/** Omitted preserves the legacy full REST/SDK response. MCP sends false by default. */
|
|
153
|
+
debug?: boolean;
|
|
154
|
+
/**
|
|
155
|
+
* A recorded decision to fetch this URL although its host's robots.txt
|
|
156
|
+
* disallows it. The reason is required; robots.txt is still read, and the
|
|
157
|
+
* override is reported in the trace, the warnings and, in the browser lane,
|
|
158
|
+
* the compliance record. A hosted server refuses the field
|
|
159
|
+
* (`unsupported_parameter`).
|
|
160
|
+
*/
|
|
161
|
+
robotsOverride?: RobotsOverride;
|
|
162
|
+
/**
|
|
163
|
+
* When W2L is stopped at a check it does not pass (a captcha, a challenge,
|
|
164
|
+
* a login wall), hand the page to the person in their own Chrome and answer
|
|
165
|
+
* with the page they get through to (`true`, or `{ waitMs }`: how long to
|
|
166
|
+
* wait for them, 10 s to 30 min, default 10 min). Offered only by a server
|
|
167
|
+
* on the person's own machine; refused elsewhere, and with `actions` or a
|
|
168
|
+
* screenshot (`unsupported_parameter`).
|
|
169
|
+
*/
|
|
170
|
+
handoff?: {
|
|
171
|
+
waitMs?: number;
|
|
172
|
+
};
|
|
173
|
+
}
|
|
174
|
+
/** A recorded robots override for one URL of a batch. */
|
|
175
|
+
export interface RobotsUrlOverride extends RobotsOverride {
|
|
176
|
+
url: string;
|
|
177
|
+
}
|
|
178
|
+
/** One scrape as the ladder ran it, before the API shapes the response: the result, its routing audit, the request's hints and, once shaped, the `warning` string. */
|
|
179
|
+
export type ScrapeRun = FetchResult & LadderRunAudit & {
|
|
180
|
+
agentHints?: AgentHints;
|
|
181
|
+
warning?: string;
|
|
182
|
+
};
|
|
183
|
+
/** The `warnings` as one string, their messages joined with a space (Firecrawl's `warning`); undefined when there are none. */
|
|
184
|
+
export declare function warningOf(warnings: readonly FetchWarning[] | undefined): string | undefined;
|
|
185
|
+
/**
|
|
186
|
+
* The full scrape response: the run, its `scrapeId`, `metadata` carrying the
|
|
187
|
+
* call's facts beside the page's declarations, and the snapshot and Evidence
|
|
188
|
+
* Record set on every response the API sends (see evidenceRecord.ts).
|
|
189
|
+
*/
|
|
190
|
+
export type ScrapeResponse = ScrapeRun & {
|
|
191
|
+
scrapeId: string;
|
|
192
|
+
metadata: ScrapeResponseMetadata;
|
|
193
|
+
snapshot?: CompactScrapeResponse['snapshot'];
|
|
194
|
+
evidenceRecord?: EvidenceRecord;
|
|
195
|
+
};
|
|
196
|
+
/**
|
|
197
|
+
* What `GET /v1/scrapes/:id` returns: the record of one scrape call, written
|
|
198
|
+
* beside the task root (`scrapes/<scrapeId>.json`) before the response was
|
|
199
|
+
* sent. It holds no page body: the request (header values replaced by their
|
|
200
|
+
* names), who made it, the verdict and the facts of the fetch.
|
|
201
|
+
*/
|
|
202
|
+
export interface ScrapeRecord extends RequestAttribution {
|
|
203
|
+
scrapeId: string;
|
|
204
|
+
/** UTC ISO time the API took the request. */
|
|
205
|
+
requestedAt: string;
|
|
206
|
+
/** The parsed request; each custom header's value is replaced by its name. */
|
|
207
|
+
request: ScrapeRequest;
|
|
208
|
+
status: FetchResult['status'];
|
|
209
|
+
failureReason: FetchResult['failureReason'];
|
|
210
|
+
blockReason: FetchResult['blockReason'];
|
|
211
|
+
budgetExceeded: FetchResult['budgetExceeded'];
|
|
212
|
+
lane: FetchResult['lane'];
|
|
213
|
+
channelsTried: readonly string[];
|
|
214
|
+
metadata: ScrapeResponseMetadata;
|
|
215
|
+
snapshot: CompactScrapeResponse['snapshot'];
|
|
216
|
+
usage: {
|
|
217
|
+
wallMs: number;
|
|
218
|
+
totalMs: number;
|
|
219
|
+
requestCount: number;
|
|
220
|
+
attemptCount: number;
|
|
221
|
+
browserMs: number;
|
|
222
|
+
};
|
|
223
|
+
warnings?: readonly FetchWarning[];
|
|
224
|
+
agentHints?: AgentHints;
|
|
225
|
+
}
|
|
226
|
+
export interface CompactScrapeResponse {
|
|
227
|
+
requestedUrl: string;
|
|
228
|
+
finalUrl: string;
|
|
229
|
+
/**
|
|
230
|
+
* Small capture identity for field audits; the HTML body remains local.
|
|
231
|
+
* `httpStatus` and `contentType` are the status and `content-type` header
|
|
232
|
+
* of the response that answered `finalUrl` (`evidence.httpStatus`,
|
|
233
|
+
* `evidence.contentType`): in the browser lane, of the document the page
|
|
234
|
+
* shows after a script or a meta refresh moved it on.
|
|
235
|
+
*/
|
|
236
|
+
snapshot: {
|
|
237
|
+
rawBodySha256: string | null;
|
|
238
|
+
artifacts: readonly string[];
|
|
239
|
+
httpStatus: number | null;
|
|
240
|
+
contentType: string | null;
|
|
241
|
+
};
|
|
242
|
+
/** The result's Evidence Record v1, the same as on the full response. */
|
|
243
|
+
evidenceRecord: EvidenceRecord;
|
|
244
|
+
status: FetchResult['status'];
|
|
245
|
+
failureReason: FetchResult['failureReason'];
|
|
246
|
+
blockReason: FetchResult['blockReason'];
|
|
247
|
+
budgetExceeded: FetchResult['budgetExceeded'];
|
|
248
|
+
retryAt?: number;
|
|
249
|
+
lane: FetchResult['lane'];
|
|
250
|
+
formats: readonly ('markdown' | 'html' | 'rawHtml' | 'links' | 'json' | 'images' | 'tables' | 'attributes' | 'screenshot' | 'list')[];
|
|
251
|
+
markdown?: string | null;
|
|
252
|
+
/** Present when `html` was asked for, as on the full response; null when the result carries none (a file, a page that was not read as content). */
|
|
253
|
+
html?: string | null;
|
|
254
|
+
/** Present when `rawHtml` was asked for, as on the full response; null when the result carries none. */
|
|
255
|
+
rawHtml?: string | null;
|
|
256
|
+
links?: readonly string[];
|
|
257
|
+
/** Present when `images` was asked for and the page was read as content: every image URL of the whole document, as on the full response. */
|
|
258
|
+
images?: readonly string[];
|
|
259
|
+
/** Present when `tables` was asked for and the page was read as content: its data tables, as on the full response. */
|
|
260
|
+
tables?: FetchResult['tables'];
|
|
261
|
+
/** Present when a `pdf` parser entry asked for `pages` and a PDF's text was read, as on the full response. */
|
|
262
|
+
pages?: FetchResult['pages'];
|
|
263
|
+
/** Present when an `attributes` entry was asked for and the page was read as content, as on the full response. */
|
|
264
|
+
attributes?: FetchResult['attributes'];
|
|
265
|
+
/** Present when a `list` entry was asked for and the page was read: its records. */
|
|
266
|
+
list?: FetchResult['list'];
|
|
267
|
+
/** Present when a `screenshot` entry was asked for, as on the full response: the capture, or null when the browser lane rendered no page or could not capture it. */
|
|
268
|
+
screenshot?: FetchResult['screenshot'];
|
|
269
|
+
/** Present when the request ran `actions`: what the steps produced, and the step that failed if one did. */
|
|
270
|
+
actions?: FetchResult['actions'];
|
|
271
|
+
document?: Pick<DocumentExtraction, 'title' | 'pageType' | 'strategy' | 'confidence' | 'adapter' | 'adapterValidation'> | null;
|
|
272
|
+
/** The call's facts (`scrapeId`, `proxyUsed`, the concurrency pair, ...) and the page's own declarations, as on the full response. */
|
|
273
|
+
metadata: ScrapeResponseMetadata;
|
|
274
|
+
json?: StructuredExtractionResult | null;
|
|
275
|
+
/** The file the response was, as on the full response; absent for a web page. */
|
|
276
|
+
file?: FetchResult['file'];
|
|
277
|
+
/** The fetch's caveats (a recorded robots override, a suspected client-rendered shell, a thin http answer kept), as on the full response; absent when it had none. */
|
|
278
|
+
warnings?: FetchResult['warnings'];
|
|
279
|
+
/** The warnings' messages joined with a space, present exactly when `warnings` is (Firecrawl's `warning`). */
|
|
280
|
+
warning?: string;
|
|
281
|
+
/** Present when the request itself left something on the table (`fastMode` declined a browser hop the http lane asked for), as on the full response. */
|
|
282
|
+
agentHints?: AgentHints;
|
|
283
|
+
/** A page stopped at a check a person can get through, on a server that hands pages to them: why, and how (`handoff: true`), as on the full response. */
|
|
284
|
+
handoff?: FetchResult['handoff'];
|
|
285
|
+
truncated: boolean;
|
|
286
|
+
truncatedAt: number | null;
|
|
287
|
+
usage: FetchResult['usage'] & {
|
|
288
|
+
totalMs: number;
|
|
289
|
+
};
|
|
290
|
+
channelsTried: readonly string[];
|
|
291
|
+
}
|
|
292
|
+
/** The events a job webhook can be sent: Firecrawl's four, plus `cancelled`, which W2L tells apart from `failed`. */
|
|
293
|
+
export declare const WEBHOOK_EVENTS: readonly ["started", "page", "completed", "failed", "cancelled"];
|
|
294
|
+
export type WebhookEvent = (typeof WEBHOOK_EVENTS)[number];
|
|
295
|
+
/** The bounds of a job webhook's configuration. */
|
|
296
|
+
export declare const MAX_WEBHOOK_URL_LENGTH = 2048;
|
|
297
|
+
export declare const MAX_WEBHOOK_HEADERS = 32;
|
|
298
|
+
export declare const MAX_WEBHOOK_HEADERS_BYTES = 8192;
|
|
299
|
+
export declare const MAX_WEBHOOK_METADATA_ENTRIES = 32;
|
|
300
|
+
export declare const MAX_WEBHOOK_METADATA_VALUE_LENGTH = 1000;
|
|
301
|
+
export declare const MAX_WEBHOOK_METADATA_BYTES = 8192;
|
|
302
|
+
/**
|
|
303
|
+
* Where a crawl or batch posts its events (`webhook` on `POST /v1/crawl` and
|
|
304
|
+
* `POST /v1/batches`; a plain string is `{ url }`). Each event is one durable
|
|
305
|
+
* delivery with retries, signed when `secretEnv` names an operator secret;
|
|
306
|
+
* `GET /v1/deliveries?jobId=<taskId>` lists them. The receiver must be https;
|
|
307
|
+
* a local server also takes plain http to a loopback receiver.
|
|
308
|
+
*/
|
|
309
|
+
export interface WebhookConfig {
|
|
310
|
+
/** The receiver: an http(s) URL of at most 2048 characters, without credentials or a fragment. */
|
|
311
|
+
url: string;
|
|
312
|
+
/**
|
|
313
|
+
* Headers sent with every delivery, retries included: at most 32, 8 KiB in
|
|
314
|
+
* all, RFC 7230 token names (lower-cased), values without line breaks.
|
|
315
|
+
* `content-type`, `content-length`, `host`, `connection`,
|
|
316
|
+
* `transfer-encoding` and every `x-w2l-*` name are W2L's and refused by
|
|
317
|
+
* name. Stored in the control database alone, never on the task or in any
|
|
318
|
+
* response, which show their names only; `secretEnv` is the signing path.
|
|
319
|
+
*/
|
|
320
|
+
headers?: Readonly<Record<string, string>>;
|
|
321
|
+
/** Strings echoed as `metadata` in every payload: at most 32, each of at most 1000 characters, 8 KiB in all. */
|
|
322
|
+
metadata?: Readonly<Record<string, string>>;
|
|
323
|
+
/** The events to deliver; default all five. A filtered event is never enqueued. */
|
|
324
|
+
events?: readonly WebhookEvent[];
|
|
325
|
+
/** An operator `W2L_WEBHOOK_SECRET_*` variable whose value signs each delivery (`x-w2l-timestamp`, `x-w2l-signature`); never a literal. */
|
|
326
|
+
secretEnv?: string;
|
|
327
|
+
}
|
|
328
|
+
/** The `webhook` field as a request may write it: a URL string, a configuration object, or null for none. */
|
|
329
|
+
export type WebhookOption = string | WebhookConfig | null;
|
|
330
|
+
/**
|
|
331
|
+
* What a crawl or batch status says about its webhook: the destination
|
|
332
|
+
* (`GET /v1/deliveries?jobId=`), the receiver as origin and path (no query),
|
|
333
|
+
* the events taken, and how its deliveries stand. `pending` counts the
|
|
334
|
+
* deliveries not yet acknowledged, those in flight included.
|
|
335
|
+
*/
|
|
336
|
+
export interface JobWebhookStatus {
|
|
337
|
+
destinationId: string;
|
|
338
|
+
url: string;
|
|
339
|
+
events: readonly WebhookEvent[];
|
|
340
|
+
pending: number;
|
|
341
|
+
delivered: number;
|
|
342
|
+
deadLetter: number;
|
|
343
|
+
}
|
|
344
|
+
export interface CrawlStartRequest extends PageOptions, RequestAttribution {
|
|
345
|
+
url: string;
|
|
346
|
+
mode?: ApiCrawlMode;
|
|
347
|
+
maxPages?: number | null;
|
|
348
|
+
maxDepth?: number | null;
|
|
349
|
+
/**
|
|
350
|
+
* A resume (`POST /v1/crawl/:id/resume`, or a restart) reuses the pages
|
|
351
|
+
* this crawl already fetched instead of fetching them again. It reaches no
|
|
352
|
+
* other request's pages: `maxAge` reuses a stored result of any request.
|
|
353
|
+
*/
|
|
354
|
+
useCached?: boolean;
|
|
355
|
+
allowlistedDomains?: readonly string[];
|
|
356
|
+
/** Formats for every page, validated as for scrape. Omitted selects Markdown. */
|
|
357
|
+
formats?: readonly ScrapeFormat[];
|
|
358
|
+
/** Include each page's outbound links, like a `links` format. */
|
|
359
|
+
includeLinks?: boolean;
|
|
360
|
+
/** Pathname regexes a discovered link must match. The seed URL is always fetched. */
|
|
361
|
+
includePaths?: readonly string[];
|
|
362
|
+
/** Pathname regexes that skip a discovered link; they win over includePaths. */
|
|
363
|
+
excludePaths?: readonly string[];
|
|
364
|
+
/**
|
|
365
|
+
* Match includePaths / excludePaths against a discovered link's canonical
|
|
366
|
+
* URL (scheme, host, path and query) instead of its pathname. Default false.
|
|
367
|
+
*/
|
|
368
|
+
regexOnFullURL?: boolean;
|
|
369
|
+
/**
|
|
370
|
+
* URLs that differ only in their query string are one page: the first
|
|
371
|
+
* variant seen is fetched, later ones are reported as collapsed. Default false.
|
|
372
|
+
*/
|
|
373
|
+
ignoreQueryParameters?: boolean;
|
|
374
|
+
/**
|
|
375
|
+
* `/a` and `/a/`, `/` and `/index.html`, `www.` and the apex, http and https
|
|
376
|
+
* name one page: the first variant seen is fetched, later ones are reported
|
|
377
|
+
* as collapsed. Default true.
|
|
378
|
+
*/
|
|
379
|
+
deduplicateSimilarURLs?: boolean;
|
|
380
|
+
/**
|
|
381
|
+
* Follow links anywhere on the start URL's host. Default false: links on
|
|
382
|
+
* that host are followed only inside the start URL's path subtree.
|
|
383
|
+
*/
|
|
384
|
+
crawlEntireDomain?: boolean;
|
|
385
|
+
/** Follow links to subdomains of the start URL's host (`*.apex`, with one leading `www.` removed). Default false. */
|
|
386
|
+
allowSubdomains?: boolean;
|
|
387
|
+
/** Follow links to any host; cannot be combined with allowlistedDomains. Default false. */
|
|
388
|
+
allowExternalLinks?: boolean;
|
|
389
|
+
/**
|
|
390
|
+
* How the crawl uses the site's sitemap: `include` (default) reads the
|
|
391
|
+
* sitemaps the start URL's robots.txt names, or `/sitemap.xml`, and queues
|
|
392
|
+
* their URLs ahead of the start page's links; `skip` reads none; `only`
|
|
393
|
+
* follows no page link, so the pages are the start URL and the sitemap's
|
|
394
|
+
* entries. Entries pass the same host, subtree, path and depth rules as
|
|
395
|
+
* links. The files read are listed in the report's `discovery.sitemap`.
|
|
396
|
+
*/
|
|
397
|
+
sitemap?: SitemapMode;
|
|
398
|
+
/**
|
|
399
|
+
* Pages this crawl fetches at once, at most: an integer >= 1, refused above
|
|
400
|
+
* the service's worker count. It can only lower the crawl's parallelism; the
|
|
401
|
+
* per-host ceiling and minimum interval still apply. Null or omitted takes
|
|
402
|
+
* the worker count.
|
|
403
|
+
*/
|
|
404
|
+
maxConcurrency?: number | null;
|
|
405
|
+
/**
|
|
406
|
+
* A client-chosen key, 1 to 200 characters without control characters,
|
|
407
|
+
* that makes a retried start return the first start's answer instead of a
|
|
408
|
+
* second job (also the `x-idempotency-key` header on REST). The same key
|
|
409
|
+
* with a different request is HTTP 409 `conflict`. Keys live 24 hours.
|
|
410
|
+
*/
|
|
411
|
+
idempotencyKey?: string;
|
|
412
|
+
/** A receiver for the crawl's events (`started`, one `page` per page recorded, then `completed`, `failed` or `cancelled`); see WebhookConfig. */
|
|
413
|
+
webhook?: WebhookOption;
|
|
414
|
+
}
|
|
415
|
+
/** What the parser hands the engine: the request plus, from the `/fc` shim, the payload shape its receiver expects. */
|
|
416
|
+
export type ParsedCrawlStartRequest = CrawlStartRequest & {
|
|
417
|
+
webhookPayloadFormat?: WebhookPayloadFormat;
|
|
418
|
+
};
|
|
419
|
+
/** A map's `limit` when the request names none, and the largest it may name: Firecrawl's documented default and maximum (read 2026-10-03). */
|
|
420
|
+
export declare const DEFAULT_MAP_LIMIT = 5000;
|
|
421
|
+
export declare const MAX_MAP_LIMIT = 100000;
|
|
422
|
+
/** One deadline for the whole map when the request names no `timeout`; a map answers synchronously. */
|
|
423
|
+
export declare const DEFAULT_MAP_TIMEOUT_MS = 60000;
|
|
424
|
+
export declare const MAX_MAP_TIMEOUT_MS = 300000;
|
|
425
|
+
/** A hosted server's map caps: Firecrawl's default limit, and the default deadline, since a map answers synchronously. */
|
|
426
|
+
export declare const HOSTED_MAP_MAX_LIMIT = 5000;
|
|
427
|
+
export declare const HOSTED_MAP_MAX_TIMEOUT_MS = 60000;
|
|
428
|
+
/** POST /v1/map: the URLs of a site from its sitemaps and its start page's links, without fetching each page. */
|
|
429
|
+
export interface MapRequest extends RequestAttribution {
|
|
430
|
+
url: string;
|
|
431
|
+
/** standard (default) or research; authed is refused: a map reads public sitemaps and one public page. */
|
|
432
|
+
mode?: 'standard' | 'research';
|
|
433
|
+
/** Links returned at most, 1 to MAX_MAP_LIMIT; default DEFAULT_MAP_LIMIT. */
|
|
434
|
+
limit?: number;
|
|
435
|
+
/** Milliseconds for the whole map, 1000 to MAX_MAP_TIMEOUT_MS; default DEFAULT_MAP_TIMEOUT_MS. At the deadline the map answers with what it found. */
|
|
436
|
+
timeout?: number;
|
|
437
|
+
/**
|
|
438
|
+
* Keep only the URLs in which every word (1 to MAP_SEARCH_MAX_WORDS words,
|
|
439
|
+
* 1 to MAP_SEARCH_MAX_CHARS characters, trimmed) appears, case-insensitively,
|
|
440
|
+
* in the percent-decoded URL or the link's title. A filter, not a ranking:
|
|
441
|
+
* discovery order is kept, and it runs before `limit` is counted.
|
|
442
|
+
*/
|
|
443
|
+
search?: string;
|
|
444
|
+
/** How the map uses the site's sitemap: include (default), skip (the start page alone), only (no page body read). */
|
|
445
|
+
sitemap?: SitemapMode;
|
|
446
|
+
/** Admit every host under the start URL's apex (the crawl's allowSubdomains). Default false. */
|
|
447
|
+
includeSubdomains?: boolean;
|
|
448
|
+
/** Fold URLs that differ only in their query string into the first one seen; the returned URL has no query. Default false. */
|
|
449
|
+
ignoreQueryParameters?: boolean;
|
|
450
|
+
/** The crawl's scope options, under their crawl names and rules. */
|
|
451
|
+
includePaths?: readonly string[];
|
|
452
|
+
excludePaths?: readonly string[];
|
|
453
|
+
regexOnFullURL?: boolean;
|
|
454
|
+
crawlEntireDomain?: boolean;
|
|
455
|
+
/** Default true, as on a crawl; a returned http link gives way to its https variant when that comes too, on an origin whose robots.txt the map read anyway and which allows it. */
|
|
456
|
+
deduplicateSimilarURLs?: boolean;
|
|
457
|
+
}
|
|
458
|
+
/** A map's `search`: at most this many characters after trimming, and this many whitespace-separated words. */
|
|
459
|
+
export declare const MAP_SEARCH_MAX_CHARS = 200;
|
|
460
|
+
export declare const MAP_SEARCH_MAX_WORDS = 10;
|
|
461
|
+
export interface CrawlAccepted {
|
|
462
|
+
taskId: string;
|
|
463
|
+
/** Present and true when `idempotencyKey` matched an earlier start and this is its stored answer; nothing was started. */
|
|
464
|
+
replayed?: boolean;
|
|
465
|
+
}
|
|
466
|
+
/** The request headers REST reads an idempotency key from, merged into the body as `idempotencyKey` before parsing. */
|
|
467
|
+
export declare const IDEMPOTENCY_KEY_HEADERS: readonly ["x-idempotency-key", "idempotency-key"];
|
|
468
|
+
export declare const MAX_IDEMPOTENCY_KEY_LENGTH = 200;
|
|
469
|
+
/** The options a running crawl was started with, as `GET /v1/crawl/active` reports them: its task's stored options plus its page budget. */
|
|
470
|
+
export interface ActiveCrawlOptions {
|
|
471
|
+
maxPages: number | null;
|
|
472
|
+
maxDepth: number | null;
|
|
473
|
+
allowlistedDomains: readonly string[];
|
|
474
|
+
includePaths: readonly string[];
|
|
475
|
+
excludePaths: readonly string[];
|
|
476
|
+
useCached: boolean;
|
|
477
|
+
sitemap: SitemapMode;
|
|
478
|
+
ignoreQueryParameters: boolean;
|
|
479
|
+
deduplicateSimilarURLs: boolean;
|
|
480
|
+
crawlEntireDomain: boolean;
|
|
481
|
+
allowSubdomains: boolean;
|
|
482
|
+
allowExternalLinks: boolean;
|
|
483
|
+
regexOnFullURL: boolean;
|
|
484
|
+
maxConcurrency: number | null;
|
|
485
|
+
/** The per-page options every page of the crawl gets: its formats, `includeLinks` and the page options. */
|
|
486
|
+
scrapeOptions: PageOptions & {
|
|
487
|
+
formats: readonly ScrapeFormat[];
|
|
488
|
+
includeLinks: boolean;
|
|
489
|
+
};
|
|
490
|
+
}
|
|
491
|
+
/** One crawl this API process is running (a crawl it resumed at startup included); batches are not listed. */
|
|
492
|
+
export interface ActiveCrawl {
|
|
493
|
+
id: string;
|
|
494
|
+
/** The start URL. */
|
|
495
|
+
url: string;
|
|
496
|
+
status: 'pending' | 'running' | 'paused';
|
|
497
|
+
/** When the latest attempt started; the task's creation time while no attempt has opened yet. */
|
|
498
|
+
startedAt: string;
|
|
499
|
+
/** The latest attempt's pages so far. */
|
|
500
|
+
pagesFetched: number;
|
|
501
|
+
options: ActiveCrawlOptions;
|
|
502
|
+
}
|
|
503
|
+
/** `GET /v1/crawl/active`: always 200, with an empty list when nothing runs. */
|
|
504
|
+
export interface ActiveCrawlList {
|
|
505
|
+
crawls: readonly ActiveCrawl[];
|
|
506
|
+
}
|
|
507
|
+
export interface BatchStartRequest extends PageOptions, RequestAttribution {
|
|
508
|
+
urls: readonly string[];
|
|
509
|
+
mode?: ApiCrawlMode;
|
|
510
|
+
formats?: readonly ScrapeFormat[];
|
|
511
|
+
includeLinks?: boolean;
|
|
512
|
+
/** Recorded robots overrides, each for one URL of `urls`. A hosted server refuses the field (`unsupported_parameter`). */
|
|
513
|
+
robotsOverrides?: readonly RobotsUrlOverride[];
|
|
514
|
+
/**
|
|
515
|
+
* Pages of this batch in flight at once, at most: an integer from 1 to 4.
|
|
516
|
+
* It only lowers the service's worker count (4 locally, 2 on the hosted
|
|
517
|
+
* MCP host); the per-host ceiling and minimum interval still apply.
|
|
518
|
+
* Omitted takes the worker count. Stored with the task, so a resumed batch
|
|
519
|
+
* runs under the same cap.
|
|
520
|
+
*/
|
|
521
|
+
maxConcurrency?: number;
|
|
522
|
+
/**
|
|
523
|
+
* Start the batch with the entries of `urls` that are http(s) URLs and
|
|
524
|
+
* report the rest as `invalidURLs` instead of refusing the request. A
|
|
525
|
+
* non-string entry is still refused (`urls[i] must be a string`), and so
|
|
526
|
+
* is a duplicate: neither is an invalid URL. Default false.
|
|
527
|
+
*/
|
|
528
|
+
ignoreInvalidURLs?: boolean;
|
|
529
|
+
/**
|
|
530
|
+
* Firecrawl's extract scope flags, accepted in their no-op form only: a
|
|
531
|
+
* batch fetches exactly the URLs given and follows no link, which is what
|
|
532
|
+
* `false` says. `true` is refused with HTTP 400 naming the alternative (a
|
|
533
|
+
* crawl's `allowExternalLinks` / `allowSubdomains`; extraction across
|
|
534
|
+
* links is the M5 multi-URL extract).
|
|
535
|
+
*/
|
|
536
|
+
allowExternalLinks?: false;
|
|
537
|
+
includeSubdomains?: false;
|
|
538
|
+
/**
|
|
539
|
+
* A client-chosen key, 1 to 200 characters without control characters,
|
|
540
|
+
* that makes a retried submission return the first one's answer (with
|
|
541
|
+
* `replayed: true`) instead of a second job; also the `x-idempotency-key`
|
|
542
|
+
* header on REST. The same key with a different request is HTTP 409
|
|
543
|
+
* `conflict`. Keys live 24 hours, per task root.
|
|
544
|
+
*/
|
|
545
|
+
idempotencyKey?: string;
|
|
546
|
+
/**
|
|
547
|
+
* The id of an existing batch to add `urls` to instead of starting a new
|
|
548
|
+
* job. The body may then carry only `urls`, `ignoreInvalidURLs`,
|
|
549
|
+
* `idempotencyKey`, `robotsOverrides` and the attribution labels: the job's
|
|
550
|
+
* `mode`, `formats`, `includeLinks`, `maxConcurrency` and page options
|
|
551
|
+
* stay as they were (`appendToId keeps the job's options; <key> cannot be
|
|
552
|
+
* changed`). The appended URLs go to the end of the job's list, in order.
|
|
553
|
+
*/
|
|
554
|
+
appendToId?: string;
|
|
555
|
+
/** A receiver for the batch's events (`started`, one `page` per item recorded, then `completed`, `failed` or `cancelled`); see WebhookConfig. */
|
|
556
|
+
webhook?: WebhookOption;
|
|
557
|
+
}
|
|
558
|
+
/** What the parser hands the engine: the request plus, when `ignoreInvalidURLs` was on, the entries it skipped (possibly none), and from a shim the payload shape its receiver expects. */
|
|
559
|
+
export type ParsedBatchStartRequest = BatchStartRequest & {
|
|
560
|
+
invalidURLs?: readonly string[];
|
|
561
|
+
webhookPayloadFormat?: WebhookPayloadFormat;
|
|
562
|
+
};
|
|
563
|
+
/**
|
|
564
|
+
* `POST /v1/batches` 202: the task id and, when `ignoreInvalidURLs` was on,
|
|
565
|
+
* the entries skipped, possibly none. An append answers with the job's id,
|
|
566
|
+
* `requested` (the job's URLs after the append) and `appended` (the URLs this
|
|
567
|
+
* request added); a replayed submission carries `replayed: true`.
|
|
568
|
+
*/
|
|
569
|
+
export interface BatchAccepted extends CrawlAccepted {
|
|
570
|
+
invalidURLs?: string[];
|
|
571
|
+
requested?: number;
|
|
572
|
+
appended?: number;
|
|
573
|
+
}
|
|
574
|
+
export interface BatchStatusResponse extends CrawlReport {
|
|
575
|
+
requested: number;
|
|
576
|
+
completed: number;
|
|
577
|
+
remaining: number;
|
|
578
|
+
/** Items recorded `success`, `partial` or `empty_verified` (a page read, with or without content), every attempt counted. */
|
|
579
|
+
succeeded: number;
|
|
580
|
+
/** Items the errors report lists: `failed`, `blocked`, `cancelled` or `budget_exceeded`, every attempt counted. With one step per URL, `completed` is `succeeded + failed`. */
|
|
581
|
+
failed: number;
|
|
582
|
+
/** The cap in force: the request's `maxConcurrency` or the service's worker count, whichever is lower. */
|
|
583
|
+
maxConcurrency: number;
|
|
584
|
+
/** The entries `ignoreInvalidURLs` skipped at submission; present exactly when the option was on. */
|
|
585
|
+
invalidURLs?: readonly string[];
|
|
586
|
+
/** Items stopped at a check a person can get through in their own Chrome (`POST /v1/batches/:id/handoff`); present on a server that offers the handoff. */
|
|
587
|
+
waitingForPerson?: number;
|
|
588
|
+
}
|
|
589
|
+
/**
|
|
590
|
+
* The checks a batch item can be handed to a person for, and the routing
|
|
591
|
+
* reason each is handed over as: a captcha, a bot check or challenge, a
|
|
592
|
+
* login wall. A rate limit or a region block is not something a person gets
|
|
593
|
+
* through in a browser.
|
|
594
|
+
*/
|
|
595
|
+
export declare const HANDOFF_REASONS: Readonly<Record<string, 'captcha_required' | 'bot_gate' | 'login_required'>>;
|
|
596
|
+
/**
|
|
597
|
+
* `POST /v1/logins/import`: save the person's login to `site` (a domain or a
|
|
598
|
+
* page URL) from the Chrome they use, as `octocrawl login import` does, on a server
|
|
599
|
+
* on their machine. `approveTimeoutMs`: how long to wait for them to click
|
|
600
|
+
* Allow in Chrome, 10 s to 10 min; default 2 min.
|
|
601
|
+
*/
|
|
602
|
+
export interface LoginImportRequest {
|
|
603
|
+
site: string;
|
|
604
|
+
approveTimeoutMs?: number;
|
|
605
|
+
}
|
|
606
|
+
export declare function parseLoginImportRequest(body: unknown): LoginImportRequest;
|
|
607
|
+
/** The localStorage a saved login holds: the origins, and how many items in all; never a value. */
|
|
608
|
+
export interface LoginStorage {
|
|
609
|
+
origins: string[];
|
|
610
|
+
itemCount: number;
|
|
611
|
+
}
|
|
612
|
+
/** A login saved for a domain: never its cookies or storage values, only how many and the hash a record names it by. */
|
|
613
|
+
export interface SavedLogin {
|
|
614
|
+
domain: string;
|
|
615
|
+
savedAt: string;
|
|
616
|
+
cookieCount: number;
|
|
617
|
+
/** The localStorage saved with it, read from the site's tabs open in Chrome when it was imported; null for none. */
|
|
618
|
+
localStorage: LoginStorage | null;
|
|
619
|
+
sessionSha256: string;
|
|
620
|
+
}
|
|
621
|
+
/** What an import saved, and whether the site's localStorage was read: false when no tab of the site was open in Chrome. */
|
|
622
|
+
export interface LoginImportResponse extends SavedLogin {
|
|
623
|
+
localStorageRead: boolean;
|
|
624
|
+
/** The origins of the site's open tabs whose localStorage Chrome did not give (a tab that crashed or was discarded): saved without it. */
|
|
625
|
+
localStorageUnread: string[];
|
|
626
|
+
/** Why each of localStorageUnread was not read, a tab at a time: the request to Chrome that failed (`Target.attachToTarget`, `Page.getFrameTree` or `DOMStorage.getDOMStorageItems`) and Chrome's answer, or the wait that ran out. */
|
|
627
|
+
localStorageUnreadReasons: {
|
|
628
|
+
origin: string;
|
|
629
|
+
step: string;
|
|
630
|
+
error: string;
|
|
631
|
+
}[];
|
|
632
|
+
}
|
|
633
|
+
/** `POST /v1/batches/:id/handoff`: how long to wait for the person on each page, 10 s to 30 min; default 10 min. */
|
|
634
|
+
export interface BatchHandoffRequest {
|
|
635
|
+
waitMs?: number;
|
|
636
|
+
}
|
|
637
|
+
export declare const MAX_HANDOFF_WAIT_MS = 1800000;
|
|
638
|
+
export declare function parseBatchHandoffRequest(body: unknown): BatchHandoffRequest;
|
|
639
|
+
/**
|
|
640
|
+
* What a handoff did: each item it handed over, in order, and whether the
|
|
641
|
+
* person got it through. An item that was through is read in their browser
|
|
642
|
+
* and its result replaces the stopped one (`status`, the item's new one); an
|
|
643
|
+
* item that was not (they did not get through in time, closed its tab, or it
|
|
644
|
+
* ended off its site) keeps its stopped result, with `reason` saying why.
|
|
645
|
+
*/
|
|
646
|
+
export interface BatchHandoffResponse {
|
|
647
|
+
id: string;
|
|
648
|
+
handedOff: number;
|
|
649
|
+
through: number;
|
|
650
|
+
notThrough: number;
|
|
651
|
+
items: Array<{
|
|
652
|
+
id: string;
|
|
653
|
+
url: string;
|
|
654
|
+
through: boolean;
|
|
655
|
+
status: string;
|
|
656
|
+
reason?: string;
|
|
657
|
+
}>;
|
|
658
|
+
}
|
|
659
|
+
/** The step statuses `GET /v1/batches/:id/errors` lists: the same ones `/v1/crawl/:id/errors` does. */
|
|
660
|
+
export declare const BATCH_ERROR_STATUSES: readonly ["failed", "blocked", "cancelled", "budget_exceeded"];
|
|
661
|
+
export type BatchErrorStatus = (typeof BATCH_ERROR_STATUSES)[number];
|
|
662
|
+
/**
|
|
663
|
+
* One item of a batch that did not succeed, under Firecrawl's names (`id`,
|
|
664
|
+
* `timestamp`, `url`, `code`, `error`) with W2L's status vocabulary beside
|
|
665
|
+
* them: `status` and `code` (the `failureReason`, `blockReason` or
|
|
666
|
+
* `budgetExceeded` the status carries, else the status itself) are the same
|
|
667
|
+
* values the item on `/items` carries.
|
|
668
|
+
*/
|
|
669
|
+
export interface BatchErrorItem {
|
|
670
|
+
/** The item's step id, as on `GET /v1/batches/:id/items`. */
|
|
671
|
+
id: string;
|
|
672
|
+
/** When the item was recorded (its `createdAt`). */
|
|
673
|
+
timestamp: string;
|
|
674
|
+
url: string;
|
|
675
|
+
status: BatchErrorStatus;
|
|
676
|
+
code: string;
|
|
677
|
+
/**
|
|
678
|
+
* A sentence for a reader: the result's first warning when it has one,
|
|
679
|
+
* else `<status>: <code>`, with ` (HTTP <n>)` when the status is known and,
|
|
680
|
+
* for a robots.txt refusal, the rule that applied.
|
|
681
|
+
*/
|
|
682
|
+
error: string;
|
|
683
|
+
httpStatus: number | null;
|
|
684
|
+
}
|
|
685
|
+
/** `GET /v1/batches/:id/errors`: the batch's errors across every attempt, one page at a time, and the URLs robots.txt refused (every attempt, not paginated). */
|
|
686
|
+
export interface BatchErrorsResponse {
|
|
687
|
+
errors: BatchErrorItem[];
|
|
688
|
+
/**
|
|
689
|
+
* URLs of the batch whose result is `policy_denied` by a `robots_disallowed`
|
|
690
|
+
* trace event that no recorded override set aside. A governance or SSRF
|
|
691
|
+
* refusal is `policy_denied` too but is not robots.txt, and stays out of
|
|
692
|
+
* this list (it is still in `errors`).
|
|
693
|
+
*/
|
|
694
|
+
robotsBlocked: string[];
|
|
695
|
+
nextCursor: string | null;
|
|
696
|
+
hasMore: boolean;
|
|
697
|
+
}
|
|
698
|
+
/** The page of errors asked for: `limit` 1 to 1000 (default 1000: a batch has at most 1000 URLs and errors carry no bodies). */
|
|
699
|
+
export interface BatchErrorsQuery {
|
|
700
|
+
cursor?: string;
|
|
701
|
+
limit?: number;
|
|
702
|
+
}
|
|
703
|
+
export declare const BATCH_ERRORS_MAX_LIMIT = 1000;
|
|
704
|
+
export type CrawlStatusResponse = CrawlReport;
|
|
705
|
+
export interface CrawlPageQuery {
|
|
706
|
+
attemptId?: string;
|
|
707
|
+
cursor?: string;
|
|
708
|
+
limit?: number;
|
|
709
|
+
debug?: boolean;
|
|
710
|
+
/** List the pages whose content repeated an earlier page's (status `duplicate`) too; left out by default. */
|
|
711
|
+
includeDuplicates?: boolean;
|
|
712
|
+
}
|
|
713
|
+
export type CrawlPagesResponse = CrawlPageList<CrawlPage>;
|
|
714
|
+
export type CrawlErrorsResponse = CrawlPageList<CrawlError>;
|
|
715
|
+
/**
|
|
716
|
+
* The events a job stream sends (`GET /v1/crawl/:id/events`,
|
|
717
|
+
* `GET /v1/batches/:id/events` as server-sent events, and the same routes'
|
|
718
|
+
* `/ws` as WebSocket frames): `catchup` with the job's report as the stream
|
|
719
|
+
* opens, one `document` per page recorded (the compact page the items routes
|
|
720
|
+
* list, with the step cursor the listing routes take, so `after=<cursor>` or
|
|
721
|
+
* `Last-Event-ID` resumes a stream), `snapshot` with the report after each
|
|
722
|
+
* page, `done` with the terminal report, `error` with `{ code, message }`.
|
|
723
|
+
* A document is sent once per step id on one stream; a client that resumes
|
|
724
|
+
* or switches transports deduplicates by it.
|
|
725
|
+
*/
|
|
726
|
+
export declare const JOB_STREAM_EVENTS: readonly ["catchup", "document", "snapshot", "done", "error"];
|
|
727
|
+
export type JobStreamEventType = (typeof JOB_STREAM_EVENTS)[number];
|
|
728
|
+
export type JobStreamReport = CrawlReport | BatchStatusResponse;
|
|
729
|
+
export type JobStreamFrame = {
|
|
730
|
+
type: 'catchup';
|
|
731
|
+
data: JobStreamReport;
|
|
732
|
+
} | {
|
|
733
|
+
type: 'document';
|
|
734
|
+
data: CrawlPage;
|
|
735
|
+
cursor: string;
|
|
736
|
+
} | {
|
|
737
|
+
type: 'snapshot';
|
|
738
|
+
data: JobStreamReport;
|
|
739
|
+
} | {
|
|
740
|
+
type: 'done';
|
|
741
|
+
data: JobStreamReport;
|
|
742
|
+
} | {
|
|
743
|
+
type: 'error';
|
|
744
|
+
error: {
|
|
745
|
+
code: string;
|
|
746
|
+
message: string;
|
|
747
|
+
};
|
|
748
|
+
};
|
|
749
|
+
/**
|
|
750
|
+
* The WebSocket subprotocol a client presents its bearer token in, since the
|
|
751
|
+
* WebSocket API sets no headers: `w2l.token.<token>`, echoed back as the
|
|
752
|
+
* selected protocol. A token must then be made of the characters a
|
|
753
|
+
* subprotocol name allows (RFC 6455 token characters); the SDK falls back to
|
|
754
|
+
* the SSE route, which carries the Authorization header, for any other token.
|
|
755
|
+
*/
|
|
756
|
+
export declare const WS_TOKEN_PROTOCOL_PREFIX = "w2l.token.";
|
|
757
|
+
export declare function isApiCrawlMode(value: string): value is ApiCrawlMode;
|
|
758
|
+
export declare function defaultApiMode(mode: CrawlMode | undefined): ApiCrawlMode;
|
|
759
|
+
/**
|
|
760
|
+
* The one set of request-error codes, shared by the REST API, /fc, the SDK and
|
|
761
|
+
* MCP. A request error means W2L refused or failed the request itself; what
|
|
762
|
+
* happened to a fetched page is its result status (blocked, failed, ...), not
|
|
763
|
+
* one of these.
|
|
764
|
+
*/
|
|
765
|
+
export declare const API_ERROR_CODES: readonly ["invalid_json", "invalid_request", "unsupported_parameter", "unsupported_format", "unauthorized", "not_found", "conflict", "internal_error"];
|
|
766
|
+
export type ApiErrorCode = (typeof API_ERROR_CODES)[number];
|
|
767
|
+
/** The HTTP status each code is returned with. */
|
|
768
|
+
export declare const API_ERROR_STATUS: Readonly<Record<ApiErrorCode, 400 | 401 | 404 | 409 | 500>>;
|
|
769
|
+
export declare function isApiErrorCode(value: unknown): value is ApiErrorCode;
|
|
770
|
+
/** What a request named that W2L cannot honour, as sent (for example `scrapeOptions.actions`). */
|
|
771
|
+
export interface ApiErrorDetails {
|
|
772
|
+
parameters?: readonly string[];
|
|
773
|
+
formats?: readonly string[];
|
|
774
|
+
}
|
|
775
|
+
/** Native error body. /fc sends the same fields after `success: false`, with the hints as `agent_hints`. */
|
|
776
|
+
export interface ApiErrorBody {
|
|
777
|
+
error: string;
|
|
778
|
+
code: ApiErrorCode;
|
|
779
|
+
details?: ApiErrorDetails;
|
|
780
|
+
/** For a refused option W2L does not offer: the supported route, one sentence each. */
|
|
781
|
+
agentHints?: AgentHints;
|
|
782
|
+
}
|
|
783
|
+
/**
|
|
784
|
+
* The code of the one answer that is not a request error: the request was
|
|
785
|
+
* well formed, and the caller is over the server's per-minute budget
|
|
786
|
+
* (HTTP 429, `Retry-After`). It is not one of API_ERROR_CODES, which name
|
|
787
|
+
* what was wrong with a request.
|
|
788
|
+
*/
|
|
789
|
+
export declare const RATE_LIMITED_CODE: "rate_limited";
|
|
790
|
+
export declare const RATE_LIMITED_STATUS = 429;
|
|
791
|
+
/** The native 429 body; /fc sends `{ success: false, error, code, agent_hints }` with the same header. */
|
|
792
|
+
export interface RateLimitedBody {
|
|
793
|
+
error: string;
|
|
794
|
+
code: typeof RATE_LIMITED_CODE;
|
|
795
|
+
/** Seconds until the window admits a request again, at least 1; also the `Retry-After` header. */
|
|
796
|
+
retryAfterSeconds: number;
|
|
797
|
+
agentHints: AgentHints;
|
|
798
|
+
}
|
|
799
|
+
export declare function rateLimitedBody(perMinute: number, retryAfterSeconds: number): RateLimitedBody;
|
|
800
|
+
export declare class RequestError extends Error {
|
|
801
|
+
readonly code: 'invalid_request' | 'unsupported_parameter' | 'unsupported_format';
|
|
802
|
+
readonly details?: ApiErrorDetails | undefined;
|
|
803
|
+
/** The supported route, when the refused option is one W2L does not offer. */
|
|
804
|
+
readonly agentHints?: AgentHints | undefined;
|
|
805
|
+
readonly status = 400;
|
|
806
|
+
constructor(message: string, code?: 'invalid_request' | 'unsupported_parameter' | 'unsupported_format', details?: ApiErrorDetails | undefined,
|
|
807
|
+
/** The supported route, when the refused option is one W2L does not offer. */
|
|
808
|
+
agentHints?: AgentHints | undefined);
|
|
809
|
+
}
|
|
810
|
+
/** The hints a refusal carries for the options W2L does not offer: the next honest step, never a way around the refusal. */
|
|
811
|
+
export declare const REFUSAL_HINTS: {
|
|
812
|
+
readonly stealth: "W2L does not offer a stealth mode or stealth proxies; a proxy or session you own (mode authed) is the supported route";
|
|
813
|
+
readonly ignoreRobotsTxt: "robots.txt is always read; a robotsOverride with a recorded reason fetches one URL past its rule, on the record";
|
|
814
|
+
readonly hostedSkipTlsVerification: "a hosted server verifies every certificate; run W2L locally to use skipTlsVerification, which is recorded in the trace and a tls_unverified warning";
|
|
815
|
+
readonly useIndex: "W2L keeps no URL index: a map reads the sitemaps the site declares and its start page, on the record; crawl reads further pages";
|
|
816
|
+
readonly actions: "actions run on scrape and batch, where each page named gets the same steps; a crawl or a map does not take them";
|
|
817
|
+
};
|
|
818
|
+
/** The hint for a refused request key, or null when the key has none (an option W2L simply does not know). */
|
|
819
|
+
export declare function refusalHint(key: string, value: unknown): string | null;
|
|
820
|
+
export declare const PAGE_KEYS: readonly ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown"];
|
|
821
|
+
export declare const ATTRIBUTION_KEYS: readonly ["origin", "integration"];
|
|
822
|
+
export declare const SCRAPE_KEYS: readonly ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
823
|
+
export declare const CRAWL_SCOPE_KEYS: readonly ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
|
|
824
|
+
export declare const CRAWL_KEYS: readonly ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", "regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks", "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
825
|
+
export declare const BATCH_KEYS: readonly ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
826
|
+
/** What a batch body may carry beside `appendToId`: the job's own options are not among them (the scope no-ops change nothing, so they may come along). */
|
|
827
|
+
export declare const BATCH_APPEND_KEYS: readonly ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", "origin", "integration"];
|
|
828
|
+
/** The scope options a map takes under their crawl names; allowSubdomains is includeSubdomains on a map, and allowExternalLinks is not offered. */
|
|
829
|
+
export declare const MAP_SCOPE_KEYS: readonly ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
|
|
830
|
+
/** What a map takes. No page option (headers, mobile, skipTlsVerification, formats, ...): a map has nothing to loosen. */
|
|
831
|
+
export declare const MAP_KEYS: readonly ["url", "mode", "limit", "timeout", "search", "sitemap", "includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs", "includePaths", "excludePaths", "origin", "integration"];
|
|
832
|
+
/** Why a webhook header (lower-cased name) cannot be sent, or null when it can: W2L's own and the transport's names are reserved. */
|
|
833
|
+
export declare function webhookHeaderRefusal(name: string): string | null;
|
|
834
|
+
/**
|
|
835
|
+
* The `webhook` of a crawl or batch request: a URL string is `{ url }`, null
|
|
836
|
+
* or undefined is none, an object takes url, headers, metadata, events and
|
|
837
|
+
* secretEnv and nothing else (`unknown webhook option: <key>`). The URL must
|
|
838
|
+
* be http(s) of at most 2048 characters without credentials or a fragment;
|
|
839
|
+
* whether http is admitted (a loopback receiver of a local service) and
|
|
840
|
+
* whether a hosted server takes the address is the engine's, mode-aware check.
|
|
841
|
+
*/
|
|
842
|
+
export declare function readWebhook(value: unknown): WebhookConfig | undefined;
|
|
843
|
+
/** The smallest window a screenshot may ask for; the largest is the declared screen (`BROWSER_FINGERPRINT.screen`, 1920x1080). */
|
|
844
|
+
export declare const MIN_SCREENSHOT_VIEWPORT: {
|
|
845
|
+
readonly width: 320;
|
|
846
|
+
readonly height: 240;
|
|
847
|
+
};
|
|
848
|
+
/** Largest number of custom request headers, and the longest value, a request may carry. */
|
|
849
|
+
export declare const MAX_REQUEST_HEADERS = 32;
|
|
850
|
+
export declare const MAX_REQUEST_HEADER_VALUE_LENGTH = 4096;
|
|
851
|
+
/**
|
|
852
|
+
* Why a request header (lower-cased name) cannot be sent on a caller's
|
|
853
|
+
* behalf, or null when it can. The User-Agent and the client hints are
|
|
854
|
+
* W2L's declared identity, which no option overrides; credentials belong to
|
|
855
|
+
* the authed session path, which the record names; transport headers are
|
|
856
|
+
* the lane's. The lanes apply the same rule to what reaches them.
|
|
857
|
+
*/
|
|
858
|
+
export declare function headerRefusal(name: string): string | null;
|
|
859
|
+
/**
|
|
860
|
+
* Custom request headers: an object of at most 32 string values, each name
|
|
861
|
+
* an RFC 7230 token given once (lower-cased here), each value at most 4096
|
|
862
|
+
* characters without CR, LF or NUL. A name the lane cannot send on the
|
|
863
|
+
* caller's behalf (headerRefusal) is refused by name.
|
|
864
|
+
*/
|
|
865
|
+
export declare function readHeaders(value: unknown, name?: string): Readonly<Record<string, string>> | undefined;
|
|
866
|
+
export declare function parseScrapeRequest(body: unknown): ScrapeRequest;
|
|
867
|
+
export declare function parseCrawlStartRequest(body: unknown): CrawlStartRequest;
|
|
868
|
+
/**
|
|
869
|
+
* A map request: url, mode (standard or research), limit, timeout, search,
|
|
870
|
+
* sitemap, includeSubdomains, the crawl's scope options under their crawl
|
|
871
|
+
* names (ignoreQueryParameters, includePaths, excludePaths, regexOnFullURL,
|
|
872
|
+
* crawlEntireDomain, deduplicateSimilarURLs), origin and integration.
|
|
873
|
+
* Anything else is refused by name, `useIndex` with the supported route;
|
|
874
|
+
* nothing is silently ignored.
|
|
875
|
+
*/
|
|
876
|
+
export declare function parseMapRequest(body: unknown): MapRequest;
|
|
877
|
+
/**
|
|
878
|
+
* A batch entry that is not an http(s) URL is refused by its index
|
|
879
|
+
* (`urls[2] must be http(s)`), or with `ignoreInvalidURLs` collected into
|
|
880
|
+
* `invalidURLs` instead; an entry that is not a string is refused either
|
|
881
|
+
* way. The 1..1000 cap counts the submitted entries; `requested` later
|
|
882
|
+
* counts the valid ones. With `appendToId` the body may carry only the
|
|
883
|
+
* entries to add and what binds to them: an option of the job itself is
|
|
884
|
+
* refused by name (`appendToId keeps the job's options; formats cannot be
|
|
885
|
+
* changed`).
|
|
886
|
+
*/
|
|
887
|
+
export declare function parseBatchStartRequest(body: unknown): ParsedBatchStartRequest;
|
|
888
|
+
/** `GET /v1/batches/:id/errors`: `cursor` and `limit` (1 to 1000) from the query string. */
|
|
889
|
+
export declare function parseBatchErrorsQuery(query: Record<string, string | undefined>): BatchErrorsQuery;
|
|
890
|
+
export declare function parseCrawlPageQuery(query: Record<string, string | undefined>): CrawlPageQuery;
|
|
891
|
+
//# sourceMappingURL=api.d.ts.map
|