@octocrawl/sdk 0.3.1 → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/index.cjs +11 -6
- package/dist/index.js +11 -6
- package/dist/types/cjs/client.d.ts +1 -1
- package/dist/types/cjs/contracts/actions.d.ts +37 -4
- package/dist/types/cjs/contracts/api.d.ts +40 -3
- package/dist/types/cjs/contracts/checkpoint.d.ts +15 -1
- package/dist/types/cjs/contracts/evidenceRecord.d.ts +100 -1
- package/dist/types/cjs/contracts/execution.d.ts +68 -0
- package/dist/types/cjs/contracts/firecrawl.d.ts +1 -1
- package/dist/types/cjs/contracts/result.d.ts +22 -1
- package/dist/types/cjs/contracts/status.d.ts +1 -1
- package/dist/types/cjs/version.d.ts +1 -1
- package/dist/types/esm/client.d.ts +1 -1
- package/dist/types/esm/contracts/actions.d.ts +37 -4
- package/dist/types/esm/contracts/api.d.ts +40 -3
- package/dist/types/esm/contracts/checkpoint.d.ts +15 -1
- package/dist/types/esm/contracts/evidenceRecord.d.ts +100 -1
- package/dist/types/esm/contracts/execution.d.ts +68 -0
- package/dist/types/esm/contracts/firecrawl.d.ts +1 -1
- package/dist/types/esm/contracts/result.d.ts +22 -1
- package/dist/types/esm/contracts/status.d.ts +1 -1
- package/dist/types/esm/version.d.ts +1 -1
- package/package.json +19 -2
package/README.md
CHANGED
|
@@ -14,4 +14,4 @@ const { report, items } = await client.batchAndWait(['https://example.com/a', 'h
|
|
|
14
14
|
|
|
15
15
|
Start a local API with `npx octocrawl serve`. Requires a runtime with `fetch` (Node.js 18 or later, browsers, Deno, Bun).
|
|
16
16
|
|
|
17
|
-
Licence: MIT. Source and the API reference: https://github.com/77777R7/Octocrawl
|
|
17
|
+
Licence: MIT. Website and docs: https://octocrawl.dev. Source and the API reference: https://github.com/77777R7/Octocrawl
|
package/dist/index.cjs
CHANGED
|
@@ -216,10 +216,10 @@ function isApiErrorCode(value) {
|
|
|
216
216
|
var RATE_LIMITED_CODE = "rate_limited";
|
|
217
217
|
var PAGE_KEYS = ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown"];
|
|
218
218
|
var ATTRIBUTION_KEYS = ["origin", "integration"];
|
|
219
|
-
var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
219
|
+
var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "lane", "access", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
220
220
|
var CRAWL_SCOPE_KEYS = ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
|
|
221
|
-
var CRAWL_KEYS = ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
222
|
-
var BATCH_KEYS = ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
221
|
+
var CRAWL_KEYS = ["url", "mode", "access", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
222
|
+
var BATCH_KEYS = ["urls", "mode", "lane", "access", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
223
223
|
var BATCH_APPEND_KEYS = ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", ...ATTRIBUTION_KEYS];
|
|
224
224
|
var MAP_SCOPE_KEYS = ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
|
|
225
225
|
var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", "ignoreRobotsTxt", ...ATTRIBUTION_KEYS];
|
|
@@ -254,11 +254,16 @@ var EVIDENCE_RECORD_KEYS = {
|
|
|
254
254
|
pageActions: keysOf()(["steps", "scriptRan"]),
|
|
255
255
|
pageActionStep: keysOf()(["type", "outcome"]),
|
|
256
256
|
requestHeader: keysOf()(["name", "valueSha256"]),
|
|
257
|
-
access: keysOf()(["route", "executor", "executorVersion", "profile", "externalCostUsd"])
|
|
257
|
+
access: keysOf()(["route", "executor", "executorVersion", "profile", "externalCostUsd", "completion", "egress", "session", "paidCalls", "grant"]),
|
|
258
|
+
accessEgress: keysOf()(["proxy", "source", "switchedFrom", "exit"]),
|
|
259
|
+
accessEgressExit: keysOf()(["ip", "country", "observedAt"]),
|
|
260
|
+
accessSession: keysOf()(["id"]),
|
|
261
|
+
accessPaidCall: keysOf()(["provider", "rung", "capabilities", "ceilingUsd", "chargedUsd", "reportedCostUsd", "outcome", "reason", "answer"]),
|
|
262
|
+
accessGrant: keysOf()(["sha256", "tier", "attestedAt"])
|
|
258
263
|
};
|
|
259
264
|
|
|
260
265
|
// packages/sdk/src/version.ts
|
|
261
|
-
var SDK_VERSION = "0.3.
|
|
266
|
+
var SDK_VERSION = "0.3.2";
|
|
262
267
|
|
|
263
268
|
// packages/sdk/src/watcher.ts
|
|
264
269
|
var DEFAULT_WATCH_POLL_INTERVAL_MS = 2e3;
|
|
@@ -791,7 +796,7 @@ var W2L = class {
|
|
|
791
796
|
}
|
|
792
797
|
async scrape(url, opts = {}, request = {}) {
|
|
793
798
|
const deadlineMs = Number.isInteger(opts.timeout) ? Math.min(Math.max(opts.timeout, 0), DEFAULT_SCRAPE_TIMEOUT_MS) : DEFAULT_SCRAPE_TIMEOUT_MS;
|
|
794
|
-
const handedOver = opts.handoff !== void 0 && opts.handoff !== false;
|
|
799
|
+
const handedOver = opts.handoff !== void 0 && opts.handoff !== false || opts.lane === "my-browser" || opts.access === "my-browser";
|
|
795
800
|
return this.post("/v1/scrape", { ...opts, url, origin: originOf(opts, request) }, 200, request, handedOver ? 0 : deadlineMs + SCRAPE_ANSWER_MARGIN_MS);
|
|
796
801
|
}
|
|
797
802
|
/** The record of one scrape call, by the `scrapeId` its response carried (`metadata.scrapeId`); a W2LError with code `not_found` for an id the server has no record of. */
|
package/dist/index.js
CHANGED
|
@@ -179,10 +179,10 @@ function isApiErrorCode(value) {
|
|
|
179
179
|
var RATE_LIMITED_CODE = "rate_limited";
|
|
180
180
|
var PAGE_KEYS = ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown"];
|
|
181
181
|
var ATTRIBUTION_KEYS = ["origin", "integration"];
|
|
182
|
-
var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
182
|
+
var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "lane", "access", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
183
183
|
var CRAWL_SCOPE_KEYS = ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
|
|
184
|
-
var CRAWL_KEYS = ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
185
|
-
var BATCH_KEYS = ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
184
|
+
var CRAWL_KEYS = ["url", "mode", "access", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
185
|
+
var BATCH_KEYS = ["urls", "mode", "lane", "access", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
186
186
|
var BATCH_APPEND_KEYS = ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", ...ATTRIBUTION_KEYS];
|
|
187
187
|
var MAP_SCOPE_KEYS = ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
|
|
188
188
|
var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", "ignoreRobotsTxt", ...ATTRIBUTION_KEYS];
|
|
@@ -217,11 +217,16 @@ var EVIDENCE_RECORD_KEYS = {
|
|
|
217
217
|
pageActions: keysOf()(["steps", "scriptRan"]),
|
|
218
218
|
pageActionStep: keysOf()(["type", "outcome"]),
|
|
219
219
|
requestHeader: keysOf()(["name", "valueSha256"]),
|
|
220
|
-
access: keysOf()(["route", "executor", "executorVersion", "profile", "externalCostUsd"])
|
|
220
|
+
access: keysOf()(["route", "executor", "executorVersion", "profile", "externalCostUsd", "completion", "egress", "session", "paidCalls", "grant"]),
|
|
221
|
+
accessEgress: keysOf()(["proxy", "source", "switchedFrom", "exit"]),
|
|
222
|
+
accessEgressExit: keysOf()(["ip", "country", "observedAt"]),
|
|
223
|
+
accessSession: keysOf()(["id"]),
|
|
224
|
+
accessPaidCall: keysOf()(["provider", "rung", "capabilities", "ceilingUsd", "chargedUsd", "reportedCostUsd", "outcome", "reason", "answer"]),
|
|
225
|
+
accessGrant: keysOf()(["sha256", "tier", "attestedAt"])
|
|
221
226
|
};
|
|
222
227
|
|
|
223
228
|
// packages/sdk/src/version.ts
|
|
224
|
-
var SDK_VERSION = "0.3.
|
|
229
|
+
var SDK_VERSION = "0.3.2";
|
|
225
230
|
|
|
226
231
|
// packages/sdk/src/watcher.ts
|
|
227
232
|
var DEFAULT_WATCH_POLL_INTERVAL_MS = 2e3;
|
|
@@ -754,7 +759,7 @@ var W2L = class {
|
|
|
754
759
|
}
|
|
755
760
|
async scrape(url, opts = {}, request = {}) {
|
|
756
761
|
const deadlineMs = Number.isInteger(opts.timeout) ? Math.min(Math.max(opts.timeout, 0), DEFAULT_SCRAPE_TIMEOUT_MS) : DEFAULT_SCRAPE_TIMEOUT_MS;
|
|
757
|
-
const handedOver = opts.handoff !== void 0 && opts.handoff !== false;
|
|
762
|
+
const handedOver = opts.handoff !== void 0 && opts.handoff !== false || opts.lane === "my-browser" || opts.access === "my-browser";
|
|
758
763
|
return this.post("/v1/scrape", { ...opts, url, origin: originOf(opts, request) }, 200, request, handedOver ? 0 : deadlineMs + SCRAPE_ANSWER_MARGIN_MS);
|
|
759
764
|
}
|
|
760
765
|
/** The record of one scrape call, by the `scrapeId` its response carried (`metadata.scrapeId`); a W2LError with code `not_found` for an id the server has no record of. */
|
|
@@ -33,7 +33,7 @@ export interface RequestOptions {
|
|
|
33
33
|
origin?: string;
|
|
34
34
|
}
|
|
35
35
|
/** What the SDK records as `origin` unless the caller or the host says otherwise. */
|
|
36
|
-
export declare const SDK_ORIGIN = "js-sdk@0.3.
|
|
36
|
+
export declare const SDK_ORIGIN = "js-sdk@0.3.2";
|
|
37
37
|
/**
|
|
38
38
|
* Polling for waitBatch and waitCrawl. A status request that fails with a
|
|
39
39
|
* network error, HTTP 408, 429 or 5xx is retried: after 1, 2, 4, 8, then
|
|
@@ -4,6 +4,7 @@
|
|
|
4
4
|
* Each step is recorded in the trace with its outcome and timing; a step
|
|
5
5
|
* that fails ends the pipeline, and the result keeps the page as it stood.
|
|
6
6
|
*/
|
|
7
|
+
import type { BlockReason } from './status.js';
|
|
7
8
|
import type { ScreenshotEvidence } from './result.js';
|
|
8
9
|
import type { ScreenshotViewport } from './structured.js';
|
|
9
10
|
/** The most steps one request may run. */
|
|
@@ -149,8 +150,20 @@ export interface ActionPdf {
|
|
|
149
150
|
path: string | null;
|
|
150
151
|
base64: string;
|
|
151
152
|
}
|
|
152
|
-
/**
|
|
153
|
-
|
|
153
|
+
/**
|
|
154
|
+
* Why a scrollToEnd, loadMore or paginate step stopped. `max` and `deadline` stop short of the list's end; so does
|
|
155
|
+
* `challenge`: a check the site put up on the next page (a Cloudflare interstitial, a page of nothing but a CAPTCHA),
|
|
156
|
+
* which paginate stops at without reading it, the result then `blocked` with the check's reason.
|
|
157
|
+
*/
|
|
158
|
+
export type ListStop = 'end' | 'no_growth' | 'repeat' | 'max' | 'deadline' | 'challenge';
|
|
159
|
+
/** The check a paginate step stopped at (ListStop `challenge`): the page it would have been, its URL, and what the gate saw. */
|
|
160
|
+
export interface ListChallenge {
|
|
161
|
+
/** 1-based position the page would have had among the pages read. */
|
|
162
|
+
page: number;
|
|
163
|
+
url: string;
|
|
164
|
+
reason: BlockReason;
|
|
165
|
+
signals: readonly string[];
|
|
166
|
+
}
|
|
154
167
|
/** What a scrollToEnd, loadMore or paginate step did. */
|
|
155
168
|
export interface ListRun {
|
|
156
169
|
index: number;
|
|
@@ -160,8 +173,27 @@ export interface ListRun {
|
|
|
160
173
|
rounds: number;
|
|
161
174
|
/** Elements matching `itemSelector` at the end (on the last page for paginate; summed over its pages in `itemsRead`); null without one. */
|
|
162
175
|
items: number | null;
|
|
163
|
-
/**
|
|
176
|
+
/**
|
|
177
|
+
* paginate: elements matching `itemSelector` over every page read; null without one. After a continuation in the person's
|
|
178
|
+
* Chrome (`continued`), the kept pages' count plus each page they showed, as their tab counted it, only when the addresses show
|
|
179
|
+
* no page can be counted twice (the kept pages each at its own, the check's page and the pages shown at none of them, the check
|
|
180
|
+
* not at the list's own address), no page shows again what another shows (its items' whole text, as the list merge tells it:
|
|
181
|
+
* a result set tied to the session that made it comes back at new addresses), and every count is known; null (unknown) otherwise.
|
|
182
|
+
*/
|
|
164
183
|
itemsRead?: number | null;
|
|
184
|
+
/** paginate: pages taken from the task's checkpoint after a run cut at page N (ExecutionContext.listResume), counted in `rounds`; absent when none. */
|
|
185
|
+
resumed?: number;
|
|
186
|
+
/** paginate: the check the step stopped at, with `stoppedBy` `challenge`; absent otherwise. */
|
|
187
|
+
challenge?: ListChallenge;
|
|
188
|
+
/**
|
|
189
|
+
* paginate: the pages read after the check at `from`, by the person paging on in their own browser once they got through
|
|
190
|
+
* it (the batch handoff), counted in `rounds`; `stoppedBy` then says how that reading ended. Absent otherwise.
|
|
191
|
+
*/
|
|
192
|
+
continued?: {
|
|
193
|
+
from: number;
|
|
194
|
+
pages: number;
|
|
195
|
+
by: 'user_browser';
|
|
196
|
+
};
|
|
165
197
|
}
|
|
166
198
|
/** What the steps produced, each list in the order of its steps. */
|
|
167
199
|
export interface ActionsResult {
|
|
@@ -170,7 +202,8 @@ export interface ActionsResult {
|
|
|
170
202
|
scrapes: {
|
|
171
203
|
url: string;
|
|
172
204
|
html: string;
|
|
173
|
-
step?: number;
|
|
205
|
+
step?: number; /** Set when the person read the page in their own browser (a list's continuation after a check); absent for W2L's own browser. */
|
|
206
|
+
by?: 'user_browser';
|
|
174
207
|
}[];
|
|
175
208
|
/** `type` is the JavaScript `typeof` of the value (`null` for null). */
|
|
176
209
|
javascriptReturns: {
|
|
@@ -170,7 +170,28 @@ export interface ScrapeRequest extends PageOptions, RequestAttribution {
|
|
|
170
170
|
handoff?: {
|
|
171
171
|
waitMs?: number;
|
|
172
172
|
};
|
|
173
|
+
/**
|
|
174
|
+
* `my-browser`: read the page in the person's own Chrome, over remote
|
|
175
|
+
* debugging, without W2L fetching it first. The person allows the
|
|
176
|
+
* connection in Chrome, then the site in a page W2L opens there; the page
|
|
177
|
+
* is read without a click of theirs only on a site they allowed, and a
|
|
178
|
+
* check it shows waits for them (`handoff.waitMs`, default 10 min). Lane
|
|
179
|
+
* `my_browser`; never cached. Offered only by a server on the person's
|
|
180
|
+
* own machine; refused elsewhere, with `actions` or a screenshot, and with
|
|
181
|
+
* a mode other than standard (`unsupported_parameter`).
|
|
182
|
+
*/
|
|
183
|
+
lane?: 'my-browser';
|
|
184
|
+
/** One of three plain choices of how the page is reached (ACCESS_CHOICES); omitted, the server's own configuration. */
|
|
185
|
+
access?: AccessChoice;
|
|
173
186
|
}
|
|
187
|
+
/**
|
|
188
|
+
* How pages are reached, as three plain choices (ROADMAP PA item 7) beside the per-route options:
|
|
189
|
+
* `standard` (Octocrawl's own lanes and none that costs a third party), `enhanced` (also what the server's
|
|
190
|
+
* access grant of tier enhanced approves, within its budget; refused on a server without one), `my-browser`
|
|
191
|
+
* (the person's own Chrome, as `lane: "my-browser"`; a scrape or a batch only).
|
|
192
|
+
*/
|
|
193
|
+
export declare const ACCESS_CHOICES: readonly ["standard", "enhanced", "my-browser"];
|
|
194
|
+
export type AccessChoice = (typeof ACCESS_CHOICES)[number];
|
|
174
195
|
/** A recorded robots override for one URL of a batch. */
|
|
175
196
|
export interface RobotsUrlOverride extends RobotsOverride {
|
|
176
197
|
url: string;
|
|
@@ -344,6 +365,8 @@ export interface JobWebhookStatus {
|
|
|
344
365
|
export interface CrawlStartRequest extends PageOptions, RequestAttribution {
|
|
345
366
|
url: string;
|
|
346
367
|
mode?: ApiCrawlMode;
|
|
368
|
+
/** `standard` or `enhanced` (ACCESS_CHOICES); a crawl does not take `my-browser`. */
|
|
369
|
+
access?: Exclude<AccessChoice, 'my-browser'>;
|
|
347
370
|
maxPages?: number | null;
|
|
348
371
|
maxDepth?: number | null;
|
|
349
372
|
/**
|
|
@@ -526,6 +549,16 @@ export interface ActiveCrawlList {
|
|
|
526
549
|
export interface BatchStartRequest extends PageOptions, RequestAttribution {
|
|
527
550
|
urls: readonly string[];
|
|
528
551
|
mode?: ApiCrawlMode;
|
|
552
|
+
/** One of three plain choices of how pages are reached (ACCESS_CHOICES); `my-browser` is `lane: "my-browser"`. */
|
|
553
|
+
access?: AccessChoice;
|
|
554
|
+
/**
|
|
555
|
+
* `my-browser`: read every page in the person's own Chrome, one at a time, as a scrape's `lane` does. The person
|
|
556
|
+
* allows the connection in Chrome, then all the batch's sites (host and port) in the page W2L opens there, once for
|
|
557
|
+
* the run; a site not among them is not read. A resumed run asks again. Offered only by a server on the person's own
|
|
558
|
+
* machine; refused elsewhere, with `actions`, a screenshot, lockdown, a mode other than standard, `maxConcurrency`
|
|
559
|
+
* above 1 or a webhook (`unsupported_parameter`).
|
|
560
|
+
*/
|
|
561
|
+
lane?: 'my-browser';
|
|
529
562
|
formats?: readonly ScrapeFormat[];
|
|
530
563
|
includeLinks?: boolean;
|
|
531
564
|
/** Recorded robots overrides, each for one URL of `urls`. A hosted server refuses the field (`unsupported_parameter`). */
|
|
@@ -604,6 +637,8 @@ export interface BatchStatusResponse extends CrawlReport {
|
|
|
604
637
|
invalidURLs?: readonly string[];
|
|
605
638
|
/** Items stopped at a check a person can get through in their own Chrome (`POST /v1/batches/:id/handoff`); present on a server that offers the handoff. */
|
|
606
639
|
waitingForPerson?: number;
|
|
640
|
+
/** A batch on the my-browser lane waiting for the person to allow its sites in the page Octocrawl opened in their Chrome; present only while it waits. */
|
|
641
|
+
waitingForApproval?: true;
|
|
607
642
|
}
|
|
608
643
|
/**
|
|
609
644
|
* The checks a batch item can be handed to a person for, and the routing
|
|
@@ -838,10 +873,12 @@ export declare const REFUSAL_HINTS: {
|
|
|
838
873
|
export declare function refusalHint(key: string, value: unknown): string | null;
|
|
839
874
|
export declare const PAGE_KEYS: readonly ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown"];
|
|
840
875
|
export declare const ATTRIBUTION_KEYS: readonly ["origin", "integration"];
|
|
841
|
-
export declare const SCRAPE_KEYS: readonly ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
876
|
+
export declare const SCRAPE_KEYS: readonly ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "lane", "access", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
877
|
+
/** The lanes a request may ask for by name. */
|
|
878
|
+
export declare const REQUEST_LANES: readonly ["my-browser"];
|
|
842
879
|
export declare const CRAWL_SCOPE_KEYS: readonly ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
|
|
843
|
-
export declare const CRAWL_KEYS: readonly ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", "regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks", "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
844
|
-
export declare const BATCH_KEYS: readonly ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
880
|
+
export declare const CRAWL_KEYS: readonly ["url", "mode", "access", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", "regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks", "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
881
|
+
export declare const BATCH_KEYS: readonly ["urls", "mode", "lane", "access", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
845
882
|
/** What a batch body may carry beside `appendToId`: the job's own options are not among them (the scope no-ops change nothing, so they may come along). */
|
|
846
883
|
export declare const BATCH_APPEND_KEYS: readonly ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", "origin", "integration"];
|
|
847
884
|
/** The scope options a map takes under their crawl names; allowSubdomains is includeSubdomains on a map, and allowExternalLinks is not offered. */
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
* Granularity is the page. A partial parse inside a page is not a step; the
|
|
8
8
|
* whole URL is retried. Block-level checkpoint is out of Phase 1.
|
|
9
9
|
*/
|
|
10
|
-
import type { PageOptions, RequestAttribution, RobotsUrlOverride, WebhookEvent } from './api.js';
|
|
10
|
+
import type { AccessChoice, PageOptions, RequestAttribution, RobotsUrlOverride, WebhookEvent } from './api.js';
|
|
11
11
|
import type { CrawlMode } from './compliance.js';
|
|
12
12
|
import type { WebhookPayloadFormat } from './delivery.js';
|
|
13
13
|
import type { CrawlDiscovery, SitemapMode } from './crawl.js';
|
|
@@ -29,6 +29,11 @@ export interface CrawlBudget {
|
|
|
29
29
|
maxWallMs: number | null;
|
|
30
30
|
maxCostUsd: number | null;
|
|
31
31
|
maxTokens: number | null;
|
|
32
|
+
/**
|
|
33
|
+
* What one page may spend on third parties (an access grant's `perRequestUsd`, ROADMAP PA item 4): its own cap
|
|
34
|
+
* within the run's, held by the run's spend ledger. Absent or null: none of its own.
|
|
35
|
+
*/
|
|
36
|
+
maxCostPerPageUsd?: number | null;
|
|
32
37
|
}
|
|
33
38
|
export declare const DEFAULT_CRAWL_BUDGET: CrawlBudget;
|
|
34
39
|
/**
|
|
@@ -72,12 +77,16 @@ export interface Task {
|
|
|
72
77
|
maxConcurrency?: number;
|
|
73
78
|
invalidURLs?: readonly string[];
|
|
74
79
|
webhook?: StoredJobWebhook;
|
|
80
|
+
lane?: 'my-browser';
|
|
81
|
+
access?: AccessChoice;
|
|
75
82
|
} & PageOptions;
|
|
76
83
|
/**
|
|
77
84
|
* Every crawl option but the page budget (`budget`), stored when the crawl
|
|
78
85
|
* starts so a resumed crawl runs with the options it was started with.
|
|
79
86
|
*/
|
|
80
87
|
crawl?: {
|
|
88
|
+
/** The plain access choice the crawl was started with (`standard` or `enhanced`); absent: the server's configuration. */
|
|
89
|
+
access?: 'standard' | 'enhanced';
|
|
81
90
|
formats?: readonly ScrapeFormat[];
|
|
82
91
|
includeLinks?: boolean;
|
|
83
92
|
includePaths?: readonly string[];
|
|
@@ -136,6 +145,11 @@ export interface Attempt {
|
|
|
136
145
|
recoveredFromAttemptId?: string | null;
|
|
137
146
|
/** What this attempt's pages offered the frontier and what became of it, written after every page of a crawl; absent for a batch and for an attempt stored before it was kept. */
|
|
138
147
|
discovery?: CrawlDiscovery | null;
|
|
148
|
+
/**
|
|
149
|
+
* What the spend ledger charged this attempt's paid calls (ROADMAP PA item 4): a resumed or appended run of the task
|
|
150
|
+
* opens its ledger with every earlier attempt's charge, so its cap is the task's, not each run's. Absent: none.
|
|
151
|
+
*/
|
|
152
|
+
chargedUsd?: number | null;
|
|
139
153
|
}
|
|
140
154
|
/**
|
|
141
155
|
* One URL inside one attempt. The atomic checkpoint unit.
|
|
@@ -56,7 +56,101 @@ export interface EvidenceAccess {
|
|
|
56
56
|
profile: string | null;
|
|
57
57
|
/** Third-party spend of the run that produced the result: 0 when no paid service was called, null when a called service stated no price. */
|
|
58
58
|
externalCostUsd: number | null;
|
|
59
|
+
/**
|
|
60
|
+
* How the page was read, counted apart (ROADMAP PA items 7 and 8): `unattended` (W2L's own lanes, no session of the
|
|
61
|
+
* person's), `authorized_session` (with the person's saved login), `user_browser` (in the person's own Chrome, on a
|
|
62
|
+
* site they allowed, without a step of theirs), `handed_to_person` (in their Chrome, after they got through a check).
|
|
63
|
+
* Null when no page was read: the result is not success, partial or empty_verified, or no lane produced it. Added with
|
|
64
|
+
* the my-browser lane: optional, so that records written before it stay valid.
|
|
65
|
+
*/
|
|
66
|
+
completion?: AccessCompletion | null;
|
|
67
|
+
/**
|
|
68
|
+
* The egress the page left through (ROADMAP PA item 3): the proxy's `host:port` and whether it came from the
|
|
69
|
+
* operator's pool (`W2L_EGRESS_PROXIES`) or the environment variables, or `direct` with no proxy when W2L's own lane
|
|
70
|
+
* recorded none and a page response shows a request was sent. `switchedFrom` names the pool egress the task last
|
|
71
|
+
* moved off before this page was read here (`egress_switched`); null otherwise. Null when nothing says where the
|
|
72
|
+
* requests left from: a vendor's service, the person's own browser, no lane at all, or a lane that stopped before
|
|
73
|
+
* a page request (robots.txt, an address check, a deadline). Added with the egress pool: optional, so that earlier
|
|
74
|
+
* records stay valid.
|
|
75
|
+
*/
|
|
76
|
+
egress?: EvidenceAccessEgress | null;
|
|
77
|
+
/**
|
|
78
|
+
* The task cookie session the page was read with (`egress_sessions`): its id alone, never its cookies. Null when
|
|
79
|
+
* the page was read with none. Added with the egress pool: optional, so that earlier records stay valid.
|
|
80
|
+
*/
|
|
81
|
+
session?: EvidenceAccessSession | null;
|
|
82
|
+
/**
|
|
83
|
+
* Every paid provider call the page was read with (ROADMAP PA item 4), in order: the provider, what the spend ledger
|
|
84
|
+
* reserved and charged, the price the provider stated, and what Octocrawl made of what came back. Null when no
|
|
85
|
+
* provider was called. Added with the spend ledger: optional, so that earlier records stay valid.
|
|
86
|
+
*/
|
|
87
|
+
paidCalls?: readonly EvidencePaidCall[] | null;
|
|
88
|
+
/** The access grant the paid calls were made under; null when no provider was called. Added with `paidCalls`. */
|
|
89
|
+
grant?: EvidenceAccessGrant | null;
|
|
90
|
+
}
|
|
91
|
+
/** One paid provider call (ROADMAP PA item 4). A provider's own word on the page is never its outcome. */
|
|
92
|
+
export interface EvidencePaidCall {
|
|
93
|
+
/** The provider's id, as the access grant's tariffs name it (`browserbase`, `steel`). */
|
|
94
|
+
provider: string;
|
|
95
|
+
/** The rung that made the call; a retry after the person's handoff is `provider(retry)`. */
|
|
96
|
+
rung: string;
|
|
97
|
+
/** The ADR 0005 capabilities the provider's session was created with: `vendor_remote_browser`, and solving or stealth when the grant named them. */
|
|
98
|
+
capabilities: readonly string[];
|
|
99
|
+
/** What the ledger reserved before the call: its price ceiling, from the grant's tariff. */
|
|
100
|
+
ceilingUsd: number;
|
|
101
|
+
/** What the ledger charged: the price the provider stated, else the ceiling (a call that threw or was cut included). */
|
|
102
|
+
chargedUsd: number;
|
|
103
|
+
/** The price the provider stated for the call; null when it stated none (Browserbase and Steel state none per call). */
|
|
104
|
+
reportedCostUsd: number | null;
|
|
105
|
+
/**
|
|
106
|
+
* What Octocrawl made of the page the call returned, by its own checks (a block page, an empty or unverified read, an
|
|
107
|
+
* identity it did not send): never the provider's word that it succeeded. Null when the call returned no page (it
|
|
108
|
+
* threw, or the deadline cut it).
|
|
109
|
+
*/
|
|
110
|
+
outcome: ResultStatus | null;
|
|
111
|
+
/** Why the outcome is not a read page, as the record's own `reason`; null otherwise. */
|
|
112
|
+
reason: FailureReason | BlockReason | BudgetKind | null;
|
|
113
|
+
/** The record's own page is this call's. */
|
|
114
|
+
answer: boolean;
|
|
115
|
+
}
|
|
116
|
+
/** The access grant paid calls were made under: enough to name it, not a copy. */
|
|
117
|
+
export interface EvidenceAccessGrant {
|
|
118
|
+
/** SHA-256 of the grant's text as Octocrawl read it (`shasum -a 256 grant.json`); null for a grant not read from text. */
|
|
119
|
+
sha256: string | null;
|
|
120
|
+
/** The grant's tier. */
|
|
121
|
+
tier: string;
|
|
122
|
+
/** When its attestation says the operator accepted the providers' terms and costs; null when it has none. */
|
|
123
|
+
attestedAt: string | null;
|
|
124
|
+
}
|
|
125
|
+
export declare const ACCESS_EGRESS_SOURCES: readonly ["pool", "environment", "direct"];
|
|
126
|
+
export type AccessEgressSource = (typeof ACCESS_EGRESS_SOURCES)[number];
|
|
127
|
+
export interface EvidenceAccessEgress {
|
|
128
|
+
/** The proxy's `host:port`, never its credentials; null when the request went direct. */
|
|
129
|
+
proxy: string | null;
|
|
130
|
+
source: AccessEgressSource;
|
|
131
|
+
/** The pool egress the task last left before this page was read here; null when it did not move. */
|
|
132
|
+
switchedFrom: string | null;
|
|
133
|
+
/**
|
|
134
|
+
* Where the pool egress leaves from, as the operator's echo URL (`W2L_EGRESS_ECHO_URL`) saw it through that proxy.
|
|
135
|
+
* Null when no echo URL is set, the echo did not answer, or the egress is not the pool's. Added after `egress`:
|
|
136
|
+
* optional, so that earlier records stay valid.
|
|
137
|
+
*/
|
|
138
|
+
exit?: EvidenceAccessEgressExit | null;
|
|
139
|
+
}
|
|
140
|
+
export interface EvidenceAccessEgressExit {
|
|
141
|
+
/** The address the echo service saw the request come from. */
|
|
142
|
+
ip: string;
|
|
143
|
+
/** Its two-letter country code, when the echo service gives one; null otherwise. */
|
|
144
|
+
country: string | null;
|
|
145
|
+
/** When the echo was asked (UTC ISO); an exit is asked again after ten minutes. */
|
|
146
|
+
observedAt: string;
|
|
147
|
+
}
|
|
148
|
+
export interface EvidenceAccessSession {
|
|
149
|
+
/** The session's id, as `session_cookies` traces it. */
|
|
150
|
+
id: string;
|
|
59
151
|
}
|
|
152
|
+
export declare const ACCESS_COMPLETIONS: readonly ["unattended", "authorized_session", "user_browser", "handed_to_person"];
|
|
153
|
+
export type AccessCompletion = (typeof ACCESS_COMPLETIONS)[number];
|
|
60
154
|
export interface EvidenceRedirectChain {
|
|
61
155
|
/**
|
|
62
156
|
* Every URL W2L requested for the page, in order: the requested URL first,
|
|
@@ -211,7 +305,12 @@ export declare const EVIDENCE_RECORD_KEYS: {
|
|
|
211
305
|
readonly pageActions: readonly ["steps", "scriptRan"];
|
|
212
306
|
readonly pageActionStep: readonly ["type", "outcome"];
|
|
213
307
|
readonly requestHeader: readonly ["name", "valueSha256"];
|
|
214
|
-
readonly access: readonly ["route", "executor", "executorVersion", "profile", "externalCostUsd"];
|
|
308
|
+
readonly access: readonly ["route", "executor", "executorVersion", "profile", "externalCostUsd", "completion", "egress", "session", "paidCalls", "grant"];
|
|
309
|
+
readonly accessEgress: readonly ["proxy", "source", "switchedFrom", "exit"];
|
|
310
|
+
readonly accessEgressExit: readonly ["ip", "country", "observedAt"];
|
|
311
|
+
readonly accessSession: readonly ["id"];
|
|
312
|
+
readonly accessPaidCall: readonly ["provider", "rung", "capabilities", "ceilingUsd", "chargedUsd", "reportedCostUsd", "outcome", "reason", "answer"];
|
|
313
|
+
readonly accessGrant: readonly ["sha256", "tier", "attestedAt"];
|
|
215
314
|
};
|
|
216
315
|
/**
|
|
217
316
|
* Keys added to v1 after it was first published: optional in the schema, so
|
|
@@ -20,6 +20,74 @@ export interface ExecutionContext {
|
|
|
20
20
|
* kept or sent, as before.
|
|
21
21
|
*/
|
|
22
22
|
cookieSession?: CookieSession;
|
|
23
|
+
/**
|
|
24
|
+
* Hear of each trace event the moment the HTTP lane records it (robots.txt checked, the response arrived, the
|
|
25
|
+
* page extracted), for a caller that shows progress while the fetch runs. The result's `trace` stays the record;
|
|
26
|
+
* a listener's error never changes the fetch. Absent: nothing is told early.
|
|
27
|
+
*/
|
|
28
|
+
onTrace?: (event: TraceEvent) => void;
|
|
29
|
+
/**
|
|
30
|
+
* Hear of each page a paginate step reads the moment it is read (ROADMAP PA item 3), for a caller that keeps
|
|
31
|
+
* them in the task's checkpoint: a run cut at page N then resumes from them through `listResume`. A listener's
|
|
32
|
+
* error never changes the fetch. Absent: nothing is told.
|
|
33
|
+
*/
|
|
34
|
+
onListPage?: (page: ListPageRead) => void;
|
|
35
|
+
/**
|
|
36
|
+
* The pages a paginate step of this URL read before an earlier run was cut, from the task's checkpoint. The
|
|
37
|
+
* browser lane counts them as read: it passes over them on its way along the site's own Next links (the only way
|
|
38
|
+
* to page N+1 when pages have no address of their own), tells and reads the pages after them, and merges every
|
|
39
|
+
* page once. Absent: the list starts from its first page.
|
|
40
|
+
*/
|
|
41
|
+
listResume?: {
|
|
42
|
+
pages: readonly ListPageRead[];
|
|
43
|
+
};
|
|
44
|
+
/**
|
|
45
|
+
* The run's spend ledger (ROADMAP PA item 4): a rung that costs a third party reserves its price ceiling here before
|
|
46
|
+
* its call and settles after, so a run's concurrent pages, retries and providers share one cap and none overruns it.
|
|
47
|
+
* Absent: nothing is reserved, as before (a run with no third-party rung).
|
|
48
|
+
*/
|
|
49
|
+
spend?: SpendLedger;
|
|
50
|
+
}
|
|
51
|
+
/**
|
|
52
|
+
* A spend ledger: the cap a run (or one page of it) may spend on third parties, what is settled and what is reserved.
|
|
53
|
+
* A paid call reserves its price ceiling first and is not made when the ceiling does not fit.
|
|
54
|
+
*/
|
|
55
|
+
export interface SpendLedger {
|
|
56
|
+
/** Reserve `ceilingUsd` against this ledger's cap and every enclosing one; null when it does not fit. */
|
|
57
|
+
reserve(ceilingUsd: number): SpendReservation | null;
|
|
58
|
+
/** A ledger over the same totals whose own reservations are also capped at `capUsd` (one page's `perRequestUsd`); null: no cap of its own. */
|
|
59
|
+
child(capUsd: number | null): SpendLedger;
|
|
60
|
+
/** Spend settled so far, in US dollars: each call at its reported price, or at its ceiling when none was reported. */
|
|
61
|
+
readonly settledUsd: number;
|
|
62
|
+
/** Ceilings reserved by calls not yet settled. */
|
|
63
|
+
readonly reservedUsd: number;
|
|
64
|
+
/** The cap; null: none. */
|
|
65
|
+
readonly capUsd: number | null;
|
|
66
|
+
}
|
|
67
|
+
/** One reserved call. */
|
|
68
|
+
export interface SpendReservation {
|
|
69
|
+
/** The call was made: settle at the price the provider reported, or at the ceiling when it reported none. Returns what was charged. */
|
|
70
|
+
settle(reportedUsd: number | null): number;
|
|
71
|
+
/** The call was not made: give the reservation back. */
|
|
72
|
+
release(): void;
|
|
73
|
+
}
|
|
74
|
+
/** One page a paginate step read: what `onListPage` tells and `listResume` gives back. */
|
|
75
|
+
export interface ListPageRead {
|
|
76
|
+
/** Index of the paginate step among the request's actions. */
|
|
77
|
+
step: number;
|
|
78
|
+
/** 1-based position among the pages the step read, the resumed ones included. */
|
|
79
|
+
page: number;
|
|
80
|
+
/** The URL the browser showed when the page was read. */
|
|
81
|
+
url: string;
|
|
82
|
+
html: string;
|
|
83
|
+
/** The page's state key as the lane computed it (its URL and its items or words), to know the page again on a resume. */
|
|
84
|
+
state: string;
|
|
85
|
+
/** The hash of the elements `itemSelector` matched, or null without one. */
|
|
86
|
+
items: string | null;
|
|
87
|
+
/** The hash of the links and sources inside those elements alone, which a changed price or date leaves as it was; null without `itemSelector` or when no item has one. */
|
|
88
|
+
itemRefs: string | null;
|
|
89
|
+
/** How many elements `itemSelector` matched on the page; null without one. */
|
|
90
|
+
count: number | null;
|
|
23
91
|
}
|
|
24
92
|
/** A cookie as a browser context takes and gives it (Playwright's shape). */
|
|
25
93
|
export interface ContextCookie {
|
|
@@ -23,7 +23,7 @@ export declare const FIRECRAWL_SHIM_SNAPSHOT: {
|
|
|
23
23
|
paths: readonly ["/scrape", "/crawl", "/crawl/:id", "/map"];
|
|
24
24
|
notCovered: readonly ["search", "interact", "agent", "monitor", "extract"];
|
|
25
25
|
};
|
|
26
|
-
export declare const FIRECRAWL_SHIM_DIFFS: readonly ["Challenge / block pages are success: false (Firecrawl often returns them as success markdown).", "A page with no main content is success: false (failed: empty_unverified) with the whole page in data.markdown as evidence; with onlyMainContent: false it is success: true.", "A page whose server HTML is a shell for data its scripts fill in is fetched again on the browser rung, and the rendered page is the answer when it holds more; otherwise the HTTP page is returned with client_rendered_suspected and low_content_yield warnings on the native response, whose messages /fc passes through as data.warning (one string, joined with a space), as it does every native warning. Firecrawl renders every page in a browser.", "No fire-engine, proxy pools or JSON extract.", "actions (scrape only; a crawl's scrapeOptions.actions is refused) run on the local browser rung alone, which such a request selects, after load, stability and waitFor and before the formats are read: wait (milliseconds up to 60000, or a selector, waited for up to 60 s within the scrape's timeout), click (all: true clicks every match), write (into the focused element), press, scroll (one screen up or down, of the page or the element a selector names), screenshot, scrape, executeJavascript (a function body; return gives the value) and pdf; at most 50 steps. data.actions holds screenshots and pdfs as data: URIs (Firecrawl returns URLs), scrapes as { url, html } and javascriptReturns as { type, value }. A step that fails stops the steps after it: success is false with data.actions.failed naming the step, its code and message, and data.markdown is the page as it stood. A step that leads the page to a URL robots.txt or the egress policy refuses fails with navigation_refused and that page is not read. A hosted server refuses actions, and the cache is not used with them.", "An omitted maxAge reuses nothing: every page is fetched live unless the request sets maxAge above 0, minAge or lockdown (Firecrawl reuses its own index by default; its Python SDK sends maxAge 4 hours). A reused page is one Octocrawl itself stored, under the same options, on this server (its task root), never a shared index; only a success is stored, and data.metadata says cacheState hit with cachedAt (its fetch time) or miss when one was looked up. lockdown with no stored result is HTTP 404 SCRAPE_LOCKDOWN_CACHE_MISS on scrape, and nothing is fetched; a crawl in lockdown needs sitemap skip, and each page with no stored result is failed with cache_miss. Mode authed neither stores nor reuses. A Firecrawl body never sets useCached, Octocrawl's reuse of a crawl's own pages on resume.", "Omitted limit / maxDepth stay unbounded on a local server; a hosted server takes its crawl limit for an omitted or null limit and refuses a larger one. Firecrawl defaults are 10000 / 10.", "maxDepth counts link hops from the start URL (Firecrawl calls that maxDiscoveryDepth); Firecrawl maxDepth counts URL path depth.", "Crawl start is mapped onto native POST /v1/crawl; the shim itself returns 200 {success,id,url}.", "creditsUsed and expiresAt are null: Octocrawl counts no credits and keeps crawl results until their task directory is deleted.", "Crawl status describes the latest attempt: completed counts its successful pages, total adds its failed, blocked and duplicate pages and, while this API process runs the crawl, the pages in flight and queued (null for a paused crawl); data lists the failed and blocked pages too (with metadata.error) but not the duplicates, whose content is an earlier entry's, up to 100 per response (limit 1 to 1000) with next carrying an Octocrawl cursor; skip is rejected.", "Scrape maps url, formats, onlyMainContent, includeTags, excludeTags, waitFor, timeout, headers, mobile, skipTlsVerification, fastMode, blockAds, removeBase64Images, maxAge, minAge, storeInCache, lockdown, actions, origin and integration; crawl maps url, limit (as maxPages), maxDepth, includePaths, excludePaths, regexOnFullURL, ignoreQueryParameters, deduplicateSimilarURLs, crawlEntireDomain (and its v1 name allowBackwardLinks), allowSubdomains, allowExternalLinks, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only), maxConcurrency, ignoreRobotsTxt (v2; a local server only), origin, integration and the same scrapeOptions (applied to every page). The formats are markdown, links, html, rawHtml, images, screenshot (also screenshot@fullPage, and { type: \"screenshot\", fullPage, quality, viewport }) and an { type: \"attributes\", selectors } entry; other formats and parameters the shim does not map (proxy, location, json, ...) are rejected by name with HTTP 400 and success: false; a refusal of stealth, proxy: stealth or enhanced, or ignoreRobotsTxt on a scrape or map names the supported route in agent_hints.", "A crawl follows links inside the start URL's path subtree on its host and www twin by default (crawlEntireDomain false), folds /a and /a/, / and /index.html, www and apex, http and https into one page (deduplicateSimilarURLs true) and reports every collapsed or refused link in the native crawl status (discovery) and each page's trace (links_offered); allowSubdomains takes every host under the start URL's apex (no public-suffix list), allowExternalLinks every host, each page with its own robots.txt read.", "sitemap (default include, as in Firecrawl) reads the sitemaps the start URL's robots.txt names, or /sitemap.xml, with the crawl's own http identity, robots.txt verdict, SSRF checks and proxy, and queues their URLs ahead of the start page's links under the same host, subtree, path and depth rules; only follows no page link; skip reads none. The native crawl status lists every sitemap file read, refused or unreadable in discovery.sitemap; the shim's status carries nothing of it, and sitemap fetches have no signed compliance record. maxConcurrency caps the pages one crawl fetches at once, at most the service's worker count (HTTP 400 above it), and never raises the per-host ceiling.", "screenshot (data.screenshot, a data:image/png;base64 string, or image/jpeg with quality 1 to 100) is captured on the local browser rung alone, which such a request selects (no http attempt; a server without a browser rung refuses the format with HTTP 400): after load, stability and waitFor, before the DOM is read, CSS-pixel sized at the declared 1280x800 viewport (device scale factor 2 is declared, not baked into the image) or at the viewport asked for (integers 320..1920 by 240..1080, within the declared screen; a window size, not a change of identity); fullPage captures the document's whole height at that width without scrolling first, so sections a page loads on scroll may show unloaded. A capture the browser could not make leaves data.screenshot null with a screenshot_unavailable warning while the page stands; a file or a page that was not rendered has null too. Firecrawl captures at its own viewport and may return a URL instead of the image.", "images (data.images) lists every image URL of the whole document as received: img src and srcset candidates, picture sources, lazy data-src/data-srcset/data-lazy-src/data-original, video posters, image_src links, og:image and twitter:image, absolute http(s) with the fragment stripped, each once, in document order, data: URIs left out; includeTags, excludeTags and onlyMainContent do not narrow it. attributes (data.attributes) gives, per selector, the named attribute's values as written, elements without it skipped; a selector Octocrawl does not match is HTTP 400 by name, as for includeTags. Both are absent for a file and for a page that is success: false. removeBase64Images (default true) keeps an image's alt text where Firecrawl writes a (<Base64-Image-Removed>) placeholder; false keeps the data: URI in the Markdown.", "origin (the Firecrawl SDKs' client label) and integration are stored, not echoed: the scrape record (GET /v1/scrapes/:id) and the crawl task carry them, and nothing sent to the target changes.", "data.metadata carries scrapeId (a UUID per call, which GET /v1/scrapes/:id looks up), proxyUsed (operator for the server's environment proxy, user for the caller's own egress, else null), timezone (the browser rung's declared zone, null on the HTTP rung), creditsUsed: null (Octocrawl counts no credits), concurrencyLimited and concurrencyQueueDurationMs (whether and how long the per-origin ceiling held the fetch back), and cacheState and cachedAt when the cache was asked (never a guessed miss).", "A page whose result Octocrawl has advice about (a login wall, a robots.txt rule, a gate, a cut, a script-filled shell) carries data.agent_hints, one sentence each; the native response calls them agentHints. A request refused for an option Octocrawl does not offer carries agent_hints in the error envelope, and a caller over the server's per-minute rate limit gets HTTP 429 { success: false, error, code: rate_limited, agent_hints } with Retry-After.", "headers never override the User-Agent, the client hints, a credential (authorization, cookie) or a transport header: such a header is HTTP 400 naming it, where Firecrawl sends it. The headers go to the requested origin after the declared identity and are on the record (the trace, the browser lane's signed sentHeaders); both rungs withhold them from a redirect hop to another origin and say so (custom_headers_withheld).", "mobile selects a declared Android Chrome identity (User-Agent, client hints, 412x915 viewport, touch) that robots.txt is evaluated against and the record carries; the page is whatever the site serves to it, with no DOM rewriting. It is refused with mode research.", "skipTlsVerification relaxes certificate verification for one local request and its robots.txt lookup, recorded in the trace (tls_verification_skipped) and a tls_unverified warning the native response carries; a hosted Octocrawl refuses it with HTTP 400. Without it a bad certificate is success: false with failed: tls_error. Firecrawl's Python SDK sends true by default; Octocrawl verifies by default.", "fastMode keeps the http rung alone: a page that needs scripts is success: false with failed: empty_unverified, never rendered; waitFor has no effect under it. Firecrawl's fast mode still renders.", "blockAds (default true) aborts requests to a bundled list of about 50 ad-serving hosts on the local browser rung and removes ad and cookie-banner elements before extraction; false keeps them. The list is curated, not EasyList: ads from hosts outside it are not blocked.", "html is the cleaned HTML the markdown is written from: the main content, the whole page without scripts, styles, form controls and embedded media when onlyMainContent is false, or a <body> holding the includeTags elements. rawHtml is the page as the answering rung received it: the response body on the HTTP rung, the rendered DOM on a browser rung. Both are null for a file and for a page that is success: false.", "includeTags keeps only the named elements, in document order, whatever onlyMainContent says; excludeTags removes elements from the main content, the whole page and an includeTags selection. A selector that does not parse, or that uses a sibling combinator, a positional pseudo-class, :has() or another pseudo-class Octocrawl does not match, is rejected with HTTP 400.", "An omitted timeout stays 300000 ms (Firecrawl: 30000). A timeout is answered with HTTP 200: success: true with the content fetched so far (native status partial), or success: false with failed: timeout; Firecrawl answers it with an error.", "waitFor skips the HTTP rung, which cannot run scripts, and starts at the browser rung; the wait counts toward timeout.", "metadata has title, description, language, keywords, robots and favicon only when the page declares them, and the Open Graph (ogTitle, ogDescription, ogUrl, ogImage, ogAudio, ogVideo, ogDeterminer, ogLocale, ogLocaleAlternate, ogSiteName), Dublin Core (dcTermsCreated, dcDateCreated, dcDate, dcTermsType, dcType, dcTermsAudience, dcTermsSubject, dcSubject, dcDescription, dcTermsKeywords) and article (publishedTime, modifiedTime, articleTag, articleSection) tags under Firecrawl's names, each only when the page states it, as written (no date normalisation, no fallback from another tag); twitter:* and other meta tags are not passed through, and a failed or blocked page has none.", "A PDF answers success: true with its text layer as markdown and metadata.numPages (the document's page count). parsers maps Firecrawl's pdf entry (the string or { type: \"pdf\", mode, maxPages, pages, pageMarkers }); mode fast and auto both read the text layer, and mode ocr and the image parser are refused by name. pageMarkers is false unless asked, as on Firecrawl; with true Octocrawl writes a <!-- page N --> line before each page (Firecrawl writes --- and the marker between pages). pages: true adds data.pages, [{ pageNumber, markdown }]. maxPages (1 to 10000) reads the first pages, and a cut it asked for stays success: true. parsers [] or v1 parsePDF false reads no PDF: success: true with markdown null and the file saved as received. A PDF without a text layer is success: false with failed: empty_unverified (no OCR). CSV, JSON and text files give their text as received; XLSX, XLS and ZIP files are success: true with markdown null. A file over W2L_MAX_FILE_BYTES is success: false with failed: body_too_large.", "Map (POST /fc/v1/map) maps url, search, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only; both v1 flags true is HTTP 400), includeSubdomains, ignoreQueryParameters, limit (1 to 100000, default 5000), timeout (1000 to 300000 ms for the whole map, default 60000; Firecrawl documents no default), origin and integration onto native POST /v1/map, and answers 200 { success: true, id, links: [url strings], warning?, agent_hints? }, or 200 { success: false, id, error, links: [] } when the map found nothing because a source failed or its deadline passed; useIndex, location, ignoreCache, threatProtection and auditMetadata are refused by name (useIndex with the hint that Octocrawl keeps no URL index). Omitted options take Octocrawl's defaults: includeSubdomains and ignoreQueryParameters are false, where Firecrawl v2 documents true for both. search keeps the URLs in which every word appears in the decoded URL or the title in hand, in discovery order; Firecrawl orders by relevance. A map reads the sitemaps the site declares and one page body (the start URL, http rung only), so a site without a sitemap maps only its start page's links; a title is the start page's own, an anchor's text or a sitemap's <news:title>, never fetched from the target; robots-disallowed URLs are left out and counted on the native response (GET /v1/maps/:id), which also records every sitemap file read.", "A crawl's webhook (a URL string or { url, headers, metadata, events }) is mapped onto the native webhook and its receiver gets Firecrawl's payload shape: { success, type: crawl.started | crawl.page | crawl.completed | crawl.failed, id, data: [page], metadata, error? }, one durable delivery per event with retries, every request carrying x-w2l-event-id, x-w2l-event-version and x-w2l-delivery-id (and the signature pair with secretEnv, a native option). A cancelled crawl is crawl.failed with error \"cancelled\". The native rules apply: https (plain http for a loopback receiver of a local server only), no content-type, host or x-w2l-* header, at most 32 headers and 32 metadata strings; a hosted server takes public https receivers only. GET /v1/deliveries?jobId=<id> on the native API lists the deliveries."];
|
|
26
|
+
export declare const FIRECRAWL_SHIM_DIFFS: readonly ["Challenge / block pages are success: false (Firecrawl often returns them as success markdown).", "A page with no main content is success: false (failed: empty_unverified) with the whole page in data.markdown as evidence; with onlyMainContent: false it is success: true.", "A page whose server HTML is a shell for data its scripts fill in is fetched again on the browser rung, and the rendered page is the answer when it holds more; otherwise the HTTP page is returned with client_rendered_suspected and low_content_yield warnings on the native response, whose messages /fc passes through as data.warning (one string, joined with a space), as it does every native warning. Firecrawl renders every page in a browser.", "No fire-engine, proxy pools or JSON extract.", "actions (scrape only; a crawl's scrapeOptions.actions is refused) run on the local browser rung alone, which such a request selects, after load, stability and waitFor and before the formats are read: wait (milliseconds up to 60000, or a selector, waited for up to 60 s within the scrape's timeout), click (all: true clicks every match), write (into the focused element), press, scroll (one screen up or down, of the page or the element a selector names), screenshot, scrape, executeJavascript (a function body; return gives the value) and pdf; at most 50 steps. data.actions holds screenshots and pdfs as data: URIs (Firecrawl returns URLs), scrapes as { url, html } and javascriptReturns as { type, value }. A step that fails stops the steps after it: success is false with data.actions.failed naming the step, its code and message, and data.markdown is the page as it stood. A step that leads the page to a URL robots.txt or the egress policy refuses fails with navigation_refused and that page is not read. A hosted server refuses actions, and the cache is not used with them.", "An omitted maxAge reuses nothing: every page is fetched live unless the request sets maxAge above 0, minAge or lockdown (Firecrawl reuses its own index by default; its Python SDK sends maxAge 4 hours). A reused page is one Octocrawl itself stored, under the same options, on this server (its task root), never a shared index; only a success is stored, and data.metadata says cacheState hit with cachedAt (its fetch time) or miss when one was looked up. lockdown with no stored result is HTTP 404 SCRAPE_LOCKDOWN_CACHE_MISS on scrape, and nothing is fetched; a crawl in lockdown needs sitemap skip, and each page with no stored result is failed with cache_miss. Mode authed neither stores nor reuses. A Firecrawl body never sets useCached, Octocrawl's reuse of a crawl's own pages on resume.", "Omitted limit / maxDepth stay unbounded on a local server; a hosted server takes its crawl limit for an omitted or null limit and refuses a larger one. Firecrawl defaults are 10000 / 10.", "maxDepth counts link hops from the start URL (Firecrawl calls that maxDiscoveryDepth); Firecrawl maxDepth counts URL path depth.", "Crawl start is mapped onto native POST /v1/crawl; the shim itself returns 200 {success,id,url}.", "creditsUsed and expiresAt are null: Octocrawl counts no credits and keeps crawl results until their task directory is deleted.", "Crawl status describes the latest attempt: completed counts its successful pages, total adds its failed, blocked and duplicate pages and, while this API process runs the crawl, the pages in flight and queued (null for a paused crawl); data lists the failed and blocked pages too (with metadata.error) but not the duplicates, whose content is an earlier entry's, up to 100 per response (limit 1 to 1000) with next carrying an Octocrawl cursor; skip is rejected.", "Scrape maps url, formats, onlyMainContent, includeTags, excludeTags, waitFor, timeout, headers, mobile, skipTlsVerification, fastMode, blockAds, removeBase64Images, maxAge, minAge, storeInCache, lockdown, actions, proxy (basic as access standard; stealth and auto as access enhanced, which a server without an access grant of tier enhanced refuses by name), origin and integration; crawl maps url, limit (as maxPages), maxDepth, includePaths, excludePaths, regexOnFullURL, ignoreQueryParameters, deduplicateSimilarURLs, crawlEntireDomain (and its v1 name allowBackwardLinks), allowSubdomains, allowExternalLinks, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only), maxConcurrency, ignoreRobotsTxt (v2; a local server only), origin, integration and the same scrapeOptions (applied to every page). The formats are markdown, links, html, rawHtml, images, screenshot (also screenshot@fullPage, and { type: \"screenshot\", fullPage, quality, viewport }) and an { type: \"attributes\", selectors } entry; other formats and parameters the shim does not map (location, json, ...) are rejected by name with HTTP 400 and success: false; a refusal of stealth, of access enhanced (proxy: stealth or auto) on a server without the grant, or of ignoreRobotsTxt on a scrape or map names the supported route in agent_hints.", "A crawl follows links inside the start URL's path subtree on its host and www twin by default (crawlEntireDomain false), folds /a and /a/, / and /index.html, www and apex, http and https into one page (deduplicateSimilarURLs true) and reports every collapsed or refused link in the native crawl status (discovery) and each page's trace (links_offered); allowSubdomains takes every host under the start URL's apex (no public-suffix list), allowExternalLinks every host, each page with its own robots.txt read.", "sitemap (default include, as in Firecrawl) reads the sitemaps the start URL's robots.txt names, or /sitemap.xml, with the crawl's own http identity, robots.txt verdict, SSRF checks and proxy, and queues their URLs ahead of the start page's links under the same host, subtree, path and depth rules; only follows no page link; skip reads none. The native crawl status lists every sitemap file read, refused or unreadable in discovery.sitemap; the shim's status carries nothing of it, and sitemap fetches have no signed compliance record. maxConcurrency caps the pages one crawl fetches at once, at most the service's worker count (HTTP 400 above it), and never raises the per-host ceiling.", "screenshot (data.screenshot, a data:image/png;base64 string, or image/jpeg with quality 1 to 100) is captured on the local browser rung alone, which such a request selects (no http attempt; a server without a browser rung refuses the format with HTTP 400): after load, stability and waitFor, before the DOM is read, CSS-pixel sized at the declared 1280x800 viewport (device scale factor 2 is declared, not baked into the image) or at the viewport asked for (integers 320..1920 by 240..1080, within the declared screen; a window size, not a change of identity); fullPage captures the document's whole height at that width without scrolling first, so sections a page loads on scroll may show unloaded. A capture the browser could not make leaves data.screenshot null with a screenshot_unavailable warning while the page stands; a file or a page that was not rendered has null too. Firecrawl captures at its own viewport and may return a URL instead of the image.", "images (data.images) lists every image URL of the whole document as received: img src and srcset candidates, picture sources, lazy data-src/data-srcset/data-lazy-src/data-original, video posters, image_src links, og:image and twitter:image, absolute http(s) with the fragment stripped, each once, in document order, data: URIs left out; includeTags, excludeTags and onlyMainContent do not narrow it. attributes (data.attributes) gives, per selector, the named attribute's values as written, elements without it skipped; a selector Octocrawl does not match is HTTP 400 by name, as for includeTags. Both are absent for a file and for a page that is success: false. removeBase64Images (default true) keeps an image's alt text where Firecrawl writes a (<Base64-Image-Removed>) placeholder; false keeps the data: URI in the Markdown.", "origin (the Firecrawl SDKs' client label) and integration are stored, not echoed: the scrape record (GET /v1/scrapes/:id) and the crawl task carry them, and nothing sent to the target changes.", "data.metadata carries scrapeId (a UUID per call, which GET /v1/scrapes/:id looks up), proxyUsed (operator for the server's environment proxy, user for the caller's own egress, else null), timezone (the browser rung's declared zone, null on the HTTP rung), creditsUsed: null (Octocrawl counts no credits), concurrencyLimited and concurrencyQueueDurationMs (whether and how long the per-origin ceiling held the fetch back), and cacheState and cachedAt when the cache was asked (never a guessed miss).", "A page whose result Octocrawl has advice about (a login wall, a robots.txt rule, a gate, a cut, a script-filled shell) carries data.agent_hints, one sentence each; the native response calls them agentHints. A request refused for an option Octocrawl does not offer carries agent_hints in the error envelope, and a caller over the server's per-minute rate limit gets HTTP 429 { success: false, error, code: rate_limited, agent_hints } with Retry-After.", "headers never override the User-Agent, the client hints, a credential (authorization, cookie) or a transport header: such a header is HTTP 400 naming it, where Firecrawl sends it. The headers go to the requested origin after the declared identity and are on the record (the trace, the browser lane's signed sentHeaders); both rungs withhold them from a redirect hop to another origin and say so (custom_headers_withheld).", "mobile selects a declared Android Chrome identity (User-Agent, client hints, 412x915 viewport, touch) that robots.txt is evaluated against and the record carries; the page is whatever the site serves to it, with no DOM rewriting. It is refused with mode research.", "skipTlsVerification relaxes certificate verification for one local request and its robots.txt lookup, recorded in the trace (tls_verification_skipped) and a tls_unverified warning the native response carries; a hosted Octocrawl refuses it with HTTP 400. Without it a bad certificate is success: false with failed: tls_error. Firecrawl's Python SDK sends true by default; Octocrawl verifies by default.", "fastMode keeps the http rung alone: a page that needs scripts is success: false with failed: empty_unverified, never rendered; waitFor has no effect under it. Firecrawl's fast mode still renders.", "blockAds (default true) aborts requests to a bundled list of about 50 ad-serving hosts on the local browser rung and removes ad and cookie-banner elements before extraction; false keeps them. The list is curated, not EasyList: ads from hosts outside it are not blocked.", "html is the cleaned HTML the markdown is written from: the main content, the whole page without scripts, styles, form controls and embedded media when onlyMainContent is false, or a <body> holding the includeTags elements. rawHtml is the page as the answering rung received it: the response body on the HTTP rung, the rendered DOM on a browser rung. Both are null for a file and for a page that is success: false.", "includeTags keeps only the named elements, in document order, whatever onlyMainContent says; excludeTags removes elements from the main content, the whole page and an includeTags selection. A selector that does not parse, or that uses a sibling combinator, a positional pseudo-class, :has() or another pseudo-class Octocrawl does not match, is rejected with HTTP 400.", "An omitted timeout stays 300000 ms (Firecrawl: 30000). A timeout is answered with HTTP 200: success: true with the content fetched so far (native status partial), or success: false with failed: timeout; Firecrawl answers it with an error.", "waitFor skips the HTTP rung, which cannot run scripts, and starts at the browser rung; the wait counts toward timeout.", "metadata has title, description, language, keywords, robots and favicon only when the page declares them, and the Open Graph (ogTitle, ogDescription, ogUrl, ogImage, ogAudio, ogVideo, ogDeterminer, ogLocale, ogLocaleAlternate, ogSiteName), Dublin Core (dcTermsCreated, dcDateCreated, dcDate, dcTermsType, dcType, dcTermsAudience, dcTermsSubject, dcSubject, dcDescription, dcTermsKeywords) and article (publishedTime, modifiedTime, articleTag, articleSection) tags under Firecrawl's names, each only when the page states it, as written (no date normalisation, no fallback from another tag); twitter:* and other meta tags are not passed through, and a failed or blocked page has none.", "A PDF answers success: true with its text layer as markdown and metadata.numPages (the document's page count). parsers maps Firecrawl's pdf entry (the string or { type: \"pdf\", mode, maxPages, pages, pageMarkers }); mode fast and auto both read the text layer, and mode ocr and the image parser are refused by name. pageMarkers is false unless asked, as on Firecrawl; with true Octocrawl writes a <!-- page N --> line before each page (Firecrawl writes --- and the marker between pages). pages: true adds data.pages, [{ pageNumber, markdown }]. maxPages (1 to 10000) reads the first pages, and a cut it asked for stays success: true. parsers [] or v1 parsePDF false reads no PDF: success: true with markdown null and the file saved as received. A PDF without a text layer is success: false with failed: empty_unverified (no OCR). CSV, JSON and text files give their text as received; XLSX, XLS and ZIP files are success: true with markdown null. A file over W2L_MAX_FILE_BYTES is success: false with failed: body_too_large.", "Map (POST /fc/v1/map) maps url, search, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only; both v1 flags true is HTTP 400), includeSubdomains, ignoreQueryParameters, limit (1 to 100000, default 5000), timeout (1000 to 300000 ms for the whole map, default 60000; Firecrawl documents no default), origin and integration onto native POST /v1/map, and answers 200 { success: true, id, links: [url strings], warning?, agent_hints? }, or 200 { success: false, id, error, links: [] } when the map found nothing because a source failed or its deadline passed; useIndex, location, ignoreCache, threatProtection and auditMetadata are refused by name (useIndex with the hint that Octocrawl keeps no URL index). Omitted options take Octocrawl's defaults: includeSubdomains and ignoreQueryParameters are false, where Firecrawl v2 documents true for both. search keeps the URLs in which every word appears in the decoded URL or the title in hand, in discovery order; Firecrawl orders by relevance. A map reads the sitemaps the site declares and one page body (the start URL, http rung only), so a site without a sitemap maps only its start page's links; a title is the start page's own, an anchor's text or a sitemap's <news:title>, never fetched from the target; robots-disallowed URLs are left out and counted on the native response (GET /v1/maps/:id), which also records every sitemap file read.", "A crawl's webhook (a URL string or { url, headers, metadata, events }) is mapped onto the native webhook and its receiver gets Firecrawl's payload shape: { success, type: crawl.started | crawl.page | crawl.completed | crawl.failed, id, data: [page], metadata, error? }, one durable delivery per event with retries, every request carrying x-w2l-event-id, x-w2l-event-version and x-w2l-delivery-id (and the signature pair with secretEnv, a native option). A cancelled crawl is crawl.failed with error \"cancelled\". The native rules apply: https (plain http for a loopback receiver of a local server only), no content-type, host or x-w2l-* header, at most 32 headers and 32 metadata strings; a hosted server takes public https receivers only. GET /v1/deliveries?jobId=<id> on the native API lists the deliveries."];
|
|
27
27
|
export interface FirecrawlPage {
|
|
28
28
|
markdown: string | null;
|
|
29
29
|
/** Present when the `html` format was asked for; null when the page has none (a file, a page that did not succeed). */
|