@octocrawl/sdk 0.3.0 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +7 -6
- package/dist/index.js +7 -6
- package/dist/types/cjs/client.d.ts +1 -1
- package/dist/types/cjs/contracts/api.d.ts +27 -8
- package/dist/types/cjs/contracts/checkpoint.d.ts +2 -0
- package/dist/types/cjs/contracts/compliance.d.ts +35 -7
- package/dist/types/cjs/contracts/crawl.d.ts +1 -1
- package/dist/types/cjs/contracts/evidenceRecord.d.ts +33 -4
- package/dist/types/cjs/contracts/execution.d.ts +54 -10
- package/dist/types/cjs/contracts/extractor.d.ts +11 -1
- package/dist/types/cjs/contracts/firecrawl.d.ts +1 -1
- package/dist/types/cjs/contracts/map.d.ts +10 -4
- package/dist/types/cjs/contracts/policy.d.ts +2 -1
- package/dist/types/cjs/contracts/proxy.d.ts +8 -0
- package/dist/types/cjs/contracts/result.d.ts +19 -3
- package/dist/types/cjs/contracts/session.d.ts +2 -0
- package/dist/types/cjs/version.d.ts +1 -1
- package/dist/types/esm/client.d.ts +1 -1
- package/dist/types/esm/contracts/api.d.ts +27 -8
- package/dist/types/esm/contracts/checkpoint.d.ts +2 -0
- package/dist/types/esm/contracts/compliance.d.ts +35 -7
- package/dist/types/esm/contracts/crawl.d.ts +1 -1
- package/dist/types/esm/contracts/evidenceRecord.d.ts +33 -4
- package/dist/types/esm/contracts/execution.d.ts +54 -10
- package/dist/types/esm/contracts/extractor.d.ts +11 -1
- package/dist/types/esm/contracts/firecrawl.d.ts +1 -1
- package/dist/types/esm/contracts/map.d.ts +10 -4
- package/dist/types/esm/contracts/policy.d.ts +2 -1
- package/dist/types/esm/contracts/proxy.d.ts +8 -0
- package/dist/types/esm/contracts/result.d.ts +19 -3
- package/dist/types/esm/contracts/session.d.ts +2 -0
- package/dist/types/esm/version.d.ts +1 -1
- package/package.json +1 -1
package/dist/index.cjs
CHANGED
|
@@ -218,11 +218,11 @@ var PAGE_KEYS = ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "inclu
|
|
|
218
218
|
var ATTRIBUTION_KEYS = ["origin", "integration"];
|
|
219
219
|
var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
220
220
|
var CRAWL_SCOPE_KEYS = ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
|
|
221
|
-
var CRAWL_KEYS = ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
221
|
+
var CRAWL_KEYS = ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
222
222
|
var BATCH_KEYS = ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
223
223
|
var BATCH_APPEND_KEYS = ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", ...ATTRIBUTION_KEYS];
|
|
224
224
|
var MAP_SCOPE_KEYS = ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
|
|
225
|
-
var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", ...ATTRIBUTION_KEYS];
|
|
225
|
+
var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", "ignoreRobotsTxt", ...ATTRIBUTION_KEYS];
|
|
226
226
|
var WEBHOOK_HEADERS_MESSAGE = `webhook.headers must be an object of at most ${MAX_WEBHOOK_HEADERS} string values`;
|
|
227
227
|
var WEBHOOK_METADATA_MESSAGE = `webhook.metadata must be an object of at most ${MAX_WEBHOOK_METADATA_ENTRIES} string values of at most ${MAX_WEBHOOK_METADATA_VALUE_LENGTH} characters`;
|
|
228
228
|
var WEBHOOK_EVENTS_MESSAGE = `webhook.events must be a non-empty array of ${WEBHOOK_EVENTS.join(", ")} without duplicates`;
|
|
@@ -243,9 +243,9 @@ var SHIM_MAP_KEYS = ["url", "search", "sitemap", "ignoreSitemap", "sitemapOnly",
|
|
|
243
243
|
// packages/contracts/dist/evidenceRecord.js
|
|
244
244
|
var keysOf = () => (keys) => keys;
|
|
245
245
|
var EVIDENCE_RECORD_KEYS = {
|
|
246
|
-
record: keysOf()(["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions"]),
|
|
246
|
+
record: keysOf()(["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions", "access"]),
|
|
247
247
|
redirectChain: keysOf()(["urls", "complete"]),
|
|
248
|
-
robotsDecision: keysOf()(["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride"]),
|
|
248
|
+
robotsDecision: keysOf()(["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride", "overrideBasis"]),
|
|
249
249
|
outputSha256: keysOf()(["markdown", "json"]),
|
|
250
250
|
extractor: keysOf()(["name", "version", "commit"]),
|
|
251
251
|
fieldEvidence: keysOf()(["source", "locator"]),
|
|
@@ -253,11 +253,12 @@ var EVIDENCE_RECORD_KEYS = {
|
|
|
253
253
|
identity: keysOf()(["userAgent", "mode", "contact", "device", "requestHeaders"]),
|
|
254
254
|
pageActions: keysOf()(["steps", "scriptRan"]),
|
|
255
255
|
pageActionStep: keysOf()(["type", "outcome"]),
|
|
256
|
-
requestHeader: keysOf()(["name", "valueSha256"])
|
|
256
|
+
requestHeader: keysOf()(["name", "valueSha256"]),
|
|
257
|
+
access: keysOf()(["route", "executor", "executorVersion", "profile", "externalCostUsd"])
|
|
257
258
|
};
|
|
258
259
|
|
|
259
260
|
// packages/sdk/src/version.ts
|
|
260
|
-
var SDK_VERSION = "0.3.
|
|
261
|
+
var SDK_VERSION = "0.3.1";
|
|
261
262
|
|
|
262
263
|
// packages/sdk/src/watcher.ts
|
|
263
264
|
var DEFAULT_WATCH_POLL_INTERVAL_MS = 2e3;
|
package/dist/index.js
CHANGED
|
@@ -181,11 +181,11 @@ var PAGE_KEYS = ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "inclu
|
|
|
181
181
|
var ATTRIBUTION_KEYS = ["origin", "integration"];
|
|
182
182
|
var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
183
183
|
var CRAWL_SCOPE_KEYS = ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
|
|
184
|
-
var CRAWL_KEYS = ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
184
|
+
var CRAWL_KEYS = ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
185
185
|
var BATCH_KEYS = ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
186
186
|
var BATCH_APPEND_KEYS = ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", ...ATTRIBUTION_KEYS];
|
|
187
187
|
var MAP_SCOPE_KEYS = ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
|
|
188
|
-
var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", ...ATTRIBUTION_KEYS];
|
|
188
|
+
var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", "ignoreRobotsTxt", ...ATTRIBUTION_KEYS];
|
|
189
189
|
var WEBHOOK_HEADERS_MESSAGE = `webhook.headers must be an object of at most ${MAX_WEBHOOK_HEADERS} string values`;
|
|
190
190
|
var WEBHOOK_METADATA_MESSAGE = `webhook.metadata must be an object of at most ${MAX_WEBHOOK_METADATA_ENTRIES} string values of at most ${MAX_WEBHOOK_METADATA_VALUE_LENGTH} characters`;
|
|
191
191
|
var WEBHOOK_EVENTS_MESSAGE = `webhook.events must be a non-empty array of ${WEBHOOK_EVENTS.join(", ")} without duplicates`;
|
|
@@ -206,9 +206,9 @@ var SHIM_MAP_KEYS = ["url", "search", "sitemap", "ignoreSitemap", "sitemapOnly",
|
|
|
206
206
|
// packages/contracts/dist/evidenceRecord.js
|
|
207
207
|
var keysOf = () => (keys) => keys;
|
|
208
208
|
var EVIDENCE_RECORD_KEYS = {
|
|
209
|
-
record: keysOf()(["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions"]),
|
|
209
|
+
record: keysOf()(["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions", "access"]),
|
|
210
210
|
redirectChain: keysOf()(["urls", "complete"]),
|
|
211
|
-
robotsDecision: keysOf()(["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride"]),
|
|
211
|
+
robotsDecision: keysOf()(["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride", "overrideBasis"]),
|
|
212
212
|
outputSha256: keysOf()(["markdown", "json"]),
|
|
213
213
|
extractor: keysOf()(["name", "version", "commit"]),
|
|
214
214
|
fieldEvidence: keysOf()(["source", "locator"]),
|
|
@@ -216,11 +216,12 @@ var EVIDENCE_RECORD_KEYS = {
|
|
|
216
216
|
identity: keysOf()(["userAgent", "mode", "contact", "device", "requestHeaders"]),
|
|
217
217
|
pageActions: keysOf()(["steps", "scriptRan"]),
|
|
218
218
|
pageActionStep: keysOf()(["type", "outcome"]),
|
|
219
|
-
requestHeader: keysOf()(["name", "valueSha256"])
|
|
219
|
+
requestHeader: keysOf()(["name", "valueSha256"]),
|
|
220
|
+
access: keysOf()(["route", "executor", "executorVersion", "profile", "externalCostUsd"])
|
|
220
221
|
};
|
|
221
222
|
|
|
222
223
|
// packages/sdk/src/version.ts
|
|
223
|
-
var SDK_VERSION = "0.3.
|
|
224
|
+
var SDK_VERSION = "0.3.1";
|
|
224
225
|
|
|
225
226
|
// packages/sdk/src/watcher.ts
|
|
226
227
|
var DEFAULT_WATCH_POLL_INTERVAL_MS = 2e3;
|
|
@@ -33,7 +33,7 @@ export interface RequestOptions {
|
|
|
33
33
|
origin?: string;
|
|
34
34
|
}
|
|
35
35
|
/** What the SDK records as `origin` unless the caller or the host says otherwise. */
|
|
36
|
-
export declare const SDK_ORIGIN = "js-sdk@0.3.
|
|
36
|
+
export declare const SDK_ORIGIN = "js-sdk@0.3.1";
|
|
37
37
|
/**
|
|
38
38
|
* Polling for waitBatch and waitCrawl. A status request that fails with a
|
|
39
39
|
* network error, HTTP 408, 429 or 5xx is retried: after 1, 2, 4, 8, then
|
|
@@ -411,6 +411,16 @@ export interface CrawlStartRequest extends PageOptions, RequestAttribution {
|
|
|
411
411
|
idempotencyKey?: string;
|
|
412
412
|
/** A receiver for the crawl's events (`started`, one `page` per page recorded, then `completed`, `failed` or `cancelled`); see WebhookConfig. */
|
|
413
413
|
webhook?: WebhookOption;
|
|
414
|
+
/**
|
|
415
|
+
* Fetch the pages and sitemap files robots.txt disallows, or whose
|
|
416
|
+
* robots.txt could not be read (Firecrawl v2's name). robots.txt is still
|
|
417
|
+
* read for every host and its verdict recorded on each page, Crawl-delay
|
|
418
|
+
* applied, with a `robots_overridden` warning and `overrideBasis:
|
|
419
|
+
* "ignore_robots_txt"` where a rule was set aside. A local server only: a
|
|
420
|
+
* hosted one refuses it by name. Default false: a crawl's links obey
|
|
421
|
+
* robots.txt.
|
|
422
|
+
*/
|
|
423
|
+
ignoreRobotsTxt?: boolean;
|
|
414
424
|
}
|
|
415
425
|
/** What the parser hands the engine: the request plus, from the `/fc` shim, the payload shape its receiver expects. */
|
|
416
426
|
export type ParsedCrawlStartRequest = CrawlStartRequest & {
|
|
@@ -454,6 +464,14 @@ export interface MapRequest extends RequestAttribution {
|
|
|
454
464
|
crawlEntireDomain?: boolean;
|
|
455
465
|
/** Default true, as on a crawl; a returned http link gives way to its https variant when that comes too, on an origin whose robots.txt the map read anyway and which allows it. */
|
|
456
466
|
deduplicateSimilarURLs?: boolean;
|
|
467
|
+
/**
|
|
468
|
+
* Return the URLs robots.txt disallows, or whose robots.txt could not be
|
|
469
|
+
* read, with that verdict on each link (`robots: "disallowed"` or
|
|
470
|
+
* `"unreachable"`), and read the start page and sitemap files past it.
|
|
471
|
+
* robots.txt is still read, within MAP_MAX_ROBOTS_HOSTS. A local server
|
|
472
|
+
* only: a hosted one refuses it by name. Default false.
|
|
473
|
+
*/
|
|
474
|
+
ignoreRobotsTxt?: boolean;
|
|
457
475
|
}
|
|
458
476
|
/** A map's `search`: at most this many characters after trimming, and this many whitespace-separated words. */
|
|
459
477
|
export declare const MAP_SEARCH_MAX_CHARS = 200;
|
|
@@ -482,6 +500,7 @@ export interface ActiveCrawlOptions {
|
|
|
482
500
|
allowExternalLinks: boolean;
|
|
483
501
|
regexOnFullURL: boolean;
|
|
484
502
|
maxConcurrency: number | null;
|
|
503
|
+
ignoreRobotsTxt: boolean;
|
|
485
504
|
/** The per-page options every page of the crawl gets: its formats, `includeLinks` and the page options. */
|
|
486
505
|
scrapeOptions: PageOptions & {
|
|
487
506
|
formats: readonly ScrapeFormat[];
|
|
@@ -809,10 +828,10 @@ export declare class RequestError extends Error {
|
|
|
809
828
|
}
|
|
810
829
|
/** The hints a refusal carries for the options W2L does not offer: the next honest step, never a way around the refusal. */
|
|
811
830
|
export declare const REFUSAL_HINTS: {
|
|
812
|
-
readonly stealth: "
|
|
813
|
-
readonly ignoreRobotsTxt: "robots.txt is always read; a
|
|
814
|
-
readonly hostedSkipTlsVerification: "a hosted server verifies every certificate; run
|
|
815
|
-
readonly useIndex: "
|
|
831
|
+
readonly stealth: "Octocrawl has no stealth option on a request: a provider's stealth or challenge solving runs only on a server started with an access grant that names it (--access-grant, ADR 0005); a proxy or session you own (mode authed) is the other route";
|
|
832
|
+
readonly ignoreRobotsTxt: "robots.txt is always read and recorded; on a local server a URL a scrape or batch names is fetched whatever it says, and ignoreRobotsTxt on a crawl or map fetches the links it disallows, on the record";
|
|
833
|
+
readonly hostedSkipTlsVerification: "a hosted server verifies every certificate; run Octocrawl locally to use skipTlsVerification, which is recorded in the trace and a tls_unverified warning";
|
|
834
|
+
readonly useIndex: "Octocrawl keeps no URL index: a map reads the sitemaps the site declares and its start page, on the record; crawl reads further pages";
|
|
816
835
|
readonly actions: "actions run on scrape and batch, where each page named gets the same steps; a crawl or a map does not take them";
|
|
817
836
|
};
|
|
818
837
|
/** The hint for a refused request key, or null when the key has none (an option W2L simply does not know). */
|
|
@@ -821,14 +840,14 @@ export declare const PAGE_KEYS: readonly ["onlyMainContent", "waitFor", "timeout
|
|
|
821
840
|
export declare const ATTRIBUTION_KEYS: readonly ["origin", "integration"];
|
|
822
841
|
export declare const SCRAPE_KEYS: readonly ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
823
842
|
export declare const CRAWL_SCOPE_KEYS: readonly ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
|
|
824
|
-
export declare const CRAWL_KEYS: readonly ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", "regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks", "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
843
|
+
export declare const CRAWL_KEYS: readonly ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", "regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks", "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
825
844
|
export declare const BATCH_KEYS: readonly ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
826
845
|
/** What a batch body may carry beside `appendToId`: the job's own options are not among them (the scope no-ops change nothing, so they may come along). */
|
|
827
846
|
export declare const BATCH_APPEND_KEYS: readonly ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", "origin", "integration"];
|
|
828
847
|
/** The scope options a map takes under their crawl names; allowSubdomains is includeSubdomains on a map, and allowExternalLinks is not offered. */
|
|
829
848
|
export declare const MAP_SCOPE_KEYS: readonly ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
|
|
830
849
|
/** What a map takes. No page option (headers, mobile, skipTlsVerification, formats, ...): a map has nothing to loosen. */
|
|
831
|
-
export declare const MAP_KEYS: readonly ["url", "mode", "limit", "timeout", "search", "sitemap", "includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs", "includePaths", "excludePaths", "origin", "integration"];
|
|
850
|
+
export declare const MAP_KEYS: readonly ["url", "mode", "limit", "timeout", "search", "sitemap", "includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs", "includePaths", "excludePaths", "ignoreRobotsTxt", "origin", "integration"];
|
|
832
851
|
/** Why a webhook header (lower-cased name) cannot be sent, or null when it can: W2L's own and the transport's names are reserved. */
|
|
833
852
|
export declare function webhookHeaderRefusal(name: string): string | null;
|
|
834
853
|
/**
|
|
@@ -869,8 +888,8 @@ export declare function parseCrawlStartRequest(body: unknown): CrawlStartRequest
|
|
|
869
888
|
* A map request: url, mode (standard or research), limit, timeout, search,
|
|
870
889
|
* sitemap, includeSubdomains, the crawl's scope options under their crawl
|
|
871
890
|
* names (ignoreQueryParameters, includePaths, excludePaths, regexOnFullURL,
|
|
872
|
-
* crawlEntireDomain, deduplicateSimilarURLs), origin and
|
|
873
|
-
* Anything else is refused by name, `useIndex` with the supported route;
|
|
891
|
+
* crawlEntireDomain, deduplicateSimilarURLs), ignoreRobotsTxt, origin and
|
|
892
|
+
* integration. Anything else is refused by name, `useIndex` with the supported route;
|
|
874
893
|
* nothing is silently ignored.
|
|
875
894
|
*/
|
|
876
895
|
export declare function parseMapRequest(body: unknown): MapRequest;
|
|
@@ -106,6 +106,8 @@ export interface Task {
|
|
|
106
106
|
maxConcurrency?: number | null;
|
|
107
107
|
/** The crawl's webhook, when the request set one. */
|
|
108
108
|
webhook?: StoredJobWebhook;
|
|
109
|
+
/** The crawl fetches what robots.txt disallows, on the record (CrawlStartRequest.ignoreRobotsTxt); absent: it obeys. A server that takes no override resumes it obeying. */
|
|
110
|
+
ignoreRobotsTxt?: boolean;
|
|
109
111
|
} & PageOptions;
|
|
110
112
|
/** Who started the task (`origin`, `integration`), stored with it and reported as `attribution` on its status; absent when the request named neither. */
|
|
111
113
|
attribution?: RequestAttribution;
|
|
@@ -109,13 +109,24 @@ export declare function researchUserAgent(contact?: string | null, host?: string
|
|
|
109
109
|
export declare function declaredContact(userAgent: string): string | null;
|
|
110
110
|
/** Whether a User-Agent is one research mode declares, in either format. */
|
|
111
111
|
export declare function isResearchUserAgent(userAgent: string): boolean;
|
|
112
|
+
/**
|
|
113
|
+
* The product token a robots.txt names to address Octocrawl itself
|
|
114
|
+
* (`User-agent: Octocrawl`), whatever User-Agent header a request sends. It
|
|
115
|
+
* is matched in every mode. A rule written for it (or for research mode's own
|
|
116
|
+
* token) is not set aside for a URL the request names or for a crawl or map
|
|
117
|
+
* started with ignoreRobotsTxt; only a recorded robotsOverride sets it aside.
|
|
118
|
+
*/
|
|
119
|
+
export declare const PRODUCT_ROBOTS_TOKEN = "octocrawl";
|
|
112
120
|
/**
|
|
113
121
|
* The text robots.txt `User-agent` lines are matched against for a
|
|
114
|
-
* User-Agent
|
|
115
|
-
*
|
|
116
|
-
* SEC.gov as it does on
|
|
122
|
+
* User-Agent Octocrawl sends: the header, with the product token added. SEC's
|
|
123
|
+
* format names no product token, so the research token is added as well: a
|
|
124
|
+
* group for w2l-research governs research requests to SEC.gov as it does on
|
|
125
|
+
* every other host.
|
|
117
126
|
*/
|
|
118
127
|
export declare function robotsAgent(userAgent: string): string;
|
|
128
|
+
/** Whether the robots.txt group that decided names Octocrawl itself (its product token, or research mode's) rather than every crawler (`*`). */
|
|
129
|
+
export declare function isOctocrawlRobotsGroup(matchedAgent: string | null | undefined): boolean;
|
|
119
130
|
/** The operator's contact from `W2L_CONTACT`, trimmed; null when unset or blank. The error never repeats the value. */
|
|
120
131
|
export declare function operatorContact(env: Readonly<Record<string, string | undefined>>): string | null;
|
|
121
132
|
/** An operator policy whose research-mode requests declare `W2L_CONTACT`, when it is set. */
|
|
@@ -254,6 +265,23 @@ export interface RobotsOverride {
|
|
|
254
265
|
/** Who recorded the decision, when the caller wants that on the record. */
|
|
255
266
|
recordedBy?: string;
|
|
256
267
|
}
|
|
268
|
+
/**
|
|
269
|
+
* On whose word a fetch went past robots.txt. `robots_override`: the
|
|
270
|
+
* caller's recorded decision for this URL (`RobotsOverride`).
|
|
271
|
+
* `user_named_url`: a local server fetching a URL the request named (a
|
|
272
|
+
* scrape, a batch entry), since robots.txt addresses crawlers that discover
|
|
273
|
+
* links, not the pages a person names. `ignore_robots_txt`: a crawl or map
|
|
274
|
+
* the caller started with `ignoreRobotsTxt` on a local server.
|
|
275
|
+
*/
|
|
276
|
+
export type RobotsOverrideBasis = 'robots_override' | 'user_named_url' | 'ignore_robots_txt';
|
|
277
|
+
/**
|
|
278
|
+
* A robots override as a lane applies it: the caller's recorded one, or one
|
|
279
|
+
* W2L applies by rule, which says so in `basis` (absent: the caller's,
|
|
280
|
+
* `robots_override`). Set by W2L, never read from a request.
|
|
281
|
+
*/
|
|
282
|
+
export interface AppliedRobotsOverride extends RobotsOverride {
|
|
283
|
+
basis?: Exclude<RobotsOverrideBasis, 'robots_override'>;
|
|
284
|
+
}
|
|
257
285
|
/**
|
|
258
286
|
* The outcome of consulting robots.txt for a single target URL. One record per
|
|
259
287
|
* fetch. `consulted` distinguishes "we checked and it said X" from "there was
|
|
@@ -288,11 +316,11 @@ export interface RobotsDecision {
|
|
|
288
316
|
*/
|
|
289
317
|
unreachable?: RobotsUnreachable;
|
|
290
318
|
/**
|
|
291
|
-
* Present when a disallow
|
|
292
|
-
*
|
|
293
|
-
*
|
|
319
|
+
* Present when a disallow was set aside, the publisher's or the one an
|
|
320
|
+
* unreachable robots.txt implies: the fetch went ahead (`skippedFetch:
|
|
321
|
+
* false`) and this says on whose word.
|
|
294
322
|
*/
|
|
295
|
-
override?:
|
|
323
|
+
override?: AppliedRobotsOverride;
|
|
296
324
|
}
|
|
297
325
|
/**
|
|
298
326
|
* What actually went on the wire. Sorted by header name, lowercased names.
|
|
@@ -86,7 +86,7 @@ export interface SitemapFileRecord {
|
|
|
86
86
|
kind: SitemapFileKind;
|
|
87
87
|
/** `<loc>` entries the file holds (child sitemaps for an index), http(s) ones only; null when the file was not parsed. */
|
|
88
88
|
entries: number | null;
|
|
89
|
-
/** The robots.txt verdict for the file's own URL under the crawl's identity (an unreachable robots.txt is `disallowed`, as for a page); null when the URL failed its egress check before robots.txt was consulted. */
|
|
89
|
+
/** The robots.txt verdict for the file's own URL under the crawl's identity (an unreachable robots.txt is `disallowed`, as for a page); null when the URL failed its egress check before robots.txt was consulted. A `disallowed` file is `refused`, unless the crawl or map was started with ignoreRobotsTxt and read it. */
|
|
90
90
|
robots: 'allowed' | 'disallowed' | 'no_robots' | null;
|
|
91
91
|
/** Whether the request left through the operator's environment proxy (local mode); a hosted server never has one. */
|
|
92
92
|
proxyUsed: boolean;
|
|
@@ -15,7 +15,7 @@ import type { PageActionType } from './actions.js';
|
|
|
15
15
|
* changes its meaning. A change that cannot follow this rule gets a new
|
|
16
16
|
* schemaVersion (`w2l.evidence/2`) and a new schema file.
|
|
17
17
|
*/
|
|
18
|
-
import type { CrawlMode, RobotsUnreachable } from './compliance.js';
|
|
18
|
+
import type { CrawlMode, RobotsOverrideBasis, RobotsUnreachable } from './compliance.js';
|
|
19
19
|
import type { BlockReason, BudgetKind, FailureReason, Lane, ResultStatus } from './status.js';
|
|
20
20
|
export declare const EVIDENCE_SCHEMA_VERSION = "w2l.evidence/1";
|
|
21
21
|
/**
|
|
@@ -33,6 +33,30 @@ export type FieldEvidenceSource = (typeof FIELD_EVIDENCE_SOURCES)[number];
|
|
|
33
33
|
*/
|
|
34
34
|
export declare const EVIDENCE_ARTIFACT_KINDS: readonly ["snapshot", "screenshot", "file"];
|
|
35
35
|
export type EvidenceArtifactKind = (typeof EVIDENCE_ARTIFACT_KINDS)[number];
|
|
36
|
+
/**
|
|
37
|
+
* The route that produced a result (ADR 0005): `http` (W2L's HTTP client, undici), `http_compat` (the
|
|
38
|
+
* browser-compatible HTTP transport, impit), `browser` (the local headless browser), `enhanced_browser`
|
|
39
|
+
* (the local browser on Patchright), `authed_browser` (the local browser with the user's saved login),
|
|
40
|
+
* `user_browser` (the person's own browser after a handoff), `vendor` (a third-party browser service).
|
|
41
|
+
*/
|
|
42
|
+
export declare const ACCESS_ROUTES: readonly ["http", "http_compat", "browser", "enhanced_browser", "authed_browser", "user_browser", "vendor"];
|
|
43
|
+
export type AccessRoute = (typeof ACCESS_ROUTES)[number];
|
|
44
|
+
/**
|
|
45
|
+
* How a result was reached, read from the result's own trace. Added to v1 with enhanced access
|
|
46
|
+
* (EVIDENCE_RECORD_ADDED_KEYS).
|
|
47
|
+
*/
|
|
48
|
+
export interface EvidenceAccess {
|
|
49
|
+
/** Null when no lane produced the result (a run cut before a rung answered, a rung that threw, a lockdown miss). */
|
|
50
|
+
route: AccessRoute | null;
|
|
51
|
+
/** The client that sent the requests: undici, impit, playwright, patchright, the person's browser, or the vendor's id; null when the result does not say. */
|
|
52
|
+
executor: string | null;
|
|
53
|
+
/** The executor's version as the lane reported it; null when it reported none. */
|
|
54
|
+
executorVersion: string | null;
|
|
55
|
+
/** The browser profile the HTTP transport sent (impit's); null on every other route. Its TLS fingerprint was not observed. */
|
|
56
|
+
profile: string | null;
|
|
57
|
+
/** Third-party spend of the run that produced the result: 0 when no paid service was called, null when a called service stated no price. */
|
|
58
|
+
externalCostUsd: number | null;
|
|
59
|
+
}
|
|
36
60
|
export interface EvidenceRedirectChain {
|
|
37
61
|
/**
|
|
38
62
|
* Every URL W2L requested for the page, in order: the requested URL first,
|
|
@@ -61,8 +85,10 @@ export interface EvidenceRobotsDecision {
|
|
|
61
85
|
/** Why robots.txt could not be fetched (then `decision` is `disallowed`, RFC 9309 §2.3.1.4); null when it was. */
|
|
62
86
|
unreachable: RobotsUnreachable | null;
|
|
63
87
|
crawlDelayMs: number | null;
|
|
64
|
-
/** Whether the fetch went ahead
|
|
88
|
+
/** Whether the fetch went ahead although `decision` is `disallowed`; on whose word is `overrideBasis`, and the reason is in the trace, the warnings and, in the browser lane, the compliance record. */
|
|
65
89
|
userOverride: boolean;
|
|
90
|
+
/** On whose word the disallow was set aside (`RobotsOverrideBasis`); null when it was not. Optional in the v1 schema file, written on every record. */
|
|
91
|
+
overrideBasis: RobotsOverrideBasis | null;
|
|
66
92
|
}
|
|
67
93
|
export interface EvidenceOutputSha256 {
|
|
68
94
|
/** SHA-256 of the UTF-8 bytes of the delivered `markdown`; null when none was delivered. */
|
|
@@ -169,12 +195,14 @@ export interface EvidenceRecord {
|
|
|
169
195
|
* the caller's ran in it, so its content may be the script's.
|
|
170
196
|
*/
|
|
171
197
|
pageActions: EvidencePageActions | null;
|
|
198
|
+
/** How the result was reached (EvidenceAccess). Added to v1 later (EVIDENCE_RECORD_ADDED_KEYS). */
|
|
199
|
+
access: EvidenceAccess;
|
|
172
200
|
}
|
|
173
201
|
/** Field order of the record and of each nested object, as in the schema file. */
|
|
174
202
|
export declare const EVIDENCE_RECORD_KEYS: {
|
|
175
|
-
readonly record: readonly ["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions"];
|
|
203
|
+
readonly record: readonly ["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions", "access"];
|
|
176
204
|
readonly redirectChain: readonly ["urls", "complete"];
|
|
177
|
-
readonly robotsDecision: readonly ["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride"];
|
|
205
|
+
readonly robotsDecision: readonly ["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride", "overrideBasis"];
|
|
178
206
|
readonly outputSha256: readonly ["markdown", "json"];
|
|
179
207
|
readonly extractor: readonly ["name", "version", "commit"];
|
|
180
208
|
readonly fieldEvidence: readonly ["source", "locator"];
|
|
@@ -183,6 +211,7 @@ export declare const EVIDENCE_RECORD_KEYS: {
|
|
|
183
211
|
readonly pageActions: readonly ["steps", "scriptRan"];
|
|
184
212
|
readonly pageActionStep: readonly ["type", "outcome"];
|
|
185
213
|
readonly requestHeader: readonly ["name", "valueSha256"];
|
|
214
|
+
readonly access: readonly ["route", "executor", "executorVersion", "profile", "externalCostUsd"];
|
|
186
215
|
};
|
|
187
216
|
/**
|
|
188
217
|
* Keys added to v1 after it was first published: optional in the schema, so
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type {
|
|
1
|
+
import type { AppliedRobotsOverride } from './compliance.js';
|
|
2
2
|
import type { FetchWarning, TraceEvent } from './result.js';
|
|
3
3
|
import type { AttributeSelector, ListFormatRequest, ScreenshotOptions } from './structured.js';
|
|
4
4
|
import type { PageAction } from './actions.js';
|
|
@@ -14,6 +14,47 @@ export interface ExecutionContext {
|
|
|
14
14
|
* when that lane never returns (a deadline) or another rung's result answers.
|
|
15
15
|
*/
|
|
16
16
|
onRobotsOverride?: (applied: RobotsOverrideApplied) => void;
|
|
17
|
+
/**
|
|
18
|
+
* The task's cookie session (ADR 0005 `egress_sessions`): the cookies a page's responses set,
|
|
19
|
+
* sent again to their site on the task's later pages, by every local rung. Absent: no cookie is
|
|
20
|
+
* kept or sent, as before.
|
|
21
|
+
*/
|
|
22
|
+
cookieSession?: CookieSession;
|
|
23
|
+
}
|
|
24
|
+
/** A cookie as a browser context takes and gives it (Playwright's shape). */
|
|
25
|
+
export interface ContextCookie {
|
|
26
|
+
name: string;
|
|
27
|
+
value: string;
|
|
28
|
+
domain: string;
|
|
29
|
+
path: string;
|
|
30
|
+
/** Seconds since the epoch; -1 for a session cookie. */
|
|
31
|
+
expires: number;
|
|
32
|
+
httpOnly: boolean;
|
|
33
|
+
secure: boolean;
|
|
34
|
+
sameSite: 'Strict' | 'Lax' | 'None';
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* One task's cookies, matched to a URL by RFC 6265 (domain, path, secure, expiry). Values never
|
|
38
|
+
* leave it into a record or a trace: a lane reports the session's `id` and counts.
|
|
39
|
+
*/
|
|
40
|
+
export interface CookieSession {
|
|
41
|
+
/** An opaque id for the record, unrelated to any cookie value. */
|
|
42
|
+
readonly id: string;
|
|
43
|
+
/** The `Cookie` header for a request to this URL; empty when none applies. */
|
|
44
|
+
cookieHeader(url: string): Promise<string>;
|
|
45
|
+
/** Keep the `Set-Cookie` lines a response to this URL carried; returns how many were kept. */
|
|
46
|
+
store(url: string, setCookies: readonly string[]): Promise<number>;
|
|
47
|
+
/** The cookies a browser context loading this URL should start with. */
|
|
48
|
+
browserCookies(url: string): Promise<ContextCookie[]>;
|
|
49
|
+
/**
|
|
50
|
+
* Keep what a browser context changed: given the cookies it started with and the ones it holds
|
|
51
|
+
* after the page, store the new and changed ones and delete the ones it dropped, unless another
|
|
52
|
+
* page of the task changed that cookie meanwhile. Unchanged cookies are left as the session has them.
|
|
53
|
+
*/
|
|
54
|
+
storeBrowserChanges(startedWith: readonly ContextCookie[], held: readonly ContextCookie[]): Promise<{
|
|
55
|
+
kept: number;
|
|
56
|
+
removed: number;
|
|
57
|
+
}>;
|
|
17
58
|
}
|
|
18
59
|
/** What a lane reports when it sets a robots.txt rule aside under a recorded override. */
|
|
19
60
|
export interface RobotsOverrideApplied {
|
|
@@ -52,15 +93,18 @@ export interface FetchOptions {
|
|
|
52
93
|
*/
|
|
53
94
|
parsers?: readonly PdfParser[];
|
|
54
95
|
/**
|
|
55
|
-
* A
|
|
56
|
-
* disallows it. robots.txt is still read and its
|
|
57
|
-
*
|
|
58
|
-
*
|
|
59
|
-
* and local browser lanes apply it; the provider lane takes none.
|
|
60
|
-
* Set
|
|
61
|
-
* `robotsOverrides` entry
|
|
62
|
-
|
|
63
|
-
|
|
96
|
+
* A decision to fetch this one URL although its host's robots.txt
|
|
97
|
+
* disallows it or could not be read. robots.txt is still read and its
|
|
98
|
+
* verdict recorded, Crawl-delay included; the override goes into the
|
|
99
|
+
* trace, the warnings and, in the browser lane, the compliance record. The
|
|
100
|
+
* HTTP and local browser lanes apply it; the provider lane takes none.
|
|
101
|
+
* Set by the engine: a scrape's `robotsOverride` or a batch's
|
|
102
|
+
* `robotsOverrides` entry, else, on a local server, `user_named_url` for
|
|
103
|
+
* every URL a scrape or batch names and `ignore_robots_txt` for a crawl's
|
|
104
|
+
* pages when the crawl asked. A navigation a page's steps make to another
|
|
105
|
+
* URL is checked against robots.txt whatever this says.
|
|
106
|
+
*/
|
|
107
|
+
robotsOverride?: AppliedRobotsOverride;
|
|
64
108
|
/**
|
|
65
109
|
* CSS selectors naming the only elements to keep. The content is those
|
|
66
110
|
* elements, in document order, copied from the page before anything is
|
|
@@ -237,7 +237,7 @@ export interface DocumentExtraction {
|
|
|
237
237
|
labelledValues?: readonly LabelledValue[];
|
|
238
238
|
}
|
|
239
239
|
/** The rule that decided a page's data is most likely rendered client-side (see RenderSignals). */
|
|
240
|
-
export type RenderReason = 'empty_table_with_scripts' | 'empty_app_root' | 'script_shell' | 'js_fallback' | 'hydration_shell' | 'aria_busy';
|
|
240
|
+
export type RenderReason = 'empty_table_with_scripts' | 'empty_app_root' | 'script_shell' | 'js_fallback' | 'hydration_shell' | 'aria_busy' | 'hydration_list_partial';
|
|
241
241
|
/** A client-side rendering marker found in the page as received (see RenderSignals). */
|
|
242
242
|
export type RenderMarker = 'hydration_state' | 'app_root_empty' | 'noscript_notice' | 'js_fallback_marker' | 'aria_busy';
|
|
243
243
|
/**
|
|
@@ -255,6 +255,16 @@ export interface RenderSignals {
|
|
|
255
255
|
emptyTables: number;
|
|
256
256
|
/** Markers found in the page as received. */
|
|
257
257
|
markers: readonly RenderMarker[];
|
|
258
|
+
/**
|
|
259
|
+
* On a listing page, the hydration data's list of named records that most
|
|
260
|
+
* outnumbers the ones its markup shows: how many it lists, and how many
|
|
261
|
+
* of their names are in the visible text. Present only when such a list
|
|
262
|
+
* decided `hydration_list_partial`.
|
|
263
|
+
*/
|
|
264
|
+
listRecords?: {
|
|
265
|
+
declared: number;
|
|
266
|
+
shown: number;
|
|
267
|
+
};
|
|
258
268
|
/** True when the signals say the data is most likely rendered client-side. */
|
|
259
269
|
clientRendered: boolean;
|
|
260
270
|
/**
|
|
@@ -23,7 +23,7 @@ export declare const FIRECRAWL_SHIM_SNAPSHOT: {
|
|
|
23
23
|
paths: readonly ["/scrape", "/crawl", "/crawl/:id", "/map"];
|
|
24
24
|
notCovered: readonly ["search", "interact", "agent", "monitor", "extract"];
|
|
25
25
|
};
|
|
26
|
-
export declare const FIRECRAWL_SHIM_DIFFS: readonly ["Challenge / block pages are success: false (Firecrawl often returns them as success markdown).", "A page with no main content is success: false (failed: empty_unverified) with the whole page in data.markdown as evidence; with onlyMainContent: false it is success: true.", "A page whose server HTML is a shell for data its scripts fill in is fetched again on the browser rung, and the rendered page is the answer when it holds more; otherwise the HTTP page is returned with client_rendered_suspected and low_content_yield warnings on the native response, whose messages /fc passes through as data.warning (one string, joined with a space), as it does every native warning. Firecrawl renders every page in a browser.", "No fire-engine, proxy pools or JSON extract.", "actions (scrape only; a crawl's scrapeOptions.actions is refused) run on the local browser rung alone, which such a request selects, after load, stability and waitFor and before the formats are read: wait (milliseconds up to 60000, or a selector, waited for up to 60 s within the scrape's timeout), click (all: true clicks every match), write (into the focused element), press, scroll (one screen up or down, of the page or the element a selector names), screenshot, scrape, executeJavascript (a function body; return gives the value) and pdf; at most 50 steps. data.actions holds screenshots and pdfs as data: URIs (Firecrawl returns URLs), scrapes as { url, html } and javascriptReturns as { type, value }. A step that fails stops the steps after it: success is false with data.actions.failed naming the step, its code and message, and data.markdown is the page as it stood. A step that leads the page to a URL robots.txt or the egress policy refuses fails with navigation_refused and that page is not read. A hosted server refuses actions, and the cache is not used with them.", "An omitted maxAge reuses nothing: every page is fetched live unless the request sets maxAge above 0, minAge or lockdown (Firecrawl reuses its own index by default; its Python SDK sends maxAge 4 hours). A reused page is one W2L itself stored, under the same options, on this server (its task root), never a shared index; only a success is stored, and data.metadata says cacheState hit with cachedAt (its fetch time) or miss when one was looked up. lockdown with no stored result is HTTP 404 SCRAPE_LOCKDOWN_CACHE_MISS on scrape, and nothing is fetched; a crawl in lockdown needs sitemap skip, and each page with no stored result is failed with cache_miss. Mode authed neither stores nor reuses. A Firecrawl body never sets useCached, W2L's reuse of a crawl's own pages on resume.", "Omitted limit / maxDepth stay unbounded on a local server; a hosted server takes its crawl limit for an omitted or null limit and refuses a larger one. Firecrawl defaults are 10000 / 10.", "maxDepth counts link hops from the start URL (Firecrawl calls that maxDiscoveryDepth); Firecrawl maxDepth counts URL path depth.", "Crawl start is mapped onto native POST /v1/crawl; the shim itself returns 200 {success,id,url}.", "creditsUsed and expiresAt are null: W2L counts no credits and keeps crawl results until their task directory is deleted.", "Crawl status describes the latest attempt: completed counts its successful pages, total adds its failed, blocked and duplicate pages and, while this API process runs the crawl, the pages in flight and queued (null for a paused crawl); data lists the failed and blocked pages too (with metadata.error) but not the duplicates, whose content is an earlier entry's, up to 100 per response (limit 1 to 1000) with next carrying a W2L cursor; skip is rejected.", "Scrape maps url, formats, onlyMainContent, includeTags, excludeTags, waitFor, timeout, headers, mobile, skipTlsVerification, fastMode, blockAds, removeBase64Images, maxAge, minAge, storeInCache, lockdown, actions, origin and integration; crawl maps url, limit (as maxPages), maxDepth, includePaths, excludePaths, regexOnFullURL, ignoreQueryParameters, deduplicateSimilarURLs, crawlEntireDomain (and its v1 name allowBackwardLinks), allowSubdomains, allowExternalLinks, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only), maxConcurrency, origin, integration and the same scrapeOptions (applied to every page). The formats are markdown, links, html, rawHtml, images, screenshot (also screenshot@fullPage, and { type: \"screenshot\", fullPage, quality, viewport }) and an { type: \"attributes\", selectors } entry; other formats and parameters the shim does not map (proxy, location, json, ...) are rejected by name with HTTP 400 and success: false; a refusal of stealth, proxy: stealth or enhanced, or ignoreRobotsTxt names the supported route in agent_hints.", "A crawl follows links inside the start URL's path subtree on its host and www twin by default (crawlEntireDomain false), folds /a and /a/, / and /index.html, www and apex, http and https into one page (deduplicateSimilarURLs true) and reports every collapsed or refused link in the native crawl status (discovery) and each page's trace (links_offered); allowSubdomains takes every host under the start URL's apex (no public-suffix list), allowExternalLinks every host, each page with its own robots.txt read.", "sitemap (default include, as in Firecrawl) reads the sitemaps the start URL's robots.txt names, or /sitemap.xml, with the crawl's own http identity, robots.txt verdict, SSRF checks and proxy, and queues their URLs ahead of the start page's links under the same host, subtree, path and depth rules; only follows no page link; skip reads none. The native crawl status lists every sitemap file read, refused or unreadable in discovery.sitemap; the shim's status carries nothing of it, and sitemap fetches have no signed compliance record. maxConcurrency caps the pages one crawl fetches at once, at most the service's worker count (HTTP 400 above it), and never raises the per-host ceiling.", "screenshot (data.screenshot, a data:image/png;base64 string, or image/jpeg with quality 1 to 100) is captured on the local browser rung alone, which such a request selects (no http attempt; a server without a browser rung refuses the format with HTTP 400): after load, stability and waitFor, before the DOM is read, CSS-pixel sized at the declared 1280x800 viewport (device scale factor 2 is declared, not baked into the image) or at the viewport asked for (integers 320..1920 by 240..1080, within the declared screen; a window size, not a change of identity); fullPage captures the document's whole height at that width without scrolling first, so sections a page loads on scroll may show unloaded. A capture the browser could not make leaves data.screenshot null with a screenshot_unavailable warning while the page stands; a file or a page that was not rendered has null too. Firecrawl captures at its own viewport and may return a URL instead of the image.", "images (data.images) lists every image URL of the whole document as received: img src and srcset candidates, picture sources, lazy data-src/data-srcset/data-lazy-src/data-original, video posters, image_src links, og:image and twitter:image, absolute http(s) with the fragment stripped, each once, in document order, data: URIs left out; includeTags, excludeTags and onlyMainContent do not narrow it. attributes (data.attributes) gives, per selector, the named attribute's values as written, elements without it skipped; a selector W2L does not match is HTTP 400 by name, as for includeTags. Both are absent for a file and for a page that is success: false. removeBase64Images (default true) keeps an image's alt text where Firecrawl writes a (<Base64-Image-Removed>) placeholder; false keeps the data: URI in the Markdown.", "origin (the Firecrawl SDKs' client label) and integration are stored, not echoed: the scrape record (GET /v1/scrapes/:id) and the crawl task carry them, and nothing sent to the target changes.", "data.metadata carries scrapeId (a UUID per call, which GET /v1/scrapes/:id looks up), proxyUsed (operator for the server's environment proxy, user for the caller's own egress, else null), timezone (the browser rung's declared zone, null on the HTTP rung), creditsUsed: null (W2L counts no credits), concurrencyLimited and concurrencyQueueDurationMs (whether and how long the per-origin ceiling held the fetch back), and cacheState and cachedAt when the cache was asked (never a guessed miss).", "A page whose result W2L has advice about (a login wall, a robots.txt rule, a gate, a cut, a script-filled shell) carries data.agent_hints, one sentence each; the native response calls them agentHints. A request refused for an option W2L does not offer carries agent_hints in the error envelope, and a caller over the server's per-minute rate limit gets HTTP 429 { success: false, error, code: rate_limited, agent_hints } with Retry-After.", "headers never override the User-Agent, the client hints, a credential (authorization, cookie) or a transport header: such a header is HTTP 400 naming it, where Firecrawl sends it. The headers go to the requested origin after the declared identity and are on the record (the trace, the browser lane's signed sentHeaders); both rungs withhold them from a redirect hop to another origin and say so (custom_headers_withheld).", "mobile selects a declared Android Chrome identity (User-Agent, client hints, 412x915 viewport, touch) that robots.txt is evaluated against and the record carries; the page is whatever the site serves to it, with no DOM rewriting. It is refused with mode research.", "skipTlsVerification relaxes certificate verification for one local request and its robots.txt lookup, recorded in the trace (tls_verification_skipped) and a tls_unverified warning the native response carries; a hosted W2L refuses it with HTTP 400. Without it a bad certificate is success: false with failed: tls_error. Firecrawl's Python SDK sends true by default; W2L verifies by default.", "fastMode keeps the http rung alone: a page that needs scripts is success: false with failed: empty_unverified, never rendered; waitFor has no effect under it. Firecrawl's fast mode still renders.", "blockAds (default true) aborts requests to a bundled list of about 50 ad-serving hosts on the local browser rung and removes ad and cookie-banner elements before extraction; false keeps them. The list is curated, not EasyList: ads from hosts outside it are not blocked.", "html is the cleaned HTML the markdown is written from: the main content, the whole page without scripts, styles, form controls and embedded media when onlyMainContent is false, or a <body> holding the includeTags elements. rawHtml is the page as the answering rung received it: the response body on the HTTP rung, the rendered DOM on a browser rung. Both are null for a file and for a page that is success: false.", "includeTags keeps only the named elements, in document order, whatever onlyMainContent says; excludeTags removes elements from the main content, the whole page and an includeTags selection. A selector that does not parse, or that uses a sibling combinator, a positional pseudo-class, :has() or another pseudo-class W2L does not match, is rejected with HTTP 400.", "An omitted timeout stays 300000 ms (Firecrawl: 30000). A timeout is answered with HTTP 200: success: true with the content fetched so far (native status partial), or success: false with failed: timeout; Firecrawl answers it with an error.", "waitFor skips the HTTP rung, which cannot run scripts, and starts at the browser rung; the wait counts toward timeout.", "metadata has title, description, language, keywords, robots and favicon only when the page declares them, and the Open Graph (ogTitle, ogDescription, ogUrl, ogImage, ogAudio, ogVideo, ogDeterminer, ogLocale, ogLocaleAlternate, ogSiteName), Dublin Core (dcTermsCreated, dcDateCreated, dcDate, dcTermsType, dcType, dcTermsAudience, dcTermsSubject, dcSubject, dcDescription, dcTermsKeywords) and article (publishedTime, modifiedTime, articleTag, articleSection) tags under Firecrawl's names, each only when the page states it, as written (no date normalisation, no fallback from another tag); twitter:* and other meta tags are not passed through, and a failed or blocked page has none.", "A PDF answers success: true with its text layer as markdown and metadata.numPages (the document's page count). parsers maps Firecrawl's pdf entry (the string or { type: \"pdf\", mode, maxPages, pages, pageMarkers }); mode fast and auto both read the text layer, and mode ocr and the image parser are refused by name. pageMarkers is false unless asked, as on Firecrawl; with true W2L writes a <!-- page N --> line before each page (Firecrawl writes --- and the marker between pages). pages: true adds data.pages, [{ pageNumber, markdown }]. maxPages (1 to 10000) reads the first pages, and a cut it asked for stays success: true. parsers [] or v1 parsePDF false reads no PDF: success: true with markdown null and the file saved as received. A PDF without a text layer is success: false with failed: empty_unverified (no OCR). CSV, JSON and text files give their text as received; XLSX, XLS and ZIP files are success: true with markdown null. A file over W2L_MAX_FILE_BYTES is success: false with failed: body_too_large.", "Map (POST /fc/v1/map) maps url, search, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only; both v1 flags true is HTTP 400), includeSubdomains, ignoreQueryParameters, limit (1 to 100000, default 5000), timeout (1000 to 300000 ms for the whole map, default 60000; Firecrawl documents no default), origin and integration onto native POST /v1/map, and answers 200 { success: true, id, links: [url strings], warning?, agent_hints? }, or 200 { success: false, id, error, links: [] } when the map found nothing because a source failed or its deadline passed; useIndex, location, ignoreCache, threatProtection and auditMetadata are refused by name (useIndex with the hint that W2L keeps no URL index). Omitted options take W2L's defaults: includeSubdomains and ignoreQueryParameters are false, where Firecrawl v2 documents true for both. search keeps the URLs in which every word appears in the decoded URL or the title in hand, in discovery order; Firecrawl orders by relevance. A map reads the sitemaps the site declares and one page body (the start URL, http rung only), so a site without a sitemap maps only its start page's links; a title is the start page's own, an anchor's text or a sitemap's <news:title>, never fetched from the target; robots-disallowed URLs are left out and counted on the native response (GET /v1/maps/:id), which also records every sitemap file read.", "A crawl's webhook (a URL string or { url, headers, metadata, events }) is mapped onto the native webhook and its receiver gets Firecrawl's payload shape: { success, type: crawl.started | crawl.page | crawl.completed | crawl.failed, id, data: [page], metadata, error? }, one durable delivery per event with retries, every request carrying x-w2l-event-id, x-w2l-event-version and x-w2l-delivery-id (and the signature pair with secretEnv, a native option). A cancelled crawl is crawl.failed with error \"cancelled\". The native rules apply: https (plain http for a loopback receiver of a local server only), no content-type, host or x-w2l-* header, at most 32 headers and 32 metadata strings; a hosted server takes public https receivers only. GET /v1/deliveries?jobId=<id> on the native API lists the deliveries."];
|
|
26
|
+
export declare const FIRECRAWL_SHIM_DIFFS: readonly ["Challenge / block pages are success: false (Firecrawl often returns them as success markdown).", "A page with no main content is success: false (failed: empty_unverified) with the whole page in data.markdown as evidence; with onlyMainContent: false it is success: true.", "A page whose server HTML is a shell for data its scripts fill in is fetched again on the browser rung, and the rendered page is the answer when it holds more; otherwise the HTTP page is returned with client_rendered_suspected and low_content_yield warnings on the native response, whose messages /fc passes through as data.warning (one string, joined with a space), as it does every native warning. Firecrawl renders every page in a browser.", "No fire-engine, proxy pools or JSON extract.", "actions (scrape only; a crawl's scrapeOptions.actions is refused) run on the local browser rung alone, which such a request selects, after load, stability and waitFor and before the formats are read: wait (milliseconds up to 60000, or a selector, waited for up to 60 s within the scrape's timeout), click (all: true clicks every match), write (into the focused element), press, scroll (one screen up or down, of the page or the element a selector names), screenshot, scrape, executeJavascript (a function body; return gives the value) and pdf; at most 50 steps. data.actions holds screenshots and pdfs as data: URIs (Firecrawl returns URLs), scrapes as { url, html } and javascriptReturns as { type, value }. A step that fails stops the steps after it: success is false with data.actions.failed naming the step, its code and message, and data.markdown is the page as it stood. A step that leads the page to a URL robots.txt or the egress policy refuses fails with navigation_refused and that page is not read. A hosted server refuses actions, and the cache is not used with them.", "An omitted maxAge reuses nothing: every page is fetched live unless the request sets maxAge above 0, minAge or lockdown (Firecrawl reuses its own index by default; its Python SDK sends maxAge 4 hours). A reused page is one Octocrawl itself stored, under the same options, on this server (its task root), never a shared index; only a success is stored, and data.metadata says cacheState hit with cachedAt (its fetch time) or miss when one was looked up. lockdown with no stored result is HTTP 404 SCRAPE_LOCKDOWN_CACHE_MISS on scrape, and nothing is fetched; a crawl in lockdown needs sitemap skip, and each page with no stored result is failed with cache_miss. Mode authed neither stores nor reuses. A Firecrawl body never sets useCached, Octocrawl's reuse of a crawl's own pages on resume.", "Omitted limit / maxDepth stay unbounded on a local server; a hosted server takes its crawl limit for an omitted or null limit and refuses a larger one. Firecrawl defaults are 10000 / 10.", "maxDepth counts link hops from the start URL (Firecrawl calls that maxDiscoveryDepth); Firecrawl maxDepth counts URL path depth.", "Crawl start is mapped onto native POST /v1/crawl; the shim itself returns 200 {success,id,url}.", "creditsUsed and expiresAt are null: Octocrawl counts no credits and keeps crawl results until their task directory is deleted.", "Crawl status describes the latest attempt: completed counts its successful pages, total adds its failed, blocked and duplicate pages and, while this API process runs the crawl, the pages in flight and queued (null for a paused crawl); data lists the failed and blocked pages too (with metadata.error) but not the duplicates, whose content is an earlier entry's, up to 100 per response (limit 1 to 1000) with next carrying an Octocrawl cursor; skip is rejected.", "Scrape maps url, formats, onlyMainContent, includeTags, excludeTags, waitFor, timeout, headers, mobile, skipTlsVerification, fastMode, blockAds, removeBase64Images, maxAge, minAge, storeInCache, lockdown, actions, origin and integration; crawl maps url, limit (as maxPages), maxDepth, includePaths, excludePaths, regexOnFullURL, ignoreQueryParameters, deduplicateSimilarURLs, crawlEntireDomain (and its v1 name allowBackwardLinks), allowSubdomains, allowExternalLinks, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only), maxConcurrency, ignoreRobotsTxt (v2; a local server only), origin, integration and the same scrapeOptions (applied to every page). The formats are markdown, links, html, rawHtml, images, screenshot (also screenshot@fullPage, and { type: \"screenshot\", fullPage, quality, viewport }) and an { type: \"attributes\", selectors } entry; other formats and parameters the shim does not map (proxy, location, json, ...) are rejected by name with HTTP 400 and success: false; a refusal of stealth, proxy: stealth or enhanced, or ignoreRobotsTxt on a scrape or map names the supported route in agent_hints.", "A crawl follows links inside the start URL's path subtree on its host and www twin by default (crawlEntireDomain false), folds /a and /a/, / and /index.html, www and apex, http and https into one page (deduplicateSimilarURLs true) and reports every collapsed or refused link in the native crawl status (discovery) and each page's trace (links_offered); allowSubdomains takes every host under the start URL's apex (no public-suffix list), allowExternalLinks every host, each page with its own robots.txt read.", "sitemap (default include, as in Firecrawl) reads the sitemaps the start URL's robots.txt names, or /sitemap.xml, with the crawl's own http identity, robots.txt verdict, SSRF checks and proxy, and queues their URLs ahead of the start page's links under the same host, subtree, path and depth rules; only follows no page link; skip reads none. The native crawl status lists every sitemap file read, refused or unreadable in discovery.sitemap; the shim's status carries nothing of it, and sitemap fetches have no signed compliance record. maxConcurrency caps the pages one crawl fetches at once, at most the service's worker count (HTTP 400 above it), and never raises the per-host ceiling.", "screenshot (data.screenshot, a data:image/png;base64 string, or image/jpeg with quality 1 to 100) is captured on the local browser rung alone, which such a request selects (no http attempt; a server without a browser rung refuses the format with HTTP 400): after load, stability and waitFor, before the DOM is read, CSS-pixel sized at the declared 1280x800 viewport (device scale factor 2 is declared, not baked into the image) or at the viewport asked for (integers 320..1920 by 240..1080, within the declared screen; a window size, not a change of identity); fullPage captures the document's whole height at that width without scrolling first, so sections a page loads on scroll may show unloaded. A capture the browser could not make leaves data.screenshot null with a screenshot_unavailable warning while the page stands; a file or a page that was not rendered has null too. Firecrawl captures at its own viewport and may return a URL instead of the image.", "images (data.images) lists every image URL of the whole document as received: img src and srcset candidates, picture sources, lazy data-src/data-srcset/data-lazy-src/data-original, video posters, image_src links, og:image and twitter:image, absolute http(s) with the fragment stripped, each once, in document order, data: URIs left out; includeTags, excludeTags and onlyMainContent do not narrow it. attributes (data.attributes) gives, per selector, the named attribute's values as written, elements without it skipped; a selector Octocrawl does not match is HTTP 400 by name, as for includeTags. Both are absent for a file and for a page that is success: false. removeBase64Images (default true) keeps an image's alt text where Firecrawl writes a (<Base64-Image-Removed>) placeholder; false keeps the data: URI in the Markdown.", "origin (the Firecrawl SDKs' client label) and integration are stored, not echoed: the scrape record (GET /v1/scrapes/:id) and the crawl task carry them, and nothing sent to the target changes.", "data.metadata carries scrapeId (a UUID per call, which GET /v1/scrapes/:id looks up), proxyUsed (operator for the server's environment proxy, user for the caller's own egress, else null), timezone (the browser rung's declared zone, null on the HTTP rung), creditsUsed: null (Octocrawl counts no credits), concurrencyLimited and concurrencyQueueDurationMs (whether and how long the per-origin ceiling held the fetch back), and cacheState and cachedAt when the cache was asked (never a guessed miss).", "A page whose result Octocrawl has advice about (a login wall, a robots.txt rule, a gate, a cut, a script-filled shell) carries data.agent_hints, one sentence each; the native response calls them agentHints. A request refused for an option Octocrawl does not offer carries agent_hints in the error envelope, and a caller over the server's per-minute rate limit gets HTTP 429 { success: false, error, code: rate_limited, agent_hints } with Retry-After.", "headers never override the User-Agent, the client hints, a credential (authorization, cookie) or a transport header: such a header is HTTP 400 naming it, where Firecrawl sends it. The headers go to the requested origin after the declared identity and are on the record (the trace, the browser lane's signed sentHeaders); both rungs withhold them from a redirect hop to another origin and say so (custom_headers_withheld).", "mobile selects a declared Android Chrome identity (User-Agent, client hints, 412x915 viewport, touch) that robots.txt is evaluated against and the record carries; the page is whatever the site serves to it, with no DOM rewriting. It is refused with mode research.", "skipTlsVerification relaxes certificate verification for one local request and its robots.txt lookup, recorded in the trace (tls_verification_skipped) and a tls_unverified warning the native response carries; a hosted Octocrawl refuses it with HTTP 400. Without it a bad certificate is success: false with failed: tls_error. Firecrawl's Python SDK sends true by default; Octocrawl verifies by default.", "fastMode keeps the http rung alone: a page that needs scripts is success: false with failed: empty_unverified, never rendered; waitFor has no effect under it. Firecrawl's fast mode still renders.", "blockAds (default true) aborts requests to a bundled list of about 50 ad-serving hosts on the local browser rung and removes ad and cookie-banner elements before extraction; false keeps them. The list is curated, not EasyList: ads from hosts outside it are not blocked.", "html is the cleaned HTML the markdown is written from: the main content, the whole page without scripts, styles, form controls and embedded media when onlyMainContent is false, or a <body> holding the includeTags elements. rawHtml is the page as the answering rung received it: the response body on the HTTP rung, the rendered DOM on a browser rung. Both are null for a file and for a page that is success: false.", "includeTags keeps only the named elements, in document order, whatever onlyMainContent says; excludeTags removes elements from the main content, the whole page and an includeTags selection. A selector that does not parse, or that uses a sibling combinator, a positional pseudo-class, :has() or another pseudo-class Octocrawl does not match, is rejected with HTTP 400.", "An omitted timeout stays 300000 ms (Firecrawl: 30000). A timeout is answered with HTTP 200: success: true with the content fetched so far (native status partial), or success: false with failed: timeout; Firecrawl answers it with an error.", "waitFor skips the HTTP rung, which cannot run scripts, and starts at the browser rung; the wait counts toward timeout.", "metadata has title, description, language, keywords, robots and favicon only when the page declares them, and the Open Graph (ogTitle, ogDescription, ogUrl, ogImage, ogAudio, ogVideo, ogDeterminer, ogLocale, ogLocaleAlternate, ogSiteName), Dublin Core (dcTermsCreated, dcDateCreated, dcDate, dcTermsType, dcType, dcTermsAudience, dcTermsSubject, dcSubject, dcDescription, dcTermsKeywords) and article (publishedTime, modifiedTime, articleTag, articleSection) tags under Firecrawl's names, each only when the page states it, as written (no date normalisation, no fallback from another tag); twitter:* and other meta tags are not passed through, and a failed or blocked page has none.", "A PDF answers success: true with its text layer as markdown and metadata.numPages (the document's page count). parsers maps Firecrawl's pdf entry (the string or { type: \"pdf\", mode, maxPages, pages, pageMarkers }); mode fast and auto both read the text layer, and mode ocr and the image parser are refused by name. pageMarkers is false unless asked, as on Firecrawl; with true Octocrawl writes a <!-- page N --> line before each page (Firecrawl writes --- and the marker between pages). pages: true adds data.pages, [{ pageNumber, markdown }]. maxPages (1 to 10000) reads the first pages, and a cut it asked for stays success: true. parsers [] or v1 parsePDF false reads no PDF: success: true with markdown null and the file saved as received. A PDF without a text layer is success: false with failed: empty_unverified (no OCR). CSV, JSON and text files give their text as received; XLSX, XLS and ZIP files are success: true with markdown null. A file over W2L_MAX_FILE_BYTES is success: false with failed: body_too_large.", "Map (POST /fc/v1/map) maps url, search, sitemap (v2; v1 ignoreSitemap true is skip and false include, sitemapOnly true is only; both v1 flags true is HTTP 400), includeSubdomains, ignoreQueryParameters, limit (1 to 100000, default 5000), timeout (1000 to 300000 ms for the whole map, default 60000; Firecrawl documents no default), origin and integration onto native POST /v1/map, and answers 200 { success: true, id, links: [url strings], warning?, agent_hints? }, or 200 { success: false, id, error, links: [] } when the map found nothing because a source failed or its deadline passed; useIndex, location, ignoreCache, threatProtection and auditMetadata are refused by name (useIndex with the hint that Octocrawl keeps no URL index). Omitted options take Octocrawl's defaults: includeSubdomains and ignoreQueryParameters are false, where Firecrawl v2 documents true for both. search keeps the URLs in which every word appears in the decoded URL or the title in hand, in discovery order; Firecrawl orders by relevance. A map reads the sitemaps the site declares and one page body (the start URL, http rung only), so a site without a sitemap maps only its start page's links; a title is the start page's own, an anchor's text or a sitemap's <news:title>, never fetched from the target; robots-disallowed URLs are left out and counted on the native response (GET /v1/maps/:id), which also records every sitemap file read.", "A crawl's webhook (a URL string or { url, headers, metadata, events }) is mapped onto the native webhook and its receiver gets Firecrawl's payload shape: { success, type: crawl.started | crawl.page | crawl.completed | crawl.failed, id, data: [page], metadata, error? }, one durable delivery per event with retries, every request carrying x-w2l-event-id, x-w2l-event-version and x-w2l-delivery-id (and the signature pair with secretEnv, a native option). A cancelled crawl is crawl.failed with error \"cancelled\". The native rules apply: https (plain http for a loopback receiver of a local server only), no content-type, host or x-w2l-* header, at most 32 headers and 32 metadata strings; a hosted server takes public https receivers only. GET /v1/deliveries?jobId=<id> on the native API lists the deliveries."];
|
|
27
27
|
export interface FirecrawlPage {
|
|
28
28
|
markdown: string | null;
|
|
29
29
|
/** Present when the `html` format was asked for; null when the page has none (a file, a page that did not succeed). */
|
|
@@ -40,11 +40,16 @@ export interface MapLink {
|
|
|
40
40
|
sitemapFile?: string;
|
|
41
41
|
/** The `<lastmod>` the sitemap gave, as written. */
|
|
42
42
|
lastmod?: string;
|
|
43
|
-
/**
|
|
44
|
-
|
|
43
|
+
/**
|
|
44
|
+
* The robots.txt verdict for the URL under the map's declared identity. A
|
|
45
|
+
* URL robots.txt disallows (`disallowed`), or on a host whose robots.txt
|
|
46
|
+
* could not be read (`unreachable`), is a link only in a map started with
|
|
47
|
+
* ignoreRobotsTxt; otherwise it is refused.
|
|
48
|
+
*/
|
|
49
|
+
robots: 'allowed' | 'no_robots' | 'disallowed' | 'unreachable';
|
|
45
50
|
}
|
|
46
51
|
export type MapStatus = 'completed' | 'partial' | 'failed';
|
|
47
|
-
/** What the start page read gave, or why it was not read (`failed/policy_denied` when robots.txt disallows the start URL or could not be read). */
|
|
52
|
+
/** What the start page read gave, or why it was not read (`failed/policy_denied` when robots.txt disallows the start URL or could not be read, unless the map was started with ignoreRobotsTxt). */
|
|
48
53
|
export interface MapStartPage {
|
|
49
54
|
url: string;
|
|
50
55
|
finalUrl: string | null;
|
|
@@ -164,7 +169,8 @@ export interface MapStartPageRead {
|
|
|
164
169
|
/** A robots.txt verdict for one URL under the map's declared identity. */
|
|
165
170
|
export type MapRobotsVerdict = 'allowed' | 'no_robots' | {
|
|
166
171
|
disallowed: true;
|
|
167
|
-
unreachable?: string;
|
|
172
|
+
unreachable?: string; /** A rule robots.txt wrote for Octocrawl itself, which ignoreRobotsTxt does not set aside. */
|
|
173
|
+
octocrawl?: true;
|
|
168
174
|
};
|
|
169
175
|
/** The fetch paths a map uses; the API engine wires the real ones, tests inject fakes. */
|
|
170
176
|
export interface MapSources {
|
|
@@ -63,7 +63,8 @@ export interface NetworkPolicy {
|
|
|
63
63
|
* because the browser lane can route through only one proxy.
|
|
64
64
|
*/
|
|
65
65
|
export interface EgressProxy {
|
|
66
|
-
|
|
66
|
+
/** `environment`: HTTPS_PROXY / HTTP_PROXY. `pool`: one of the operator's W2L_EGRESS_PROXIES (ADR 0005 egress_sessions). */
|
|
67
|
+
source: 'environment' | 'pool';
|
|
67
68
|
/** From HTTPS_PROXY / https_proxy. Null sends https: URLs direct. */
|
|
68
69
|
https: ProxyServer | null;
|
|
69
70
|
/** From HTTP_PROXY / http_proxy. Null sends http: URLs direct. */
|
|
@@ -25,6 +25,14 @@ export declare class ProxyConfigError extends Error {
|
|
|
25
25
|
export declare function environmentProxy(env: Env): EgressProxy | null;
|
|
26
26
|
/** A local-mode policy that routes through the environment's proxy, when one is set. */
|
|
27
27
|
export declare function withEnvironmentProxy(policy: NetworkPolicy, env: Env): NetworkPolicy;
|
|
28
|
+
/**
|
|
29
|
+
* The operator's egress proxies (`W2L_EGRESS_PROXIES`, ADR 0005 `egress_sessions`): http:// or https://
|
|
30
|
+
* URLs, credentials in their userinfo, separated by commas or white space. Each one is a whole egress:
|
|
31
|
+
* every URL's scheme goes through it. Duplicates (the same endpoint) are kept once.
|
|
32
|
+
*/
|
|
33
|
+
export declare function egressProxies(raw: string | undefined): ProxyServer[];
|
|
34
|
+
/** A policy whose every request leaves through this egress proxy; NO_PROXY entries and loopback still go direct. */
|
|
35
|
+
export declare function withPoolProxy(policy: NetworkPolicy, server: ProxyServer): NetworkPolicy;
|
|
28
36
|
/** The operator proxy a URL leaves through under this policy; null means a direct connection. */
|
|
29
37
|
export declare function proxyFor(url: string | URL, policy: Pick<NetworkPolicy, 'origin' | 'egressProxy'>): ProxyServer | null;
|
|
30
38
|
export type NoProxyEntry = {
|