@octocrawl/sdk 0.3.0 → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/index.cjs +15 -9
- package/dist/index.js +15 -9
- package/dist/types/cjs/client.d.ts +1 -1
- package/dist/types/cjs/contracts/actions.d.ts +37 -4
- package/dist/types/cjs/contracts/api.d.ts +66 -10
- package/dist/types/cjs/contracts/checkpoint.d.ts +17 -1
- package/dist/types/cjs/contracts/compliance.d.ts +35 -7
- package/dist/types/cjs/contracts/crawl.d.ts +1 -1
- package/dist/types/cjs/contracts/evidenceRecord.d.ts +132 -4
- package/dist/types/cjs/contracts/execution.d.ts +121 -9
- package/dist/types/cjs/contracts/extractor.d.ts +11 -1
- package/dist/types/cjs/contracts/firecrawl.d.ts +1 -1
- package/dist/types/cjs/contracts/map.d.ts +10 -4
- package/dist/types/cjs/contracts/policy.d.ts +2 -1
- package/dist/types/cjs/contracts/proxy.d.ts +8 -0
- package/dist/types/cjs/contracts/result.d.ts +41 -4
- package/dist/types/cjs/contracts/session.d.ts +2 -0
- package/dist/types/cjs/contracts/status.d.ts +1 -1
- package/dist/types/cjs/version.d.ts +1 -1
- package/dist/types/esm/client.d.ts +1 -1
- package/dist/types/esm/contracts/actions.d.ts +37 -4
- package/dist/types/esm/contracts/api.d.ts +66 -10
- package/dist/types/esm/contracts/checkpoint.d.ts +17 -1
- package/dist/types/esm/contracts/compliance.d.ts +35 -7
- package/dist/types/esm/contracts/crawl.d.ts +1 -1
- package/dist/types/esm/contracts/evidenceRecord.d.ts +132 -4
- package/dist/types/esm/contracts/execution.d.ts +121 -9
- package/dist/types/esm/contracts/extractor.d.ts +11 -1
- package/dist/types/esm/contracts/firecrawl.d.ts +1 -1
- package/dist/types/esm/contracts/map.d.ts +10 -4
- package/dist/types/esm/contracts/policy.d.ts +2 -1
- package/dist/types/esm/contracts/proxy.d.ts +8 -0
- package/dist/types/esm/contracts/result.d.ts +41 -4
- package/dist/types/esm/contracts/session.d.ts +2 -0
- package/dist/types/esm/contracts/status.d.ts +1 -1
- package/dist/types/esm/version.d.ts +1 -1
- package/package.json +19 -2
|
@@ -4,6 +4,7 @@
|
|
|
4
4
|
* Each step is recorded in the trace with its outcome and timing; a step
|
|
5
5
|
* that fails ends the pipeline, and the result keeps the page as it stood.
|
|
6
6
|
*/
|
|
7
|
+
import type { BlockReason } from './status.js';
|
|
7
8
|
import type { ScreenshotEvidence } from './result.js';
|
|
8
9
|
import type { ScreenshotViewport } from './structured.js';
|
|
9
10
|
/** The most steps one request may run. */
|
|
@@ -149,8 +150,20 @@ export interface ActionPdf {
|
|
|
149
150
|
path: string | null;
|
|
150
151
|
base64: string;
|
|
151
152
|
}
|
|
152
|
-
/**
|
|
153
|
-
|
|
153
|
+
/**
|
|
154
|
+
* Why a scrollToEnd, loadMore or paginate step stopped. `max` and `deadline` stop short of the list's end; so does
|
|
155
|
+
* `challenge`: a check the site put up on the next page (a Cloudflare interstitial, a page of nothing but a CAPTCHA),
|
|
156
|
+
* which paginate stops at without reading it, the result then `blocked` with the check's reason.
|
|
157
|
+
*/
|
|
158
|
+
export type ListStop = 'end' | 'no_growth' | 'repeat' | 'max' | 'deadline' | 'challenge';
|
|
159
|
+
/** The check a paginate step stopped at (ListStop `challenge`): the page it would have been, its URL, and what the gate saw. */
|
|
160
|
+
export interface ListChallenge {
|
|
161
|
+
/** 1-based position the page would have had among the pages read. */
|
|
162
|
+
page: number;
|
|
163
|
+
url: string;
|
|
164
|
+
reason: BlockReason;
|
|
165
|
+
signals: readonly string[];
|
|
166
|
+
}
|
|
154
167
|
/** What a scrollToEnd, loadMore or paginate step did. */
|
|
155
168
|
export interface ListRun {
|
|
156
169
|
index: number;
|
|
@@ -160,8 +173,27 @@ export interface ListRun {
|
|
|
160
173
|
rounds: number;
|
|
161
174
|
/** Elements matching `itemSelector` at the end (on the last page for paginate; summed over its pages in `itemsRead`); null without one. */
|
|
162
175
|
items: number | null;
|
|
163
|
-
/**
|
|
176
|
+
/**
|
|
177
|
+
* paginate: elements matching `itemSelector` over every page read; null without one. After a continuation in the person's
|
|
178
|
+
* Chrome (`continued`), the kept pages' count plus each page they showed, as their tab counted it, only when the addresses show
|
|
179
|
+
* no page can be counted twice (the kept pages each at its own, the check's page and the pages shown at none of them, the check
|
|
180
|
+
* not at the list's own address), no page shows again what another shows (its items' whole text, as the list merge tells it:
|
|
181
|
+
* a result set tied to the session that made it comes back at new addresses), and every count is known; null (unknown) otherwise.
|
|
182
|
+
*/
|
|
164
183
|
itemsRead?: number | null;
|
|
184
|
+
/** paginate: pages taken from the task's checkpoint after a run cut at page N (ExecutionContext.listResume), counted in `rounds`; absent when none. */
|
|
185
|
+
resumed?: number;
|
|
186
|
+
/** paginate: the check the step stopped at, with `stoppedBy` `challenge`; absent otherwise. */
|
|
187
|
+
challenge?: ListChallenge;
|
|
188
|
+
/**
|
|
189
|
+
* paginate: the pages read after the check at `from`, by the person paging on in their own browser once they got through
|
|
190
|
+
* it (the batch handoff), counted in `rounds`; `stoppedBy` then says how that reading ended. Absent otherwise.
|
|
191
|
+
*/
|
|
192
|
+
continued?: {
|
|
193
|
+
from: number;
|
|
194
|
+
pages: number;
|
|
195
|
+
by: 'user_browser';
|
|
196
|
+
};
|
|
165
197
|
}
|
|
166
198
|
/** What the steps produced, each list in the order of its steps. */
|
|
167
199
|
export interface ActionsResult {
|
|
@@ -170,7 +202,8 @@ export interface ActionsResult {
|
|
|
170
202
|
scrapes: {
|
|
171
203
|
url: string;
|
|
172
204
|
html: string;
|
|
173
|
-
step?: number;
|
|
205
|
+
step?: number; /** Set when the person read the page in their own browser (a list's continuation after a check); absent for W2L's own browser. */
|
|
206
|
+
by?: 'user_browser';
|
|
174
207
|
}[];
|
|
175
208
|
/** `type` is the JavaScript `typeof` of the value (`null` for null). */
|
|
176
209
|
javascriptReturns: {
|
|
@@ -170,7 +170,28 @@ export interface ScrapeRequest extends PageOptions, RequestAttribution {
|
|
|
170
170
|
handoff?: {
|
|
171
171
|
waitMs?: number;
|
|
172
172
|
};
|
|
173
|
+
/**
|
|
174
|
+
* `my-browser`: read the page in the person's own Chrome, over remote
|
|
175
|
+
* debugging, without W2L fetching it first. The person allows the
|
|
176
|
+
* connection in Chrome, then the site in a page W2L opens there; the page
|
|
177
|
+
* is read without a click of theirs only on a site they allowed, and a
|
|
178
|
+
* check it shows waits for them (`handoff.waitMs`, default 10 min). Lane
|
|
179
|
+
* `my_browser`; never cached. Offered only by a server on the person's
|
|
180
|
+
* own machine; refused elsewhere, with `actions` or a screenshot, and with
|
|
181
|
+
* a mode other than standard (`unsupported_parameter`).
|
|
182
|
+
*/
|
|
183
|
+
lane?: 'my-browser';
|
|
184
|
+
/** One of three plain choices of how the page is reached (ACCESS_CHOICES); omitted, the server's own configuration. */
|
|
185
|
+
access?: AccessChoice;
|
|
173
186
|
}
|
|
187
|
+
/**
|
|
188
|
+
* How pages are reached, as three plain choices (ROADMAP PA item 7) beside the per-route options:
|
|
189
|
+
* `standard` (Octocrawl's own lanes and none that costs a third party), `enhanced` (also what the server's
|
|
190
|
+
* access grant of tier enhanced approves, within its budget; refused on a server without one), `my-browser`
|
|
191
|
+
* (the person's own Chrome, as `lane: "my-browser"`; a scrape or a batch only).
|
|
192
|
+
*/
|
|
193
|
+
export declare const ACCESS_CHOICES: readonly ["standard", "enhanced", "my-browser"];
|
|
194
|
+
export type AccessChoice = (typeof ACCESS_CHOICES)[number];
|
|
174
195
|
/** A recorded robots override for one URL of a batch. */
|
|
175
196
|
export interface RobotsUrlOverride extends RobotsOverride {
|
|
176
197
|
url: string;
|
|
@@ -344,6 +365,8 @@ export interface JobWebhookStatus {
|
|
|
344
365
|
export interface CrawlStartRequest extends PageOptions, RequestAttribution {
|
|
345
366
|
url: string;
|
|
346
367
|
mode?: ApiCrawlMode;
|
|
368
|
+
/** `standard` or `enhanced` (ACCESS_CHOICES); a crawl does not take `my-browser`. */
|
|
369
|
+
access?: Exclude<AccessChoice, 'my-browser'>;
|
|
347
370
|
maxPages?: number | null;
|
|
348
371
|
maxDepth?: number | null;
|
|
349
372
|
/**
|
|
@@ -411,6 +434,16 @@ export interface CrawlStartRequest extends PageOptions, RequestAttribution {
|
|
|
411
434
|
idempotencyKey?: string;
|
|
412
435
|
/** A receiver for the crawl's events (`started`, one `page` per page recorded, then `completed`, `failed` or `cancelled`); see WebhookConfig. */
|
|
413
436
|
webhook?: WebhookOption;
|
|
437
|
+
/**
|
|
438
|
+
* Fetch the pages and sitemap files robots.txt disallows, or whose
|
|
439
|
+
* robots.txt could not be read (Firecrawl v2's name). robots.txt is still
|
|
440
|
+
* read for every host and its verdict recorded on each page, Crawl-delay
|
|
441
|
+
* applied, with a `robots_overridden` warning and `overrideBasis:
|
|
442
|
+
* "ignore_robots_txt"` where a rule was set aside. A local server only: a
|
|
443
|
+
* hosted one refuses it by name. Default false: a crawl's links obey
|
|
444
|
+
* robots.txt.
|
|
445
|
+
*/
|
|
446
|
+
ignoreRobotsTxt?: boolean;
|
|
414
447
|
}
|
|
415
448
|
/** What the parser hands the engine: the request plus, from the `/fc` shim, the payload shape its receiver expects. */
|
|
416
449
|
export type ParsedCrawlStartRequest = CrawlStartRequest & {
|
|
@@ -454,6 +487,14 @@ export interface MapRequest extends RequestAttribution {
|
|
|
454
487
|
crawlEntireDomain?: boolean;
|
|
455
488
|
/** Default true, as on a crawl; a returned http link gives way to its https variant when that comes too, on an origin whose robots.txt the map read anyway and which allows it. */
|
|
456
489
|
deduplicateSimilarURLs?: boolean;
|
|
490
|
+
/**
|
|
491
|
+
* Return the URLs robots.txt disallows, or whose robots.txt could not be
|
|
492
|
+
* read, with that verdict on each link (`robots: "disallowed"` or
|
|
493
|
+
* `"unreachable"`), and read the start page and sitemap files past it.
|
|
494
|
+
* robots.txt is still read, within MAP_MAX_ROBOTS_HOSTS. A local server
|
|
495
|
+
* only: a hosted one refuses it by name. Default false.
|
|
496
|
+
*/
|
|
497
|
+
ignoreRobotsTxt?: boolean;
|
|
457
498
|
}
|
|
458
499
|
/** A map's `search`: at most this many characters after trimming, and this many whitespace-separated words. */
|
|
459
500
|
export declare const MAP_SEARCH_MAX_CHARS = 200;
|
|
@@ -482,6 +523,7 @@ export interface ActiveCrawlOptions {
|
|
|
482
523
|
allowExternalLinks: boolean;
|
|
483
524
|
regexOnFullURL: boolean;
|
|
484
525
|
maxConcurrency: number | null;
|
|
526
|
+
ignoreRobotsTxt: boolean;
|
|
485
527
|
/** The per-page options every page of the crawl gets: its formats, `includeLinks` and the page options. */
|
|
486
528
|
scrapeOptions: PageOptions & {
|
|
487
529
|
formats: readonly ScrapeFormat[];
|
|
@@ -507,6 +549,16 @@ export interface ActiveCrawlList {
|
|
|
507
549
|
export interface BatchStartRequest extends PageOptions, RequestAttribution {
|
|
508
550
|
urls: readonly string[];
|
|
509
551
|
mode?: ApiCrawlMode;
|
|
552
|
+
/** One of three plain choices of how pages are reached (ACCESS_CHOICES); `my-browser` is `lane: "my-browser"`. */
|
|
553
|
+
access?: AccessChoice;
|
|
554
|
+
/**
|
|
555
|
+
* `my-browser`: read every page in the person's own Chrome, one at a time, as a scrape's `lane` does. The person
|
|
556
|
+
* allows the connection in Chrome, then all the batch's sites (host and port) in the page W2L opens there, once for
|
|
557
|
+
* the run; a site not among them is not read. A resumed run asks again. Offered only by a server on the person's own
|
|
558
|
+
* machine; refused elsewhere, with `actions`, a screenshot, lockdown, a mode other than standard, `maxConcurrency`
|
|
559
|
+
* above 1 or a webhook (`unsupported_parameter`).
|
|
560
|
+
*/
|
|
561
|
+
lane?: 'my-browser';
|
|
510
562
|
formats?: readonly ScrapeFormat[];
|
|
511
563
|
includeLinks?: boolean;
|
|
512
564
|
/** Recorded robots overrides, each for one URL of `urls`. A hosted server refuses the field (`unsupported_parameter`). */
|
|
@@ -585,6 +637,8 @@ export interface BatchStatusResponse extends CrawlReport {
|
|
|
585
637
|
invalidURLs?: readonly string[];
|
|
586
638
|
/** Items stopped at a check a person can get through in their own Chrome (`POST /v1/batches/:id/handoff`); present on a server that offers the handoff. */
|
|
587
639
|
waitingForPerson?: number;
|
|
640
|
+
/** A batch on the my-browser lane waiting for the person to allow its sites in the page Octocrawl opened in their Chrome; present only while it waits. */
|
|
641
|
+
waitingForApproval?: true;
|
|
588
642
|
}
|
|
589
643
|
/**
|
|
590
644
|
* The checks a batch item can be handed to a person for, and the routing
|
|
@@ -809,26 +863,28 @@ export declare class RequestError extends Error {
|
|
|
809
863
|
}
|
|
810
864
|
/** The hints a refusal carries for the options W2L does not offer: the next honest step, never a way around the refusal. */
|
|
811
865
|
export declare const REFUSAL_HINTS: {
|
|
812
|
-
readonly stealth: "
|
|
813
|
-
readonly ignoreRobotsTxt: "robots.txt is always read; a
|
|
814
|
-
readonly hostedSkipTlsVerification: "a hosted server verifies every certificate; run
|
|
815
|
-
readonly useIndex: "
|
|
866
|
+
readonly stealth: "Octocrawl has no stealth option on a request: a provider's stealth or challenge solving runs only on a server started with an access grant that names it (--access-grant, ADR 0005); a proxy or session you own (mode authed) is the other route";
|
|
867
|
+
readonly ignoreRobotsTxt: "robots.txt is always read and recorded; on a local server a URL a scrape or batch names is fetched whatever it says, and ignoreRobotsTxt on a crawl or map fetches the links it disallows, on the record";
|
|
868
|
+
readonly hostedSkipTlsVerification: "a hosted server verifies every certificate; run Octocrawl locally to use skipTlsVerification, which is recorded in the trace and a tls_unverified warning";
|
|
869
|
+
readonly useIndex: "Octocrawl keeps no URL index: a map reads the sitemaps the site declares and its start page, on the record; crawl reads further pages";
|
|
816
870
|
readonly actions: "actions run on scrape and batch, where each page named gets the same steps; a crawl or a map does not take them";
|
|
817
871
|
};
|
|
818
872
|
/** The hint for a refused request key, or null when the key has none (an option W2L simply does not know). */
|
|
819
873
|
export declare function refusalHint(key: string, value: unknown): string | null;
|
|
820
874
|
export declare const PAGE_KEYS: readonly ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown"];
|
|
821
875
|
export declare const ATTRIBUTION_KEYS: readonly ["origin", "integration"];
|
|
822
|
-
export declare const SCRAPE_KEYS: readonly ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
876
|
+
export declare const SCRAPE_KEYS: readonly ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "lane", "access", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
877
|
+
/** The lanes a request may ask for by name. */
|
|
878
|
+
export declare const REQUEST_LANES: readonly ["my-browser"];
|
|
823
879
|
export declare const CRAWL_SCOPE_KEYS: readonly ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
|
|
824
|
-
export declare const CRAWL_KEYS: readonly ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", "regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks", "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
825
|
-
export declare const BATCH_KEYS: readonly ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
880
|
+
export declare const CRAWL_KEYS: readonly ["url", "mode", "access", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", "regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks", "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
881
|
+
export declare const BATCH_KEYS: readonly ["urls", "mode", "lane", "access", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
826
882
|
/** What a batch body may carry beside `appendToId`: the job's own options are not among them (the scope no-ops change nothing, so they may come along). */
|
|
827
883
|
export declare const BATCH_APPEND_KEYS: readonly ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", "origin", "integration"];
|
|
828
884
|
/** The scope options a map takes under their crawl names; allowSubdomains is includeSubdomains on a map, and allowExternalLinks is not offered. */
|
|
829
885
|
export declare const MAP_SCOPE_KEYS: readonly ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
|
|
830
886
|
/** What a map takes. No page option (headers, mobile, skipTlsVerification, formats, ...): a map has nothing to loosen. */
|
|
831
|
-
export declare const MAP_KEYS: readonly ["url", "mode", "limit", "timeout", "search", "sitemap", "includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs", "includePaths", "excludePaths", "origin", "integration"];
|
|
887
|
+
export declare const MAP_KEYS: readonly ["url", "mode", "limit", "timeout", "search", "sitemap", "includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs", "includePaths", "excludePaths", "ignoreRobotsTxt", "origin", "integration"];
|
|
832
888
|
/** Why a webhook header (lower-cased name) cannot be sent, or null when it can: W2L's own and the transport's names are reserved. */
|
|
833
889
|
export declare function webhookHeaderRefusal(name: string): string | null;
|
|
834
890
|
/**
|
|
@@ -869,8 +925,8 @@ export declare function parseCrawlStartRequest(body: unknown): CrawlStartRequest
|
|
|
869
925
|
* A map request: url, mode (standard or research), limit, timeout, search,
|
|
870
926
|
* sitemap, includeSubdomains, the crawl's scope options under their crawl
|
|
871
927
|
* names (ignoreQueryParameters, includePaths, excludePaths, regexOnFullURL,
|
|
872
|
-
* crawlEntireDomain, deduplicateSimilarURLs), origin and
|
|
873
|
-
* Anything else is refused by name, `useIndex` with the supported route;
|
|
928
|
+
* crawlEntireDomain, deduplicateSimilarURLs), ignoreRobotsTxt, origin and
|
|
929
|
+
* integration. Anything else is refused by name, `useIndex` with the supported route;
|
|
874
930
|
* nothing is silently ignored.
|
|
875
931
|
*/
|
|
876
932
|
export declare function parseMapRequest(body: unknown): MapRequest;
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
* Granularity is the page. A partial parse inside a page is not a step; the
|
|
8
8
|
* whole URL is retried. Block-level checkpoint is out of Phase 1.
|
|
9
9
|
*/
|
|
10
|
-
import type { PageOptions, RequestAttribution, RobotsUrlOverride, WebhookEvent } from './api.js';
|
|
10
|
+
import type { AccessChoice, PageOptions, RequestAttribution, RobotsUrlOverride, WebhookEvent } from './api.js';
|
|
11
11
|
import type { CrawlMode } from './compliance.js';
|
|
12
12
|
import type { WebhookPayloadFormat } from './delivery.js';
|
|
13
13
|
import type { CrawlDiscovery, SitemapMode } from './crawl.js';
|
|
@@ -29,6 +29,11 @@ export interface CrawlBudget {
|
|
|
29
29
|
maxWallMs: number | null;
|
|
30
30
|
maxCostUsd: number | null;
|
|
31
31
|
maxTokens: number | null;
|
|
32
|
+
/**
|
|
33
|
+
* What one page may spend on third parties (an access grant's `perRequestUsd`, ROADMAP PA item 4): its own cap
|
|
34
|
+
* within the run's, held by the run's spend ledger. Absent or null: none of its own.
|
|
35
|
+
*/
|
|
36
|
+
maxCostPerPageUsd?: number | null;
|
|
32
37
|
}
|
|
33
38
|
export declare const DEFAULT_CRAWL_BUDGET: CrawlBudget;
|
|
34
39
|
/**
|
|
@@ -72,12 +77,16 @@ export interface Task {
|
|
|
72
77
|
maxConcurrency?: number;
|
|
73
78
|
invalidURLs?: readonly string[];
|
|
74
79
|
webhook?: StoredJobWebhook;
|
|
80
|
+
lane?: 'my-browser';
|
|
81
|
+
access?: AccessChoice;
|
|
75
82
|
} & PageOptions;
|
|
76
83
|
/**
|
|
77
84
|
* Every crawl option but the page budget (`budget`), stored when the crawl
|
|
78
85
|
* starts so a resumed crawl runs with the options it was started with.
|
|
79
86
|
*/
|
|
80
87
|
crawl?: {
|
|
88
|
+
/** The plain access choice the crawl was started with (`standard` or `enhanced`); absent: the server's configuration. */
|
|
89
|
+
access?: 'standard' | 'enhanced';
|
|
81
90
|
formats?: readonly ScrapeFormat[];
|
|
82
91
|
includeLinks?: boolean;
|
|
83
92
|
includePaths?: readonly string[];
|
|
@@ -106,6 +115,8 @@ export interface Task {
|
|
|
106
115
|
maxConcurrency?: number | null;
|
|
107
116
|
/** The crawl's webhook, when the request set one. */
|
|
108
117
|
webhook?: StoredJobWebhook;
|
|
118
|
+
/** The crawl fetches what robots.txt disallows, on the record (CrawlStartRequest.ignoreRobotsTxt); absent: it obeys. A server that takes no override resumes it obeying. */
|
|
119
|
+
ignoreRobotsTxt?: boolean;
|
|
109
120
|
} & PageOptions;
|
|
110
121
|
/** Who started the task (`origin`, `integration`), stored with it and reported as `attribution` on its status; absent when the request named neither. */
|
|
111
122
|
attribution?: RequestAttribution;
|
|
@@ -134,6 +145,11 @@ export interface Attempt {
|
|
|
134
145
|
recoveredFromAttemptId?: string | null;
|
|
135
146
|
/** What this attempt's pages offered the frontier and what became of it, written after every page of a crawl; absent for a batch and for an attempt stored before it was kept. */
|
|
136
147
|
discovery?: CrawlDiscovery | null;
|
|
148
|
+
/**
|
|
149
|
+
* What the spend ledger charged this attempt's paid calls (ROADMAP PA item 4): a resumed or appended run of the task
|
|
150
|
+
* opens its ledger with every earlier attempt's charge, so its cap is the task's, not each run's. Absent: none.
|
|
151
|
+
*/
|
|
152
|
+
chargedUsd?: number | null;
|
|
137
153
|
}
|
|
138
154
|
/**
|
|
139
155
|
* One URL inside one attempt. The atomic checkpoint unit.
|
|
@@ -109,13 +109,24 @@ export declare function researchUserAgent(contact?: string | null, host?: string
|
|
|
109
109
|
export declare function declaredContact(userAgent: string): string | null;
|
|
110
110
|
/** Whether a User-Agent is one research mode declares, in either format. */
|
|
111
111
|
export declare function isResearchUserAgent(userAgent: string): boolean;
|
|
112
|
+
/**
|
|
113
|
+
* The product token a robots.txt names to address Octocrawl itself
|
|
114
|
+
* (`User-agent: Octocrawl`), whatever User-Agent header a request sends. It
|
|
115
|
+
* is matched in every mode. A rule written for it (or for research mode's own
|
|
116
|
+
* token) is not set aside for a URL the request names or for a crawl or map
|
|
117
|
+
* started with ignoreRobotsTxt; only a recorded robotsOverride sets it aside.
|
|
118
|
+
*/
|
|
119
|
+
export declare const PRODUCT_ROBOTS_TOKEN = "octocrawl";
|
|
112
120
|
/**
|
|
113
121
|
* The text robots.txt `User-agent` lines are matched against for a
|
|
114
|
-
* User-Agent
|
|
115
|
-
*
|
|
116
|
-
* SEC.gov as it does on
|
|
122
|
+
* User-Agent Octocrawl sends: the header, with the product token added. SEC's
|
|
123
|
+
* format names no product token, so the research token is added as well: a
|
|
124
|
+
* group for w2l-research governs research requests to SEC.gov as it does on
|
|
125
|
+
* every other host.
|
|
117
126
|
*/
|
|
118
127
|
export declare function robotsAgent(userAgent: string): string;
|
|
128
|
+
/** Whether the robots.txt group that decided names Octocrawl itself (its product token, or research mode's) rather than every crawler (`*`). */
|
|
129
|
+
export declare function isOctocrawlRobotsGroup(matchedAgent: string | null | undefined): boolean;
|
|
119
130
|
/** The operator's contact from `W2L_CONTACT`, trimmed; null when unset or blank. The error never repeats the value. */
|
|
120
131
|
export declare function operatorContact(env: Readonly<Record<string, string | undefined>>): string | null;
|
|
121
132
|
/** An operator policy whose research-mode requests declare `W2L_CONTACT`, when it is set. */
|
|
@@ -254,6 +265,23 @@ export interface RobotsOverride {
|
|
|
254
265
|
/** Who recorded the decision, when the caller wants that on the record. */
|
|
255
266
|
recordedBy?: string;
|
|
256
267
|
}
|
|
268
|
+
/**
|
|
269
|
+
* On whose word a fetch went past robots.txt. `robots_override`: the
|
|
270
|
+
* caller's recorded decision for this URL (`RobotsOverride`).
|
|
271
|
+
* `user_named_url`: a local server fetching a URL the request named (a
|
|
272
|
+
* scrape, a batch entry), since robots.txt addresses crawlers that discover
|
|
273
|
+
* links, not the pages a person names. `ignore_robots_txt`: a crawl or map
|
|
274
|
+
* the caller started with `ignoreRobotsTxt` on a local server.
|
|
275
|
+
*/
|
|
276
|
+
export type RobotsOverrideBasis = 'robots_override' | 'user_named_url' | 'ignore_robots_txt';
|
|
277
|
+
/**
|
|
278
|
+
* A robots override as a lane applies it: the caller's recorded one, or one
|
|
279
|
+
* W2L applies by rule, which says so in `basis` (absent: the caller's,
|
|
280
|
+
* `robots_override`). Set by W2L, never read from a request.
|
|
281
|
+
*/
|
|
282
|
+
export interface AppliedRobotsOverride extends RobotsOverride {
|
|
283
|
+
basis?: Exclude<RobotsOverrideBasis, 'robots_override'>;
|
|
284
|
+
}
|
|
257
285
|
/**
|
|
258
286
|
* The outcome of consulting robots.txt for a single target URL. One record per
|
|
259
287
|
* fetch. `consulted` distinguishes "we checked and it said X" from "there was
|
|
@@ -288,11 +316,11 @@ export interface RobotsDecision {
|
|
|
288
316
|
*/
|
|
289
317
|
unreachable?: RobotsUnreachable;
|
|
290
318
|
/**
|
|
291
|
-
* Present when a disallow
|
|
292
|
-
*
|
|
293
|
-
*
|
|
319
|
+
* Present when a disallow was set aside, the publisher's or the one an
|
|
320
|
+
* unreachable robots.txt implies: the fetch went ahead (`skippedFetch:
|
|
321
|
+
* false`) and this says on whose word.
|
|
294
322
|
*/
|
|
295
|
-
override?:
|
|
323
|
+
override?: AppliedRobotsOverride;
|
|
296
324
|
}
|
|
297
325
|
/**
|
|
298
326
|
* What actually went on the wire. Sorted by header name, lowercased names.
|
|
@@ -86,7 +86,7 @@ export interface SitemapFileRecord {
|
|
|
86
86
|
kind: SitemapFileKind;
|
|
87
87
|
/** `<loc>` entries the file holds (child sitemaps for an index), http(s) ones only; null when the file was not parsed. */
|
|
88
88
|
entries: number | null;
|
|
89
|
-
/** The robots.txt verdict for the file's own URL under the crawl's identity (an unreachable robots.txt is `disallowed`, as for a page); null when the URL failed its egress check before robots.txt was consulted. */
|
|
89
|
+
/** The robots.txt verdict for the file's own URL under the crawl's identity (an unreachable robots.txt is `disallowed`, as for a page); null when the URL failed its egress check before robots.txt was consulted. A `disallowed` file is `refused`, unless the crawl or map was started with ignoreRobotsTxt and read it. */
|
|
90
90
|
robots: 'allowed' | 'disallowed' | 'no_robots' | null;
|
|
91
91
|
/** Whether the request left through the operator's environment proxy (local mode); a hosted server never has one. */
|
|
92
92
|
proxyUsed: boolean;
|
|
@@ -15,7 +15,7 @@ import type { PageActionType } from './actions.js';
|
|
|
15
15
|
* changes its meaning. A change that cannot follow this rule gets a new
|
|
16
16
|
* schemaVersion (`w2l.evidence/2`) and a new schema file.
|
|
17
17
|
*/
|
|
18
|
-
import type { CrawlMode, RobotsUnreachable } from './compliance.js';
|
|
18
|
+
import type { CrawlMode, RobotsOverrideBasis, RobotsUnreachable } from './compliance.js';
|
|
19
19
|
import type { BlockReason, BudgetKind, FailureReason, Lane, ResultStatus } from './status.js';
|
|
20
20
|
export declare const EVIDENCE_SCHEMA_VERSION = "w2l.evidence/1";
|
|
21
21
|
/**
|
|
@@ -33,6 +33,124 @@ export type FieldEvidenceSource = (typeof FIELD_EVIDENCE_SOURCES)[number];
|
|
|
33
33
|
*/
|
|
34
34
|
export declare const EVIDENCE_ARTIFACT_KINDS: readonly ["snapshot", "screenshot", "file"];
|
|
35
35
|
export type EvidenceArtifactKind = (typeof EVIDENCE_ARTIFACT_KINDS)[number];
|
|
36
|
+
/**
|
|
37
|
+
* The route that produced a result (ADR 0005): `http` (W2L's HTTP client, undici), `http_compat` (the
|
|
38
|
+
* browser-compatible HTTP transport, impit), `browser` (the local headless browser), `enhanced_browser`
|
|
39
|
+
* (the local browser on Patchright), `authed_browser` (the local browser with the user's saved login),
|
|
40
|
+
* `user_browser` (the person's own browser after a handoff), `vendor` (a third-party browser service).
|
|
41
|
+
*/
|
|
42
|
+
export declare const ACCESS_ROUTES: readonly ["http", "http_compat", "browser", "enhanced_browser", "authed_browser", "user_browser", "vendor"];
|
|
43
|
+
export type AccessRoute = (typeof ACCESS_ROUTES)[number];
|
|
44
|
+
/**
|
|
45
|
+
* How a result was reached, read from the result's own trace. Added to v1 with enhanced access
|
|
46
|
+
* (EVIDENCE_RECORD_ADDED_KEYS).
|
|
47
|
+
*/
|
|
48
|
+
export interface EvidenceAccess {
|
|
49
|
+
/** Null when no lane produced the result (a run cut before a rung answered, a rung that threw, a lockdown miss). */
|
|
50
|
+
route: AccessRoute | null;
|
|
51
|
+
/** The client that sent the requests: undici, impit, playwright, patchright, the person's browser, or the vendor's id; null when the result does not say. */
|
|
52
|
+
executor: string | null;
|
|
53
|
+
/** The executor's version as the lane reported it; null when it reported none. */
|
|
54
|
+
executorVersion: string | null;
|
|
55
|
+
/** The browser profile the HTTP transport sent (impit's); null on every other route. Its TLS fingerprint was not observed. */
|
|
56
|
+
profile: string | null;
|
|
57
|
+
/** Third-party spend of the run that produced the result: 0 when no paid service was called, null when a called service stated no price. */
|
|
58
|
+
externalCostUsd: number | null;
|
|
59
|
+
/**
|
|
60
|
+
* How the page was read, counted apart (ROADMAP PA items 7 and 8): `unattended` (W2L's own lanes, no session of the
|
|
61
|
+
* person's), `authorized_session` (with the person's saved login), `user_browser` (in the person's own Chrome, on a
|
|
62
|
+
* site they allowed, without a step of theirs), `handed_to_person` (in their Chrome, after they got through a check).
|
|
63
|
+
* Null when no page was read: the result is not success, partial or empty_verified, or no lane produced it. Added with
|
|
64
|
+
* the my-browser lane: optional, so that records written before it stay valid.
|
|
65
|
+
*/
|
|
66
|
+
completion?: AccessCompletion | null;
|
|
67
|
+
/**
|
|
68
|
+
* The egress the page left through (ROADMAP PA item 3): the proxy's `host:port` and whether it came from the
|
|
69
|
+
* operator's pool (`W2L_EGRESS_PROXIES`) or the environment variables, or `direct` with no proxy when W2L's own lane
|
|
70
|
+
* recorded none and a page response shows a request was sent. `switchedFrom` names the pool egress the task last
|
|
71
|
+
* moved off before this page was read here (`egress_switched`); null otherwise. Null when nothing says where the
|
|
72
|
+
* requests left from: a vendor's service, the person's own browser, no lane at all, or a lane that stopped before
|
|
73
|
+
* a page request (robots.txt, an address check, a deadline). Added with the egress pool: optional, so that earlier
|
|
74
|
+
* records stay valid.
|
|
75
|
+
*/
|
|
76
|
+
egress?: EvidenceAccessEgress | null;
|
|
77
|
+
/**
|
|
78
|
+
* The task cookie session the page was read with (`egress_sessions`): its id alone, never its cookies. Null when
|
|
79
|
+
* the page was read with none. Added with the egress pool: optional, so that earlier records stay valid.
|
|
80
|
+
*/
|
|
81
|
+
session?: EvidenceAccessSession | null;
|
|
82
|
+
/**
|
|
83
|
+
* Every paid provider call the page was read with (ROADMAP PA item 4), in order: the provider, what the spend ledger
|
|
84
|
+
* reserved and charged, the price the provider stated, and what Octocrawl made of what came back. Null when no
|
|
85
|
+
* provider was called. Added with the spend ledger: optional, so that earlier records stay valid.
|
|
86
|
+
*/
|
|
87
|
+
paidCalls?: readonly EvidencePaidCall[] | null;
|
|
88
|
+
/** The access grant the paid calls were made under; null when no provider was called. Added with `paidCalls`. */
|
|
89
|
+
grant?: EvidenceAccessGrant | null;
|
|
90
|
+
}
|
|
91
|
+
/** One paid provider call (ROADMAP PA item 4). A provider's own word on the page is never its outcome. */
|
|
92
|
+
export interface EvidencePaidCall {
|
|
93
|
+
/** The provider's id, as the access grant's tariffs name it (`browserbase`, `steel`). */
|
|
94
|
+
provider: string;
|
|
95
|
+
/** The rung that made the call; a retry after the person's handoff is `provider(retry)`. */
|
|
96
|
+
rung: string;
|
|
97
|
+
/** The ADR 0005 capabilities the provider's session was created with: `vendor_remote_browser`, and solving or stealth when the grant named them. */
|
|
98
|
+
capabilities: readonly string[];
|
|
99
|
+
/** What the ledger reserved before the call: its price ceiling, from the grant's tariff. */
|
|
100
|
+
ceilingUsd: number;
|
|
101
|
+
/** What the ledger charged: the price the provider stated, else the ceiling (a call that threw or was cut included). */
|
|
102
|
+
chargedUsd: number;
|
|
103
|
+
/** The price the provider stated for the call; null when it stated none (Browserbase and Steel state none per call). */
|
|
104
|
+
reportedCostUsd: number | null;
|
|
105
|
+
/**
|
|
106
|
+
* What Octocrawl made of the page the call returned, by its own checks (a block page, an empty or unverified read, an
|
|
107
|
+
* identity it did not send): never the provider's word that it succeeded. Null when the call returned no page (it
|
|
108
|
+
* threw, or the deadline cut it).
|
|
109
|
+
*/
|
|
110
|
+
outcome: ResultStatus | null;
|
|
111
|
+
/** Why the outcome is not a read page, as the record's own `reason`; null otherwise. */
|
|
112
|
+
reason: FailureReason | BlockReason | BudgetKind | null;
|
|
113
|
+
/** The record's own page is this call's. */
|
|
114
|
+
answer: boolean;
|
|
115
|
+
}
|
|
116
|
+
/** The access grant paid calls were made under: enough to name it, not a copy. */
|
|
117
|
+
export interface EvidenceAccessGrant {
|
|
118
|
+
/** SHA-256 of the grant's text as Octocrawl read it (`shasum -a 256 grant.json`); null for a grant not read from text. */
|
|
119
|
+
sha256: string | null;
|
|
120
|
+
/** The grant's tier. */
|
|
121
|
+
tier: string;
|
|
122
|
+
/** When its attestation says the operator accepted the providers' terms and costs; null when it has none. */
|
|
123
|
+
attestedAt: string | null;
|
|
124
|
+
}
|
|
125
|
+
export declare const ACCESS_EGRESS_SOURCES: readonly ["pool", "environment", "direct"];
|
|
126
|
+
export type AccessEgressSource = (typeof ACCESS_EGRESS_SOURCES)[number];
|
|
127
|
+
export interface EvidenceAccessEgress {
|
|
128
|
+
/** The proxy's `host:port`, never its credentials; null when the request went direct. */
|
|
129
|
+
proxy: string | null;
|
|
130
|
+
source: AccessEgressSource;
|
|
131
|
+
/** The pool egress the task last left before this page was read here; null when it did not move. */
|
|
132
|
+
switchedFrom: string | null;
|
|
133
|
+
/**
|
|
134
|
+
* Where the pool egress leaves from, as the operator's echo URL (`W2L_EGRESS_ECHO_URL`) saw it through that proxy.
|
|
135
|
+
* Null when no echo URL is set, the echo did not answer, or the egress is not the pool's. Added after `egress`:
|
|
136
|
+
* optional, so that earlier records stay valid.
|
|
137
|
+
*/
|
|
138
|
+
exit?: EvidenceAccessEgressExit | null;
|
|
139
|
+
}
|
|
140
|
+
export interface EvidenceAccessEgressExit {
|
|
141
|
+
/** The address the echo service saw the request come from. */
|
|
142
|
+
ip: string;
|
|
143
|
+
/** Its two-letter country code, when the echo service gives one; null otherwise. */
|
|
144
|
+
country: string | null;
|
|
145
|
+
/** When the echo was asked (UTC ISO); an exit is asked again after ten minutes. */
|
|
146
|
+
observedAt: string;
|
|
147
|
+
}
|
|
148
|
+
export interface EvidenceAccessSession {
|
|
149
|
+
/** The session's id, as `session_cookies` traces it. */
|
|
150
|
+
id: string;
|
|
151
|
+
}
|
|
152
|
+
export declare const ACCESS_COMPLETIONS: readonly ["unattended", "authorized_session", "user_browser", "handed_to_person"];
|
|
153
|
+
export type AccessCompletion = (typeof ACCESS_COMPLETIONS)[number];
|
|
36
154
|
export interface EvidenceRedirectChain {
|
|
37
155
|
/**
|
|
38
156
|
* Every URL W2L requested for the page, in order: the requested URL first,
|
|
@@ -61,8 +179,10 @@ export interface EvidenceRobotsDecision {
|
|
|
61
179
|
/** Why robots.txt could not be fetched (then `decision` is `disallowed`, RFC 9309 §2.3.1.4); null when it was. */
|
|
62
180
|
unreachable: RobotsUnreachable | null;
|
|
63
181
|
crawlDelayMs: number | null;
|
|
64
|
-
/** Whether the fetch went ahead
|
|
182
|
+
/** Whether the fetch went ahead although `decision` is `disallowed`; on whose word is `overrideBasis`, and the reason is in the trace, the warnings and, in the browser lane, the compliance record. */
|
|
65
183
|
userOverride: boolean;
|
|
184
|
+
/** On whose word the disallow was set aside (`RobotsOverrideBasis`); null when it was not. Optional in the v1 schema file, written on every record. */
|
|
185
|
+
overrideBasis: RobotsOverrideBasis | null;
|
|
66
186
|
}
|
|
67
187
|
export interface EvidenceOutputSha256 {
|
|
68
188
|
/** SHA-256 of the UTF-8 bytes of the delivered `markdown`; null when none was delivered. */
|
|
@@ -169,12 +289,14 @@ export interface EvidenceRecord {
|
|
|
169
289
|
* the caller's ran in it, so its content may be the script's.
|
|
170
290
|
*/
|
|
171
291
|
pageActions: EvidencePageActions | null;
|
|
292
|
+
/** How the result was reached (EvidenceAccess). Added to v1 later (EVIDENCE_RECORD_ADDED_KEYS). */
|
|
293
|
+
access: EvidenceAccess;
|
|
172
294
|
}
|
|
173
295
|
/** Field order of the record and of each nested object, as in the schema file. */
|
|
174
296
|
export declare const EVIDENCE_RECORD_KEYS: {
|
|
175
|
-
readonly record: readonly ["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions"];
|
|
297
|
+
readonly record: readonly ["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions", "access"];
|
|
176
298
|
readonly redirectChain: readonly ["urls", "complete"];
|
|
177
|
-
readonly robotsDecision: readonly ["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride"];
|
|
299
|
+
readonly robotsDecision: readonly ["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride", "overrideBasis"];
|
|
178
300
|
readonly outputSha256: readonly ["markdown", "json"];
|
|
179
301
|
readonly extractor: readonly ["name", "version", "commit"];
|
|
180
302
|
readonly fieldEvidence: readonly ["source", "locator"];
|
|
@@ -183,6 +305,12 @@ export declare const EVIDENCE_RECORD_KEYS: {
|
|
|
183
305
|
readonly pageActions: readonly ["steps", "scriptRan"];
|
|
184
306
|
readonly pageActionStep: readonly ["type", "outcome"];
|
|
185
307
|
readonly requestHeader: readonly ["name", "valueSha256"];
|
|
308
|
+
readonly access: readonly ["route", "executor", "executorVersion", "profile", "externalCostUsd", "completion", "egress", "session", "paidCalls", "grant"];
|
|
309
|
+
readonly accessEgress: readonly ["proxy", "source", "switchedFrom", "exit"];
|
|
310
|
+
readonly accessEgressExit: readonly ["ip", "country", "observedAt"];
|
|
311
|
+
readonly accessSession: readonly ["id"];
|
|
312
|
+
readonly accessPaidCall: readonly ["provider", "rung", "capabilities", "ceilingUsd", "chargedUsd", "reportedCostUsd", "outcome", "reason", "answer"];
|
|
313
|
+
readonly accessGrant: readonly ["sha256", "tier", "attestedAt"];
|
|
186
314
|
};
|
|
187
315
|
/**
|
|
188
316
|
* Keys added to v1 after it was first published: optional in the schema, so
|