@octocrawl/sdk 0.3.0 → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/README.md +1 -1
  2. package/dist/index.cjs +15 -9
  3. package/dist/index.js +15 -9
  4. package/dist/types/cjs/client.d.ts +1 -1
  5. package/dist/types/cjs/contracts/actions.d.ts +37 -4
  6. package/dist/types/cjs/contracts/api.d.ts +66 -10
  7. package/dist/types/cjs/contracts/checkpoint.d.ts +17 -1
  8. package/dist/types/cjs/contracts/compliance.d.ts +35 -7
  9. package/dist/types/cjs/contracts/crawl.d.ts +1 -1
  10. package/dist/types/cjs/contracts/evidenceRecord.d.ts +132 -4
  11. package/dist/types/cjs/contracts/execution.d.ts +121 -9
  12. package/dist/types/cjs/contracts/extractor.d.ts +11 -1
  13. package/dist/types/cjs/contracts/firecrawl.d.ts +1 -1
  14. package/dist/types/cjs/contracts/map.d.ts +10 -4
  15. package/dist/types/cjs/contracts/policy.d.ts +2 -1
  16. package/dist/types/cjs/contracts/proxy.d.ts +8 -0
  17. package/dist/types/cjs/contracts/result.d.ts +41 -4
  18. package/dist/types/cjs/contracts/session.d.ts +2 -0
  19. package/dist/types/cjs/contracts/status.d.ts +1 -1
  20. package/dist/types/cjs/version.d.ts +1 -1
  21. package/dist/types/esm/client.d.ts +1 -1
  22. package/dist/types/esm/contracts/actions.d.ts +37 -4
  23. package/dist/types/esm/contracts/api.d.ts +66 -10
  24. package/dist/types/esm/contracts/checkpoint.d.ts +17 -1
  25. package/dist/types/esm/contracts/compliance.d.ts +35 -7
  26. package/dist/types/esm/contracts/crawl.d.ts +1 -1
  27. package/dist/types/esm/contracts/evidenceRecord.d.ts +132 -4
  28. package/dist/types/esm/contracts/execution.d.ts +121 -9
  29. package/dist/types/esm/contracts/extractor.d.ts +11 -1
  30. package/dist/types/esm/contracts/firecrawl.d.ts +1 -1
  31. package/dist/types/esm/contracts/map.d.ts +10 -4
  32. package/dist/types/esm/contracts/policy.d.ts +2 -1
  33. package/dist/types/esm/contracts/proxy.d.ts +8 -0
  34. package/dist/types/esm/contracts/result.d.ts +41 -4
  35. package/dist/types/esm/contracts/session.d.ts +2 -0
  36. package/dist/types/esm/contracts/status.d.ts +1 -1
  37. package/dist/types/esm/version.d.ts +1 -1
  38. package/package.json +19 -2
package/README.md CHANGED
@@ -14,4 +14,4 @@ const { report, items } = await client.batchAndWait(['https://example.com/a', 'h
14
14
 
15
15
  Start a local API with `npx octocrawl serve`. Requires a runtime with `fetch` (Node.js 18 or later, browsers, Deno, Bun).
16
16
 
17
- Licence: MIT. Source and the API reference: https://github.com/77777R7/Octocrawl
17
+ Licence: MIT. Website and docs: https://octocrawl.dev. Source and the API reference: https://github.com/77777R7/Octocrawl
package/dist/index.cjs CHANGED
@@ -216,13 +216,13 @@ function isApiErrorCode(value) {
216
216
  var RATE_LIMITED_CODE = "rate_limited";
217
217
  var PAGE_KEYS = ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown"];
218
218
  var ATTRIBUTION_KEYS = ["origin", "integration"];
219
- var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
219
+ var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "lane", "access", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
220
220
  var CRAWL_SCOPE_KEYS = ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
221
- var CRAWL_KEYS = ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
222
- var BATCH_KEYS = ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
221
+ var CRAWL_KEYS = ["url", "mode", "access", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
222
+ var BATCH_KEYS = ["urls", "mode", "lane", "access", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
223
223
  var BATCH_APPEND_KEYS = ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", ...ATTRIBUTION_KEYS];
224
224
  var MAP_SCOPE_KEYS = ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
225
- var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", ...ATTRIBUTION_KEYS];
225
+ var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", "ignoreRobotsTxt", ...ATTRIBUTION_KEYS];
226
226
  var WEBHOOK_HEADERS_MESSAGE = `webhook.headers must be an object of at most ${MAX_WEBHOOK_HEADERS} string values`;
227
227
  var WEBHOOK_METADATA_MESSAGE = `webhook.metadata must be an object of at most ${MAX_WEBHOOK_METADATA_ENTRIES} string values of at most ${MAX_WEBHOOK_METADATA_VALUE_LENGTH} characters`;
228
228
  var WEBHOOK_EVENTS_MESSAGE = `webhook.events must be a non-empty array of ${WEBHOOK_EVENTS.join(", ")} without duplicates`;
@@ -243,9 +243,9 @@ var SHIM_MAP_KEYS = ["url", "search", "sitemap", "ignoreSitemap", "sitemapOnly",
243
243
  // packages/contracts/dist/evidenceRecord.js
244
244
  var keysOf = () => (keys) => keys;
245
245
  var EVIDENCE_RECORD_KEYS = {
246
- record: keysOf()(["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions"]),
246
+ record: keysOf()(["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions", "access"]),
247
247
  redirectChain: keysOf()(["urls", "complete"]),
248
- robotsDecision: keysOf()(["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride"]),
248
+ robotsDecision: keysOf()(["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride", "overrideBasis"]),
249
249
  outputSha256: keysOf()(["markdown", "json"]),
250
250
  extractor: keysOf()(["name", "version", "commit"]),
251
251
  fieldEvidence: keysOf()(["source", "locator"]),
@@ -253,11 +253,17 @@ var EVIDENCE_RECORD_KEYS = {
253
253
  identity: keysOf()(["userAgent", "mode", "contact", "device", "requestHeaders"]),
254
254
  pageActions: keysOf()(["steps", "scriptRan"]),
255
255
  pageActionStep: keysOf()(["type", "outcome"]),
256
- requestHeader: keysOf()(["name", "valueSha256"])
256
+ requestHeader: keysOf()(["name", "valueSha256"]),
257
+ access: keysOf()(["route", "executor", "executorVersion", "profile", "externalCostUsd", "completion", "egress", "session", "paidCalls", "grant"]),
258
+ accessEgress: keysOf()(["proxy", "source", "switchedFrom", "exit"]),
259
+ accessEgressExit: keysOf()(["ip", "country", "observedAt"]),
260
+ accessSession: keysOf()(["id"]),
261
+ accessPaidCall: keysOf()(["provider", "rung", "capabilities", "ceilingUsd", "chargedUsd", "reportedCostUsd", "outcome", "reason", "answer"]),
262
+ accessGrant: keysOf()(["sha256", "tier", "attestedAt"])
257
263
  };
258
264
 
259
265
  // packages/sdk/src/version.ts
260
- var SDK_VERSION = "0.3.0";
266
+ var SDK_VERSION = "0.3.2";
261
267
 
262
268
  // packages/sdk/src/watcher.ts
263
269
  var DEFAULT_WATCH_POLL_INTERVAL_MS = 2e3;
@@ -790,7 +796,7 @@ var W2L = class {
790
796
  }
791
797
  async scrape(url, opts = {}, request = {}) {
792
798
  const deadlineMs = Number.isInteger(opts.timeout) ? Math.min(Math.max(opts.timeout, 0), DEFAULT_SCRAPE_TIMEOUT_MS) : DEFAULT_SCRAPE_TIMEOUT_MS;
793
- const handedOver = opts.handoff !== void 0 && opts.handoff !== false;
799
+ const handedOver = opts.handoff !== void 0 && opts.handoff !== false || opts.lane === "my-browser" || opts.access === "my-browser";
794
800
  return this.post("/v1/scrape", { ...opts, url, origin: originOf(opts, request) }, 200, request, handedOver ? 0 : deadlineMs + SCRAPE_ANSWER_MARGIN_MS);
795
801
  }
796
802
  /** The record of one scrape call, by the `scrapeId` its response carried (`metadata.scrapeId`); a W2LError with code `not_found` for an id the server has no record of. */
package/dist/index.js CHANGED
@@ -179,13 +179,13 @@ function isApiErrorCode(value) {
179
179
  var RATE_LIMITED_CODE = "rate_limited";
180
180
  var PAGE_KEYS = ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown"];
181
181
  var ATTRIBUTION_KEYS = ["origin", "integration"];
182
- var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
182
+ var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "lane", "access", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
183
183
  var CRAWL_SCOPE_KEYS = ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
184
- var CRAWL_KEYS = ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
185
- var BATCH_KEYS = ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
184
+ var CRAWL_KEYS = ["url", "mode", "access", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
185
+ var BATCH_KEYS = ["urls", "mode", "lane", "access", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
186
186
  var BATCH_APPEND_KEYS = ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", ...ATTRIBUTION_KEYS];
187
187
  var MAP_SCOPE_KEYS = ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
188
- var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", ...ATTRIBUTION_KEYS];
188
+ var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", "ignoreRobotsTxt", ...ATTRIBUTION_KEYS];
189
189
  var WEBHOOK_HEADERS_MESSAGE = `webhook.headers must be an object of at most ${MAX_WEBHOOK_HEADERS} string values`;
190
190
  var WEBHOOK_METADATA_MESSAGE = `webhook.metadata must be an object of at most ${MAX_WEBHOOK_METADATA_ENTRIES} string values of at most ${MAX_WEBHOOK_METADATA_VALUE_LENGTH} characters`;
191
191
  var WEBHOOK_EVENTS_MESSAGE = `webhook.events must be a non-empty array of ${WEBHOOK_EVENTS.join(", ")} without duplicates`;
@@ -206,9 +206,9 @@ var SHIM_MAP_KEYS = ["url", "search", "sitemap", "ignoreSitemap", "sitemapOnly",
206
206
  // packages/contracts/dist/evidenceRecord.js
207
207
  var keysOf = () => (keys) => keys;
208
208
  var EVIDENCE_RECORD_KEYS = {
209
- record: keysOf()(["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions"]),
209
+ record: keysOf()(["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions", "access"]),
210
210
  redirectChain: keysOf()(["urls", "complete"]),
211
- robotsDecision: keysOf()(["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride"]),
211
+ robotsDecision: keysOf()(["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride", "overrideBasis"]),
212
212
  outputSha256: keysOf()(["markdown", "json"]),
213
213
  extractor: keysOf()(["name", "version", "commit"]),
214
214
  fieldEvidence: keysOf()(["source", "locator"]),
@@ -216,11 +216,17 @@ var EVIDENCE_RECORD_KEYS = {
216
216
  identity: keysOf()(["userAgent", "mode", "contact", "device", "requestHeaders"]),
217
217
  pageActions: keysOf()(["steps", "scriptRan"]),
218
218
  pageActionStep: keysOf()(["type", "outcome"]),
219
- requestHeader: keysOf()(["name", "valueSha256"])
219
+ requestHeader: keysOf()(["name", "valueSha256"]),
220
+ access: keysOf()(["route", "executor", "executorVersion", "profile", "externalCostUsd", "completion", "egress", "session", "paidCalls", "grant"]),
221
+ accessEgress: keysOf()(["proxy", "source", "switchedFrom", "exit"]),
222
+ accessEgressExit: keysOf()(["ip", "country", "observedAt"]),
223
+ accessSession: keysOf()(["id"]),
224
+ accessPaidCall: keysOf()(["provider", "rung", "capabilities", "ceilingUsd", "chargedUsd", "reportedCostUsd", "outcome", "reason", "answer"]),
225
+ accessGrant: keysOf()(["sha256", "tier", "attestedAt"])
220
226
  };
221
227
 
222
228
  // packages/sdk/src/version.ts
223
- var SDK_VERSION = "0.3.0";
229
+ var SDK_VERSION = "0.3.2";
224
230
 
225
231
  // packages/sdk/src/watcher.ts
226
232
  var DEFAULT_WATCH_POLL_INTERVAL_MS = 2e3;
@@ -753,7 +759,7 @@ var W2L = class {
753
759
  }
754
760
  async scrape(url, opts = {}, request = {}) {
755
761
  const deadlineMs = Number.isInteger(opts.timeout) ? Math.min(Math.max(opts.timeout, 0), DEFAULT_SCRAPE_TIMEOUT_MS) : DEFAULT_SCRAPE_TIMEOUT_MS;
756
- const handedOver = opts.handoff !== void 0 && opts.handoff !== false;
762
+ const handedOver = opts.handoff !== void 0 && opts.handoff !== false || opts.lane === "my-browser" || opts.access === "my-browser";
757
763
  return this.post("/v1/scrape", { ...opts, url, origin: originOf(opts, request) }, 200, request, handedOver ? 0 : deadlineMs + SCRAPE_ANSWER_MARGIN_MS);
758
764
  }
759
765
  /** The record of one scrape call, by the `scrapeId` its response carried (`metadata.scrapeId`); a W2LError with code `not_found` for an id the server has no record of. */
@@ -33,7 +33,7 @@ export interface RequestOptions {
33
33
  origin?: string;
34
34
  }
35
35
  /** What the SDK records as `origin` unless the caller or the host says otherwise. */
36
- export declare const SDK_ORIGIN = "js-sdk@0.3.0";
36
+ export declare const SDK_ORIGIN = "js-sdk@0.3.2";
37
37
  /**
38
38
  * Polling for waitBatch and waitCrawl. A status request that fails with a
39
39
  * network error, HTTP 408, 429 or 5xx is retried: after 1, 2, 4, 8, then
@@ -4,6 +4,7 @@
4
4
  * Each step is recorded in the trace with its outcome and timing; a step
5
5
  * that fails ends the pipeline, and the result keeps the page as it stood.
6
6
  */
7
+ import type { BlockReason } from './status.js';
7
8
  import type { ScreenshotEvidence } from './result.js';
8
9
  import type { ScreenshotViewport } from './structured.js';
9
10
  /** The most steps one request may run. */
@@ -149,8 +150,20 @@ export interface ActionPdf {
149
150
  path: string | null;
150
151
  base64: string;
151
152
  }
152
- /** Why a scrollToEnd, loadMore or paginate step stopped. `max` and `deadline` stop short of the list's end. */
153
- export type ListStop = 'end' | 'no_growth' | 'repeat' | 'max' | 'deadline';
153
+ /**
154
+ * Why a scrollToEnd, loadMore or paginate step stopped. `max` and `deadline` stop short of the list's end; so does
155
+ * `challenge`: a check the site put up on the next page (a Cloudflare interstitial, a page of nothing but a CAPTCHA),
156
+ * which paginate stops at without reading it, the result then `blocked` with the check's reason.
157
+ */
158
+ export type ListStop = 'end' | 'no_growth' | 'repeat' | 'max' | 'deadline' | 'challenge';
159
+ /** The check a paginate step stopped at (ListStop `challenge`): the page it would have been, its URL, and what the gate saw. */
160
+ export interface ListChallenge {
161
+ /** 1-based position the page would have had among the pages read. */
162
+ page: number;
163
+ url: string;
164
+ reason: BlockReason;
165
+ signals: readonly string[];
166
+ }
154
167
  /** What a scrollToEnd, loadMore or paginate step did. */
155
168
  export interface ListRun {
156
169
  index: number;
@@ -160,8 +173,27 @@ export interface ListRun {
160
173
  rounds: number;
161
174
  /** Elements matching `itemSelector` at the end (on the last page for paginate; summed over its pages in `itemsRead`); null without one. */
162
175
  items: number | null;
163
- /** paginate: elements matching `itemSelector` over every page read; null without one. */
176
+ /**
177
+ * paginate: elements matching `itemSelector` over every page read; null without one. After a continuation in the person's
178
+ * Chrome (`continued`), the kept pages' count plus each page they showed, as their tab counted it, only when the addresses show
179
+ * no page can be counted twice (the kept pages each at its own, the check's page and the pages shown at none of them, the check
180
+ * not at the list's own address), no page shows again what another shows (its items' whole text, as the list merge tells it:
181
+ * a result set tied to the session that made it comes back at new addresses), and every count is known; null (unknown) otherwise.
182
+ */
164
183
  itemsRead?: number | null;
184
+ /** paginate: pages taken from the task's checkpoint after a run cut at page N (ExecutionContext.listResume), counted in `rounds`; absent when none. */
185
+ resumed?: number;
186
+ /** paginate: the check the step stopped at, with `stoppedBy` `challenge`; absent otherwise. */
187
+ challenge?: ListChallenge;
188
+ /**
189
+ * paginate: the pages read after the check at `from`, by the person paging on in their own browser once they got through
190
+ * it (the batch handoff), counted in `rounds`; `stoppedBy` then says how that reading ended. Absent otherwise.
191
+ */
192
+ continued?: {
193
+ from: number;
194
+ pages: number;
195
+ by: 'user_browser';
196
+ };
165
197
  }
166
198
  /** What the steps produced, each list in the order of its steps. */
167
199
  export interface ActionsResult {
@@ -170,7 +202,8 @@ export interface ActionsResult {
170
202
  scrapes: {
171
203
  url: string;
172
204
  html: string;
173
- step?: number;
205
+ step?: number; /** Set when the person read the page in their own browser (a list's continuation after a check); absent for W2L's own browser. */
206
+ by?: 'user_browser';
174
207
  }[];
175
208
  /** `type` is the JavaScript `typeof` of the value (`null` for null). */
176
209
  javascriptReturns: {
@@ -170,7 +170,28 @@ export interface ScrapeRequest extends PageOptions, RequestAttribution {
170
170
  handoff?: {
171
171
  waitMs?: number;
172
172
  };
173
+ /**
174
+ * `my-browser`: read the page in the person's own Chrome, over remote
175
+ * debugging, without W2L fetching it first. The person allows the
176
+ * connection in Chrome, then the site in a page W2L opens there; the page
177
+ * is read without a click of theirs only on a site they allowed, and a
178
+ * check it shows waits for them (`handoff.waitMs`, default 10 min). Lane
179
+ * `my_browser`; never cached. Offered only by a server on the person's
180
+ * own machine; refused elsewhere, with `actions` or a screenshot, and with
181
+ * a mode other than standard (`unsupported_parameter`).
182
+ */
183
+ lane?: 'my-browser';
184
+ /** One of three plain choices of how the page is reached (ACCESS_CHOICES); omitted, the server's own configuration. */
185
+ access?: AccessChoice;
173
186
  }
187
+ /**
188
+ * How pages are reached, as three plain choices (ROADMAP PA item 7) beside the per-route options:
189
+ * `standard` (Octocrawl's own lanes and none that costs a third party), `enhanced` (also what the server's
190
+ * access grant of tier enhanced approves, within its budget; refused on a server without one), `my-browser`
191
+ * (the person's own Chrome, as `lane: "my-browser"`; a scrape or a batch only).
192
+ */
193
+ export declare const ACCESS_CHOICES: readonly ["standard", "enhanced", "my-browser"];
194
+ export type AccessChoice = (typeof ACCESS_CHOICES)[number];
174
195
  /** A recorded robots override for one URL of a batch. */
175
196
  export interface RobotsUrlOverride extends RobotsOverride {
176
197
  url: string;
@@ -344,6 +365,8 @@ export interface JobWebhookStatus {
344
365
  export interface CrawlStartRequest extends PageOptions, RequestAttribution {
345
366
  url: string;
346
367
  mode?: ApiCrawlMode;
368
+ /** `standard` or `enhanced` (ACCESS_CHOICES); a crawl does not take `my-browser`. */
369
+ access?: Exclude<AccessChoice, 'my-browser'>;
347
370
  maxPages?: number | null;
348
371
  maxDepth?: number | null;
349
372
  /**
@@ -411,6 +434,16 @@ export interface CrawlStartRequest extends PageOptions, RequestAttribution {
411
434
  idempotencyKey?: string;
412
435
  /** A receiver for the crawl's events (`started`, one `page` per page recorded, then `completed`, `failed` or `cancelled`); see WebhookConfig. */
413
436
  webhook?: WebhookOption;
437
+ /**
438
+ * Fetch the pages and sitemap files robots.txt disallows, or whose
439
+ * robots.txt could not be read (Firecrawl v2's name). robots.txt is still
440
+ * read for every host and its verdict recorded on each page, Crawl-delay
441
+ * applied, with a `robots_overridden` warning and `overrideBasis:
442
+ * "ignore_robots_txt"` where a rule was set aside. A local server only: a
443
+ * hosted one refuses it by name. Default false: a crawl's links obey
444
+ * robots.txt.
445
+ */
446
+ ignoreRobotsTxt?: boolean;
414
447
  }
415
448
  /** What the parser hands the engine: the request plus, from the `/fc` shim, the payload shape its receiver expects. */
416
449
  export type ParsedCrawlStartRequest = CrawlStartRequest & {
@@ -454,6 +487,14 @@ export interface MapRequest extends RequestAttribution {
454
487
  crawlEntireDomain?: boolean;
455
488
  /** Default true, as on a crawl; a returned http link gives way to its https variant when that comes too, on an origin whose robots.txt the map read anyway and which allows it. */
456
489
  deduplicateSimilarURLs?: boolean;
490
+ /**
491
+ * Return the URLs robots.txt disallows, or whose robots.txt could not be
492
+ * read, with that verdict on each link (`robots: "disallowed"` or
493
+ * `"unreachable"`), and read the start page and sitemap files past it.
494
+ * robots.txt is still read, within MAP_MAX_ROBOTS_HOSTS. A local server
495
+ * only: a hosted one refuses it by name. Default false.
496
+ */
497
+ ignoreRobotsTxt?: boolean;
457
498
  }
458
499
  /** A map's `search`: at most this many characters after trimming, and this many whitespace-separated words. */
459
500
  export declare const MAP_SEARCH_MAX_CHARS = 200;
@@ -482,6 +523,7 @@ export interface ActiveCrawlOptions {
482
523
  allowExternalLinks: boolean;
483
524
  regexOnFullURL: boolean;
484
525
  maxConcurrency: number | null;
526
+ ignoreRobotsTxt: boolean;
485
527
  /** The per-page options every page of the crawl gets: its formats, `includeLinks` and the page options. */
486
528
  scrapeOptions: PageOptions & {
487
529
  formats: readonly ScrapeFormat[];
@@ -507,6 +549,16 @@ export interface ActiveCrawlList {
507
549
  export interface BatchStartRequest extends PageOptions, RequestAttribution {
508
550
  urls: readonly string[];
509
551
  mode?: ApiCrawlMode;
552
+ /** One of three plain choices of how pages are reached (ACCESS_CHOICES); `my-browser` is `lane: "my-browser"`. */
553
+ access?: AccessChoice;
554
+ /**
555
+ * `my-browser`: read every page in the person's own Chrome, one at a time, as a scrape's `lane` does. The person
556
+ * allows the connection in Chrome, then all the batch's sites (host and port) in the page W2L opens there, once for
557
+ * the run; a site not among them is not read. A resumed run asks again. Offered only by a server on the person's own
558
+ * machine; refused elsewhere, with `actions`, a screenshot, lockdown, a mode other than standard, `maxConcurrency`
559
+ * above 1 or a webhook (`unsupported_parameter`).
560
+ */
561
+ lane?: 'my-browser';
510
562
  formats?: readonly ScrapeFormat[];
511
563
  includeLinks?: boolean;
512
564
  /** Recorded robots overrides, each for one URL of `urls`. A hosted server refuses the field (`unsupported_parameter`). */
@@ -585,6 +637,8 @@ export interface BatchStatusResponse extends CrawlReport {
585
637
  invalidURLs?: readonly string[];
586
638
  /** Items stopped at a check a person can get through in their own Chrome (`POST /v1/batches/:id/handoff`); present on a server that offers the handoff. */
587
639
  waitingForPerson?: number;
640
+ /** A batch on the my-browser lane waiting for the person to allow its sites in the page Octocrawl opened in their Chrome; present only while it waits. */
641
+ waitingForApproval?: true;
588
642
  }
589
643
  /**
590
644
  * The checks a batch item can be handed to a person for, and the routing
@@ -809,26 +863,28 @@ export declare class RequestError extends Error {
809
863
  }
810
864
  /** The hints a refusal carries for the options W2L does not offer: the next honest step, never a way around the refusal. */
811
865
  export declare const REFUSAL_HINTS: {
812
- readonly stealth: "W2L does not offer a stealth mode or stealth proxies; a proxy or session you own (mode authed) is the supported route";
813
- readonly ignoreRobotsTxt: "robots.txt is always read; a robotsOverride with a recorded reason fetches one URL past its rule, on the record";
814
- readonly hostedSkipTlsVerification: "a hosted server verifies every certificate; run W2L locally to use skipTlsVerification, which is recorded in the trace and a tls_unverified warning";
815
- readonly useIndex: "W2L keeps no URL index: a map reads the sitemaps the site declares and its start page, on the record; crawl reads further pages";
866
+ readonly stealth: "Octocrawl has no stealth option on a request: a provider's stealth or challenge solving runs only on a server started with an access grant that names it (--access-grant, ADR 0005); a proxy or session you own (mode authed) is the other route";
867
+ readonly ignoreRobotsTxt: "robots.txt is always read and recorded; on a local server a URL a scrape or batch names is fetched whatever it says, and ignoreRobotsTxt on a crawl or map fetches the links it disallows, on the record";
868
+ readonly hostedSkipTlsVerification: "a hosted server verifies every certificate; run Octocrawl locally to use skipTlsVerification, which is recorded in the trace and a tls_unverified warning";
869
+ readonly useIndex: "Octocrawl keeps no URL index: a map reads the sitemaps the site declares and its start page, on the record; crawl reads further pages";
816
870
  readonly actions: "actions run on scrape and batch, where each page named gets the same steps; a crawl or a map does not take them";
817
871
  };
818
872
  /** The hint for a refused request key, or null when the key has none (an option W2L simply does not know). */
819
873
  export declare function refusalHint(key: string, value: unknown): string | null;
820
874
  export declare const PAGE_KEYS: readonly ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown"];
821
875
  export declare const ATTRIBUTION_KEYS: readonly ["origin", "integration"];
822
- export declare const SCRAPE_KEYS: readonly ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
876
+ export declare const SCRAPE_KEYS: readonly ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "lane", "access", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
877
+ /** The lanes a request may ask for by name. */
878
+ export declare const REQUEST_LANES: readonly ["my-browser"];
823
879
  export declare const CRAWL_SCOPE_KEYS: readonly ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
824
- export declare const CRAWL_KEYS: readonly ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", "regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks", "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
825
- export declare const BATCH_KEYS: readonly ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
880
+ export declare const CRAWL_KEYS: readonly ["url", "mode", "access", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", "regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks", "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
881
+ export declare const BATCH_KEYS: readonly ["urls", "mode", "lane", "access", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
826
882
  /** What a batch body may carry beside `appendToId`: the job's own options are not among them (the scope no-ops change nothing, so they may come along). */
827
883
  export declare const BATCH_APPEND_KEYS: readonly ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", "origin", "integration"];
828
884
  /** The scope options a map takes under their crawl names; allowSubdomains is includeSubdomains on a map, and allowExternalLinks is not offered. */
829
885
  export declare const MAP_SCOPE_KEYS: readonly ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
830
886
  /** What a map takes. No page option (headers, mobile, skipTlsVerification, formats, ...): a map has nothing to loosen. */
831
- export declare const MAP_KEYS: readonly ["url", "mode", "limit", "timeout", "search", "sitemap", "includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs", "includePaths", "excludePaths", "origin", "integration"];
887
+ export declare const MAP_KEYS: readonly ["url", "mode", "limit", "timeout", "search", "sitemap", "includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs", "includePaths", "excludePaths", "ignoreRobotsTxt", "origin", "integration"];
832
888
  /** Why a webhook header (lower-cased name) cannot be sent, or null when it can: W2L's own and the transport's names are reserved. */
833
889
  export declare function webhookHeaderRefusal(name: string): string | null;
834
890
  /**
@@ -869,8 +925,8 @@ export declare function parseCrawlStartRequest(body: unknown): CrawlStartRequest
869
925
  * A map request: url, mode (standard or research), limit, timeout, search,
870
926
  * sitemap, includeSubdomains, the crawl's scope options under their crawl
871
927
  * names (ignoreQueryParameters, includePaths, excludePaths, regexOnFullURL,
872
- * crawlEntireDomain, deduplicateSimilarURLs), origin and integration.
873
- * Anything else is refused by name, `useIndex` with the supported route;
928
+ * crawlEntireDomain, deduplicateSimilarURLs), ignoreRobotsTxt, origin and
929
+ * integration. Anything else is refused by name, `useIndex` with the supported route;
874
930
  * nothing is silently ignored.
875
931
  */
876
932
  export declare function parseMapRequest(body: unknown): MapRequest;
@@ -7,7 +7,7 @@
7
7
  * Granularity is the page. A partial parse inside a page is not a step; the
8
8
  * whole URL is retried. Block-level checkpoint is out of Phase 1.
9
9
  */
10
- import type { PageOptions, RequestAttribution, RobotsUrlOverride, WebhookEvent } from './api.js';
10
+ import type { AccessChoice, PageOptions, RequestAttribution, RobotsUrlOverride, WebhookEvent } from './api.js';
11
11
  import type { CrawlMode } from './compliance.js';
12
12
  import type { WebhookPayloadFormat } from './delivery.js';
13
13
  import type { CrawlDiscovery, SitemapMode } from './crawl.js';
@@ -29,6 +29,11 @@ export interface CrawlBudget {
29
29
  maxWallMs: number | null;
30
30
  maxCostUsd: number | null;
31
31
  maxTokens: number | null;
32
+ /**
33
+ * What one page may spend on third parties (an access grant's `perRequestUsd`, ROADMAP PA item 4): its own cap
34
+ * within the run's, held by the run's spend ledger. Absent or null: none of its own.
35
+ */
36
+ maxCostPerPageUsd?: number | null;
32
37
  }
33
38
  export declare const DEFAULT_CRAWL_BUDGET: CrawlBudget;
34
39
  /**
@@ -72,12 +77,16 @@ export interface Task {
72
77
  maxConcurrency?: number;
73
78
  invalidURLs?: readonly string[];
74
79
  webhook?: StoredJobWebhook;
80
+ lane?: 'my-browser';
81
+ access?: AccessChoice;
75
82
  } & PageOptions;
76
83
  /**
77
84
  * Every crawl option but the page budget (`budget`), stored when the crawl
78
85
  * starts so a resumed crawl runs with the options it was started with.
79
86
  */
80
87
  crawl?: {
88
+ /** The plain access choice the crawl was started with (`standard` or `enhanced`); absent: the server's configuration. */
89
+ access?: 'standard' | 'enhanced';
81
90
  formats?: readonly ScrapeFormat[];
82
91
  includeLinks?: boolean;
83
92
  includePaths?: readonly string[];
@@ -106,6 +115,8 @@ export interface Task {
106
115
  maxConcurrency?: number | null;
107
116
  /** The crawl's webhook, when the request set one. */
108
117
  webhook?: StoredJobWebhook;
118
+ /** The crawl fetches what robots.txt disallows, on the record (CrawlStartRequest.ignoreRobotsTxt); absent: it obeys. A server that takes no override resumes it obeying. */
119
+ ignoreRobotsTxt?: boolean;
109
120
  } & PageOptions;
110
121
  /** Who started the task (`origin`, `integration`), stored with it and reported as `attribution` on its status; absent when the request named neither. */
111
122
  attribution?: RequestAttribution;
@@ -134,6 +145,11 @@ export interface Attempt {
134
145
  recoveredFromAttemptId?: string | null;
135
146
  /** What this attempt's pages offered the frontier and what became of it, written after every page of a crawl; absent for a batch and for an attempt stored before it was kept. */
136
147
  discovery?: CrawlDiscovery | null;
148
+ /**
149
+ * What the spend ledger charged this attempt's paid calls (ROADMAP PA item 4): a resumed or appended run of the task
150
+ * opens its ledger with every earlier attempt's charge, so its cap is the task's, not each run's. Absent: none.
151
+ */
152
+ chargedUsd?: number | null;
137
153
  }
138
154
  /**
139
155
  * One URL inside one attempt. The atomic checkpoint unit.
@@ -109,13 +109,24 @@ export declare function researchUserAgent(contact?: string | null, host?: string
109
109
  export declare function declaredContact(userAgent: string): string | null;
110
110
  /** Whether a User-Agent is one research mode declares, in either format. */
111
111
  export declare function isResearchUserAgent(userAgent: string): boolean;
112
+ /**
113
+ * The product token a robots.txt names to address Octocrawl itself
114
+ * (`User-agent: Octocrawl`), whatever User-Agent header a request sends. It
115
+ * is matched in every mode. A rule written for it (or for research mode's own
116
+ * token) is not set aside for a URL the request names or for a crawl or map
117
+ * started with ignoreRobotsTxt; only a recorded robotsOverride sets it aside.
118
+ */
119
+ export declare const PRODUCT_ROBOTS_TOKEN = "octocrawl";
112
120
  /**
113
121
  * The text robots.txt `User-agent` lines are matched against for a
114
- * User-Agent W2L sends. SEC's format names no product token, so the research
115
- * token is added: a group for w2l-research governs research requests to
116
- * SEC.gov as it does on every other host.
122
+ * User-Agent Octocrawl sends: the header, with the product token added. SEC's
123
+ * format names no product token, so the research token is added as well: a
124
+ * group for w2l-research governs research requests to SEC.gov as it does on
125
+ * every other host.
117
126
  */
118
127
  export declare function robotsAgent(userAgent: string): string;
128
+ /** Whether the robots.txt group that decided names Octocrawl itself (its product token, or research mode's) rather than every crawler (`*`). */
129
+ export declare function isOctocrawlRobotsGroup(matchedAgent: string | null | undefined): boolean;
119
130
  /** The operator's contact from `W2L_CONTACT`, trimmed; null when unset or blank. The error never repeats the value. */
120
131
  export declare function operatorContact(env: Readonly<Record<string, string | undefined>>): string | null;
121
132
  /** An operator policy whose research-mode requests declare `W2L_CONTACT`, when it is set. */
@@ -254,6 +265,23 @@ export interface RobotsOverride {
254
265
  /** Who recorded the decision, when the caller wants that on the record. */
255
266
  recordedBy?: string;
256
267
  }
268
+ /**
269
+ * On whose word a fetch went past robots.txt. `robots_override`: the
270
+ * caller's recorded decision for this URL (`RobotsOverride`).
271
+ * `user_named_url`: a local server fetching a URL the request named (a
272
+ * scrape, a batch entry), since robots.txt addresses crawlers that discover
273
+ * links, not the pages a person names. `ignore_robots_txt`: a crawl or map
274
+ * the caller started with `ignoreRobotsTxt` on a local server.
275
+ */
276
+ export type RobotsOverrideBasis = 'robots_override' | 'user_named_url' | 'ignore_robots_txt';
277
+ /**
278
+ * A robots override as a lane applies it: the caller's recorded one, or one
279
+ * W2L applies by rule, which says so in `basis` (absent: the caller's,
280
+ * `robots_override`). Set by W2L, never read from a request.
281
+ */
282
+ export interface AppliedRobotsOverride extends RobotsOverride {
283
+ basis?: Exclude<RobotsOverrideBasis, 'robots_override'>;
284
+ }
257
285
  /**
258
286
  * The outcome of consulting robots.txt for a single target URL. One record per
259
287
  * fetch. `consulted` distinguishes "we checked and it said X" from "there was
@@ -288,11 +316,11 @@ export interface RobotsDecision {
288
316
  */
289
317
  unreachable?: RobotsUnreachable;
290
318
  /**
291
- * Present when a disallow the publisher wrote was set aside by a recorded
292
- * decision: the fetch went ahead (`skippedFetch: false`) and this says on
293
- * whose word. Never set for an unreachable robots.txt.
319
+ * Present when a disallow was set aside, the publisher's or the one an
320
+ * unreachable robots.txt implies: the fetch went ahead (`skippedFetch:
321
+ * false`) and this says on whose word.
294
322
  */
295
- override?: RobotsOverride;
323
+ override?: AppliedRobotsOverride;
296
324
  }
297
325
  /**
298
326
  * What actually went on the wire. Sorted by header name, lowercased names.
@@ -86,7 +86,7 @@ export interface SitemapFileRecord {
86
86
  kind: SitemapFileKind;
87
87
  /** `<loc>` entries the file holds (child sitemaps for an index), http(s) ones only; null when the file was not parsed. */
88
88
  entries: number | null;
89
- /** The robots.txt verdict for the file's own URL under the crawl's identity (an unreachable robots.txt is `disallowed`, as for a page); null when the URL failed its egress check before robots.txt was consulted. */
89
+ /** The robots.txt verdict for the file's own URL under the crawl's identity (an unreachable robots.txt is `disallowed`, as for a page); null when the URL failed its egress check before robots.txt was consulted. A `disallowed` file is `refused`, unless the crawl or map was started with ignoreRobotsTxt and read it. */
90
90
  robots: 'allowed' | 'disallowed' | 'no_robots' | null;
91
91
  /** Whether the request left through the operator's environment proxy (local mode); a hosted server never has one. */
92
92
  proxyUsed: boolean;