@octocrawl/sdk 0.3.0 → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/index.cjs +15 -9
- package/dist/index.js +15 -9
- package/dist/types/cjs/client.d.ts +1 -1
- package/dist/types/cjs/contracts/actions.d.ts +37 -4
- package/dist/types/cjs/contracts/api.d.ts +66 -10
- package/dist/types/cjs/contracts/checkpoint.d.ts +17 -1
- package/dist/types/cjs/contracts/compliance.d.ts +35 -7
- package/dist/types/cjs/contracts/crawl.d.ts +1 -1
- package/dist/types/cjs/contracts/evidenceRecord.d.ts +132 -4
- package/dist/types/cjs/contracts/execution.d.ts +121 -9
- package/dist/types/cjs/contracts/extractor.d.ts +11 -1
- package/dist/types/cjs/contracts/firecrawl.d.ts +1 -1
- package/dist/types/cjs/contracts/map.d.ts +10 -4
- package/dist/types/cjs/contracts/policy.d.ts +2 -1
- package/dist/types/cjs/contracts/proxy.d.ts +8 -0
- package/dist/types/cjs/contracts/result.d.ts +41 -4
- package/dist/types/cjs/contracts/session.d.ts +2 -0
- package/dist/types/cjs/contracts/status.d.ts +1 -1
- package/dist/types/cjs/version.d.ts +1 -1
- package/dist/types/esm/client.d.ts +1 -1
- package/dist/types/esm/contracts/actions.d.ts +37 -4
- package/dist/types/esm/contracts/api.d.ts +66 -10
- package/dist/types/esm/contracts/checkpoint.d.ts +17 -1
- package/dist/types/esm/contracts/compliance.d.ts +35 -7
- package/dist/types/esm/contracts/crawl.d.ts +1 -1
- package/dist/types/esm/contracts/evidenceRecord.d.ts +132 -4
- package/dist/types/esm/contracts/execution.d.ts +121 -9
- package/dist/types/esm/contracts/extractor.d.ts +11 -1
- package/dist/types/esm/contracts/firecrawl.d.ts +1 -1
- package/dist/types/esm/contracts/map.d.ts +10 -4
- package/dist/types/esm/contracts/policy.d.ts +2 -1
- package/dist/types/esm/contracts/proxy.d.ts +8 -0
- package/dist/types/esm/contracts/result.d.ts +41 -4
- package/dist/types/esm/contracts/session.d.ts +2 -0
- package/dist/types/esm/contracts/status.d.ts +1 -1
- package/dist/types/esm/version.d.ts +1 -1
- package/package.json +19 -2
package/README.md
CHANGED
|
@@ -14,4 +14,4 @@ const { report, items } = await client.batchAndWait(['https://example.com/a', 'h
|
|
|
14
14
|
|
|
15
15
|
Start a local API with `npx octocrawl serve`. Requires a runtime with `fetch` (Node.js 18 or later, browsers, Deno, Bun).
|
|
16
16
|
|
|
17
|
-
Licence: MIT. Source and the API reference: https://github.com/77777R7/Octocrawl
|
|
17
|
+
Licence: MIT. Website and docs: https://octocrawl.dev. Source and the API reference: https://github.com/77777R7/Octocrawl
|
package/dist/index.cjs
CHANGED
|
@@ -216,13 +216,13 @@ function isApiErrorCode(value) {
|
|
|
216
216
|
var RATE_LIMITED_CODE = "rate_limited";
|
|
217
217
|
var PAGE_KEYS = ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown"];
|
|
218
218
|
var ATTRIBUTION_KEYS = ["origin", "integration"];
|
|
219
|
-
var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
219
|
+
var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "lane", "access", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
220
220
|
var CRAWL_SCOPE_KEYS = ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
|
|
221
|
-
var CRAWL_KEYS = ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
222
|
-
var BATCH_KEYS = ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
221
|
+
var CRAWL_KEYS = ["url", "mode", "access", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
222
|
+
var BATCH_KEYS = ["urls", "mode", "lane", "access", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
223
223
|
var BATCH_APPEND_KEYS = ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", ...ATTRIBUTION_KEYS];
|
|
224
224
|
var MAP_SCOPE_KEYS = ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
|
|
225
|
-
var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", ...ATTRIBUTION_KEYS];
|
|
225
|
+
var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", "ignoreRobotsTxt", ...ATTRIBUTION_KEYS];
|
|
226
226
|
var WEBHOOK_HEADERS_MESSAGE = `webhook.headers must be an object of at most ${MAX_WEBHOOK_HEADERS} string values`;
|
|
227
227
|
var WEBHOOK_METADATA_MESSAGE = `webhook.metadata must be an object of at most ${MAX_WEBHOOK_METADATA_ENTRIES} string values of at most ${MAX_WEBHOOK_METADATA_VALUE_LENGTH} characters`;
|
|
228
228
|
var WEBHOOK_EVENTS_MESSAGE = `webhook.events must be a non-empty array of ${WEBHOOK_EVENTS.join(", ")} without duplicates`;
|
|
@@ -243,9 +243,9 @@ var SHIM_MAP_KEYS = ["url", "search", "sitemap", "ignoreSitemap", "sitemapOnly",
|
|
|
243
243
|
// packages/contracts/dist/evidenceRecord.js
|
|
244
244
|
var keysOf = () => (keys) => keys;
|
|
245
245
|
var EVIDENCE_RECORD_KEYS = {
|
|
246
|
-
record: keysOf()(["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions"]),
|
|
246
|
+
record: keysOf()(["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions", "access"]),
|
|
247
247
|
redirectChain: keysOf()(["urls", "complete"]),
|
|
248
|
-
robotsDecision: keysOf()(["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride"]),
|
|
248
|
+
robotsDecision: keysOf()(["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride", "overrideBasis"]),
|
|
249
249
|
outputSha256: keysOf()(["markdown", "json"]),
|
|
250
250
|
extractor: keysOf()(["name", "version", "commit"]),
|
|
251
251
|
fieldEvidence: keysOf()(["source", "locator"]),
|
|
@@ -253,11 +253,17 @@ var EVIDENCE_RECORD_KEYS = {
|
|
|
253
253
|
identity: keysOf()(["userAgent", "mode", "contact", "device", "requestHeaders"]),
|
|
254
254
|
pageActions: keysOf()(["steps", "scriptRan"]),
|
|
255
255
|
pageActionStep: keysOf()(["type", "outcome"]),
|
|
256
|
-
requestHeader: keysOf()(["name", "valueSha256"])
|
|
256
|
+
requestHeader: keysOf()(["name", "valueSha256"]),
|
|
257
|
+
access: keysOf()(["route", "executor", "executorVersion", "profile", "externalCostUsd", "completion", "egress", "session", "paidCalls", "grant"]),
|
|
258
|
+
accessEgress: keysOf()(["proxy", "source", "switchedFrom", "exit"]),
|
|
259
|
+
accessEgressExit: keysOf()(["ip", "country", "observedAt"]),
|
|
260
|
+
accessSession: keysOf()(["id"]),
|
|
261
|
+
accessPaidCall: keysOf()(["provider", "rung", "capabilities", "ceilingUsd", "chargedUsd", "reportedCostUsd", "outcome", "reason", "answer"]),
|
|
262
|
+
accessGrant: keysOf()(["sha256", "tier", "attestedAt"])
|
|
257
263
|
};
|
|
258
264
|
|
|
259
265
|
// packages/sdk/src/version.ts
|
|
260
|
-
var SDK_VERSION = "0.3.
|
|
266
|
+
var SDK_VERSION = "0.3.2";
|
|
261
267
|
|
|
262
268
|
// packages/sdk/src/watcher.ts
|
|
263
269
|
var DEFAULT_WATCH_POLL_INTERVAL_MS = 2e3;
|
|
@@ -790,7 +796,7 @@ var W2L = class {
|
|
|
790
796
|
}
|
|
791
797
|
async scrape(url, opts = {}, request = {}) {
|
|
792
798
|
const deadlineMs = Number.isInteger(opts.timeout) ? Math.min(Math.max(opts.timeout, 0), DEFAULT_SCRAPE_TIMEOUT_MS) : DEFAULT_SCRAPE_TIMEOUT_MS;
|
|
793
|
-
const handedOver = opts.handoff !== void 0 && opts.handoff !== false;
|
|
799
|
+
const handedOver = opts.handoff !== void 0 && opts.handoff !== false || opts.lane === "my-browser" || opts.access === "my-browser";
|
|
794
800
|
return this.post("/v1/scrape", { ...opts, url, origin: originOf(opts, request) }, 200, request, handedOver ? 0 : deadlineMs + SCRAPE_ANSWER_MARGIN_MS);
|
|
795
801
|
}
|
|
796
802
|
/** The record of one scrape call, by the `scrapeId` its response carried (`metadata.scrapeId`); a W2LError with code `not_found` for an id the server has no record of. */
|
package/dist/index.js
CHANGED
|
@@ -179,13 +179,13 @@ function isApiErrorCode(value) {
|
|
|
179
179
|
var RATE_LIMITED_CODE = "rate_limited";
|
|
180
180
|
var PAGE_KEYS = ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown"];
|
|
181
181
|
var ATTRIBUTION_KEYS = ["origin", "integration"];
|
|
182
|
-
var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
182
|
+
var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "lane", "access", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
183
183
|
var CRAWL_SCOPE_KEYS = ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
|
|
184
|
-
var CRAWL_KEYS = ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
185
|
-
var BATCH_KEYS = ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
184
|
+
var CRAWL_KEYS = ["url", "mode", "access", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
185
|
+
var BATCH_KEYS = ["urls", "mode", "lane", "access", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
186
186
|
var BATCH_APPEND_KEYS = ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", ...ATTRIBUTION_KEYS];
|
|
187
187
|
var MAP_SCOPE_KEYS = ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
|
|
188
|
-
var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", ...ATTRIBUTION_KEYS];
|
|
188
|
+
var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", "ignoreRobotsTxt", ...ATTRIBUTION_KEYS];
|
|
189
189
|
var WEBHOOK_HEADERS_MESSAGE = `webhook.headers must be an object of at most ${MAX_WEBHOOK_HEADERS} string values`;
|
|
190
190
|
var WEBHOOK_METADATA_MESSAGE = `webhook.metadata must be an object of at most ${MAX_WEBHOOK_METADATA_ENTRIES} string values of at most ${MAX_WEBHOOK_METADATA_VALUE_LENGTH} characters`;
|
|
191
191
|
var WEBHOOK_EVENTS_MESSAGE = `webhook.events must be a non-empty array of ${WEBHOOK_EVENTS.join(", ")} without duplicates`;
|
|
@@ -206,9 +206,9 @@ var SHIM_MAP_KEYS = ["url", "search", "sitemap", "ignoreSitemap", "sitemapOnly",
|
|
|
206
206
|
// packages/contracts/dist/evidenceRecord.js
|
|
207
207
|
var keysOf = () => (keys) => keys;
|
|
208
208
|
var EVIDENCE_RECORD_KEYS = {
|
|
209
|
-
record: keysOf()(["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions"]),
|
|
209
|
+
record: keysOf()(["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions", "access"]),
|
|
210
210
|
redirectChain: keysOf()(["urls", "complete"]),
|
|
211
|
-
robotsDecision: keysOf()(["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride"]),
|
|
211
|
+
robotsDecision: keysOf()(["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride", "overrideBasis"]),
|
|
212
212
|
outputSha256: keysOf()(["markdown", "json"]),
|
|
213
213
|
extractor: keysOf()(["name", "version", "commit"]),
|
|
214
214
|
fieldEvidence: keysOf()(["source", "locator"]),
|
|
@@ -216,11 +216,17 @@ var EVIDENCE_RECORD_KEYS = {
|
|
|
216
216
|
identity: keysOf()(["userAgent", "mode", "contact", "device", "requestHeaders"]),
|
|
217
217
|
pageActions: keysOf()(["steps", "scriptRan"]),
|
|
218
218
|
pageActionStep: keysOf()(["type", "outcome"]),
|
|
219
|
-
requestHeader: keysOf()(["name", "valueSha256"])
|
|
219
|
+
requestHeader: keysOf()(["name", "valueSha256"]),
|
|
220
|
+
access: keysOf()(["route", "executor", "executorVersion", "profile", "externalCostUsd", "completion", "egress", "session", "paidCalls", "grant"]),
|
|
221
|
+
accessEgress: keysOf()(["proxy", "source", "switchedFrom", "exit"]),
|
|
222
|
+
accessEgressExit: keysOf()(["ip", "country", "observedAt"]),
|
|
223
|
+
accessSession: keysOf()(["id"]),
|
|
224
|
+
accessPaidCall: keysOf()(["provider", "rung", "capabilities", "ceilingUsd", "chargedUsd", "reportedCostUsd", "outcome", "reason", "answer"]),
|
|
225
|
+
accessGrant: keysOf()(["sha256", "tier", "attestedAt"])
|
|
220
226
|
};
|
|
221
227
|
|
|
222
228
|
// packages/sdk/src/version.ts
|
|
223
|
-
var SDK_VERSION = "0.3.
|
|
229
|
+
var SDK_VERSION = "0.3.2";
|
|
224
230
|
|
|
225
231
|
// packages/sdk/src/watcher.ts
|
|
226
232
|
var DEFAULT_WATCH_POLL_INTERVAL_MS = 2e3;
|
|
@@ -753,7 +759,7 @@ var W2L = class {
|
|
|
753
759
|
}
|
|
754
760
|
async scrape(url, opts = {}, request = {}) {
|
|
755
761
|
const deadlineMs = Number.isInteger(opts.timeout) ? Math.min(Math.max(opts.timeout, 0), DEFAULT_SCRAPE_TIMEOUT_MS) : DEFAULT_SCRAPE_TIMEOUT_MS;
|
|
756
|
-
const handedOver = opts.handoff !== void 0 && opts.handoff !== false;
|
|
762
|
+
const handedOver = opts.handoff !== void 0 && opts.handoff !== false || opts.lane === "my-browser" || opts.access === "my-browser";
|
|
757
763
|
return this.post("/v1/scrape", { ...opts, url, origin: originOf(opts, request) }, 200, request, handedOver ? 0 : deadlineMs + SCRAPE_ANSWER_MARGIN_MS);
|
|
758
764
|
}
|
|
759
765
|
/** The record of one scrape call, by the `scrapeId` its response carried (`metadata.scrapeId`); a W2LError with code `not_found` for an id the server has no record of. */
|
|
@@ -33,7 +33,7 @@ export interface RequestOptions {
|
|
|
33
33
|
origin?: string;
|
|
34
34
|
}
|
|
35
35
|
/** What the SDK records as `origin` unless the caller or the host says otherwise. */
|
|
36
|
-
export declare const SDK_ORIGIN = "js-sdk@0.3.
|
|
36
|
+
export declare const SDK_ORIGIN = "js-sdk@0.3.2";
|
|
37
37
|
/**
|
|
38
38
|
* Polling for waitBatch and waitCrawl. A status request that fails with a
|
|
39
39
|
* network error, HTTP 408, 429 or 5xx is retried: after 1, 2, 4, 8, then
|
|
@@ -4,6 +4,7 @@
|
|
|
4
4
|
* Each step is recorded in the trace with its outcome and timing; a step
|
|
5
5
|
* that fails ends the pipeline, and the result keeps the page as it stood.
|
|
6
6
|
*/
|
|
7
|
+
import type { BlockReason } from './status.js';
|
|
7
8
|
import type { ScreenshotEvidence } from './result.js';
|
|
8
9
|
import type { ScreenshotViewport } from './structured.js';
|
|
9
10
|
/** The most steps one request may run. */
|
|
@@ -149,8 +150,20 @@ export interface ActionPdf {
|
|
|
149
150
|
path: string | null;
|
|
150
151
|
base64: string;
|
|
151
152
|
}
|
|
152
|
-
/**
|
|
153
|
-
|
|
153
|
+
/**
|
|
154
|
+
* Why a scrollToEnd, loadMore or paginate step stopped. `max` and `deadline` stop short of the list's end; so does
|
|
155
|
+
* `challenge`: a check the site put up on the next page (a Cloudflare interstitial, a page of nothing but a CAPTCHA),
|
|
156
|
+
* which paginate stops at without reading it, the result then `blocked` with the check's reason.
|
|
157
|
+
*/
|
|
158
|
+
export type ListStop = 'end' | 'no_growth' | 'repeat' | 'max' | 'deadline' | 'challenge';
|
|
159
|
+
/** The check a paginate step stopped at (ListStop `challenge`): the page it would have been, its URL, and what the gate saw. */
|
|
160
|
+
export interface ListChallenge {
|
|
161
|
+
/** 1-based position the page would have had among the pages read. */
|
|
162
|
+
page: number;
|
|
163
|
+
url: string;
|
|
164
|
+
reason: BlockReason;
|
|
165
|
+
signals: readonly string[];
|
|
166
|
+
}
|
|
154
167
|
/** What a scrollToEnd, loadMore or paginate step did. */
|
|
155
168
|
export interface ListRun {
|
|
156
169
|
index: number;
|
|
@@ -160,8 +173,27 @@ export interface ListRun {
|
|
|
160
173
|
rounds: number;
|
|
161
174
|
/** Elements matching `itemSelector` at the end (on the last page for paginate; summed over its pages in `itemsRead`); null without one. */
|
|
162
175
|
items: number | null;
|
|
163
|
-
/**
|
|
176
|
+
/**
|
|
177
|
+
* paginate: elements matching `itemSelector` over every page read; null without one. After a continuation in the person's
|
|
178
|
+
* Chrome (`continued`), the kept pages' count plus each page they showed, as their tab counted it, only when the addresses show
|
|
179
|
+
* no page can be counted twice (the kept pages each at its own, the check's page and the pages shown at none of them, the check
|
|
180
|
+
* not at the list's own address), no page shows again what another shows (its items' whole text, as the list merge tells it:
|
|
181
|
+
* a result set tied to the session that made it comes back at new addresses), and every count is known; null (unknown) otherwise.
|
|
182
|
+
*/
|
|
164
183
|
itemsRead?: number | null;
|
|
184
|
+
/** paginate: pages taken from the task's checkpoint after a run cut at page N (ExecutionContext.listResume), counted in `rounds`; absent when none. */
|
|
185
|
+
resumed?: number;
|
|
186
|
+
/** paginate: the check the step stopped at, with `stoppedBy` `challenge`; absent otherwise. */
|
|
187
|
+
challenge?: ListChallenge;
|
|
188
|
+
/**
|
|
189
|
+
* paginate: the pages read after the check at `from`, by the person paging on in their own browser once they got through
|
|
190
|
+
* it (the batch handoff), counted in `rounds`; `stoppedBy` then says how that reading ended. Absent otherwise.
|
|
191
|
+
*/
|
|
192
|
+
continued?: {
|
|
193
|
+
from: number;
|
|
194
|
+
pages: number;
|
|
195
|
+
by: 'user_browser';
|
|
196
|
+
};
|
|
165
197
|
}
|
|
166
198
|
/** What the steps produced, each list in the order of its steps. */
|
|
167
199
|
export interface ActionsResult {
|
|
@@ -170,7 +202,8 @@ export interface ActionsResult {
|
|
|
170
202
|
scrapes: {
|
|
171
203
|
url: string;
|
|
172
204
|
html: string;
|
|
173
|
-
step?: number;
|
|
205
|
+
step?: number; /** Set when the person read the page in their own browser (a list's continuation after a check); absent for W2L's own browser. */
|
|
206
|
+
by?: 'user_browser';
|
|
174
207
|
}[];
|
|
175
208
|
/** `type` is the JavaScript `typeof` of the value (`null` for null). */
|
|
176
209
|
javascriptReturns: {
|
|
@@ -170,7 +170,28 @@ export interface ScrapeRequest extends PageOptions, RequestAttribution {
|
|
|
170
170
|
handoff?: {
|
|
171
171
|
waitMs?: number;
|
|
172
172
|
};
|
|
173
|
+
/**
|
|
174
|
+
* `my-browser`: read the page in the person's own Chrome, over remote
|
|
175
|
+
* debugging, without W2L fetching it first. The person allows the
|
|
176
|
+
* connection in Chrome, then the site in a page W2L opens there; the page
|
|
177
|
+
* is read without a click of theirs only on a site they allowed, and a
|
|
178
|
+
* check it shows waits for them (`handoff.waitMs`, default 10 min). Lane
|
|
179
|
+
* `my_browser`; never cached. Offered only by a server on the person's
|
|
180
|
+
* own machine; refused elsewhere, with `actions` or a screenshot, and with
|
|
181
|
+
* a mode other than standard (`unsupported_parameter`).
|
|
182
|
+
*/
|
|
183
|
+
lane?: 'my-browser';
|
|
184
|
+
/** One of three plain choices of how the page is reached (ACCESS_CHOICES); omitted, the server's own configuration. */
|
|
185
|
+
access?: AccessChoice;
|
|
173
186
|
}
|
|
187
|
+
/**
|
|
188
|
+
* How pages are reached, as three plain choices (ROADMAP PA item 7) beside the per-route options:
|
|
189
|
+
* `standard` (Octocrawl's own lanes and none that costs a third party), `enhanced` (also what the server's
|
|
190
|
+
* access grant of tier enhanced approves, within its budget; refused on a server without one), `my-browser`
|
|
191
|
+
* (the person's own Chrome, as `lane: "my-browser"`; a scrape or a batch only).
|
|
192
|
+
*/
|
|
193
|
+
export declare const ACCESS_CHOICES: readonly ["standard", "enhanced", "my-browser"];
|
|
194
|
+
export type AccessChoice = (typeof ACCESS_CHOICES)[number];
|
|
174
195
|
/** A recorded robots override for one URL of a batch. */
|
|
175
196
|
export interface RobotsUrlOverride extends RobotsOverride {
|
|
176
197
|
url: string;
|
|
@@ -344,6 +365,8 @@ export interface JobWebhookStatus {
|
|
|
344
365
|
export interface CrawlStartRequest extends PageOptions, RequestAttribution {
|
|
345
366
|
url: string;
|
|
346
367
|
mode?: ApiCrawlMode;
|
|
368
|
+
/** `standard` or `enhanced` (ACCESS_CHOICES); a crawl does not take `my-browser`. */
|
|
369
|
+
access?: Exclude<AccessChoice, 'my-browser'>;
|
|
347
370
|
maxPages?: number | null;
|
|
348
371
|
maxDepth?: number | null;
|
|
349
372
|
/**
|
|
@@ -411,6 +434,16 @@ export interface CrawlStartRequest extends PageOptions, RequestAttribution {
|
|
|
411
434
|
idempotencyKey?: string;
|
|
412
435
|
/** A receiver for the crawl's events (`started`, one `page` per page recorded, then `completed`, `failed` or `cancelled`); see WebhookConfig. */
|
|
413
436
|
webhook?: WebhookOption;
|
|
437
|
+
/**
|
|
438
|
+
* Fetch the pages and sitemap files robots.txt disallows, or whose
|
|
439
|
+
* robots.txt could not be read (Firecrawl v2's name). robots.txt is still
|
|
440
|
+
* read for every host and its verdict recorded on each page, Crawl-delay
|
|
441
|
+
* applied, with a `robots_overridden` warning and `overrideBasis:
|
|
442
|
+
* "ignore_robots_txt"` where a rule was set aside. A local server only: a
|
|
443
|
+
* hosted one refuses it by name. Default false: a crawl's links obey
|
|
444
|
+
* robots.txt.
|
|
445
|
+
*/
|
|
446
|
+
ignoreRobotsTxt?: boolean;
|
|
414
447
|
}
|
|
415
448
|
/** What the parser hands the engine: the request plus, from the `/fc` shim, the payload shape its receiver expects. */
|
|
416
449
|
export type ParsedCrawlStartRequest = CrawlStartRequest & {
|
|
@@ -454,6 +487,14 @@ export interface MapRequest extends RequestAttribution {
|
|
|
454
487
|
crawlEntireDomain?: boolean;
|
|
455
488
|
/** Default true, as on a crawl; a returned http link gives way to its https variant when that comes too, on an origin whose robots.txt the map read anyway and which allows it. */
|
|
456
489
|
deduplicateSimilarURLs?: boolean;
|
|
490
|
+
/**
|
|
491
|
+
* Return the URLs robots.txt disallows, or whose robots.txt could not be
|
|
492
|
+
* read, with that verdict on each link (`robots: "disallowed"` or
|
|
493
|
+
* `"unreachable"`), and read the start page and sitemap files past it.
|
|
494
|
+
* robots.txt is still read, within MAP_MAX_ROBOTS_HOSTS. A local server
|
|
495
|
+
* only: a hosted one refuses it by name. Default false.
|
|
496
|
+
*/
|
|
497
|
+
ignoreRobotsTxt?: boolean;
|
|
457
498
|
}
|
|
458
499
|
/** A map's `search`: at most this many characters after trimming, and this many whitespace-separated words. */
|
|
459
500
|
export declare const MAP_SEARCH_MAX_CHARS = 200;
|
|
@@ -482,6 +523,7 @@ export interface ActiveCrawlOptions {
|
|
|
482
523
|
allowExternalLinks: boolean;
|
|
483
524
|
regexOnFullURL: boolean;
|
|
484
525
|
maxConcurrency: number | null;
|
|
526
|
+
ignoreRobotsTxt: boolean;
|
|
485
527
|
/** The per-page options every page of the crawl gets: its formats, `includeLinks` and the page options. */
|
|
486
528
|
scrapeOptions: PageOptions & {
|
|
487
529
|
formats: readonly ScrapeFormat[];
|
|
@@ -507,6 +549,16 @@ export interface ActiveCrawlList {
|
|
|
507
549
|
export interface BatchStartRequest extends PageOptions, RequestAttribution {
|
|
508
550
|
urls: readonly string[];
|
|
509
551
|
mode?: ApiCrawlMode;
|
|
552
|
+
/** One of three plain choices of how pages are reached (ACCESS_CHOICES); `my-browser` is `lane: "my-browser"`. */
|
|
553
|
+
access?: AccessChoice;
|
|
554
|
+
/**
|
|
555
|
+
* `my-browser`: read every page in the person's own Chrome, one at a time, as a scrape's `lane` does. The person
|
|
556
|
+
* allows the connection in Chrome, then all the batch's sites (host and port) in the page W2L opens there, once for
|
|
557
|
+
* the run; a site not among them is not read. A resumed run asks again. Offered only by a server on the person's own
|
|
558
|
+
* machine; refused elsewhere, with `actions`, a screenshot, lockdown, a mode other than standard, `maxConcurrency`
|
|
559
|
+
* above 1 or a webhook (`unsupported_parameter`).
|
|
560
|
+
*/
|
|
561
|
+
lane?: 'my-browser';
|
|
510
562
|
formats?: readonly ScrapeFormat[];
|
|
511
563
|
includeLinks?: boolean;
|
|
512
564
|
/** Recorded robots overrides, each for one URL of `urls`. A hosted server refuses the field (`unsupported_parameter`). */
|
|
@@ -585,6 +637,8 @@ export interface BatchStatusResponse extends CrawlReport {
|
|
|
585
637
|
invalidURLs?: readonly string[];
|
|
586
638
|
/** Items stopped at a check a person can get through in their own Chrome (`POST /v1/batches/:id/handoff`); present on a server that offers the handoff. */
|
|
587
639
|
waitingForPerson?: number;
|
|
640
|
+
/** A batch on the my-browser lane waiting for the person to allow its sites in the page Octocrawl opened in their Chrome; present only while it waits. */
|
|
641
|
+
waitingForApproval?: true;
|
|
588
642
|
}
|
|
589
643
|
/**
|
|
590
644
|
* The checks a batch item can be handed to a person for, and the routing
|
|
@@ -809,26 +863,28 @@ export declare class RequestError extends Error {
|
|
|
809
863
|
}
|
|
810
864
|
/** The hints a refusal carries for the options W2L does not offer: the next honest step, never a way around the refusal. */
|
|
811
865
|
export declare const REFUSAL_HINTS: {
|
|
812
|
-
readonly stealth: "
|
|
813
|
-
readonly ignoreRobotsTxt: "robots.txt is always read; a
|
|
814
|
-
readonly hostedSkipTlsVerification: "a hosted server verifies every certificate; run
|
|
815
|
-
readonly useIndex: "
|
|
866
|
+
readonly stealth: "Octocrawl has no stealth option on a request: a provider's stealth or challenge solving runs only on a server started with an access grant that names it (--access-grant, ADR 0005); a proxy or session you own (mode authed) is the other route";
|
|
867
|
+
readonly ignoreRobotsTxt: "robots.txt is always read and recorded; on a local server a URL a scrape or batch names is fetched whatever it says, and ignoreRobotsTxt on a crawl or map fetches the links it disallows, on the record";
|
|
868
|
+
readonly hostedSkipTlsVerification: "a hosted server verifies every certificate; run Octocrawl locally to use skipTlsVerification, which is recorded in the trace and a tls_unverified warning";
|
|
869
|
+
readonly useIndex: "Octocrawl keeps no URL index: a map reads the sitemaps the site declares and its start page, on the record; crawl reads further pages";
|
|
816
870
|
readonly actions: "actions run on scrape and batch, where each page named gets the same steps; a crawl or a map does not take them";
|
|
817
871
|
};
|
|
818
872
|
/** The hint for a refused request key, or null when the key has none (an option W2L simply does not know). */
|
|
819
873
|
export declare function refusalHint(key: string, value: unknown): string | null;
|
|
820
874
|
export declare const PAGE_KEYS: readonly ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown"];
|
|
821
875
|
export declare const ATTRIBUTION_KEYS: readonly ["origin", "integration"];
|
|
822
|
-
export declare const SCRAPE_KEYS: readonly ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
876
|
+
export declare const SCRAPE_KEYS: readonly ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", "lane", "access", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
877
|
+
/** The lanes a request may ask for by name. */
|
|
878
|
+
export declare const REQUEST_LANES: readonly ["my-browser"];
|
|
823
879
|
export declare const CRAWL_SCOPE_KEYS: readonly ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
|
|
824
|
-
export declare const CRAWL_KEYS: readonly ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", "regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks", "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
825
|
-
export declare const BATCH_KEYS: readonly ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
880
|
+
export declare const CRAWL_KEYS: readonly ["url", "mode", "access", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", "regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks", "sitemap", "maxConcurrency", "idempotencyKey", "webhook", "ignoreRobotsTxt", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
881
|
+
export declare const BATCH_KEYS: readonly ["urls", "mode", "lane", "access", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", "onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown", "origin", "integration"];
|
|
826
882
|
/** What a batch body may carry beside `appendToId`: the job's own options are not among them (the scope no-ops change nothing, so they may come along). */
|
|
827
883
|
export declare const BATCH_APPEND_KEYS: readonly ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", "origin", "integration"];
|
|
828
884
|
/** The scope options a map takes under their crawl names; allowSubdomains is includeSubdomains on a map, and allowExternalLinks is not offered. */
|
|
829
885
|
export declare const MAP_SCOPE_KEYS: readonly ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
|
|
830
886
|
/** What a map takes. No page option (headers, mobile, skipTlsVerification, formats, ...): a map has nothing to loosen. */
|
|
831
|
-
export declare const MAP_KEYS: readonly ["url", "mode", "limit", "timeout", "search", "sitemap", "includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs", "includePaths", "excludePaths", "origin", "integration"];
|
|
887
|
+
export declare const MAP_KEYS: readonly ["url", "mode", "limit", "timeout", "search", "sitemap", "includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs", "includePaths", "excludePaths", "ignoreRobotsTxt", "origin", "integration"];
|
|
832
888
|
/** Why a webhook header (lower-cased name) cannot be sent, or null when it can: W2L's own and the transport's names are reserved. */
|
|
833
889
|
export declare function webhookHeaderRefusal(name: string): string | null;
|
|
834
890
|
/**
|
|
@@ -869,8 +925,8 @@ export declare function parseCrawlStartRequest(body: unknown): CrawlStartRequest
|
|
|
869
925
|
* A map request: url, mode (standard or research), limit, timeout, search,
|
|
870
926
|
* sitemap, includeSubdomains, the crawl's scope options under their crawl
|
|
871
927
|
* names (ignoreQueryParameters, includePaths, excludePaths, regexOnFullURL,
|
|
872
|
-
* crawlEntireDomain, deduplicateSimilarURLs), origin and
|
|
873
|
-
* Anything else is refused by name, `useIndex` with the supported route;
|
|
928
|
+
* crawlEntireDomain, deduplicateSimilarURLs), ignoreRobotsTxt, origin and
|
|
929
|
+
* integration. Anything else is refused by name, `useIndex` with the supported route;
|
|
874
930
|
* nothing is silently ignored.
|
|
875
931
|
*/
|
|
876
932
|
export declare function parseMapRequest(body: unknown): MapRequest;
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
* Granularity is the page. A partial parse inside a page is not a step; the
|
|
8
8
|
* whole URL is retried. Block-level checkpoint is out of Phase 1.
|
|
9
9
|
*/
|
|
10
|
-
import type { PageOptions, RequestAttribution, RobotsUrlOverride, WebhookEvent } from './api.js';
|
|
10
|
+
import type { AccessChoice, PageOptions, RequestAttribution, RobotsUrlOverride, WebhookEvent } from './api.js';
|
|
11
11
|
import type { CrawlMode } from './compliance.js';
|
|
12
12
|
import type { WebhookPayloadFormat } from './delivery.js';
|
|
13
13
|
import type { CrawlDiscovery, SitemapMode } from './crawl.js';
|
|
@@ -29,6 +29,11 @@ export interface CrawlBudget {
|
|
|
29
29
|
maxWallMs: number | null;
|
|
30
30
|
maxCostUsd: number | null;
|
|
31
31
|
maxTokens: number | null;
|
|
32
|
+
/**
|
|
33
|
+
* What one page may spend on third parties (an access grant's `perRequestUsd`, ROADMAP PA item 4): its own cap
|
|
34
|
+
* within the run's, held by the run's spend ledger. Absent or null: none of its own.
|
|
35
|
+
*/
|
|
36
|
+
maxCostPerPageUsd?: number | null;
|
|
32
37
|
}
|
|
33
38
|
export declare const DEFAULT_CRAWL_BUDGET: CrawlBudget;
|
|
34
39
|
/**
|
|
@@ -72,12 +77,16 @@ export interface Task {
|
|
|
72
77
|
maxConcurrency?: number;
|
|
73
78
|
invalidURLs?: readonly string[];
|
|
74
79
|
webhook?: StoredJobWebhook;
|
|
80
|
+
lane?: 'my-browser';
|
|
81
|
+
access?: AccessChoice;
|
|
75
82
|
} & PageOptions;
|
|
76
83
|
/**
|
|
77
84
|
* Every crawl option but the page budget (`budget`), stored when the crawl
|
|
78
85
|
* starts so a resumed crawl runs with the options it was started with.
|
|
79
86
|
*/
|
|
80
87
|
crawl?: {
|
|
88
|
+
/** The plain access choice the crawl was started with (`standard` or `enhanced`); absent: the server's configuration. */
|
|
89
|
+
access?: 'standard' | 'enhanced';
|
|
81
90
|
formats?: readonly ScrapeFormat[];
|
|
82
91
|
includeLinks?: boolean;
|
|
83
92
|
includePaths?: readonly string[];
|
|
@@ -106,6 +115,8 @@ export interface Task {
|
|
|
106
115
|
maxConcurrency?: number | null;
|
|
107
116
|
/** The crawl's webhook, when the request set one. */
|
|
108
117
|
webhook?: StoredJobWebhook;
|
|
118
|
+
/** The crawl fetches what robots.txt disallows, on the record (CrawlStartRequest.ignoreRobotsTxt); absent: it obeys. A server that takes no override resumes it obeying. */
|
|
119
|
+
ignoreRobotsTxt?: boolean;
|
|
109
120
|
} & PageOptions;
|
|
110
121
|
/** Who started the task (`origin`, `integration`), stored with it and reported as `attribution` on its status; absent when the request named neither. */
|
|
111
122
|
attribution?: RequestAttribution;
|
|
@@ -134,6 +145,11 @@ export interface Attempt {
|
|
|
134
145
|
recoveredFromAttemptId?: string | null;
|
|
135
146
|
/** What this attempt's pages offered the frontier and what became of it, written after every page of a crawl; absent for a batch and for an attempt stored before it was kept. */
|
|
136
147
|
discovery?: CrawlDiscovery | null;
|
|
148
|
+
/**
|
|
149
|
+
* What the spend ledger charged this attempt's paid calls (ROADMAP PA item 4): a resumed or appended run of the task
|
|
150
|
+
* opens its ledger with every earlier attempt's charge, so its cap is the task's, not each run's. Absent: none.
|
|
151
|
+
*/
|
|
152
|
+
chargedUsd?: number | null;
|
|
137
153
|
}
|
|
138
154
|
/**
|
|
139
155
|
* One URL inside one attempt. The atomic checkpoint unit.
|
|
@@ -109,13 +109,24 @@ export declare function researchUserAgent(contact?: string | null, host?: string
|
|
|
109
109
|
export declare function declaredContact(userAgent: string): string | null;
|
|
110
110
|
/** Whether a User-Agent is one research mode declares, in either format. */
|
|
111
111
|
export declare function isResearchUserAgent(userAgent: string): boolean;
|
|
112
|
+
/**
|
|
113
|
+
* The product token a robots.txt names to address Octocrawl itself
|
|
114
|
+
* (`User-agent: Octocrawl`), whatever User-Agent header a request sends. It
|
|
115
|
+
* is matched in every mode. A rule written for it (or for research mode's own
|
|
116
|
+
* token) is not set aside for a URL the request names or for a crawl or map
|
|
117
|
+
* started with ignoreRobotsTxt; only a recorded robotsOverride sets it aside.
|
|
118
|
+
*/
|
|
119
|
+
export declare const PRODUCT_ROBOTS_TOKEN = "octocrawl";
|
|
112
120
|
/**
|
|
113
121
|
* The text robots.txt `User-agent` lines are matched against for a
|
|
114
|
-
* User-Agent
|
|
115
|
-
*
|
|
116
|
-
* SEC.gov as it does on
|
|
122
|
+
* User-Agent Octocrawl sends: the header, with the product token added. SEC's
|
|
123
|
+
* format names no product token, so the research token is added as well: a
|
|
124
|
+
* group for w2l-research governs research requests to SEC.gov as it does on
|
|
125
|
+
* every other host.
|
|
117
126
|
*/
|
|
118
127
|
export declare function robotsAgent(userAgent: string): string;
|
|
128
|
+
/** Whether the robots.txt group that decided names Octocrawl itself (its product token, or research mode's) rather than every crawler (`*`). */
|
|
129
|
+
export declare function isOctocrawlRobotsGroup(matchedAgent: string | null | undefined): boolean;
|
|
119
130
|
/** The operator's contact from `W2L_CONTACT`, trimmed; null when unset or blank. The error never repeats the value. */
|
|
120
131
|
export declare function operatorContact(env: Readonly<Record<string, string | undefined>>): string | null;
|
|
121
132
|
/** An operator policy whose research-mode requests declare `W2L_CONTACT`, when it is set. */
|
|
@@ -254,6 +265,23 @@ export interface RobotsOverride {
|
|
|
254
265
|
/** Who recorded the decision, when the caller wants that on the record. */
|
|
255
266
|
recordedBy?: string;
|
|
256
267
|
}
|
|
268
|
+
/**
|
|
269
|
+
* On whose word a fetch went past robots.txt. `robots_override`: the
|
|
270
|
+
* caller's recorded decision for this URL (`RobotsOverride`).
|
|
271
|
+
* `user_named_url`: a local server fetching a URL the request named (a
|
|
272
|
+
* scrape, a batch entry), since robots.txt addresses crawlers that discover
|
|
273
|
+
* links, not the pages a person names. `ignore_robots_txt`: a crawl or map
|
|
274
|
+
* the caller started with `ignoreRobotsTxt` on a local server.
|
|
275
|
+
*/
|
|
276
|
+
export type RobotsOverrideBasis = 'robots_override' | 'user_named_url' | 'ignore_robots_txt';
|
|
277
|
+
/**
|
|
278
|
+
* A robots override as a lane applies it: the caller's recorded one, or one
|
|
279
|
+
* W2L applies by rule, which says so in `basis` (absent: the caller's,
|
|
280
|
+
* `robots_override`). Set by W2L, never read from a request.
|
|
281
|
+
*/
|
|
282
|
+
export interface AppliedRobotsOverride extends RobotsOverride {
|
|
283
|
+
basis?: Exclude<RobotsOverrideBasis, 'robots_override'>;
|
|
284
|
+
}
|
|
257
285
|
/**
|
|
258
286
|
* The outcome of consulting robots.txt for a single target URL. One record per
|
|
259
287
|
* fetch. `consulted` distinguishes "we checked and it said X" from "there was
|
|
@@ -288,11 +316,11 @@ export interface RobotsDecision {
|
|
|
288
316
|
*/
|
|
289
317
|
unreachable?: RobotsUnreachable;
|
|
290
318
|
/**
|
|
291
|
-
* Present when a disallow
|
|
292
|
-
*
|
|
293
|
-
*
|
|
319
|
+
* Present when a disallow was set aside, the publisher's or the one an
|
|
320
|
+
* unreachable robots.txt implies: the fetch went ahead (`skippedFetch:
|
|
321
|
+
* false`) and this says on whose word.
|
|
294
322
|
*/
|
|
295
|
-
override?:
|
|
323
|
+
override?: AppliedRobotsOverride;
|
|
296
324
|
}
|
|
297
325
|
/**
|
|
298
326
|
* What actually went on the wire. Sorted by header name, lowercased names.
|
|
@@ -86,7 +86,7 @@ export interface SitemapFileRecord {
|
|
|
86
86
|
kind: SitemapFileKind;
|
|
87
87
|
/** `<loc>` entries the file holds (child sitemaps for an index), http(s) ones only; null when the file was not parsed. */
|
|
88
88
|
entries: number | null;
|
|
89
|
-
/** The robots.txt verdict for the file's own URL under the crawl's identity (an unreachable robots.txt is `disallowed`, as for a page); null when the URL failed its egress check before robots.txt was consulted. */
|
|
89
|
+
/** The robots.txt verdict for the file's own URL under the crawl's identity (an unreachable robots.txt is `disallowed`, as for a page); null when the URL failed its egress check before robots.txt was consulted. A `disallowed` file is `refused`, unless the crawl or map was started with ignoreRobotsTxt and read it. */
|
|
90
90
|
robots: 'allowed' | 'disallowed' | 'no_robots' | null;
|
|
91
91
|
/** Whether the request left through the operator's environment proxy (local mode); a hosted server never has one. */
|
|
92
92
|
proxyUsed: boolean;
|