@octocrawl/sdk 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +17 -0
- package/dist/index.cjs +1278 -0
- package/dist/index.js +1240 -0
- package/dist/types/cjs/client.d.ts +362 -0
- package/dist/types/cjs/contracts/access.d.ts +166 -0
- package/dist/types/cjs/contracts/actions.d.ts +191 -0
- package/dist/types/cjs/contracts/api.d.ts +891 -0
- package/dist/types/cjs/contracts/benchmark.d.ts +116 -0
- package/dist/types/cjs/contracts/checkpoint.d.ts +164 -0
- package/dist/types/cjs/contracts/compliance.d.ts +412 -0
- package/dist/types/cjs/contracts/crawl.d.ts +302 -0
- package/dist/types/cjs/contracts/delivery.d.ts +136 -0
- package/dist/types/cjs/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/cjs/contracts/execution.d.ts +197 -0
- package/dist/types/cjs/contracts/extractor.d.ts +379 -0
- package/dist/types/cjs/contracts/file.d.ts +117 -0
- package/dist/types/cjs/contracts/firecrawl.d.ts +258 -0
- package/dist/types/cjs/contracts/groundTruth.d.ts +77 -0
- package/dist/types/cjs/contracts/identityBundle.d.ts +70 -0
- package/dist/types/cjs/contracts/index.d.ts +30 -0
- package/dist/types/cjs/contracts/map.d.ts +180 -0
- package/dist/types/cjs/contracts/monitor.d.ts +217 -0
- package/dist/types/cjs/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/cjs/contracts/policy.d.ts +93 -0
- package/dist/types/cjs/contracts/proxy.d.ts +52 -0
- package/dist/types/cjs/contracts/recipe.d.ts +74 -0
- package/dist/types/cjs/contracts/regexSafety.d.ts +31 -0
- package/dist/types/cjs/contracts/result.d.ts +503 -0
- package/dist/types/cjs/contracts/session.d.ts +51 -0
- package/dist/types/cjs/contracts/ssrf.d.ts +16 -0
- package/dist/types/cjs/contracts/status.d.ts +27 -0
- package/dist/types/cjs/contracts/structured.d.ts +185 -0
- package/dist/types/cjs/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/cjs/contracts/tokens.d.ts +47 -0
- package/dist/types/cjs/index.d.ts +9 -0
- package/dist/types/cjs/package.json +1 -0
- package/dist/types/cjs/version.d.ts +8 -0
- package/dist/types/cjs/watcher.d.ts +151 -0
- package/dist/types/esm/client.d.ts +362 -0
- package/dist/types/esm/contracts/access.d.ts +166 -0
- package/dist/types/esm/contracts/actions.d.ts +191 -0
- package/dist/types/esm/contracts/api.d.ts +891 -0
- package/dist/types/esm/contracts/benchmark.d.ts +116 -0
- package/dist/types/esm/contracts/checkpoint.d.ts +164 -0
- package/dist/types/esm/contracts/compliance.d.ts +412 -0
- package/dist/types/esm/contracts/crawl.d.ts +302 -0
- package/dist/types/esm/contracts/delivery.d.ts +136 -0
- package/dist/types/esm/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/esm/contracts/execution.d.ts +197 -0
- package/dist/types/esm/contracts/extractor.d.ts +379 -0
- package/dist/types/esm/contracts/file.d.ts +117 -0
- package/dist/types/esm/contracts/firecrawl.d.ts +258 -0
- package/dist/types/esm/contracts/groundTruth.d.ts +77 -0
- package/dist/types/esm/contracts/identityBundle.d.ts +70 -0
- package/dist/types/esm/contracts/index.d.ts +30 -0
- package/dist/types/esm/contracts/map.d.ts +180 -0
- package/dist/types/esm/contracts/monitor.d.ts +217 -0
- package/dist/types/esm/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/esm/contracts/policy.d.ts +93 -0
- package/dist/types/esm/contracts/proxy.d.ts +52 -0
- package/dist/types/esm/contracts/recipe.d.ts +74 -0
- package/dist/types/esm/contracts/regexSafety.d.ts +31 -0
- package/dist/types/esm/contracts/result.d.ts +503 -0
- package/dist/types/esm/contracts/session.d.ts +51 -0
- package/dist/types/esm/contracts/ssrf.d.ts +16 -0
- package/dist/types/esm/contracts/status.d.ts +27 -0
- package/dist/types/esm/contracts/structured.d.ts +185 -0
- package/dist/types/esm/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/esm/contracts/tokens.d.ts +47 -0
- package/dist/types/esm/index.d.ts +9 -0
- package/dist/types/esm/version.d.ts +8 -0
- package/dist/types/esm/watcher.d.ts +151 -0
- package/package.json +40 -0
|
@@ -0,0 +1,302 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Crawl composition contract: one scrape atom plus a crawl-level report.
|
|
3
|
+
*
|
|
4
|
+
* The orchestrator never opens Playwright. A page is one `scrape(url)`.
|
|
5
|
+
* Types only — no I/O.
|
|
6
|
+
*/
|
|
7
|
+
import type { AgentHints, JobWebhookStatus, RequestAttribution } from './api.js';
|
|
8
|
+
import type { CrawlBudget, StepStatus, TaskStatus } from './checkpoint.js';
|
|
9
|
+
import type { CrawlMode } from './compliance.js';
|
|
10
|
+
import type { Evidence, FetchResult, FetchWarning, HandoffRequest, LadderRunAudit, TraceEvent } from './result.js';
|
|
11
|
+
import type { EvidenceRecord } from './evidenceRecord.js';
|
|
12
|
+
import type { Lane } from './status.js';
|
|
13
|
+
import type { BudgetKind } from './status.js';
|
|
14
|
+
import type { ExecutionContext } from './execution.js';
|
|
15
|
+
/**
|
|
16
|
+
* How a crawl uses the site's sitemap: beside the links it finds (`include`,
|
|
17
|
+
* the default), not at all (`skip`), or as its only source of URLs beside the
|
|
18
|
+
* start URL (`only`: page links are returned when asked for, never followed).
|
|
19
|
+
*/
|
|
20
|
+
export declare const SITEMAP_MODES: readonly ["include", "skip", "only"];
|
|
21
|
+
export type SitemapMode = (typeof SITEMAP_MODES)[number];
|
|
22
|
+
export interface ScrapeOutcome {
|
|
23
|
+
result: FetchResult;
|
|
24
|
+
links: readonly string[];
|
|
25
|
+
audit?: LadderRunAudit;
|
|
26
|
+
crawlDelayMs?: number | null;
|
|
27
|
+
/**
|
|
28
|
+
* True when the result is a stored one the cache answered with: nothing
|
|
29
|
+
* was fetched, so the page costs no fetch and says nothing new about its
|
|
30
|
+
* host's robots.txt (its Crawl-delay is left as it was).
|
|
31
|
+
*/
|
|
32
|
+
cached?: boolean;
|
|
33
|
+
}
|
|
34
|
+
/**
|
|
35
|
+
* One URL in, one page outcome out. The ladder is the production
|
|
36
|
+
* implementation; Phase 4 tests inject a fake.
|
|
37
|
+
*/
|
|
38
|
+
export interface ScrapeAtom {
|
|
39
|
+
scrape(url: string, context?: ExecutionContext): Promise<ScrapeOutcome>;
|
|
40
|
+
close(): Promise<void>;
|
|
41
|
+
}
|
|
42
|
+
export interface SitemapLoadRequest {
|
|
43
|
+
seedUrl: string;
|
|
44
|
+
/** Stop collecting entries once this many are in hand: the crawl's maxPages, or 50 000 when it is unbounded. */
|
|
45
|
+
maxUrls: number;
|
|
46
|
+
/** At most this many sitemap files are fetched for one load, an index and its children each counting as one. */
|
|
47
|
+
maxFiles: number;
|
|
48
|
+
/**
|
|
49
|
+
* Opt-in (a map passes it, a crawl does not): judges each entry before it is
|
|
50
|
+
* collected. An entry it refuses is not collected and does not count toward
|
|
51
|
+
* `maxUrls`. Once `maxUrls` entries are in hand, the rest of the file in
|
|
52
|
+
* flight is still offered, and those it accepts are counted as left over
|
|
53
|
+
* (`truncated: 'urls'`), never collected.
|
|
54
|
+
*/
|
|
55
|
+
accept?: (entry: SitemapEntry) => boolean | Promise<boolean>;
|
|
56
|
+
/**
|
|
57
|
+
* Opt-in (a map passes it, a crawl does not): an absolute UTC time after
|
|
58
|
+
* which the load stops and returns what it has, `truncated: 'time'`. The
|
|
59
|
+
* file in flight is aborted and recorded as `unreadable`, error `timeout`.
|
|
60
|
+
* The caller's own cancellation still throws.
|
|
61
|
+
*/
|
|
62
|
+
softDeadlineAt?: number;
|
|
63
|
+
}
|
|
64
|
+
/** Where a load looked for sitemaps: the `Sitemap:` lines of the start URL's robots.txt, or the conventional `/sitemap.xml` when it lists none. */
|
|
65
|
+
export type SitemapSourceKind = 'robots' | 'guess';
|
|
66
|
+
/**
|
|
67
|
+
* What one sitemap file turned out to be: a `<sitemapindex>`, a `<urlset>`, a
|
|
68
|
+
* 4xx (`absent`), a 2xx body that is neither (`not_sitemap`), a file that
|
|
69
|
+
* could not be read (`unreadable`: too large, over the decompression cap, a
|
|
70
|
+
* Content-Encoding W2L does not decode or bytes that do not decode as theirs,
|
|
71
|
+
* a 5xx, a transport failure; `error` says which) or one its host's robots.txt
|
|
72
|
+
* disallows for the crawl's identity (`refused`, never requested).
|
|
73
|
+
*/
|
|
74
|
+
export type SitemapFileKind = 'index' | 'urlset' | 'absent' | 'not_sitemap' | 'unreadable' | 'refused';
|
|
75
|
+
/** One sitemap file a load fetched or refused, on the record of the crawl that used it. These fetches carry no signed compliance record. */
|
|
76
|
+
export interface SitemapFileRecord {
|
|
77
|
+
url: string;
|
|
78
|
+
/** Where the file was read from after redirects; null when no response answered. */
|
|
79
|
+
finalUrl: string | null;
|
|
80
|
+
status: number | null;
|
|
81
|
+
contentType: string | null;
|
|
82
|
+
/** Bytes on the wire, before any gzip inflation; null when no body was read. */
|
|
83
|
+
bytes: number | null;
|
|
84
|
+
/** SHA-256 of the bytes as received; null when no body was read. */
|
|
85
|
+
sha256: string | null;
|
|
86
|
+
kind: SitemapFileKind;
|
|
87
|
+
/** `<loc>` entries the file holds (child sitemaps for an index), http(s) ones only; null when the file was not parsed. */
|
|
88
|
+
entries: number | null;
|
|
89
|
+
/** The robots.txt verdict for the file's own URL under the crawl's identity (an unreachable robots.txt is `disallowed`, as for a page); null when the URL failed its egress check before robots.txt was consulted. */
|
|
90
|
+
robots: 'allowed' | 'disallowed' | 'no_robots' | null;
|
|
91
|
+
/** Whether the request left through the operator's environment proxy (local mode); a hosted server never has one. */
|
|
92
|
+
proxyUsed: boolean;
|
|
93
|
+
error: string | null;
|
|
94
|
+
}
|
|
95
|
+
/** A URL a sitemap listed and the file that listed it. */
|
|
96
|
+
export interface SitemapEntry {
|
|
97
|
+
url: string;
|
|
98
|
+
file: string;
|
|
99
|
+
/** The entry's `<lastmod>`, as written (trimmed, entities decoded, not normalised); absent when it has none. A crawl ignores it. */
|
|
100
|
+
lastmod?: string;
|
|
101
|
+
/** The entry's `<news:title>`, entity-decoded and trimmed; absent when it has none. A crawl ignores it. */
|
|
102
|
+
title?: string;
|
|
103
|
+
}
|
|
104
|
+
export interface SitemapLoadResult {
|
|
105
|
+
/** The declared identity the files were requested with: the crawl mode's http identity. */
|
|
106
|
+
identity: {
|
|
107
|
+
mode: CrawlMode;
|
|
108
|
+
userAgent: string;
|
|
109
|
+
};
|
|
110
|
+
sources: SitemapSourceKind[];
|
|
111
|
+
files: SitemapFileRecord[];
|
|
112
|
+
/** The entries collected, in listed order, each once, up to `maxUrls`. */
|
|
113
|
+
urls: SitemapEntry[];
|
|
114
|
+
/** Set when the load stopped before reading everything: at `maxFiles` with files unread, at `maxUrls` with entries uncollected, or at `softDeadlineAt` (`time`, never on a crawl). */
|
|
115
|
+
truncated: 'files' | 'urls' | 'time' | null;
|
|
116
|
+
/** Only with `truncated: 'time'`: the files located (declared, or children of an index read) and not requested, the aborted one excluded; absent when the deadline came before any was located. */
|
|
117
|
+
unreadFiles?: number;
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Reads a site's sitemaps for a crawl: an auxiliary fetch path beside
|
|
121
|
+
* robots.txt, never a page fetcher. The bench's HttpSitemapSource is the
|
|
122
|
+
* production implementation; tests inject a fake. One source serves one crawl
|
|
123
|
+
* and is closed with it.
|
|
124
|
+
*/
|
|
125
|
+
export interface SitemapSource {
|
|
126
|
+
load(request: SitemapLoadRequest, context?: ExecutionContext): Promise<SitemapLoadResult>;
|
|
127
|
+
close(): Promise<void>;
|
|
128
|
+
}
|
|
129
|
+
/** What a crawl attempt's sitemap load found and what the frontier made of it. Null when the crawl read no sitemap. */
|
|
130
|
+
export interface SitemapDiscovery {
|
|
131
|
+
mode: SitemapMode;
|
|
132
|
+
sources: SitemapSourceKind[];
|
|
133
|
+
files: SitemapFileRecord[];
|
|
134
|
+
/** Entries the load returned and offered to the frontier. */
|
|
135
|
+
listed: number;
|
|
136
|
+
/** Entries the frontier accepted as pages to fetch. */
|
|
137
|
+
enqueued: number;
|
|
138
|
+
/** As SitemapLoadResult says; a crawl never produces `time`. */
|
|
139
|
+
truncated: 'files' | 'urls' | 'time' | null;
|
|
140
|
+
/** Why the load itself failed, when it threw before returning; `files` then holds what it had read. Null otherwise. */
|
|
141
|
+
error: string | null;
|
|
142
|
+
}
|
|
143
|
+
/**
|
|
144
|
+
* What one run of the orchestrator crawls. A new task stores its budget,
|
|
145
|
+
* maxDepth, allowlistedDomains, includePaths, excludePaths, URL-scope options,
|
|
146
|
+
* sitemap mode and concurrency cap; a resumed (`resumeFrom`) or existing
|
|
147
|
+
* (`taskId`) task runs with the ones it stored, so a resume never widens the
|
|
148
|
+
* crawl it continues.
|
|
149
|
+
*/
|
|
150
|
+
export interface CrawlSpec {
|
|
151
|
+
seedUrl: string;
|
|
152
|
+
seedUrls?: readonly string[];
|
|
153
|
+
taskDir: string;
|
|
154
|
+
mode: CrawlMode;
|
|
155
|
+
budget: CrawlBudget;
|
|
156
|
+
maxDepth: number | null;
|
|
157
|
+
/** Hosts links may lead to beside the seed's host, its www twin and where the seed redirected. */
|
|
158
|
+
allowlistedDomains: readonly string[];
|
|
159
|
+
resumeFrom: string | null;
|
|
160
|
+
useCached: boolean;
|
|
161
|
+
/** When set, openRun updates this existing task instead of inserting a new id. */
|
|
162
|
+
taskId?: string;
|
|
163
|
+
/**
|
|
164
|
+
* Pathname regexes for discovered links (the seed is always fetched); a
|
|
165
|
+
* match in excludePaths wins.
|
|
166
|
+
*/
|
|
167
|
+
includePaths?: readonly string[];
|
|
168
|
+
excludePaths?: readonly string[];
|
|
169
|
+
/** The URL-scope options, as CrawlStartRequest names them; DEFAULT_CRAWL_SPEC has their defaults. */
|
|
170
|
+
regexOnFullURL?: boolean;
|
|
171
|
+
ignoreQueryParameters?: boolean;
|
|
172
|
+
deduplicateSimilarURLs?: boolean;
|
|
173
|
+
crawlEntireDomain?: boolean;
|
|
174
|
+
allowSubdomains?: boolean;
|
|
175
|
+
allowExternalLinks?: boolean;
|
|
176
|
+
/** How the crawl uses the site's sitemap; `skip` when the run has no SitemapSource. */
|
|
177
|
+
sitemap?: SitemapMode;
|
|
178
|
+
/** Pages this crawl fetches at once, at most; null takes the service's worker count. Never raises the per-host ceiling. */
|
|
179
|
+
maxConcurrency?: number | null;
|
|
180
|
+
}
|
|
181
|
+
/**
|
|
182
|
+
* What a crawl attempt's pages offered the frontier and what became of each
|
|
183
|
+
* link: enqueued, an exact repeat (`duplicate`), a variant folded into a
|
|
184
|
+
* first-seen page (`collapsed`: a query the crawl ignores, `/a/` after `/a`,
|
|
185
|
+
* `/index.html` after `/`, the www twin), or refused by the host scope, the
|
|
186
|
+
* start URL's path subtree, includePaths / excludePaths or maxDepth. The
|
|
187
|
+
* `offered` total also counts links no counter names (assets, non-http
|
|
188
|
+
* schemes, a path a filter could not decide). Sitemap entries the crawl
|
|
189
|
+
* offered count here too, and `sitemap` says what the load read. `duplicateContent`
|
|
190
|
+
* counts pages fetched and then found to repeat an earlier page's body. Null
|
|
191
|
+
* on a batch, which discovers nothing.
|
|
192
|
+
*/
|
|
193
|
+
export interface CrawlDiscovery {
|
|
194
|
+
offered: number;
|
|
195
|
+
enqueued: number;
|
|
196
|
+
duplicate: number;
|
|
197
|
+
collapsed: number;
|
|
198
|
+
hostDenied: number;
|
|
199
|
+
subtreeDenied: number;
|
|
200
|
+
pathDenied: number;
|
|
201
|
+
depthDenied: number;
|
|
202
|
+
duplicateContent: number;
|
|
203
|
+
/** The attempt's sitemap load; null when the crawl read no sitemap (`sitemap: skip`, or a task stored before the option). */
|
|
204
|
+
sitemap: SitemapDiscovery | null;
|
|
205
|
+
}
|
|
206
|
+
export declare const EMPTY_CRAWL_DISCOVERY: CrawlDiscovery;
|
|
207
|
+
export interface CrawlReport {
|
|
208
|
+
taskId: string;
|
|
209
|
+
attemptId: string;
|
|
210
|
+
status: TaskStatus;
|
|
211
|
+
pagesFetched: number;
|
|
212
|
+
cachedPages: number;
|
|
213
|
+
budgetExceeded: BudgetKind | null;
|
|
214
|
+
loopDetected: boolean;
|
|
215
|
+
wallMs: number;
|
|
216
|
+
costUsd: number | null;
|
|
217
|
+
costUnknown?: boolean;
|
|
218
|
+
contentTokens: number | null;
|
|
219
|
+
contentTokensUnknown?: boolean;
|
|
220
|
+
/** The latest attempt's link discovery counters; null for a batch and for a crawl stored before they were kept. */
|
|
221
|
+
discovery: CrawlDiscovery | null;
|
|
222
|
+
/** Who started the task, as the request said (`origin`, `integration`); absent when it named neither. */
|
|
223
|
+
attribution?: RequestAttribution;
|
|
224
|
+
/** The task's webhook and how its deliveries stand; absent when the request set none. */
|
|
225
|
+
webhook?: JobWebhookStatus;
|
|
226
|
+
}
|
|
227
|
+
export interface CrawlPage {
|
|
228
|
+
id: string;
|
|
229
|
+
url: string;
|
|
230
|
+
canonicalUrl: string;
|
|
231
|
+
depth: number;
|
|
232
|
+
status: StepStatus;
|
|
233
|
+
lane: Lane | null;
|
|
234
|
+
markdown: string | null;
|
|
235
|
+
/** The fetch's caveats (a recorded robots override), as on a scrape result; absent when it had none. */
|
|
236
|
+
warnings?: readonly FetchWarning[];
|
|
237
|
+
/** The warnings' messages joined with a space, present exactly when `warnings` is, as on a scrape response. */
|
|
238
|
+
warning?: string;
|
|
239
|
+
/** What to change about the request next time (a login wall, a robots.txt rule, a cut), as on a scrape response; absent when nothing applies. */
|
|
240
|
+
agentHints?: AgentHints;
|
|
241
|
+
/**
|
|
242
|
+
* A batch item stopped at a check W2L does not pass (a captcha, a
|
|
243
|
+
* challenge, a login wall), on a server that can hand it to a person in
|
|
244
|
+
* their own Chrome (`POST /v1/batches/:id/handoff`): why, and how. Absent
|
|
245
|
+
* otherwise.
|
|
246
|
+
*/
|
|
247
|
+
handoff?: HandoffRequest;
|
|
248
|
+
/** Present when the task asked for the `html` format, as on a scrape result; null when the page has none. */
|
|
249
|
+
html?: string | null;
|
|
250
|
+
/** Present when the task asked for the `rawHtml` format, as on a scrape result; null when the page has none. */
|
|
251
|
+
rawHtml?: string | null;
|
|
252
|
+
/** Present when the task asked for the `images` format and the page was read as content, as on a scrape result. */
|
|
253
|
+
images?: readonly string[];
|
|
254
|
+
/** Present when the task asked for the `tables` format and the page was read as content, as on a scrape result. */
|
|
255
|
+
tables?: FetchResult['tables'];
|
|
256
|
+
/** Present when the task's `pdf` parser asked for `pages` and the page is a PDF whose text was read, as on a scrape result. */
|
|
257
|
+
pages?: FetchResult['pages'];
|
|
258
|
+
/** Present when the task asked for an `attributes` format and the page was read as content, as on a scrape result. */
|
|
259
|
+
attributes?: FetchResult['attributes'];
|
|
260
|
+
/** Present when the task asked for the `screenshot` format and the page rendered, as on a scrape result: the capture, or null when the browser lane could not capture it. */
|
|
261
|
+
screenshot?: FetchResult['screenshot'];
|
|
262
|
+
/** Present when the batch ran `actions` on the page: what the steps produced, and the step that failed if one did. */
|
|
263
|
+
actions?: FetchResult['actions'];
|
|
264
|
+
/** Present when the task asked for a `list` entry and the page was read: its records. */
|
|
265
|
+
list?: FetchResult['list'];
|
|
266
|
+
/** Absolute outbound links; present when the task requested links. */
|
|
267
|
+
links?: readonly string[];
|
|
268
|
+
/** The page's own title, description, language, ... as on a scrape result; absent when no page was extracted. */
|
|
269
|
+
metadata?: FetchResult['metadata'];
|
|
270
|
+
json?: import('./structured.js').StructuredExtractionResult | null;
|
|
271
|
+
/** The file the page was (PDF, CSV, ...), as on a scrape result; absent for a web page. */
|
|
272
|
+
file?: FetchResult['file'];
|
|
273
|
+
failureReason: string | null;
|
|
274
|
+
blockReason: string | null;
|
|
275
|
+
budgetExceeded: BudgetKind | null;
|
|
276
|
+
evidence: Evidence | null;
|
|
277
|
+
/** The page's Evidence Record v1; null while the page has no result yet. */
|
|
278
|
+
evidenceRecord: EvidenceRecord | null;
|
|
279
|
+
/** Per-URL latency, attempts and metering without loading the full audit. */
|
|
280
|
+
usage?: import('./result.js').ResourceUsage | null;
|
|
281
|
+
trace: readonly TraceEvent[];
|
|
282
|
+
audit?: LadderRunAudit;
|
|
283
|
+
/** True when the page was not fetched in this attempt: a stored result was reused (the cache, or a resume's own pages). */
|
|
284
|
+
cached: boolean;
|
|
285
|
+
/** Whether the cache answered, as on a scrape response's `metadata`: present when the task looked pages up (`maxAge`, `minAge`, `lockdown`). */
|
|
286
|
+
cacheState?: 'hit' | 'miss';
|
|
287
|
+
/** On a hit, when the reused result was fetched (its `evidenceRecord.fetchedAt`). */
|
|
288
|
+
cachedAt?: string;
|
|
289
|
+
contentHash: string | null;
|
|
290
|
+
createdAt: string;
|
|
291
|
+
updatedAt: string;
|
|
292
|
+
}
|
|
293
|
+
export interface CrawlError extends CrawlPage {
|
|
294
|
+
trace: readonly TraceEvent[];
|
|
295
|
+
}
|
|
296
|
+
export interface CrawlPageList<T> {
|
|
297
|
+
items: readonly T[];
|
|
298
|
+
nextCursor: string | null;
|
|
299
|
+
hasMore: boolean;
|
|
300
|
+
}
|
|
301
|
+
export declare const DEFAULT_CRAWL_SPEC: Omit<CrawlSpec, 'seedUrl' | 'taskDir'>;
|
|
302
|
+
//# sourceMappingURL=crawl.d.ts.map
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
import type { BatchStatusResponse, WebhookEvent } from './api.js';
|
|
2
|
+
import type { CrawlPage, CrawlReport } from './crawl.js';
|
|
3
|
+
import type { FirecrawlWebhookPayload } from './firecrawl.js';
|
|
4
|
+
import type { MonitorEvent, MonitorSnapshot } from './monitor.js';
|
|
5
|
+
/** Whose events a destination receives: a Monitor's, or a job's (a crawl or batch, destination `job:<taskId>`). */
|
|
6
|
+
export type DeliveryDestinationKind = 'monitor' | 'job';
|
|
7
|
+
/** The shape of a job destination's payloads: W2L's JobWebhookEnvelope, or Firecrawl's for a job started through the `/fc` shim. */
|
|
8
|
+
export type WebhookPayloadFormat = 'w2l' | 'firecrawl';
|
|
9
|
+
export interface DeliveryDestinationInput {
|
|
10
|
+
id: string;
|
|
11
|
+
/** The Monitor the destination belongs to; for a job destination, `job:<taskId>`. */
|
|
12
|
+
monitorId: string;
|
|
13
|
+
url: string;
|
|
14
|
+
maxAttempts?: number;
|
|
15
|
+
/** Operator environment reference; literal secrets are never stored or returned. */
|
|
16
|
+
secretEnv?: string;
|
|
17
|
+
enabled?: boolean;
|
|
18
|
+
/** Default `monitor`. A `job` destination takes the four options below; a Monitor destination takes none of them yet. */
|
|
19
|
+
kind?: DeliveryDestinationKind;
|
|
20
|
+
/** The job events to deliver; default all. */
|
|
21
|
+
events?: readonly WebhookEvent[];
|
|
22
|
+
/** Headers sent with every delivery (lower-cased names); stored in the control database alone and never returned. */
|
|
23
|
+
headers?: Readonly<Record<string, string>>;
|
|
24
|
+
/** Strings copied into every payload's `metadata`. */
|
|
25
|
+
metadata?: Readonly<Record<string, string>>;
|
|
26
|
+
payloadFormat?: WebhookPayloadFormat;
|
|
27
|
+
}
|
|
28
|
+
export interface DeliveryDestination {
|
|
29
|
+
id: string;
|
|
30
|
+
monitorId: string;
|
|
31
|
+
url: string;
|
|
32
|
+
maxAttempts: number;
|
|
33
|
+
secretEnv?: string;
|
|
34
|
+
enabled: boolean;
|
|
35
|
+
createdAt: number;
|
|
36
|
+
kind: DeliveryDestinationKind;
|
|
37
|
+
/** The job the destination belongs to (`kind: 'job'`). */
|
|
38
|
+
jobId?: string;
|
|
39
|
+
events?: readonly WebhookEvent[];
|
|
40
|
+
/** The names of the custom headers sent with each delivery; their values are never returned. */
|
|
41
|
+
headerNames: string[];
|
|
42
|
+
metadata?: Readonly<Record<string, string>>;
|
|
43
|
+
payloadFormat?: WebhookPayloadFormat;
|
|
44
|
+
}
|
|
45
|
+
/** eventVersion is the committed snapshot version, scoped to this monitor/entity/view. */
|
|
46
|
+
export interface WebhookEventEnvelope {
|
|
47
|
+
schemaVersion: 'w2l.monitor-event/v1';
|
|
48
|
+
eventId: string;
|
|
49
|
+
eventVersion: number;
|
|
50
|
+
monitorId: string;
|
|
51
|
+
workspaceId: string;
|
|
52
|
+
entityKey: string;
|
|
53
|
+
viewKey: string;
|
|
54
|
+
event: MonitorEvent;
|
|
55
|
+
snapshot: MonitorSnapshot;
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* One event of a crawl or batch, as its webhook receiver gets it. `eventId`
|
|
59
|
+
* is `<taskId>:started`, `<taskId>:page:<stepId>` or `<taskId>:<status>`
|
|
60
|
+
* (a job run again, a batch appended after completion, suffixes its later
|
|
61
|
+
* terminal ids with the attempt id); `sequence` is 0 for `started`, then
|
|
62
|
+
* one more for every page and terminal event in the order they were
|
|
63
|
+
* enqueued, so a job of n pages ends at n+1. Also the `x-w2l-event-id` and
|
|
64
|
+
* `x-w2l-event-version` headers of the delivery.
|
|
65
|
+
*/
|
|
66
|
+
export interface JobWebhookEnvelope {
|
|
67
|
+
schemaVersion: 'w2l.job-event/v1';
|
|
68
|
+
eventId: string;
|
|
69
|
+
sequence: number;
|
|
70
|
+
jobId: string;
|
|
71
|
+
jobKind: 'crawl' | 'batch';
|
|
72
|
+
event: WebhookEvent;
|
|
73
|
+
/** When the event was recorded, ISO 8601. */
|
|
74
|
+
at: string;
|
|
75
|
+
/** The request's `webhook.metadata`, `{}` when none. */
|
|
76
|
+
metadata: Readonly<Record<string, string>>;
|
|
77
|
+
/** On a `page` event: the page as `GET /v1/crawl/:id/pages` or `/v1/batches/:id/items` lists it (no audit, empty trace). */
|
|
78
|
+
page?: CrawlPage;
|
|
79
|
+
/** On a terminal event: the job's status as `GET /v1/crawl/:id` or `GET /v1/batches/:id` reports it then. */
|
|
80
|
+
report?: CrawlReport | BatchStatusResponse;
|
|
81
|
+
/** On `failed`: why. */
|
|
82
|
+
error?: string;
|
|
83
|
+
}
|
|
84
|
+
/** What a delivery carries: a Monitor event, a job event, or a job event in Firecrawl's shape. */
|
|
85
|
+
export type WebhookPayload = WebhookEventEnvelope | JobWebhookEnvelope | FirecrawlWebhookPayload;
|
|
86
|
+
export type DeliveryState = 'pending' | 'delivering' | 'delivered' | 'dead_letter';
|
|
87
|
+
export interface WebhookDelivery {
|
|
88
|
+
id: string;
|
|
89
|
+
destinationId: string;
|
|
90
|
+
monitorId: string;
|
|
91
|
+
eventId: string;
|
|
92
|
+
eventVersion: number;
|
|
93
|
+
state: DeliveryState;
|
|
94
|
+
attemptCount: number;
|
|
95
|
+
maxAttempts: number;
|
|
96
|
+
nextAttemptAt: number;
|
|
97
|
+
leaseUntil: number | null;
|
|
98
|
+
fencingToken: number;
|
|
99
|
+
createdAt: number;
|
|
100
|
+
deliveredAt: number | null;
|
|
101
|
+
lastStatus: number | null;
|
|
102
|
+
lastError: string | null;
|
|
103
|
+
payload: WebhookPayload;
|
|
104
|
+
}
|
|
105
|
+
export interface DeliveryAttempt {
|
|
106
|
+
id: string;
|
|
107
|
+
deliveryId: string;
|
|
108
|
+
fencingToken: number;
|
|
109
|
+
startedAt: number;
|
|
110
|
+
endedAt: number | null;
|
|
111
|
+
outcome: 'sending' | 'delivered' | 'retry' | 'dead_letter' | 'lease_expired';
|
|
112
|
+
status: number | null;
|
|
113
|
+
error: string | null;
|
|
114
|
+
retryAfterAt: number | null;
|
|
115
|
+
}
|
|
116
|
+
export interface DeliveryQuery {
|
|
117
|
+
monitorId?: string;
|
|
118
|
+
/** The crawl or batch whose deliveries to list: `monitorId` `job:<taskId>` under its own name; not beside `monitorId`. */
|
|
119
|
+
jobId?: string;
|
|
120
|
+
destinationId?: string;
|
|
121
|
+
state?: DeliveryState;
|
|
122
|
+
}
|
|
123
|
+
export interface DeliveryPageQuery extends DeliveryQuery {
|
|
124
|
+
cursor?: string;
|
|
125
|
+
limit?: number;
|
|
126
|
+
}
|
|
127
|
+
export interface DeliveryPage {
|
|
128
|
+
items: WebhookDelivery[];
|
|
129
|
+
nextCursor: string | null;
|
|
130
|
+
hasMore: boolean;
|
|
131
|
+
}
|
|
132
|
+
export interface DeliveryDetail {
|
|
133
|
+
delivery: WebhookDelivery;
|
|
134
|
+
attempts: DeliveryAttempt[];
|
|
135
|
+
}
|
|
136
|
+
//# sourceMappingURL=delivery.d.ts.map
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
import type { PageActionType } from './actions.js';
|
|
2
|
+
/**
|
|
3
|
+
* Evidence Record v1: the evidence every result carries, stated the same way
|
|
4
|
+
* whichever lane produced it. The published JSON Schema is
|
|
5
|
+
* `schemas/evidence-record.v1.json` in this package; the key lists below and
|
|
6
|
+
* `test/evidenceRecord.test.ts` keep the file and this type in step. Types
|
|
7
|
+
* only: the builder (`toEvidenceRecord`) lives in @w2l/runtime.
|
|
8
|
+
*
|
|
9
|
+
* Unknown is null, never zero or a guess. Every field is always present.
|
|
10
|
+
*
|
|
11
|
+
* Versioning: `w2l.evidence/1` changes only additively. A later revision of
|
|
12
|
+
* the schema file accepts every record an earlier one accepted: a field added
|
|
13
|
+
* later is optional in the schema (EVIDENCE_RECORD_ADDED_KEYS), though W2L
|
|
14
|
+
* writes it on every record from then on; enums only gain values; no field
|
|
15
|
+
* changes its meaning. A change that cannot follow this rule gets a new
|
|
16
|
+
* schemaVersion (`w2l.evidence/2`) and a new schema file.
|
|
17
|
+
*/
|
|
18
|
+
import type { CrawlMode, RobotsUnreachable } from './compliance.js';
|
|
19
|
+
import type { BlockReason, BudgetKind, FailureReason, Lane, ResultStatus } from './status.js';
|
|
20
|
+
export declare const EVIDENCE_SCHEMA_VERSION = "w2l.evidence/1";
|
|
21
|
+
/**
|
|
22
|
+
* Where a JSON field's value was read. `fetch` is the fetch's own value (its
|
|
23
|
+
* final or requested URL), not read from the page. `pdf`: a `Label: value`
|
|
24
|
+
* line of a PDF's text, whose locator names the page and the label.
|
|
25
|
+
*/
|
|
26
|
+
export declare const FIELD_EVIDENCE_SOURCES: readonly ["jsonld", "microdata", "meta", "hydration", "dom", "text", "inferred", "model", "fetch", "pdf"];
|
|
27
|
+
export type FieldEvidenceSource = (typeof FIELD_EVIDENCE_SOURCES)[number];
|
|
28
|
+
/**
|
|
29
|
+
* `snapshot` is the page's HTML as W2L read it (`W2L_CAPTURE_RAW_DIR`);
|
|
30
|
+
* `file` is a file (PDF, CSV, ...) saved as received; `screenshot` is the
|
|
31
|
+
* browser lane's capture for the `screenshot` format, saved under
|
|
32
|
+
* `W2L_CAPTURE_RAW_DIR` as `<sha256>.png` or `.jpg`, with its size and type.
|
|
33
|
+
*/
|
|
34
|
+
export declare const EVIDENCE_ARTIFACT_KINDS: readonly ["snapshot", "screenshot", "file"];
|
|
35
|
+
export type EvidenceArtifactKind = (typeof EVIDENCE_ARTIFACT_KINDS)[number];
|
|
36
|
+
export interface EvidenceRedirectChain {
|
|
37
|
+
/**
|
|
38
|
+
* Every URL W2L requested for the page, in order: the requested URL first,
|
|
39
|
+
* the final URL last, with each redirect between (in the browser lane, a
|
|
40
|
+
* document a script or a meta refresh loaded is one). `[requestedUrl]` when
|
|
41
|
+
* there was no redirect; empty when W2L sent no request for the page.
|
|
42
|
+
*/
|
|
43
|
+
urls: readonly string[];
|
|
44
|
+
/**
|
|
45
|
+
* True when every hop is listed. The HTTP lane follows redirects itself and
|
|
46
|
+
* lists each; the browser lane lists each redirect Chromium followed and
|
|
47
|
+
* each document a script or a meta refresh loaded, and is false only when a
|
|
48
|
+
* follow-up navigation did not start at the requested URL, a document came
|
|
49
|
+
* without a request, or the chain of a page that kept moving on was cut to
|
|
50
|
+
* its first URL and last 20. The provider lane sees only where its vendor
|
|
51
|
+
* started and ended, so its chain is `[requested, final]` and this is false.
|
|
52
|
+
*/
|
|
53
|
+
complete: boolean;
|
|
54
|
+
}
|
|
55
|
+
export interface EvidenceRobotsDecision {
|
|
56
|
+
/** `no_robots`: robots.txt was requested and the site has none (a 4xx, or a body that is not text/plain). */
|
|
57
|
+
decision: 'allowed' | 'disallowed' | 'no_robots';
|
|
58
|
+
robotsUrl: string | null;
|
|
59
|
+
/** SHA-256 of the robots.txt bytes parsed; null when none was parsed. */
|
|
60
|
+
robotsSha256: string | null;
|
|
61
|
+
/** Why robots.txt could not be fetched (then `decision` is `disallowed`, RFC 9309 §2.3.1.4); null when it was. */
|
|
62
|
+
unreachable: RobotsUnreachable | null;
|
|
63
|
+
crawlDelayMs: number | null;
|
|
64
|
+
/** Whether the fetch went ahead under a recorded robots override although `decision` is `disallowed`; the reason is in the trace, the warnings and, in the browser lane, the compliance record. */
|
|
65
|
+
userOverride: boolean;
|
|
66
|
+
}
|
|
67
|
+
export interface EvidenceOutputSha256 {
|
|
68
|
+
/** SHA-256 of the UTF-8 bytes of the delivered `markdown`; null when none was delivered. */
|
|
69
|
+
markdown: string | null;
|
|
70
|
+
/** SHA-256 of the canonical JSON of the delivered `json.data`; null when JSON was not requested or has no data. */
|
|
71
|
+
json: string | null;
|
|
72
|
+
}
|
|
73
|
+
export interface EvidenceExtractor {
|
|
74
|
+
name: string;
|
|
75
|
+
/** EXTRACTOR_VERSION of @w2l/extract-tf for a web page; PDF_TEXT_VERSION for a PDF; FILE_TEXT_VERSION for another file. */
|
|
76
|
+
version: string;
|
|
77
|
+
/** The source commit of the running W2L (`W2L_SOURCE_COMMIT`); null when not declared. */
|
|
78
|
+
commit: string | null;
|
|
79
|
+
}
|
|
80
|
+
export interface EvidenceFieldLocation {
|
|
81
|
+
source: FieldEvidenceSource;
|
|
82
|
+
/** JSON-LD path, DOM selector, `table[i] tr[j] "label"`, `h1[0]`, a result field such as `finalUrl`, or `page N "label"` for a PDF; null when the source gives none. */
|
|
83
|
+
locator: string | null;
|
|
84
|
+
}
|
|
85
|
+
export interface EvidenceArtifact {
|
|
86
|
+
/** Null when W2L cannot tell what the file is. */
|
|
87
|
+
kind: EvidenceArtifactKind | null;
|
|
88
|
+
path: string;
|
|
89
|
+
sha256: string | null;
|
|
90
|
+
/** Size of the saved file in bytes; null when unknown. Added to v1 for files (EVIDENCE_RECORD_ADDED_KEYS). */
|
|
91
|
+
bytes: number | null;
|
|
92
|
+
/** The Content-Type the file was received with; null when unknown or none was sent. Added to v1 for files. */
|
|
93
|
+
contentType: string | null;
|
|
94
|
+
}
|
|
95
|
+
export interface EvidenceIdentity {
|
|
96
|
+
/** The User-Agent observed on the wire; null when the lane did not observe it. */
|
|
97
|
+
userAgent: string | null;
|
|
98
|
+
mode: CrawlMode;
|
|
99
|
+
/** The contact the User-Agent declares (research mode with `W2L_CONTACT`); null when none. */
|
|
100
|
+
contact: string | null;
|
|
101
|
+
/**
|
|
102
|
+
* The device the answering lane's identity declared (`mobile` for a
|
|
103
|
+
* request with `mobile: true`); null when W2L sent no request for the page,
|
|
104
|
+
* in mode `research` (a bot declares no device) or when the lane recorded
|
|
105
|
+
* none (the provider lane). Added to v1 (EVIDENCE_RECORD_ADDED_KEYS).
|
|
106
|
+
*/
|
|
107
|
+
device: 'desktop' | 'mobile' | null;
|
|
108
|
+
/**
|
|
109
|
+
* The caller's custom headers (`headers`) the answering lane sent, names
|
|
110
|
+
* lower-cased and sorted, each with the SHA-256 of its value, never the
|
|
111
|
+
* value: a value may be a key the caller would not publish with the data,
|
|
112
|
+
* as the scrape record keeps names only. Empty when it sent none; null when
|
|
113
|
+
* W2L sent no request for the page. Added to v1 (EVIDENCE_RECORD_ADDED_KEYS).
|
|
114
|
+
*/
|
|
115
|
+
requestHeaders: readonly EvidenceRequestHeader[] | null;
|
|
116
|
+
}
|
|
117
|
+
/** One custom request header as sent: its name and the SHA-256 (hex) of its value's UTF-8 bytes. */
|
|
118
|
+
export interface EvidenceRequestHeader {
|
|
119
|
+
name: string;
|
|
120
|
+
valueSha256: string;
|
|
121
|
+
}
|
|
122
|
+
/** The steps a request ran on the page before it was read. */
|
|
123
|
+
export interface EvidencePageActions {
|
|
124
|
+
steps: readonly EvidencePageActionStep[];
|
|
125
|
+
/** Whether an `executeJavascript` step ran in the page. */
|
|
126
|
+
scriptRan: boolean;
|
|
127
|
+
}
|
|
128
|
+
export interface EvidencePageActionStep {
|
|
129
|
+
type: PageActionType;
|
|
130
|
+
outcome: 'ok' | 'failed';
|
|
131
|
+
}
|
|
132
|
+
export interface EvidenceRecord {
|
|
133
|
+
schemaVersion: typeof EVIDENCE_SCHEMA_VERSION;
|
|
134
|
+
requestedUrl: string;
|
|
135
|
+
/** The last URL W2L requested for the page; null when it sent none (robots.txt disallow, DNS failure, policy). */
|
|
136
|
+
finalUrl: string | null;
|
|
137
|
+
redirectChain: EvidenceRedirectChain;
|
|
138
|
+
/** UTC ISO 8601 time the response W2L reports was received; null when there was none. */
|
|
139
|
+
fetchedAt: string | null;
|
|
140
|
+
httpStatus: number | null;
|
|
141
|
+
status: ResultStatus;
|
|
142
|
+
/** The failure, block or budget reason; null for other statuses. */
|
|
143
|
+
reason: FailureReason | BlockReason | BudgetKind | null;
|
|
144
|
+
lane: Lane;
|
|
145
|
+
/** Null when no robots.txt decision was made for this result. */
|
|
146
|
+
robotsDecision: EvidenceRobotsDecision | null;
|
|
147
|
+
/** `evidence.rawBodySha256`: the body W2L read (see the schema for what each lane reads). */
|
|
148
|
+
rawSha256: string | null;
|
|
149
|
+
/**
|
|
150
|
+
* The Content-Encoding the body was read from, as the HTTP lane received it
|
|
151
|
+
* (lower-cased codings in applied order, `x-gzip` read as `gzip`), or
|
|
152
|
+
* `identity` when it had none: what was decoded before `rawSha256` was
|
|
153
|
+
* taken. Null when no body was read, and in the browser and provider lanes.
|
|
154
|
+
* Added to v1 later (EVIDENCE_RECORD_ADDED_KEYS).
|
|
155
|
+
*/
|
|
156
|
+
contentEncoding: string | null;
|
|
157
|
+
outputSha256: EvidenceOutputSha256;
|
|
158
|
+
extractor: EvidenceExtractor;
|
|
159
|
+
/** JSON Pointer into `json.data` → where that value was read; null when JSON was not requested. */
|
|
160
|
+
fieldEvidence: Readonly<Record<string, EvidenceFieldLocation>> | null;
|
|
161
|
+
artifacts: readonly EvidenceArtifact[];
|
|
162
|
+
/** `host:port` of the operator's environment proxy the request went through; null when it went direct or the lane does not report its route. */
|
|
163
|
+
proxy: string | null;
|
|
164
|
+
identity: EvidenceIdentity;
|
|
165
|
+
/**
|
|
166
|
+
* The request's `actions` that ran on the page before it was read, in order,
|
|
167
|
+
* with their outcome; null when the request had none. The hashes above are
|
|
168
|
+
* of the page as the steps left it, and `scriptRan` says when a script of
|
|
169
|
+
* the caller's ran in it, so its content may be the script's.
|
|
170
|
+
*/
|
|
171
|
+
pageActions: EvidencePageActions | null;
|
|
172
|
+
}
|
|
173
|
+
/** Field order of the record and of each nested object, as in the schema file. */
|
|
174
|
+
export declare const EVIDENCE_RECORD_KEYS: {
|
|
175
|
+
readonly record: readonly ["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions"];
|
|
176
|
+
readonly redirectChain: readonly ["urls", "complete"];
|
|
177
|
+
readonly robotsDecision: readonly ["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride"];
|
|
178
|
+
readonly outputSha256: readonly ["markdown", "json"];
|
|
179
|
+
readonly extractor: readonly ["name", "version", "commit"];
|
|
180
|
+
readonly fieldEvidence: readonly ["source", "locator"];
|
|
181
|
+
readonly artifact: readonly ["kind", "path", "sha256", "bytes", "contentType"];
|
|
182
|
+
readonly identity: readonly ["userAgent", "mode", "contact", "device", "requestHeaders"];
|
|
183
|
+
readonly pageActions: readonly ["steps", "scriptRan"];
|
|
184
|
+
readonly pageActionStep: readonly ["type", "outcome"];
|
|
185
|
+
readonly requestHeader: readonly ["name", "valueSha256"];
|
|
186
|
+
};
|
|
187
|
+
/**
|
|
188
|
+
* Keys added to v1 after it was first published: optional in the schema, so
|
|
189
|
+
* that records written before them stay valid, though W2L always writes them.
|
|
190
|
+
*/
|
|
191
|
+
export declare const EVIDENCE_RECORD_ADDED_KEYS: Partial<Record<keyof typeof EVIDENCE_RECORD_KEYS, readonly string[]>>;
|
|
192
|
+
//# sourceMappingURL=evidenceRecord.d.ts.map
|