@octocrawl/sdk 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +17 -0
- package/dist/index.cjs +1278 -0
- package/dist/index.js +1240 -0
- package/dist/types/cjs/client.d.ts +362 -0
- package/dist/types/cjs/contracts/access.d.ts +166 -0
- package/dist/types/cjs/contracts/actions.d.ts +191 -0
- package/dist/types/cjs/contracts/api.d.ts +891 -0
- package/dist/types/cjs/contracts/benchmark.d.ts +116 -0
- package/dist/types/cjs/contracts/checkpoint.d.ts +164 -0
- package/dist/types/cjs/contracts/compliance.d.ts +412 -0
- package/dist/types/cjs/contracts/crawl.d.ts +302 -0
- package/dist/types/cjs/contracts/delivery.d.ts +136 -0
- package/dist/types/cjs/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/cjs/contracts/execution.d.ts +197 -0
- package/dist/types/cjs/contracts/extractor.d.ts +379 -0
- package/dist/types/cjs/contracts/file.d.ts +117 -0
- package/dist/types/cjs/contracts/firecrawl.d.ts +258 -0
- package/dist/types/cjs/contracts/groundTruth.d.ts +77 -0
- package/dist/types/cjs/contracts/identityBundle.d.ts +70 -0
- package/dist/types/cjs/contracts/index.d.ts +30 -0
- package/dist/types/cjs/contracts/map.d.ts +180 -0
- package/dist/types/cjs/contracts/monitor.d.ts +217 -0
- package/dist/types/cjs/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/cjs/contracts/policy.d.ts +93 -0
- package/dist/types/cjs/contracts/proxy.d.ts +52 -0
- package/dist/types/cjs/contracts/recipe.d.ts +74 -0
- package/dist/types/cjs/contracts/regexSafety.d.ts +31 -0
- package/dist/types/cjs/contracts/result.d.ts +503 -0
- package/dist/types/cjs/contracts/session.d.ts +51 -0
- package/dist/types/cjs/contracts/ssrf.d.ts +16 -0
- package/dist/types/cjs/contracts/status.d.ts +27 -0
- package/dist/types/cjs/contracts/structured.d.ts +185 -0
- package/dist/types/cjs/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/cjs/contracts/tokens.d.ts +47 -0
- package/dist/types/cjs/index.d.ts +9 -0
- package/dist/types/cjs/package.json +1 -0
- package/dist/types/cjs/version.d.ts +8 -0
- package/dist/types/cjs/watcher.d.ts +151 -0
- package/dist/types/esm/client.d.ts +362 -0
- package/dist/types/esm/contracts/access.d.ts +166 -0
- package/dist/types/esm/contracts/actions.d.ts +191 -0
- package/dist/types/esm/contracts/api.d.ts +891 -0
- package/dist/types/esm/contracts/benchmark.d.ts +116 -0
- package/dist/types/esm/contracts/checkpoint.d.ts +164 -0
- package/dist/types/esm/contracts/compliance.d.ts +412 -0
- package/dist/types/esm/contracts/crawl.d.ts +302 -0
- package/dist/types/esm/contracts/delivery.d.ts +136 -0
- package/dist/types/esm/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/esm/contracts/execution.d.ts +197 -0
- package/dist/types/esm/contracts/extractor.d.ts +379 -0
- package/dist/types/esm/contracts/file.d.ts +117 -0
- package/dist/types/esm/contracts/firecrawl.d.ts +258 -0
- package/dist/types/esm/contracts/groundTruth.d.ts +77 -0
- package/dist/types/esm/contracts/identityBundle.d.ts +70 -0
- package/dist/types/esm/contracts/index.d.ts +30 -0
- package/dist/types/esm/contracts/map.d.ts +180 -0
- package/dist/types/esm/contracts/monitor.d.ts +217 -0
- package/dist/types/esm/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/esm/contracts/policy.d.ts +93 -0
- package/dist/types/esm/contracts/proxy.d.ts +52 -0
- package/dist/types/esm/contracts/recipe.d.ts +74 -0
- package/dist/types/esm/contracts/regexSafety.d.ts +31 -0
- package/dist/types/esm/contracts/result.d.ts +503 -0
- package/dist/types/esm/contracts/session.d.ts +51 -0
- package/dist/types/esm/contracts/ssrf.d.ts +16 -0
- package/dist/types/esm/contracts/status.d.ts +27 -0
- package/dist/types/esm/contracts/structured.d.ts +185 -0
- package/dist/types/esm/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/esm/contracts/tokens.d.ts +47 -0
- package/dist/types/esm/index.d.ts +9 -0
- package/dist/types/esm/version.d.ts +8 -0
- package/dist/types/esm/watcher.d.ts +151 -0
- package/package.json +40 -0
|
@@ -0,0 +1,503 @@
|
|
|
1
|
+
import type { BlockReason, BudgetKind, FailureReason, Lane, ResultStatus } from './status.js';
|
|
2
|
+
import type { ComplianceRecord } from './compliance.js';
|
|
3
|
+
import type { DocumentExtraction, PageMetadata } from './extractor.js';
|
|
4
|
+
import type { ListField, StructuredExtractionResult } from './structured.js';
|
|
5
|
+
import type { FileDescription } from './file.js';
|
|
6
|
+
import type { ActionsResult } from './actions.js';
|
|
7
|
+
export interface ResourceTimings {
|
|
8
|
+
/** Every wait in the origin scheduler but a cooldown: the concurrency ceiling and the minimum interval between requests. */
|
|
9
|
+
queueMs?: number;
|
|
10
|
+
robotsMs?: number;
|
|
11
|
+
cooldownWaitMs?: number;
|
|
12
|
+
/**
|
|
13
|
+
* Present when the per-origin concurrency ceiling held this lane's permit
|
|
14
|
+
* back: the milliseconds it waited for a slot, cooldown and pacing
|
|
15
|
+
* excluded (both stay in `cooldownWaitMs` / `queueMs`). Absent when the
|
|
16
|
+
* permit started at once, or when the lane acquired none.
|
|
17
|
+
*/
|
|
18
|
+
concurrencyWaitMs?: number;
|
|
19
|
+
retryWaitMs?: number;
|
|
20
|
+
/** Initial request/headers time, excluding body read and retry sleep. */
|
|
21
|
+
requestMs?: number;
|
|
22
|
+
bodyReadMs?: number;
|
|
23
|
+
/** Total network work excluding retry sleep. */
|
|
24
|
+
transportMs?: number;
|
|
25
|
+
parseMs?: number;
|
|
26
|
+
extractMs?: number;
|
|
27
|
+
formatMs?: number;
|
|
28
|
+
serializeMs?: number;
|
|
29
|
+
modelMs?: number;
|
|
30
|
+
totalMs: number;
|
|
31
|
+
}
|
|
32
|
+
/** Why the runtime moved from one lane to the next. Logged for the escalation corpus. */
|
|
33
|
+
export interface Escalation {
|
|
34
|
+
from: Lane;
|
|
35
|
+
to: Lane;
|
|
36
|
+
/** Machine-readable trigger, e.g. 'low_text_yield' | 'spa_marker' | 'blocked' */
|
|
37
|
+
trigger: string;
|
|
38
|
+
/** Whether the escalation actually improved the outcome. Null until known. */
|
|
39
|
+
improved: boolean | null;
|
|
40
|
+
}
|
|
41
|
+
export interface ResourceUsage {
|
|
42
|
+
wallMs: number;
|
|
43
|
+
/** Bytes received on the wire (compressed). */
|
|
44
|
+
bytesWire: number | null;
|
|
45
|
+
/** Bytes after decompression. Guarded by a decompressed-size cap. */
|
|
46
|
+
bytesDecompressed: number;
|
|
47
|
+
requestCount: number;
|
|
48
|
+
attemptCount: number;
|
|
49
|
+
/** Actual status-driven retries; not variant-selection navigations. */
|
|
50
|
+
statusRetryCount?: number;
|
|
51
|
+
navigationFollowupCount?: number;
|
|
52
|
+
/** Token count of the emitted main content. Null if not tokenized. */
|
|
53
|
+
contentTokens: number | null;
|
|
54
|
+
browserMs: number;
|
|
55
|
+
/**
|
|
56
|
+
* Cost incurred outside this process, paid by the user to a third party
|
|
57
|
+
* (BYO proxy egress, provider browser minutes, model calls).
|
|
58
|
+
* `null` means no external cost path was used — never means "free".
|
|
59
|
+
*/
|
|
60
|
+
externalCostUsd: number | null;
|
|
61
|
+
/** Stage timings use a monotonic clock. Optional for legacy producers. */
|
|
62
|
+
timings?: ResourceTimings;
|
|
63
|
+
/**
|
|
64
|
+
* True when the caller's deadline (a scrape's `timeout`) ended this fetch
|
|
65
|
+
* before it finished: the result is then `partial` with the content
|
|
66
|
+
* fetched so far, or `failed` with `timeout`. Absent otherwise.
|
|
67
|
+
*/
|
|
68
|
+
deadlineExceeded?: boolean;
|
|
69
|
+
}
|
|
70
|
+
export interface Meter {
|
|
71
|
+
knownSubtotal: number;
|
|
72
|
+
unknown: boolean;
|
|
73
|
+
}
|
|
74
|
+
export interface Evidence {
|
|
75
|
+
/**
|
|
76
|
+
* Final URL after redirects: the last URL requested for the page, whose
|
|
77
|
+
* response `httpStatus` and `contentType` are from. In the browser lane a
|
|
78
|
+
* script or a meta refresh that loaded another document is a redirect; a
|
|
79
|
+
* URL the page set with the history API (pushState, replaceState), which
|
|
80
|
+
* nothing requested, is not: the page keeps it as the base of its links.
|
|
81
|
+
*/
|
|
82
|
+
finalUrl: string;
|
|
83
|
+
/**
|
|
84
|
+
* The status of the response that answered `finalUrl`: in the browser lane,
|
|
85
|
+
* of the document the page shows when it is read, not of the navigation W2L
|
|
86
|
+
* started. Null when there was none (no request, a transport failure, a
|
|
87
|
+
* document that came without a response).
|
|
88
|
+
*/
|
|
89
|
+
httpStatus: number | null;
|
|
90
|
+
/**
|
|
91
|
+
* The URLs of a redirect, the requested URL first and `finalUrl` last;
|
|
92
|
+
* empty when nothing redirected.
|
|
93
|
+
*/
|
|
94
|
+
redirectChain: readonly string[];
|
|
95
|
+
/**
|
|
96
|
+
* True when `redirectChain` lists every hop the lane requested: the HTTP
|
|
97
|
+
* lane follows each redirect itself, and the browser lane lists each
|
|
98
|
+
* redirect Chromium followed and each document a script or a meta refresh
|
|
99
|
+
* loaded, for a page it shows or a file it displays or downloads. The
|
|
100
|
+
* browser lane says false when a follow-up navigation did not start at the
|
|
101
|
+
* requested URL, a document came without a request, or it cut the chain of
|
|
102
|
+
* a page that kept moving on to its first URL and last 20. Absent when the
|
|
103
|
+
* lane does not say (the provider lane, which sees where its vendor started
|
|
104
|
+
* and ended, and results stored before lanes recorded it).
|
|
105
|
+
*/
|
|
106
|
+
redirectChainComplete?: boolean;
|
|
107
|
+
/**
|
|
108
|
+
* The `content-type` header of the response `httpStatus` is from, as the
|
|
109
|
+
* server sent it, in every lane (the browser lane reads the rendered page,
|
|
110
|
+
* whatever it says); null when there was no response or no such header.
|
|
111
|
+
*/
|
|
112
|
+
contentType: string | null;
|
|
113
|
+
/** sha256 of the raw response body. Null only when no body was read. */
|
|
114
|
+
rawBodySha256: string | null;
|
|
115
|
+
/**
|
|
116
|
+
* The `content-encoding` of the response `httpStatus` is from, as the HTTP
|
|
117
|
+
* lane received it (lower-cased codings in applied order, `x-gzip` read as
|
|
118
|
+
* `gzip`), or `identity` when it had none. The lane decodes gzip, deflate
|
|
119
|
+
* and br, so the body behind `rawBodySha256` is the decoded one; any other
|
|
120
|
+
* coding fails with `unsupported_content_encoding`. Absent when no response
|
|
121
|
+
* body was read, and in lanes that do not report it (browser, provider).
|
|
122
|
+
*/
|
|
123
|
+
contentEncoding?: string;
|
|
124
|
+
/** Relative artifact paths (raw body, screenshot, DOM snapshot). */
|
|
125
|
+
artifacts: readonly string[];
|
|
126
|
+
/**
|
|
127
|
+
* UTC ISO time the lane received what it reports: the final response's
|
|
128
|
+
* headers (HTTP), the vendor's answer (provider), the rendered page's
|
|
129
|
+
* capture (browser). Absent when no response was read, and on results
|
|
130
|
+
* stored before lanes recorded it.
|
|
131
|
+
*/
|
|
132
|
+
fetchedAt?: string;
|
|
133
|
+
/** HTTP validators observed for the representation, when exposed. */
|
|
134
|
+
etag?: string | null;
|
|
135
|
+
lastModified?: string | null;
|
|
136
|
+
cacheControl?: string | null;
|
|
137
|
+
vary?: string | null;
|
|
138
|
+
/** A response setting cookies cannot enter the public monitor cache. */
|
|
139
|
+
setsCookie?: boolean;
|
|
140
|
+
/**
|
|
141
|
+
* `host:port` of the operator's environment proxy (local mode) that the
|
|
142
|
+
* request for `finalUrl` went through; never its credentials. Null when that
|
|
143
|
+
* request did not use it (NO_PROXY, loopback). Absent when no environment
|
|
144
|
+
* proxy was configured, no request was answered, or the lane does not
|
|
145
|
+
* report its route (the provider lane).
|
|
146
|
+
*/
|
|
147
|
+
envProxy?: string | null;
|
|
148
|
+
}
|
|
149
|
+
export interface TraceEvent {
|
|
150
|
+
at: number;
|
|
151
|
+
lane: Lane;
|
|
152
|
+
event: string;
|
|
153
|
+
detail?: Record<string, unknown>;
|
|
154
|
+
}
|
|
155
|
+
export interface LadderAttempt {
|
|
156
|
+
channel: string;
|
|
157
|
+
result: FetchResult;
|
|
158
|
+
}
|
|
159
|
+
export interface LadderExecutionSummary {
|
|
160
|
+
channelsTried: readonly string[];
|
|
161
|
+
attempts: readonly LadderAttempt[];
|
|
162
|
+
wallMs: number;
|
|
163
|
+
browserMs: number;
|
|
164
|
+
bytesWire: number | null;
|
|
165
|
+
bytesDecompressed: number;
|
|
166
|
+
requestCount: number;
|
|
167
|
+
attemptCount: number;
|
|
168
|
+
contentTokens: number | null;
|
|
169
|
+
externalCostUsd: number | null;
|
|
170
|
+
externalCost: Meter;
|
|
171
|
+
contentTokenMeter: Meter;
|
|
172
|
+
artifacts: readonly string[];
|
|
173
|
+
/** Actual caller wait across all ladder work, including routing overhead. */
|
|
174
|
+
totalMs?: number;
|
|
175
|
+
}
|
|
176
|
+
export interface LadderRunAudit {
|
|
177
|
+
channelsTried: readonly string[];
|
|
178
|
+
ladderTrace: readonly {
|
|
179
|
+
at: number;
|
|
180
|
+
event: string;
|
|
181
|
+
channel: string;
|
|
182
|
+
detail: Record<string, unknown>;
|
|
183
|
+
}[];
|
|
184
|
+
summary: LadderExecutionSummary;
|
|
185
|
+
}
|
|
186
|
+
/**
|
|
187
|
+
* A request for human takeover: the lane hit a captcha or login wall it will
|
|
188
|
+
* not defeat, a live-view door exists, and the task pauses here until a
|
|
189
|
+
* human returns (or the run aborts). Carrying this on the result instead of
|
|
190
|
+
* throwing keeps the decision trail in the signed record: the run did not
|
|
191
|
+
* silently skip the page, it stopped and asked.
|
|
192
|
+
*/
|
|
193
|
+
export interface HandoffRequest {
|
|
194
|
+
/** The seven-class routing reason, e.g. 'captcha_required'. */
|
|
195
|
+
reason: string;
|
|
196
|
+
/** Live view URL for the human, when one was opened. */
|
|
197
|
+
liveViewUrl: string | null;
|
|
198
|
+
/** Why this specific result asks for a human, one sentence. */
|
|
199
|
+
rationale: string;
|
|
200
|
+
}
|
|
201
|
+
/** The values of one HTML attribute on the elements one CSS selector names (the `attributes` format). */
|
|
202
|
+
export interface AttributeExtraction {
|
|
203
|
+
selector: string;
|
|
204
|
+
attribute: string;
|
|
205
|
+
/** As written in the HTML, in document order; an element without the attribute is skipped; `[]` when nothing matches. */
|
|
206
|
+
values: readonly string[];
|
|
207
|
+
}
|
|
208
|
+
/**
|
|
209
|
+
* The `screenshot` format: the rendered page as the browser lane captured it
|
|
210
|
+
* after load, stability and `waitFor`, before the DOM was read, so the image
|
|
211
|
+
* and the Markdown show the same page. The bytes are inline (`base64`) and
|
|
212
|
+
* hash to `sha256`; `path` names the file under W2L_CAPTURE_RAW_DIR when
|
|
213
|
+
* that is set (`<sha256>.png` or `.jpg`, listed in `evidence.artifacts`
|
|
214
|
+
* too), else null. `width` and `height` are CSS pixels (`scale: 'css'`):
|
|
215
|
+
* the viewport's for a viewport capture, the viewport's width and the
|
|
216
|
+
* document's height for `fullPage`; `deviceScaleFactor` is what the context
|
|
217
|
+
* declared, not baked into the image. The page itself is unchanged: nothing
|
|
218
|
+
* is scrolled, clicked or hidden for the capture.
|
|
219
|
+
*/
|
|
220
|
+
export interface ScreenshotEvidence {
|
|
221
|
+
contentType: 'image/png' | 'image/jpeg';
|
|
222
|
+
width: number;
|
|
223
|
+
height: number;
|
|
224
|
+
fullPage: boolean;
|
|
225
|
+
/** The window the page was laid out in, in CSS pixels: the request's viewport, or the declared one. */
|
|
226
|
+
viewport: {
|
|
227
|
+
width: number;
|
|
228
|
+
height: number;
|
|
229
|
+
};
|
|
230
|
+
/** Device pixels per CSS pixel the context declared (2 for the desktop identity, 2.625 for the mobile one). */
|
|
231
|
+
deviceScaleFactor: number;
|
|
232
|
+
/** The JPEG quality asked for; null for a PNG. */
|
|
233
|
+
quality: number | null;
|
|
234
|
+
bytes: number;
|
|
235
|
+
sha256: string;
|
|
236
|
+
path: string | null;
|
|
237
|
+
base64: string;
|
|
238
|
+
}
|
|
239
|
+
/**
|
|
240
|
+
* A caveat a reader of the result must see without opening the trace. Never
|
|
241
|
+
* a failure reason, which `status` and its reason fields carry.
|
|
242
|
+
*/
|
|
243
|
+
export interface FetchWarning {
|
|
244
|
+
/**
|
|
245
|
+
* Machine-readable code. `robots_overridden`: a robots.txt rule was set
|
|
246
|
+
* aside by a recorded override. `tls_unverified`: the certificate was not
|
|
247
|
+
* verified at the caller's request (`skipTlsVerification`), so the content
|
|
248
|
+
* cannot be attributed to the host with certainty. `client_rendered_suspected`:
|
|
249
|
+
* the HTTP lane's page looks like a shell for data its scripts fill in
|
|
250
|
+
* (see RenderSignals), so the capture may not be the page a browser shows.
|
|
251
|
+
* `low_content_yield`: a thin answer stayed the run's answer: the http
|
|
252
|
+
* lane's, which the browser lane did not improve on or was not offered,
|
|
253
|
+
* or a rendered one (the browser or a provider lane's) its own extraction
|
|
254
|
+
* found thin and low-confidence.
|
|
255
|
+
* `screenshot_unavailable`: the `screenshot` format was asked for and the
|
|
256
|
+
* browser lane rendered the page but could not capture it
|
|
257
|
+
* (`screenshot_failed` in the trace); `screenshot` is null and the page
|
|
258
|
+
* result stands.
|
|
259
|
+
*/
|
|
260
|
+
code: string;
|
|
261
|
+
message: string;
|
|
262
|
+
}
|
|
263
|
+
/**
|
|
264
|
+
* A page-level fetch outcome. `status` is the single source of truth
|
|
265
|
+
* (see RESULT_STATUS); the reason fields narrow it.
|
|
266
|
+
*/
|
|
267
|
+
export interface FetchResult {
|
|
268
|
+
/** Earliest permitted next request after a deferred Retry-After, UTC milliseconds. */
|
|
269
|
+
retryAt?: number;
|
|
270
|
+
requestedUrl: string;
|
|
271
|
+
status: ResultStatus;
|
|
272
|
+
/** Set iff status === 'failed'. Duplicate content uses status `duplicate`. */
|
|
273
|
+
failureReason: FailureReason | null;
|
|
274
|
+
/** Set iff status === 'blocked'. */
|
|
275
|
+
blockReason: BlockReason | null;
|
|
276
|
+
/** Set iff status === 'budget_exceeded'. */
|
|
277
|
+
budgetExceeded: BudgetKind | null;
|
|
278
|
+
/** The lane that produced this result. */
|
|
279
|
+
lane: Lane;
|
|
280
|
+
escalations: readonly Escalation[];
|
|
281
|
+
/** Set when the result asks for human takeover. Never on a success.
|
|
282
|
+
* Optional for backward compatibility with existing result producers;
|
|
283
|
+
* the router and provider lanes always populate it. */
|
|
284
|
+
handoff?: HandoffRequest | null;
|
|
285
|
+
/**
|
|
286
|
+
* Vendor session resume material (context/profile/storage) that the
|
|
287
|
+
* provider lane produced, so the ladder can persist it for the next run.
|
|
288
|
+
* Shape is vendor-specific; it is a credential-free continuation token.
|
|
289
|
+
*/
|
|
290
|
+
resumeContext?: unknown | null;
|
|
291
|
+
/**
|
|
292
|
+
* The page as Markdown: its main content, the whole page when
|
|
293
|
+
* `onlyMainContent` is false, or the elements `includeTags` names, in each
|
|
294
|
+
* case without `excludeTags`. `data:` link and image targets are dropped,
|
|
295
|
+
* the link text and alt text kept. Null unless status is contentful,
|
|
296
|
+
* except on a failed or blocked result that kept a page as evidence, never
|
|
297
|
+
* content: the page an error status carried, or the whole page when the
|
|
298
|
+
* extractor found no main content (`empty_unverified`, and `timeout` when
|
|
299
|
+
* the deadline then ended a later rung).
|
|
300
|
+
*/
|
|
301
|
+
markdown: string | null;
|
|
302
|
+
/** HTML-derived page/product facts; never reconstructed from Markdown. */
|
|
303
|
+
document?: DocumentExtraction | null;
|
|
304
|
+
/**
|
|
305
|
+
* What the page's HTML declares about itself (its `<title>`, description,
|
|
306
|
+
* language, keywords, robots, icon and canonical URL), present with
|
|
307
|
+
* `document`. `metadata.title` is the page's `<title>`; `document.title` is
|
|
308
|
+
* the content's title, usually its first heading.
|
|
309
|
+
*/
|
|
310
|
+
metadata?: PageMetadata;
|
|
311
|
+
/**
|
|
312
|
+
* The cleaned HTML the Markdown was written from, present only when the
|
|
313
|
+
* `html` format was asked for: the main content; with
|
|
314
|
+
* `onlyMainContent: false` the whole page without what Markdown never
|
|
315
|
+
* shows (scripts, styles, form controls, embedded media) and without the
|
|
316
|
+
* caller's `excludeTags`; with `includeTags` a `<body>` holding the named
|
|
317
|
+
* elements. A lane sets it on a contentful page only; the API returns
|
|
318
|
+
* null for a file and for a page that was not read as content.
|
|
319
|
+
*/
|
|
320
|
+
html?: string | null;
|
|
321
|
+
/**
|
|
322
|
+
* The page as the lane received it, present only when the `rawHtml` format
|
|
323
|
+
* was asked for: the response body on the HTTP lane, the rendered DOM on a
|
|
324
|
+
* browser lane, scripts and all. Its UTF-8 bytes hash to
|
|
325
|
+
* `evidence.rawBodySha256`. Set and returned as `html` is.
|
|
326
|
+
*/
|
|
327
|
+
rawHtml?: string | null;
|
|
328
|
+
/** Present only when a JSON format was requested. */
|
|
329
|
+
json?: StructuredExtractionResult | null;
|
|
330
|
+
/**
|
|
331
|
+
* Present when the response was a file (PDF, CSV, JSON, text, XLSX, XLS,
|
|
332
|
+
* ZIP) rather than a web page: what it was, its size, SHA-256 and where it
|
|
333
|
+
* was saved, and for a PDF its pages. Such a result has no `document` or
|
|
334
|
+
* `metadata`.
|
|
335
|
+
*/
|
|
336
|
+
file?: FileDescription;
|
|
337
|
+
/**
|
|
338
|
+
* Outbound http(s) links from the FULL document, collected after extract
|
|
339
|
+
* and before the raw HTML is dropped. Not from `mainHtml` — prune strips
|
|
340
|
+
* nav. Empty / omitted when the fetch never produced HTML. Never the page
|
|
341
|
+
* HTML itself.
|
|
342
|
+
*/
|
|
343
|
+
links?: readonly string[];
|
|
344
|
+
/**
|
|
345
|
+
* The image URLs of the FULL document (`img` src and srcset candidates,
|
|
346
|
+
* `picture` sources, lazy-loading attributes, video posters, `image_src`
|
|
347
|
+
* links, og:image and twitter:image), absolute http(s), fragment stripped,
|
|
348
|
+
* each once, in document order; `data:` URIs left out. Present only when
|
|
349
|
+
* the `images` format was asked for (`FetchOptions.includeImages`), on a
|
|
350
|
+
* contentful page: absent for a file and for a page that was not read as
|
|
351
|
+
* content. `[]` for a page without images.
|
|
352
|
+
*/
|
|
353
|
+
images?: readonly string[];
|
|
354
|
+
/**
|
|
355
|
+
* The `tables` format: every data table of the content the Markdown was
|
|
356
|
+
* written from (the main content, the whole page with `onlyMainContent:
|
|
357
|
+
* false`, or the `includeTags` selection), one entry per GFM table of the
|
|
358
|
+
* Markdown, in its order. Present only when asked for
|
|
359
|
+
* (`FetchOptions.includeTables`), on a contentful page: absent for a file
|
|
360
|
+
* and for a page that was not read as content. `[]` for a page without one.
|
|
361
|
+
*/
|
|
362
|
+
tables?: readonly PageTable[];
|
|
363
|
+
/**
|
|
364
|
+
* The `list` format, when asked for and the page was read: its records,
|
|
365
|
+
* one per element `itemSelector` matched, from every page a paginate step
|
|
366
|
+
* read when one ran, else from the page as it stands.
|
|
367
|
+
*/
|
|
368
|
+
list?: ListExtraction;
|
|
369
|
+
/**
|
|
370
|
+
* A PDF's pages, each as the Markdown has it (without its marker), when a
|
|
371
|
+
* `pdf` parser entry asked for them (`pages: true`) and the text layer was
|
|
372
|
+
* read; absent otherwise.
|
|
373
|
+
*/
|
|
374
|
+
pages?: readonly PdfPageMarkdown[];
|
|
375
|
+
/**
|
|
376
|
+
* The `attributes` format: one entry per selector of the request, in its
|
|
377
|
+
* order, with the attribute's values as written in the HTML. Present only
|
|
378
|
+
* when asked for (`FetchOptions.attributes`), on a contentful page, like
|
|
379
|
+
* `images`.
|
|
380
|
+
*/
|
|
381
|
+
attributes?: readonly AttributeExtraction[];
|
|
382
|
+
/**
|
|
383
|
+
* The `screenshot` format, present only when asked for
|
|
384
|
+
* (`FetchOptions.screenshot`), on every result the browser lane built from
|
|
385
|
+
* the rendered document: a success or partial page, and an error-status or
|
|
386
|
+
* blocked page kept as evidence. Null when the page rendered but the
|
|
387
|
+
* capture failed (`screenshot_failed` in the trace, a
|
|
388
|
+
* `screenshot_unavailable` warning). Absent when no page rendered (a file,
|
|
389
|
+
* a robots.txt denial, a navigation failure), as the API then answers
|
|
390
|
+
* null. The attempt copies in a run's audit carry null, so the image
|
|
391
|
+
* travels once.
|
|
392
|
+
*/
|
|
393
|
+
screenshot?: ScreenshotEvidence | null;
|
|
394
|
+
/**
|
|
395
|
+
* What the request's `actions` produced (screenshots, HTML snapshots,
|
|
396
|
+
* script returns, PDFs) and the step that failed, if one did. Present on
|
|
397
|
+
* every result of the browser lane that ran the steps; absent otherwise.
|
|
398
|
+
*/
|
|
399
|
+
actions?: ActionsResult;
|
|
400
|
+
/**
|
|
401
|
+
* The fetch's caveats, present only when it has any: a `robots_overridden`
|
|
402
|
+
* warning first when a recorded override set a robots.txt rule aside, then
|
|
403
|
+
* `tls_unverified` when the fetch skipped certificate verification, then
|
|
404
|
+
* `client_rendered_suspected` when the HTTP lane read the page as a shell
|
|
405
|
+
* its scripts fill in, or `screenshot_unavailable` when the browser lane
|
|
406
|
+
* could not capture the screenshot asked for. Kept on batch items and the
|
|
407
|
+
* compact scrape response too.
|
|
408
|
+
*/
|
|
409
|
+
warnings?: readonly FetchWarning[];
|
|
410
|
+
/** True when content was cut to fit a token budget. */
|
|
411
|
+
truncated: boolean;
|
|
412
|
+
/** Character offset where truncation occurred; null when not truncated. */
|
|
413
|
+
truncatedAt: number | null;
|
|
414
|
+
/**
|
|
415
|
+
* Tamper-evident record of what the fetch actually did (robots.txt decision,
|
|
416
|
+
* exact headers sent, rate-limit facts). Null until a subject wires the
|
|
417
|
+
* record builder in; contentful subjects are expected to produce one per
|
|
418
|
+
* fetch so the premium "provable politeness" tier is not a bolt-on.
|
|
419
|
+
*/
|
|
420
|
+
compliance: ComplianceRecord | null;
|
|
421
|
+
evidence: Evidence;
|
|
422
|
+
usage: ResourceUsage;
|
|
423
|
+
trace: readonly TraceEvent[];
|
|
424
|
+
}
|
|
425
|
+
/**
|
|
426
|
+
* One data table of a page (the `tables` format). `tableIndex` counts the
|
|
427
|
+
* data tables from 0 in document order: table N is the Nth GFM table of the
|
|
428
|
+
* Markdown made from the same request. Cells are plain text: a link is its
|
|
429
|
+
* text, an image its alt text, whitespace collapsed, nothing escaped; a cell
|
|
430
|
+
* that spans rows or columns gives its value to every slot it covers, so
|
|
431
|
+
* every row has `columns` cells and none is shifted.
|
|
432
|
+
*/
|
|
433
|
+
/** One record of the `list` format. */
|
|
434
|
+
export interface ListRecord {
|
|
435
|
+
/** Each field's value, by name: the text (whitespace collapsed) or the attribute; null when the record has none. */
|
|
436
|
+
values: Record<string, string | null>;
|
|
437
|
+
/** The fields that are null, in field order: what this record lacks, never filled in. */
|
|
438
|
+
missing: string[];
|
|
439
|
+
/** Where it was read: the page's URL, the page's number (1 for the page read, or each page a paginate step read, in order), and the record's place on it (0-based, document order). */
|
|
440
|
+
source: {
|
|
441
|
+
url: string;
|
|
442
|
+
page: number;
|
|
443
|
+
index: number;
|
|
444
|
+
};
|
|
445
|
+
}
|
|
446
|
+
/** The `list` format: the records of the page, or of every page a paginate step read. */
|
|
447
|
+
export interface ListExtraction {
|
|
448
|
+
/** The items' selector: as asked, or the one W2L found (`detected`); null when it found no list on the page (a `list_not_detected` warning). */
|
|
449
|
+
itemSelector: string | null;
|
|
450
|
+
fields: string[];
|
|
451
|
+
/**
|
|
452
|
+
* Present when W2L chose the itemSelector or the fields: the fields as a
|
|
453
|
+
* request names them, to send back as they are or changed, and the other
|
|
454
|
+
* lists it found on the page, best first.
|
|
455
|
+
*/
|
|
456
|
+
detected?: ListDetection;
|
|
457
|
+
records: ListRecord[];
|
|
458
|
+
/** Pages the records were read from. */
|
|
459
|
+
pages: number;
|
|
460
|
+
/** Records with at least one field missing. */
|
|
461
|
+
incomplete: number;
|
|
462
|
+
/** True when a limit cut the list (10,000 records, 5,000,000 characters of values): the page had more records than these. */
|
|
463
|
+
truncated: boolean;
|
|
464
|
+
/** The records as RFC 4180 CSV: the fields, then source_url, page and index. */
|
|
465
|
+
csv: string;
|
|
466
|
+
/** SHA-256 (hex) of the UTF-8 bytes of `csv`. */
|
|
467
|
+
csvSha256: string;
|
|
468
|
+
}
|
|
469
|
+
export interface ListDetection {
|
|
470
|
+
fields: ListField[];
|
|
471
|
+
alternatives: Array<{
|
|
472
|
+
itemSelector: string;
|
|
473
|
+
count: number;
|
|
474
|
+
}>;
|
|
475
|
+
}
|
|
476
|
+
export interface PageTable {
|
|
477
|
+
tableIndex: number;
|
|
478
|
+
/** The table's `<caption>` as plain text; null when it has none (a title written above the table is not one). */
|
|
479
|
+
caption: string | null;
|
|
480
|
+
/** The final URL of the page the table was read from (`evidence.finalUrl`). */
|
|
481
|
+
sourceUrl: string;
|
|
482
|
+
/** Leading rows in `<thead>` or made of `<th>` cells alone; 0 when none. */
|
|
483
|
+
headerRows: number;
|
|
484
|
+
columns: number;
|
|
485
|
+
rows: readonly (readonly string[])[];
|
|
486
|
+
/** The rows as RFC 4180 CSV: CRLF line ends, a field quoted when it holds a comma, a quote or a line break, quotes doubled. */
|
|
487
|
+
csv: string;
|
|
488
|
+
/** SHA-256 (hex) of the UTF-8 bytes of `csv`. */
|
|
489
|
+
csvSha256: string;
|
|
490
|
+
/**
|
|
491
|
+
* Present when the table is too large to give: its cells, each spanned
|
|
492
|
+
* value repeated, would exceed 2,000,000 characters, or what is left of
|
|
493
|
+
* 5,000,000 for all of the page's tables. `rows` is then empty,
|
|
494
|
+
* `csv` is `''` and `columns` 0; the table keeps its `tableIndex`.
|
|
495
|
+
*/
|
|
496
|
+
omitted?: 'too_large';
|
|
497
|
+
}
|
|
498
|
+
/** One page of a PDF's text (`pages`): its number in the document, from 1, and its Markdown. */
|
|
499
|
+
export interface PdfPageMarkdown {
|
|
500
|
+
pageNumber: number;
|
|
501
|
+
markdown: string;
|
|
502
|
+
}
|
|
503
|
+
//# sourceMappingURL=result.d.ts.map
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
export declare const SESSION_STATE: readonly ["active", "waiting_user", "revoked", "expired"];
|
|
2
|
+
export type SessionState = (typeof SESSION_STATE)[number];
|
|
3
|
+
export interface ManagedSessionRef {
|
|
4
|
+
sessionRef: string;
|
|
5
|
+
workspaceId: string;
|
|
6
|
+
accountRef: string;
|
|
7
|
+
originScope: string;
|
|
8
|
+
profileId: string;
|
|
9
|
+
profileDir: string;
|
|
10
|
+
grantEpoch: number;
|
|
11
|
+
state: SessionState;
|
|
12
|
+
createdAt: string;
|
|
13
|
+
updatedAt: string;
|
|
14
|
+
revokedAt: string | null;
|
|
15
|
+
expiresAt: string | null;
|
|
16
|
+
handoff: SessionHandoff | null;
|
|
17
|
+
backend?: 'managed' | 'existing_chrome';
|
|
18
|
+
/** Private registry only; never include in public status/logs. */
|
|
19
|
+
cdpEndpoint?: string;
|
|
20
|
+
}
|
|
21
|
+
export interface SessionHandoff {
|
|
22
|
+
handoffId: string;
|
|
23
|
+
reason: string;
|
|
24
|
+
createdAt: string;
|
|
25
|
+
expiresAt: string;
|
|
26
|
+
}
|
|
27
|
+
export interface SessionGrant {
|
|
28
|
+
sessionRef: string;
|
|
29
|
+
workspaceId: string;
|
|
30
|
+
accountRef: string;
|
|
31
|
+
originScope: string;
|
|
32
|
+
grantEpoch: number;
|
|
33
|
+
expiresAt: string | null;
|
|
34
|
+
}
|
|
35
|
+
export type SessionAccessResult = {
|
|
36
|
+
kind: 'granted';
|
|
37
|
+
grant: SessionGrant;
|
|
38
|
+
} | {
|
|
39
|
+
kind: 'waiting_user';
|
|
40
|
+
sessionRef: string;
|
|
41
|
+
reason: string;
|
|
42
|
+
} | {
|
|
43
|
+
kind: 'revoked';
|
|
44
|
+
sessionRef: string;
|
|
45
|
+
reason: string;
|
|
46
|
+
} | {
|
|
47
|
+
kind: 'expired';
|
|
48
|
+
sessionRef: string;
|
|
49
|
+
reason: string;
|
|
50
|
+
};
|
|
51
|
+
//# sourceMappingURL=session.d.ts.map
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { type NetworkPolicy, type PolicyDecision, type PolicyViolation } from './policy.js';
|
|
2
|
+
export declare const LOCAL_PRIVATE_ALLOWLIST: readonly ["127.0.0.0/8", "::1/128", "10.0.0.0/8", "172.16.0.0/12", "192.168.0.0/16", "fc00::/7"];
|
|
3
|
+
export declare function localNetworkPolicy(): NetworkPolicy;
|
|
4
|
+
export declare function hostedNetworkPolicy(): NetworkPolicy;
|
|
5
|
+
export declare function classifyIp(address: string): PolicyViolation | null;
|
|
6
|
+
/** True for an IPv4 or IPv6 literal (brackets allowed). */
|
|
7
|
+
export declare function isIpAddress(text: string): boolean;
|
|
8
|
+
/** Whether an IP literal lies in a CIDR block such as `10.0.0.0/8` or `::1/128`. */
|
|
9
|
+
export declare function ipInCidr(address: string, cidr: string): boolean;
|
|
10
|
+
export declare function evaluateHostname(hostname: string, policy: NetworkPolicy): PolicyDecision | null;
|
|
11
|
+
export declare function evaluateAddress(hostname: string, address: string, policy: NetworkPolicy): PolicyDecision;
|
|
12
|
+
export declare function evaluateResolved(hostname: string, addresses: readonly string[], policy: NetworkPolicy): PolicyDecision;
|
|
13
|
+
export declare function evaluateUrl(url: string, policy: NetworkPolicy): PolicyDecision | {
|
|
14
|
+
hostname: string;
|
|
15
|
+
};
|
|
16
|
+
//# sourceMappingURL=ssrf.d.ts.map
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The single canonical result status. There is no second enum.
|
|
3
|
+
*
|
|
4
|
+
* Resolution of the earlier empty_legit/empty_suspicious vs empty_verified/partial
|
|
5
|
+
* conflict: a suspected-bad empty result is `failed` with reason `empty_unverified`,
|
|
6
|
+
* not a status of its own. Only *proven* emptiness gets a status.
|
|
7
|
+
*/
|
|
8
|
+
export declare const RESULT_STATUS: readonly ["success", "partial", "empty_verified", "blocked", "failed", "cancelled", "budget_exceeded", "duplicate"];
|
|
9
|
+
export type ResultStatus = (typeof RESULT_STATUS)[number];
|
|
10
|
+
/** Statuses that carry usable extracted content. */
|
|
11
|
+
export declare const CONTENTFUL_STATUS: ReadonlySet<ResultStatus>;
|
|
12
|
+
export declare const FAILURE_REASON: readonly ["empty_unverified", "timeout", "dns_error", "connection_error", "tls_error", "http_error", "redirect_limit", "redirect_loop", "body_too_large", "decompressed_too_large", "unsupported_content_type", "unsupported_content_encoding", "parse_error", "loop_detected", "policy_denied", "provider_error", "identity_compromised", "internal_error", "cache_miss", "action_failed"];
|
|
13
|
+
export type FailureReason = (typeof FAILURE_REASON)[number];
|
|
14
|
+
export declare const BLOCK_REASON: readonly ["cloudflare_challenge", "captcha", "rate_limit", "login_wall", "geo_restricted", "bot_detected_generic"];
|
|
15
|
+
export type BlockReason = (typeof BLOCK_REASON)[number];
|
|
16
|
+
export declare const BUDGET_KIND: readonly ["tokens", "tokens_unknown", "time", "cost", "cost_unknown", "pages", "retries"];
|
|
17
|
+
export type BudgetKind = (typeof BUDGET_KIND)[number];
|
|
18
|
+
/** Execution tiers of the escalation ladder (PHASE1_ENGINEERING_NOTES §2.5). */
|
|
19
|
+
export declare const LANE: readonly ["http", "browser_local", "browser_local_authed", "browser_proxy", "provider"];
|
|
20
|
+
export type Lane = (typeof LANE)[number];
|
|
21
|
+
/**
|
|
22
|
+
* Public canary benchmarks only evaluate these lanes. Authenticated and
|
|
23
|
+
* proxied/provider lanes are reported separately with owned test accounts.
|
|
24
|
+
*/
|
|
25
|
+
export declare const PUBLIC_CANARY_LANES: readonly Lane[];
|
|
26
|
+
export declare function isPublicCanaryLane(lane: Lane): boolean;
|
|
27
|
+
//# sourceMappingURL=status.d.ts.map
|