@octocrawl/sdk 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +17 -0
  3. package/dist/index.cjs +1278 -0
  4. package/dist/index.js +1240 -0
  5. package/dist/types/cjs/client.d.ts +362 -0
  6. package/dist/types/cjs/contracts/access.d.ts +166 -0
  7. package/dist/types/cjs/contracts/actions.d.ts +191 -0
  8. package/dist/types/cjs/contracts/api.d.ts +891 -0
  9. package/dist/types/cjs/contracts/benchmark.d.ts +116 -0
  10. package/dist/types/cjs/contracts/checkpoint.d.ts +164 -0
  11. package/dist/types/cjs/contracts/compliance.d.ts +412 -0
  12. package/dist/types/cjs/contracts/crawl.d.ts +302 -0
  13. package/dist/types/cjs/contracts/delivery.d.ts +136 -0
  14. package/dist/types/cjs/contracts/evidenceRecord.d.ts +192 -0
  15. package/dist/types/cjs/contracts/execution.d.ts +197 -0
  16. package/dist/types/cjs/contracts/extractor.d.ts +379 -0
  17. package/dist/types/cjs/contracts/file.d.ts +117 -0
  18. package/dist/types/cjs/contracts/firecrawl.d.ts +258 -0
  19. package/dist/types/cjs/contracts/groundTruth.d.ts +77 -0
  20. package/dist/types/cjs/contracts/identityBundle.d.ts +70 -0
  21. package/dist/types/cjs/contracts/index.d.ts +30 -0
  22. package/dist/types/cjs/contracts/map.d.ts +180 -0
  23. package/dist/types/cjs/contracts/monitor.d.ts +217 -0
  24. package/dist/types/cjs/contracts/monitorConfig.d.ts +9 -0
  25. package/dist/types/cjs/contracts/policy.d.ts +93 -0
  26. package/dist/types/cjs/contracts/proxy.d.ts +52 -0
  27. package/dist/types/cjs/contracts/recipe.d.ts +74 -0
  28. package/dist/types/cjs/contracts/regexSafety.d.ts +31 -0
  29. package/dist/types/cjs/contracts/result.d.ts +503 -0
  30. package/dist/types/cjs/contracts/session.d.ts +51 -0
  31. package/dist/types/cjs/contracts/ssrf.d.ts +16 -0
  32. package/dist/types/cjs/contracts/status.d.ts +27 -0
  33. package/dist/types/cjs/contracts/structured.d.ts +185 -0
  34. package/dist/types/cjs/contracts/tableMarkdown.d.ts +58 -0
  35. package/dist/types/cjs/contracts/tokens.d.ts +47 -0
  36. package/dist/types/cjs/index.d.ts +9 -0
  37. package/dist/types/cjs/package.json +1 -0
  38. package/dist/types/cjs/version.d.ts +8 -0
  39. package/dist/types/cjs/watcher.d.ts +151 -0
  40. package/dist/types/esm/client.d.ts +362 -0
  41. package/dist/types/esm/contracts/access.d.ts +166 -0
  42. package/dist/types/esm/contracts/actions.d.ts +191 -0
  43. package/dist/types/esm/contracts/api.d.ts +891 -0
  44. package/dist/types/esm/contracts/benchmark.d.ts +116 -0
  45. package/dist/types/esm/contracts/checkpoint.d.ts +164 -0
  46. package/dist/types/esm/contracts/compliance.d.ts +412 -0
  47. package/dist/types/esm/contracts/crawl.d.ts +302 -0
  48. package/dist/types/esm/contracts/delivery.d.ts +136 -0
  49. package/dist/types/esm/contracts/evidenceRecord.d.ts +192 -0
  50. package/dist/types/esm/contracts/execution.d.ts +197 -0
  51. package/dist/types/esm/contracts/extractor.d.ts +379 -0
  52. package/dist/types/esm/contracts/file.d.ts +117 -0
  53. package/dist/types/esm/contracts/firecrawl.d.ts +258 -0
  54. package/dist/types/esm/contracts/groundTruth.d.ts +77 -0
  55. package/dist/types/esm/contracts/identityBundle.d.ts +70 -0
  56. package/dist/types/esm/contracts/index.d.ts +30 -0
  57. package/dist/types/esm/contracts/map.d.ts +180 -0
  58. package/dist/types/esm/contracts/monitor.d.ts +217 -0
  59. package/dist/types/esm/contracts/monitorConfig.d.ts +9 -0
  60. package/dist/types/esm/contracts/policy.d.ts +93 -0
  61. package/dist/types/esm/contracts/proxy.d.ts +52 -0
  62. package/dist/types/esm/contracts/recipe.d.ts +74 -0
  63. package/dist/types/esm/contracts/regexSafety.d.ts +31 -0
  64. package/dist/types/esm/contracts/result.d.ts +503 -0
  65. package/dist/types/esm/contracts/session.d.ts +51 -0
  66. package/dist/types/esm/contracts/ssrf.d.ts +16 -0
  67. package/dist/types/esm/contracts/status.d.ts +27 -0
  68. package/dist/types/esm/contracts/structured.d.ts +185 -0
  69. package/dist/types/esm/contracts/tableMarkdown.d.ts +58 -0
  70. package/dist/types/esm/contracts/tokens.d.ts +47 -0
  71. package/dist/types/esm/index.d.ts +9 -0
  72. package/dist/types/esm/version.d.ts +8 -0
  73. package/dist/types/esm/watcher.d.ts +151 -0
  74. package/package.json +40 -0
@@ -0,0 +1,503 @@
1
+ import type { BlockReason, BudgetKind, FailureReason, Lane, ResultStatus } from './status.js';
2
+ import type { ComplianceRecord } from './compliance.js';
3
+ import type { DocumentExtraction, PageMetadata } from './extractor.js';
4
+ import type { ListField, StructuredExtractionResult } from './structured.js';
5
+ import type { FileDescription } from './file.js';
6
+ import type { ActionsResult } from './actions.js';
7
+ export interface ResourceTimings {
8
+ /** Every wait in the origin scheduler but a cooldown: the concurrency ceiling and the minimum interval between requests. */
9
+ queueMs?: number;
10
+ robotsMs?: number;
11
+ cooldownWaitMs?: number;
12
+ /**
13
+ * Present when the per-origin concurrency ceiling held this lane's permit
14
+ * back: the milliseconds it waited for a slot, cooldown and pacing
15
+ * excluded (both stay in `cooldownWaitMs` / `queueMs`). Absent when the
16
+ * permit started at once, or when the lane acquired none.
17
+ */
18
+ concurrencyWaitMs?: number;
19
+ retryWaitMs?: number;
20
+ /** Initial request/headers time, excluding body read and retry sleep. */
21
+ requestMs?: number;
22
+ bodyReadMs?: number;
23
+ /** Total network work excluding retry sleep. */
24
+ transportMs?: number;
25
+ parseMs?: number;
26
+ extractMs?: number;
27
+ formatMs?: number;
28
+ serializeMs?: number;
29
+ modelMs?: number;
30
+ totalMs: number;
31
+ }
32
+ /** Why the runtime moved from one lane to the next. Logged for the escalation corpus. */
33
+ export interface Escalation {
34
+ from: Lane;
35
+ to: Lane;
36
+ /** Machine-readable trigger, e.g. 'low_text_yield' | 'spa_marker' | 'blocked' */
37
+ trigger: string;
38
+ /** Whether the escalation actually improved the outcome. Null until known. */
39
+ improved: boolean | null;
40
+ }
41
+ export interface ResourceUsage {
42
+ wallMs: number;
43
+ /** Bytes received on the wire (compressed). */
44
+ bytesWire: number | null;
45
+ /** Bytes after decompression. Guarded by a decompressed-size cap. */
46
+ bytesDecompressed: number;
47
+ requestCount: number;
48
+ attemptCount: number;
49
+ /** Actual status-driven retries; not variant-selection navigations. */
50
+ statusRetryCount?: number;
51
+ navigationFollowupCount?: number;
52
+ /** Token count of the emitted main content. Null if not tokenized. */
53
+ contentTokens: number | null;
54
+ browserMs: number;
55
+ /**
56
+ * Cost incurred outside this process, paid by the user to a third party
57
+ * (BYO proxy egress, provider browser minutes, model calls).
58
+ * `null` means no external cost path was used — never means "free".
59
+ */
60
+ externalCostUsd: number | null;
61
+ /** Stage timings use a monotonic clock. Optional for legacy producers. */
62
+ timings?: ResourceTimings;
63
+ /**
64
+ * True when the caller's deadline (a scrape's `timeout`) ended this fetch
65
+ * before it finished: the result is then `partial` with the content
66
+ * fetched so far, or `failed` with `timeout`. Absent otherwise.
67
+ */
68
+ deadlineExceeded?: boolean;
69
+ }
70
+ export interface Meter {
71
+ knownSubtotal: number;
72
+ unknown: boolean;
73
+ }
74
+ export interface Evidence {
75
+ /**
76
+ * Final URL after redirects: the last URL requested for the page, whose
77
+ * response `httpStatus` and `contentType` are from. In the browser lane a
78
+ * script or a meta refresh that loaded another document is a redirect; a
79
+ * URL the page set with the history API (pushState, replaceState), which
80
+ * nothing requested, is not: the page keeps it as the base of its links.
81
+ */
82
+ finalUrl: string;
83
+ /**
84
+ * The status of the response that answered `finalUrl`: in the browser lane,
85
+ * of the document the page shows when it is read, not of the navigation W2L
86
+ * started. Null when there was none (no request, a transport failure, a
87
+ * document that came without a response).
88
+ */
89
+ httpStatus: number | null;
90
+ /**
91
+ * The URLs of a redirect, the requested URL first and `finalUrl` last;
92
+ * empty when nothing redirected.
93
+ */
94
+ redirectChain: readonly string[];
95
+ /**
96
+ * True when `redirectChain` lists every hop the lane requested: the HTTP
97
+ * lane follows each redirect itself, and the browser lane lists each
98
+ * redirect Chromium followed and each document a script or a meta refresh
99
+ * loaded, for a page it shows or a file it displays or downloads. The
100
+ * browser lane says false when a follow-up navigation did not start at the
101
+ * requested URL, a document came without a request, or it cut the chain of
102
+ * a page that kept moving on to its first URL and last 20. Absent when the
103
+ * lane does not say (the provider lane, which sees where its vendor started
104
+ * and ended, and results stored before lanes recorded it).
105
+ */
106
+ redirectChainComplete?: boolean;
107
+ /**
108
+ * The `content-type` header of the response `httpStatus` is from, as the
109
+ * server sent it, in every lane (the browser lane reads the rendered page,
110
+ * whatever it says); null when there was no response or no such header.
111
+ */
112
+ contentType: string | null;
113
+ /** sha256 of the raw response body. Null only when no body was read. */
114
+ rawBodySha256: string | null;
115
+ /**
116
+ * The `content-encoding` of the response `httpStatus` is from, as the HTTP
117
+ * lane received it (lower-cased codings in applied order, `x-gzip` read as
118
+ * `gzip`), or `identity` when it had none. The lane decodes gzip, deflate
119
+ * and br, so the body behind `rawBodySha256` is the decoded one; any other
120
+ * coding fails with `unsupported_content_encoding`. Absent when no response
121
+ * body was read, and in lanes that do not report it (browser, provider).
122
+ */
123
+ contentEncoding?: string;
124
+ /** Relative artifact paths (raw body, screenshot, DOM snapshot). */
125
+ artifacts: readonly string[];
126
+ /**
127
+ * UTC ISO time the lane received what it reports: the final response's
128
+ * headers (HTTP), the vendor's answer (provider), the rendered page's
129
+ * capture (browser). Absent when no response was read, and on results
130
+ * stored before lanes recorded it.
131
+ */
132
+ fetchedAt?: string;
133
+ /** HTTP validators observed for the representation, when exposed. */
134
+ etag?: string | null;
135
+ lastModified?: string | null;
136
+ cacheControl?: string | null;
137
+ vary?: string | null;
138
+ /** A response setting cookies cannot enter the public monitor cache. */
139
+ setsCookie?: boolean;
140
+ /**
141
+ * `host:port` of the operator's environment proxy (local mode) that the
142
+ * request for `finalUrl` went through; never its credentials. Null when that
143
+ * request did not use it (NO_PROXY, loopback). Absent when no environment
144
+ * proxy was configured, no request was answered, or the lane does not
145
+ * report its route (the provider lane).
146
+ */
147
+ envProxy?: string | null;
148
+ }
149
+ export interface TraceEvent {
150
+ at: number;
151
+ lane: Lane;
152
+ event: string;
153
+ detail?: Record<string, unknown>;
154
+ }
155
+ export interface LadderAttempt {
156
+ channel: string;
157
+ result: FetchResult;
158
+ }
159
+ export interface LadderExecutionSummary {
160
+ channelsTried: readonly string[];
161
+ attempts: readonly LadderAttempt[];
162
+ wallMs: number;
163
+ browserMs: number;
164
+ bytesWire: number | null;
165
+ bytesDecompressed: number;
166
+ requestCount: number;
167
+ attemptCount: number;
168
+ contentTokens: number | null;
169
+ externalCostUsd: number | null;
170
+ externalCost: Meter;
171
+ contentTokenMeter: Meter;
172
+ artifacts: readonly string[];
173
+ /** Actual caller wait across all ladder work, including routing overhead. */
174
+ totalMs?: number;
175
+ }
176
+ export interface LadderRunAudit {
177
+ channelsTried: readonly string[];
178
+ ladderTrace: readonly {
179
+ at: number;
180
+ event: string;
181
+ channel: string;
182
+ detail: Record<string, unknown>;
183
+ }[];
184
+ summary: LadderExecutionSummary;
185
+ }
186
+ /**
187
+ * A request for human takeover: the lane hit a captcha or login wall it will
188
+ * not defeat, a live-view door exists, and the task pauses here until a
189
+ * human returns (or the run aborts). Carrying this on the result instead of
190
+ * throwing keeps the decision trail in the signed record: the run did not
191
+ * silently skip the page, it stopped and asked.
192
+ */
193
+ export interface HandoffRequest {
194
+ /** The seven-class routing reason, e.g. 'captcha_required'. */
195
+ reason: string;
196
+ /** Live view URL for the human, when one was opened. */
197
+ liveViewUrl: string | null;
198
+ /** Why this specific result asks for a human, one sentence. */
199
+ rationale: string;
200
+ }
201
+ /** The values of one HTML attribute on the elements one CSS selector names (the `attributes` format). */
202
+ export interface AttributeExtraction {
203
+ selector: string;
204
+ attribute: string;
205
+ /** As written in the HTML, in document order; an element without the attribute is skipped; `[]` when nothing matches. */
206
+ values: readonly string[];
207
+ }
208
+ /**
209
+ * The `screenshot` format: the rendered page as the browser lane captured it
210
+ * after load, stability and `waitFor`, before the DOM was read, so the image
211
+ * and the Markdown show the same page. The bytes are inline (`base64`) and
212
+ * hash to `sha256`; `path` names the file under W2L_CAPTURE_RAW_DIR when
213
+ * that is set (`<sha256>.png` or `.jpg`, listed in `evidence.artifacts`
214
+ * too), else null. `width` and `height` are CSS pixels (`scale: 'css'`):
215
+ * the viewport's for a viewport capture, the viewport's width and the
216
+ * document's height for `fullPage`; `deviceScaleFactor` is what the context
217
+ * declared, not baked into the image. The page itself is unchanged: nothing
218
+ * is scrolled, clicked or hidden for the capture.
219
+ */
220
+ export interface ScreenshotEvidence {
221
+ contentType: 'image/png' | 'image/jpeg';
222
+ width: number;
223
+ height: number;
224
+ fullPage: boolean;
225
+ /** The window the page was laid out in, in CSS pixels: the request's viewport, or the declared one. */
226
+ viewport: {
227
+ width: number;
228
+ height: number;
229
+ };
230
+ /** Device pixels per CSS pixel the context declared (2 for the desktop identity, 2.625 for the mobile one). */
231
+ deviceScaleFactor: number;
232
+ /** The JPEG quality asked for; null for a PNG. */
233
+ quality: number | null;
234
+ bytes: number;
235
+ sha256: string;
236
+ path: string | null;
237
+ base64: string;
238
+ }
239
+ /**
240
+ * A caveat a reader of the result must see without opening the trace. Never
241
+ * a failure reason, which `status` and its reason fields carry.
242
+ */
243
+ export interface FetchWarning {
244
+ /**
245
+ * Machine-readable code. `robots_overridden`: a robots.txt rule was set
246
+ * aside by a recorded override. `tls_unverified`: the certificate was not
247
+ * verified at the caller's request (`skipTlsVerification`), so the content
248
+ * cannot be attributed to the host with certainty. `client_rendered_suspected`:
249
+ * the HTTP lane's page looks like a shell for data its scripts fill in
250
+ * (see RenderSignals), so the capture may not be the page a browser shows.
251
+ * `low_content_yield`: a thin answer stayed the run's answer: the http
252
+ * lane's, which the browser lane did not improve on or was not offered,
253
+ * or a rendered one (the browser or a provider lane's) its own extraction
254
+ * found thin and low-confidence.
255
+ * `screenshot_unavailable`: the `screenshot` format was asked for and the
256
+ * browser lane rendered the page but could not capture it
257
+ * (`screenshot_failed` in the trace); `screenshot` is null and the page
258
+ * result stands.
259
+ */
260
+ code: string;
261
+ message: string;
262
+ }
263
+ /**
264
+ * A page-level fetch outcome. `status` is the single source of truth
265
+ * (see RESULT_STATUS); the reason fields narrow it.
266
+ */
267
+ export interface FetchResult {
268
+ /** Earliest permitted next request after a deferred Retry-After, UTC milliseconds. */
269
+ retryAt?: number;
270
+ requestedUrl: string;
271
+ status: ResultStatus;
272
+ /** Set iff status === 'failed'. Duplicate content uses status `duplicate`. */
273
+ failureReason: FailureReason | null;
274
+ /** Set iff status === 'blocked'. */
275
+ blockReason: BlockReason | null;
276
+ /** Set iff status === 'budget_exceeded'. */
277
+ budgetExceeded: BudgetKind | null;
278
+ /** The lane that produced this result. */
279
+ lane: Lane;
280
+ escalations: readonly Escalation[];
281
+ /** Set when the result asks for human takeover. Never on a success.
282
+ * Optional for backward compatibility with existing result producers;
283
+ * the router and provider lanes always populate it. */
284
+ handoff?: HandoffRequest | null;
285
+ /**
286
+ * Vendor session resume material (context/profile/storage) that the
287
+ * provider lane produced, so the ladder can persist it for the next run.
288
+ * Shape is vendor-specific; it is a credential-free continuation token.
289
+ */
290
+ resumeContext?: unknown | null;
291
+ /**
292
+ * The page as Markdown: its main content, the whole page when
293
+ * `onlyMainContent` is false, or the elements `includeTags` names, in each
294
+ * case without `excludeTags`. `data:` link and image targets are dropped,
295
+ * the link text and alt text kept. Null unless status is contentful,
296
+ * except on a failed or blocked result that kept a page as evidence, never
297
+ * content: the page an error status carried, or the whole page when the
298
+ * extractor found no main content (`empty_unverified`, and `timeout` when
299
+ * the deadline then ended a later rung).
300
+ */
301
+ markdown: string | null;
302
+ /** HTML-derived page/product facts; never reconstructed from Markdown. */
303
+ document?: DocumentExtraction | null;
304
+ /**
305
+ * What the page's HTML declares about itself (its `<title>`, description,
306
+ * language, keywords, robots, icon and canonical URL), present with
307
+ * `document`. `metadata.title` is the page's `<title>`; `document.title` is
308
+ * the content's title, usually its first heading.
309
+ */
310
+ metadata?: PageMetadata;
311
+ /**
312
+ * The cleaned HTML the Markdown was written from, present only when the
313
+ * `html` format was asked for: the main content; with
314
+ * `onlyMainContent: false` the whole page without what Markdown never
315
+ * shows (scripts, styles, form controls, embedded media) and without the
316
+ * caller's `excludeTags`; with `includeTags` a `<body>` holding the named
317
+ * elements. A lane sets it on a contentful page only; the API returns
318
+ * null for a file and for a page that was not read as content.
319
+ */
320
+ html?: string | null;
321
+ /**
322
+ * The page as the lane received it, present only when the `rawHtml` format
323
+ * was asked for: the response body on the HTTP lane, the rendered DOM on a
324
+ * browser lane, scripts and all. Its UTF-8 bytes hash to
325
+ * `evidence.rawBodySha256`. Set and returned as `html` is.
326
+ */
327
+ rawHtml?: string | null;
328
+ /** Present only when a JSON format was requested. */
329
+ json?: StructuredExtractionResult | null;
330
+ /**
331
+ * Present when the response was a file (PDF, CSV, JSON, text, XLSX, XLS,
332
+ * ZIP) rather than a web page: what it was, its size, SHA-256 and where it
333
+ * was saved, and for a PDF its pages. Such a result has no `document` or
334
+ * `metadata`.
335
+ */
336
+ file?: FileDescription;
337
+ /**
338
+ * Outbound http(s) links from the FULL document, collected after extract
339
+ * and before the raw HTML is dropped. Not from `mainHtml` — prune strips
340
+ * nav. Empty / omitted when the fetch never produced HTML. Never the page
341
+ * HTML itself.
342
+ */
343
+ links?: readonly string[];
344
+ /**
345
+ * The image URLs of the FULL document (`img` src and srcset candidates,
346
+ * `picture` sources, lazy-loading attributes, video posters, `image_src`
347
+ * links, og:image and twitter:image), absolute http(s), fragment stripped,
348
+ * each once, in document order; `data:` URIs left out. Present only when
349
+ * the `images` format was asked for (`FetchOptions.includeImages`), on a
350
+ * contentful page: absent for a file and for a page that was not read as
351
+ * content. `[]` for a page without images.
352
+ */
353
+ images?: readonly string[];
354
+ /**
355
+ * The `tables` format: every data table of the content the Markdown was
356
+ * written from (the main content, the whole page with `onlyMainContent:
357
+ * false`, or the `includeTags` selection), one entry per GFM table of the
358
+ * Markdown, in its order. Present only when asked for
359
+ * (`FetchOptions.includeTables`), on a contentful page: absent for a file
360
+ * and for a page that was not read as content. `[]` for a page without one.
361
+ */
362
+ tables?: readonly PageTable[];
363
+ /**
364
+ * The `list` format, when asked for and the page was read: its records,
365
+ * one per element `itemSelector` matched, from every page a paginate step
366
+ * read when one ran, else from the page as it stands.
367
+ */
368
+ list?: ListExtraction;
369
+ /**
370
+ * A PDF's pages, each as the Markdown has it (without its marker), when a
371
+ * `pdf` parser entry asked for them (`pages: true`) and the text layer was
372
+ * read; absent otherwise.
373
+ */
374
+ pages?: readonly PdfPageMarkdown[];
375
+ /**
376
+ * The `attributes` format: one entry per selector of the request, in its
377
+ * order, with the attribute's values as written in the HTML. Present only
378
+ * when asked for (`FetchOptions.attributes`), on a contentful page, like
379
+ * `images`.
380
+ */
381
+ attributes?: readonly AttributeExtraction[];
382
+ /**
383
+ * The `screenshot` format, present only when asked for
384
+ * (`FetchOptions.screenshot`), on every result the browser lane built from
385
+ * the rendered document: a success or partial page, and an error-status or
386
+ * blocked page kept as evidence. Null when the page rendered but the
387
+ * capture failed (`screenshot_failed` in the trace, a
388
+ * `screenshot_unavailable` warning). Absent when no page rendered (a file,
389
+ * a robots.txt denial, a navigation failure), as the API then answers
390
+ * null. The attempt copies in a run's audit carry null, so the image
391
+ * travels once.
392
+ */
393
+ screenshot?: ScreenshotEvidence | null;
394
+ /**
395
+ * What the request's `actions` produced (screenshots, HTML snapshots,
396
+ * script returns, PDFs) and the step that failed, if one did. Present on
397
+ * every result of the browser lane that ran the steps; absent otherwise.
398
+ */
399
+ actions?: ActionsResult;
400
+ /**
401
+ * The fetch's caveats, present only when it has any: a `robots_overridden`
402
+ * warning first when a recorded override set a robots.txt rule aside, then
403
+ * `tls_unverified` when the fetch skipped certificate verification, then
404
+ * `client_rendered_suspected` when the HTTP lane read the page as a shell
405
+ * its scripts fill in, or `screenshot_unavailable` when the browser lane
406
+ * could not capture the screenshot asked for. Kept on batch items and the
407
+ * compact scrape response too.
408
+ */
409
+ warnings?: readonly FetchWarning[];
410
+ /** True when content was cut to fit a token budget. */
411
+ truncated: boolean;
412
+ /** Character offset where truncation occurred; null when not truncated. */
413
+ truncatedAt: number | null;
414
+ /**
415
+ * Tamper-evident record of what the fetch actually did (robots.txt decision,
416
+ * exact headers sent, rate-limit facts). Null until a subject wires the
417
+ * record builder in; contentful subjects are expected to produce one per
418
+ * fetch so the premium "provable politeness" tier is not a bolt-on.
419
+ */
420
+ compliance: ComplianceRecord | null;
421
+ evidence: Evidence;
422
+ usage: ResourceUsage;
423
+ trace: readonly TraceEvent[];
424
+ }
425
+ /**
426
+ * One data table of a page (the `tables` format). `tableIndex` counts the
427
+ * data tables from 0 in document order: table N is the Nth GFM table of the
428
+ * Markdown made from the same request. Cells are plain text: a link is its
429
+ * text, an image its alt text, whitespace collapsed, nothing escaped; a cell
430
+ * that spans rows or columns gives its value to every slot it covers, so
431
+ * every row has `columns` cells and none is shifted.
432
+ */
433
+ /** One record of the `list` format. */
434
+ export interface ListRecord {
435
+ /** Each field's value, by name: the text (whitespace collapsed) or the attribute; null when the record has none. */
436
+ values: Record<string, string | null>;
437
+ /** The fields that are null, in field order: what this record lacks, never filled in. */
438
+ missing: string[];
439
+ /** Where it was read: the page's URL, the page's number (1 for the page read, or each page a paginate step read, in order), and the record's place on it (0-based, document order). */
440
+ source: {
441
+ url: string;
442
+ page: number;
443
+ index: number;
444
+ };
445
+ }
446
+ /** The `list` format: the records of the page, or of every page a paginate step read. */
447
+ export interface ListExtraction {
448
+ /** The items' selector: as asked, or the one W2L found (`detected`); null when it found no list on the page (a `list_not_detected` warning). */
449
+ itemSelector: string | null;
450
+ fields: string[];
451
+ /**
452
+ * Present when W2L chose the itemSelector or the fields: the fields as a
453
+ * request names them, to send back as they are or changed, and the other
454
+ * lists it found on the page, best first.
455
+ */
456
+ detected?: ListDetection;
457
+ records: ListRecord[];
458
+ /** Pages the records were read from. */
459
+ pages: number;
460
+ /** Records with at least one field missing. */
461
+ incomplete: number;
462
+ /** True when a limit cut the list (10,000 records, 5,000,000 characters of values): the page had more records than these. */
463
+ truncated: boolean;
464
+ /** The records as RFC 4180 CSV: the fields, then source_url, page and index. */
465
+ csv: string;
466
+ /** SHA-256 (hex) of the UTF-8 bytes of `csv`. */
467
+ csvSha256: string;
468
+ }
469
+ export interface ListDetection {
470
+ fields: ListField[];
471
+ alternatives: Array<{
472
+ itemSelector: string;
473
+ count: number;
474
+ }>;
475
+ }
476
+ export interface PageTable {
477
+ tableIndex: number;
478
+ /** The table's `<caption>` as plain text; null when it has none (a title written above the table is not one). */
479
+ caption: string | null;
480
+ /** The final URL of the page the table was read from (`evidence.finalUrl`). */
481
+ sourceUrl: string;
482
+ /** Leading rows in `<thead>` or made of `<th>` cells alone; 0 when none. */
483
+ headerRows: number;
484
+ columns: number;
485
+ rows: readonly (readonly string[])[];
486
+ /** The rows as RFC 4180 CSV: CRLF line ends, a field quoted when it holds a comma, a quote or a line break, quotes doubled. */
487
+ csv: string;
488
+ /** SHA-256 (hex) of the UTF-8 bytes of `csv`. */
489
+ csvSha256: string;
490
+ /**
491
+ * Present when the table is too large to give: its cells, each spanned
492
+ * value repeated, would exceed 2,000,000 characters, or what is left of
493
+ * 5,000,000 for all of the page's tables. `rows` is then empty,
494
+ * `csv` is `''` and `columns` 0; the table keeps its `tableIndex`.
495
+ */
496
+ omitted?: 'too_large';
497
+ }
498
+ /** One page of a PDF's text (`pages`): its number in the document, from 1, and its Markdown. */
499
+ export interface PdfPageMarkdown {
500
+ pageNumber: number;
501
+ markdown: string;
502
+ }
503
+ //# sourceMappingURL=result.d.ts.map
@@ -0,0 +1,51 @@
1
+ export declare const SESSION_STATE: readonly ["active", "waiting_user", "revoked", "expired"];
2
+ export type SessionState = (typeof SESSION_STATE)[number];
3
+ export interface ManagedSessionRef {
4
+ sessionRef: string;
5
+ workspaceId: string;
6
+ accountRef: string;
7
+ originScope: string;
8
+ profileId: string;
9
+ profileDir: string;
10
+ grantEpoch: number;
11
+ state: SessionState;
12
+ createdAt: string;
13
+ updatedAt: string;
14
+ revokedAt: string | null;
15
+ expiresAt: string | null;
16
+ handoff: SessionHandoff | null;
17
+ backend?: 'managed' | 'existing_chrome';
18
+ /** Private registry only; never include in public status/logs. */
19
+ cdpEndpoint?: string;
20
+ }
21
+ export interface SessionHandoff {
22
+ handoffId: string;
23
+ reason: string;
24
+ createdAt: string;
25
+ expiresAt: string;
26
+ }
27
+ export interface SessionGrant {
28
+ sessionRef: string;
29
+ workspaceId: string;
30
+ accountRef: string;
31
+ originScope: string;
32
+ grantEpoch: number;
33
+ expiresAt: string | null;
34
+ }
35
+ export type SessionAccessResult = {
36
+ kind: 'granted';
37
+ grant: SessionGrant;
38
+ } | {
39
+ kind: 'waiting_user';
40
+ sessionRef: string;
41
+ reason: string;
42
+ } | {
43
+ kind: 'revoked';
44
+ sessionRef: string;
45
+ reason: string;
46
+ } | {
47
+ kind: 'expired';
48
+ sessionRef: string;
49
+ reason: string;
50
+ };
51
+ //# sourceMappingURL=session.d.ts.map
@@ -0,0 +1,16 @@
1
+ import { type NetworkPolicy, type PolicyDecision, type PolicyViolation } from './policy.js';
2
+ export declare const LOCAL_PRIVATE_ALLOWLIST: readonly ["127.0.0.0/8", "::1/128", "10.0.0.0/8", "172.16.0.0/12", "192.168.0.0/16", "fc00::/7"];
3
+ export declare function localNetworkPolicy(): NetworkPolicy;
4
+ export declare function hostedNetworkPolicy(): NetworkPolicy;
5
+ export declare function classifyIp(address: string): PolicyViolation | null;
6
+ /** True for an IPv4 or IPv6 literal (brackets allowed). */
7
+ export declare function isIpAddress(text: string): boolean;
8
+ /** Whether an IP literal lies in a CIDR block such as `10.0.0.0/8` or `::1/128`. */
9
+ export declare function ipInCidr(address: string, cidr: string): boolean;
10
+ export declare function evaluateHostname(hostname: string, policy: NetworkPolicy): PolicyDecision | null;
11
+ export declare function evaluateAddress(hostname: string, address: string, policy: NetworkPolicy): PolicyDecision;
12
+ export declare function evaluateResolved(hostname: string, addresses: readonly string[], policy: NetworkPolicy): PolicyDecision;
13
+ export declare function evaluateUrl(url: string, policy: NetworkPolicy): PolicyDecision | {
14
+ hostname: string;
15
+ };
16
+ //# sourceMappingURL=ssrf.d.ts.map
@@ -0,0 +1,27 @@
1
+ /**
2
+ * The single canonical result status. There is no second enum.
3
+ *
4
+ * Resolution of the earlier empty_legit/empty_suspicious vs empty_verified/partial
5
+ * conflict: a suspected-bad empty result is `failed` with reason `empty_unverified`,
6
+ * not a status of its own. Only *proven* emptiness gets a status.
7
+ */
8
+ export declare const RESULT_STATUS: readonly ["success", "partial", "empty_verified", "blocked", "failed", "cancelled", "budget_exceeded", "duplicate"];
9
+ export type ResultStatus = (typeof RESULT_STATUS)[number];
10
+ /** Statuses that carry usable extracted content. */
11
+ export declare const CONTENTFUL_STATUS: ReadonlySet<ResultStatus>;
12
+ export declare const FAILURE_REASON: readonly ["empty_unverified", "timeout", "dns_error", "connection_error", "tls_error", "http_error", "redirect_limit", "redirect_loop", "body_too_large", "decompressed_too_large", "unsupported_content_type", "unsupported_content_encoding", "parse_error", "loop_detected", "policy_denied", "provider_error", "identity_compromised", "internal_error", "cache_miss", "action_failed"];
13
+ export type FailureReason = (typeof FAILURE_REASON)[number];
14
+ export declare const BLOCK_REASON: readonly ["cloudflare_challenge", "captcha", "rate_limit", "login_wall", "geo_restricted", "bot_detected_generic"];
15
+ export type BlockReason = (typeof BLOCK_REASON)[number];
16
+ export declare const BUDGET_KIND: readonly ["tokens", "tokens_unknown", "time", "cost", "cost_unknown", "pages", "retries"];
17
+ export type BudgetKind = (typeof BUDGET_KIND)[number];
18
+ /** Execution tiers of the escalation ladder (PHASE1_ENGINEERING_NOTES §2.5). */
19
+ export declare const LANE: readonly ["http", "browser_local", "browser_local_authed", "browser_proxy", "provider"];
20
+ export type Lane = (typeof LANE)[number];
21
+ /**
22
+ * Public canary benchmarks only evaluate these lanes. Authenticated and
23
+ * proxied/provider lanes are reported separately with owned test accounts.
24
+ */
25
+ export declare const PUBLIC_CANARY_LANES: readonly Lane[];
26
+ export declare function isPublicCanaryLane(lane: Lane): boolean;
27
+ //# sourceMappingURL=status.d.ts.map