@octocrawl/sdk 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +17 -0
  3. package/dist/index.cjs +1278 -0
  4. package/dist/index.js +1240 -0
  5. package/dist/types/cjs/client.d.ts +362 -0
  6. package/dist/types/cjs/contracts/access.d.ts +166 -0
  7. package/dist/types/cjs/contracts/actions.d.ts +191 -0
  8. package/dist/types/cjs/contracts/api.d.ts +891 -0
  9. package/dist/types/cjs/contracts/benchmark.d.ts +116 -0
  10. package/dist/types/cjs/contracts/checkpoint.d.ts +164 -0
  11. package/dist/types/cjs/contracts/compliance.d.ts +412 -0
  12. package/dist/types/cjs/contracts/crawl.d.ts +302 -0
  13. package/dist/types/cjs/contracts/delivery.d.ts +136 -0
  14. package/dist/types/cjs/contracts/evidenceRecord.d.ts +192 -0
  15. package/dist/types/cjs/contracts/execution.d.ts +197 -0
  16. package/dist/types/cjs/contracts/extractor.d.ts +379 -0
  17. package/dist/types/cjs/contracts/file.d.ts +117 -0
  18. package/dist/types/cjs/contracts/firecrawl.d.ts +258 -0
  19. package/dist/types/cjs/contracts/groundTruth.d.ts +77 -0
  20. package/dist/types/cjs/contracts/identityBundle.d.ts +70 -0
  21. package/dist/types/cjs/contracts/index.d.ts +30 -0
  22. package/dist/types/cjs/contracts/map.d.ts +180 -0
  23. package/dist/types/cjs/contracts/monitor.d.ts +217 -0
  24. package/dist/types/cjs/contracts/monitorConfig.d.ts +9 -0
  25. package/dist/types/cjs/contracts/policy.d.ts +93 -0
  26. package/dist/types/cjs/contracts/proxy.d.ts +52 -0
  27. package/dist/types/cjs/contracts/recipe.d.ts +74 -0
  28. package/dist/types/cjs/contracts/regexSafety.d.ts +31 -0
  29. package/dist/types/cjs/contracts/result.d.ts +503 -0
  30. package/dist/types/cjs/contracts/session.d.ts +51 -0
  31. package/dist/types/cjs/contracts/ssrf.d.ts +16 -0
  32. package/dist/types/cjs/contracts/status.d.ts +27 -0
  33. package/dist/types/cjs/contracts/structured.d.ts +185 -0
  34. package/dist/types/cjs/contracts/tableMarkdown.d.ts +58 -0
  35. package/dist/types/cjs/contracts/tokens.d.ts +47 -0
  36. package/dist/types/cjs/index.d.ts +9 -0
  37. package/dist/types/cjs/package.json +1 -0
  38. package/dist/types/cjs/version.d.ts +8 -0
  39. package/dist/types/cjs/watcher.d.ts +151 -0
  40. package/dist/types/esm/client.d.ts +362 -0
  41. package/dist/types/esm/contracts/access.d.ts +166 -0
  42. package/dist/types/esm/contracts/actions.d.ts +191 -0
  43. package/dist/types/esm/contracts/api.d.ts +891 -0
  44. package/dist/types/esm/contracts/benchmark.d.ts +116 -0
  45. package/dist/types/esm/contracts/checkpoint.d.ts +164 -0
  46. package/dist/types/esm/contracts/compliance.d.ts +412 -0
  47. package/dist/types/esm/contracts/crawl.d.ts +302 -0
  48. package/dist/types/esm/contracts/delivery.d.ts +136 -0
  49. package/dist/types/esm/contracts/evidenceRecord.d.ts +192 -0
  50. package/dist/types/esm/contracts/execution.d.ts +197 -0
  51. package/dist/types/esm/contracts/extractor.d.ts +379 -0
  52. package/dist/types/esm/contracts/file.d.ts +117 -0
  53. package/dist/types/esm/contracts/firecrawl.d.ts +258 -0
  54. package/dist/types/esm/contracts/groundTruth.d.ts +77 -0
  55. package/dist/types/esm/contracts/identityBundle.d.ts +70 -0
  56. package/dist/types/esm/contracts/index.d.ts +30 -0
  57. package/dist/types/esm/contracts/map.d.ts +180 -0
  58. package/dist/types/esm/contracts/monitor.d.ts +217 -0
  59. package/dist/types/esm/contracts/monitorConfig.d.ts +9 -0
  60. package/dist/types/esm/contracts/policy.d.ts +93 -0
  61. package/dist/types/esm/contracts/proxy.d.ts +52 -0
  62. package/dist/types/esm/contracts/recipe.d.ts +74 -0
  63. package/dist/types/esm/contracts/regexSafety.d.ts +31 -0
  64. package/dist/types/esm/contracts/result.d.ts +503 -0
  65. package/dist/types/esm/contracts/session.d.ts +51 -0
  66. package/dist/types/esm/contracts/ssrf.d.ts +16 -0
  67. package/dist/types/esm/contracts/status.d.ts +27 -0
  68. package/dist/types/esm/contracts/structured.d.ts +185 -0
  69. package/dist/types/esm/contracts/tableMarkdown.d.ts +58 -0
  70. package/dist/types/esm/contracts/tokens.d.ts +47 -0
  71. package/dist/types/esm/index.d.ts +9 -0
  72. package/dist/types/esm/version.d.ts +8 -0
  73. package/dist/types/esm/watcher.d.ts +151 -0
  74. package/package.json +40 -0
@@ -0,0 +1,180 @@
1
+ /**
2
+ * Map contract: the URLs of a site, discovered without fetching each page.
3
+ *
4
+ * A map reads robots.txt, at most one page body (the start URL, on the http
5
+ * rung) and the sitemaps the site declares, inside one deadline, and returns
6
+ * what it found as links with the evidence for each: how it was found, the
7
+ * sitemap file that listed it, the robots.txt verdict, and a title only where
8
+ * one was already in hand (the start page's own metadata, an anchor's text, a
9
+ * sitemap's `<news:title>`). Deeper discovery is a crawl. Types only, no I/O:
10
+ * the runtime's MapRunner composes the sources, the API engine wires the real
11
+ * ones.
12
+ */
13
+ import type { AgentHints, MapRequest } from './api.js';
14
+ import type { CrawlMode } from './compliance.js';
15
+ import type { SitemapFileRecord, SitemapMode, SitemapSource, SitemapSourceKind } from './crawl.js';
16
+ import type { ExecutionContext } from './execution.js';
17
+ import type { Evidence, FetchResult } from './result.js';
18
+ import type { FailureReason, ResultStatus } from './status.js';
19
+ /** Sitemap files one map reads at most, an index and its children each counting as one (gov.uk's index lists 35 children). A crawl keeps its own 20. */
20
+ export declare const MAP_SITEMAP_MAX_FILES = 50;
21
+ /** Hosts beside the start host whose robots.txt one map reads at most; URLs on further hosts are refused unchecked. */
22
+ export declare const MAP_MAX_ROBOTS_HOSTS = 20;
23
+ /** Samples each refused counter keeps at most. */
24
+ export declare const MAP_REFUSED_SAMPLES = 20;
25
+ /** The longest anchor-text title a map link carries, in characters. */
26
+ export declare const MAP_TITLE_MAX_CHARS = 300;
27
+ /** How a map link was found: it is the start URL, a link on the start page, or a sitemap entry. */
28
+ export type MapLinkVia = 'start' | 'link' | 'sitemap';
29
+ export interface MapLink {
30
+ /** The canonical URL: fragment and tracking parameters dropped (the whole query with ignoreQueryParameters). */
31
+ url: string;
32
+ /** Present only when a title was in hand: the start page's own, an anchor's text, or a sitemap's `<news:title>`. Never generated, never empty. */
33
+ title?: string;
34
+ /** The start page's own meta description; no other link has one. */
35
+ description?: string;
36
+ /** Where `title` came from; present exactly when `title` is. */
37
+ titleSource?: 'page' | 'anchor' | 'sitemap';
38
+ via: MapLinkVia[];
39
+ /** The first sitemap file that listed the URL. */
40
+ sitemapFile?: string;
41
+ /** The `<lastmod>` the sitemap gave, as written. */
42
+ lastmod?: string;
43
+ /** The robots.txt verdict for the URL under the map's declared identity; a disallowed URL is never a link. */
44
+ robots: 'allowed' | 'no_robots';
45
+ }
46
+ export type MapStatus = 'completed' | 'partial' | 'failed';
47
+ /** What the start page read gave, or why it was not read (`failed/policy_denied` when robots.txt disallows the start URL or could not be read). */
48
+ export interface MapStartPage {
49
+ url: string;
50
+ finalUrl: string | null;
51
+ httpStatus: number | null;
52
+ status: ResultStatus;
53
+ failureReason: FailureReason | null;
54
+ lane: 'http';
55
+ /**
56
+ * The start URL's robots.txt verdict; `unreachable` when its robots.txt
57
+ * could not be read (a 5xx, a network error, its lookup's timeout or the
58
+ * egress policy), which counts as a complete disallow but is no rule the
59
+ * publisher wrote; null when it was not read before the deadline.
60
+ */
61
+ robots: 'allowed' | 'no_robots' | 'disallowed' | 'unreachable' | null;
62
+ /** Why robots.txt could not be read (`server_error`, `network_error`, `timeout`, ...); present only with robots `unreachable`. */
63
+ robotsUnreachable?: string;
64
+ rawBodySha256: string | null;
65
+ /** http(s) links the page holds, each once, before scope, robots and limit. */
66
+ linksFound: number;
67
+ title: string | null;
68
+ description: string | null;
69
+ }
70
+ export interface MapSitemapSource {
71
+ mode: SitemapMode;
72
+ sources: SitemapSourceKind[];
73
+ files: SitemapFileRecord[];
74
+ /** Entries the reader offered the map. */
75
+ listed: number;
76
+ /** Entries the map returned as new links (a URL already found on the page is merged, not counted). */
77
+ accepted: number;
78
+ truncated: 'files' | 'urls' | 'time' | null;
79
+ /** Why the load itself failed, when it threw; null otherwise. */
80
+ error: string | null;
81
+ }
82
+ /** Candidates the map did not return, by reason, with samples. */
83
+ export interface MapRefused {
84
+ /** The same URL again (merged into the link it repeats when that link was returned). */
85
+ duplicate: number;
86
+ /** A variant folded into a URL seen first (`/a/` after `/a`, the www twin, ...), or an http link replaced by its https variant; `samples.collapsed` names both, `into` being the URL returned. */
87
+ collapsed: number;
88
+ hostDenied: number;
89
+ subtreeDenied: number;
90
+ pathDenied: number;
91
+ assetDenied: number;
92
+ /** Disallowed by robots.txt, or on a host whose robots.txt was unreachable or denied by the egress policy. */
93
+ robots: number;
94
+ /** On a host whose robots.txt was not read: past MAP_MAX_ROBOTS_HOSTS, or after the deadline. */
95
+ robotsUnchecked: number;
96
+ /** In scope but left out by `search`: not every word is in the URL or the title in hand. Counted before robots.txt and `limit`. */
97
+ searchFiltered: number;
98
+ /** Accepted candidates offered after `limit` links were in hand. */
99
+ overLimit: number;
100
+ samples: {
101
+ collapsed: Array<{
102
+ url: string;
103
+ into: string;
104
+ }>;
105
+ hostDenied: string[];
106
+ robots: string[];
107
+ };
108
+ }
109
+ export declare const MAP_WARNING_CODES: readonly ["map_timeout", "start_page_unreadable", "start_page_client_rendered", "sitemap_unreadable", "sitemap_files_capped", "robots_host_cap", "robots_unreachable", "map_record_unwritten"];
110
+ export type MapWarningCode = (typeof MAP_WARNING_CODES)[number];
111
+ export interface MapWarning {
112
+ code: MapWarningCode;
113
+ message: string;
114
+ }
115
+ export interface MapResponse {
116
+ /** The map run's id: its record key (GET /v1/maps/:id). */
117
+ id: string;
118
+ url: string;
119
+ /**
120
+ * `completed`: every source asked for was read or is definitively absent,
121
+ * and the deadline did not cut the run (reaching `limit` is completed).
122
+ * `partial`: links came back, but the deadline cut the run or a source
123
+ * failed. `failed`: no links, and a source failed or the deadline fired.
124
+ */
125
+ status: MapStatus;
126
+ stoppedBy: 'limit' | 'timeout' | null;
127
+ /** The start URL, then the start page's links in document order, then sitemap entries not already present, in listed order. */
128
+ links: MapLink[];
129
+ sources: {
130
+ startPage: MapStartPage | null;
131
+ sitemap: MapSitemapSource | null;
132
+ };
133
+ refused: MapRefused;
134
+ /** The declared identity robots.txt, the start page and the sitemaps were read under. */
135
+ identity: {
136
+ mode: CrawlMode;
137
+ userAgent: string;
138
+ };
139
+ warnings: MapWarning[];
140
+ agentHints?: AgentHints;
141
+ elapsedMs: number;
142
+ }
143
+ /** One map call as stored under `<taskRoot>/maps/<id>.json` and read back by GET /v1/maps/:id. */
144
+ export interface MapRecord {
145
+ requestedAt: string;
146
+ /** The request as parsed, `origin` and `integration` included. */
147
+ request: MapRequest;
148
+ response: MapResponse;
149
+ }
150
+ /** One link of the start page: its absolute URL (fragment stripped) and the first non-empty text an anchor gave it, or null. */
151
+ export interface MapPageLink {
152
+ url: string;
153
+ text: string | null;
154
+ }
155
+ /** What the start page read gave a map. */
156
+ export interface MapStartPageRead {
157
+ result: Pick<FetchResult, 'status' | 'failureReason' | 'metadata'> & {
158
+ evidence: Pick<Evidence, 'finalUrl' | 'httpStatus' | 'rawBodySha256'>;
159
+ };
160
+ links: readonly MapPageLink[];
161
+ /** The http lane found the page filled by script, so its links may be incomplete. */
162
+ clientRendered?: boolean;
163
+ }
164
+ /** A robots.txt verdict for one URL under the map's declared identity. */
165
+ export type MapRobotsVerdict = 'allowed' | 'no_robots' | {
166
+ disallowed: true;
167
+ unreachable?: string;
168
+ };
169
+ /** The fetch paths a map uses; the API engine wires the real ones, tests inject fakes. */
170
+ export interface MapSources {
171
+ /** The start URL on the http rung alone; never a browser. */
172
+ readStartPage(url: string, context: ExecutionContext): Promise<MapStartPageRead>;
173
+ sitemap: SitemapSource;
174
+ robotsVerdict(url: string, context: ExecutionContext): Promise<MapRobotsVerdict>;
175
+ identity: {
176
+ mode: CrawlMode;
177
+ userAgent: string;
178
+ };
179
+ }
180
+ //# sourceMappingURL=map.d.ts.map
@@ -0,0 +1,217 @@
1
+ import type { ScrapeOutcome } from './crawl.js';
2
+ /** Deliberately one public document adapter for the first B1/B2 slice. */
3
+ export declare const FIRECRAWL_INTRO_URL = "https://docs.firecrawl.dev/introduction";
4
+ export declare const FIRECRAWL_MONITOR_ID = "firecrawl-introduction";
5
+ export declare const DOCUMENT_RULE_VERSION = "firecrawl-introduction/v1";
6
+ export type FieldValue<T = string | boolean> = {
7
+ state: 'present';
8
+ value: T;
9
+ evidenceRefs: string[];
10
+ } | {
11
+ state: 'explicit_null';
12
+ reason: string;
13
+ evidenceRefs: string[];
14
+ } | {
15
+ state: 'unobserved';
16
+ reason: string;
17
+ } | {
18
+ state: 'conflicting';
19
+ candidates: {
20
+ value: T;
21
+ evidenceRefs: string[];
22
+ }[];
23
+ } | {
24
+ state: 'redacted';
25
+ reason: string;
26
+ };
27
+ export type DocumentFields = Record<string, string | FieldValue>;
28
+ export interface MonitorFieldRule {
29
+ name: string;
30
+ heading: string;
31
+ type: 'text' | 'code' | 'decimal' | 'boolean';
32
+ required: boolean;
33
+ nullMarker?: string;
34
+ redactedMarker?: string;
35
+ /** Changes in field meaning are schema migrations, not price changes. */
36
+ unit?: string;
37
+ currency?: string;
38
+ }
39
+ export interface DocumentMonitorConfig {
40
+ adapter: 'markdown-sections/v1';
41
+ workspaceId: string;
42
+ entityKey: string;
43
+ viewKey: string;
44
+ expectedTitle: string;
45
+ schemaVersion: string;
46
+ fields: MonitorFieldRule[];
47
+ conditionalRequests: boolean;
48
+ /** Acquisition capability is independent of transport caching. Defaults to ladder. */
49
+ captureMode?: 'http' | 'ladder';
50
+ }
51
+ export type DocumentField = keyof DocumentFields;
52
+ export interface FieldEvidence {
53
+ field: DocumentField;
54
+ start: number;
55
+ end: number;
56
+ quote: string;
57
+ }
58
+ export interface DocumentAssessment {
59
+ ruleVersion: string;
60
+ quality: 'valid' | 'partial' | 'invalid' | 'unknown';
61
+ reasons: string[];
62
+ fields: DocumentFields | null;
63
+ evidence: FieldEvidence[];
64
+ }
65
+ export interface MonitorRevision {
66
+ monitorId: string;
67
+ revision: number;
68
+ url: string;
69
+ ruleVersion: string;
70
+ config?: DocumentMonitorConfig;
71
+ intervalMs: number;
72
+ staleAfterMs: number;
73
+ createdAt: number;
74
+ }
75
+ export type MonitorChange = 'initialized' | 'changed' | 'unchanged' | 'cannot_verify';
76
+ /**
77
+ * Why a committed run moved the baseline. `extraction_reprocessed`: W2L's
78
+ * reading changed, either a revised rule or fields that changed over a
79
+ * byte-identical raw body (an extractor upgrade).
80
+ */
81
+ export type MonitorChangeReason = 'source_changed' | 'initialized' | 'extraction_reprocessed' | 'schema_migrated';
82
+ export interface DocumentDiff {
83
+ field: DocumentField;
84
+ before: string | FieldValue | null;
85
+ after: string | FieldValue | null;
86
+ }
87
+ export interface MonitorRun {
88
+ id: string;
89
+ monitorId: string;
90
+ revision: number;
91
+ triggerKey: string;
92
+ state: 'queued' | 'running' | 'waiting_retry' | 'completed' | 'failed' | 'cancelled' | 'expired';
93
+ epoch: number;
94
+ fencingToken: number;
95
+ attemptId: string | null;
96
+ leaseUntil: number | null;
97
+ deadlineAt: number | null;
98
+ nextAttemptAt?: number | null;
99
+ expectedBaselineId: string | null;
100
+ createdAt: number;
101
+ endedAt: number | null;
102
+ quality: DocumentAssessment['quality'] | null;
103
+ change: MonitorChange | null;
104
+ /**
105
+ * Set when the run initialized or changed the baseline, also when no event
106
+ * was emitted for it (an extractor upgrade over an unchanged raw body). Null
107
+ * otherwise and for runs committed before W2L recorded it.
108
+ */
109
+ changeReason?: MonitorChangeReason | null;
110
+ error: string | null;
111
+ }
112
+ export interface MonitorAttempt {
113
+ id: string;
114
+ runId: string;
115
+ fencingToken: number;
116
+ state: 'running' | 'succeeded' | 'failed' | 'interrupted' | 'cancelled';
117
+ startedAt: number;
118
+ endedAt: number | null;
119
+ recoveredFromAttemptId: string | null;
120
+ }
121
+ export interface MonitorObservation {
122
+ id: string;
123
+ runId: string;
124
+ attemptId: string;
125
+ observedAt: number;
126
+ clientWallMs: number;
127
+ markdownSha256: string | null;
128
+ /** sha256 of the raw body the assessed Markdown came from; after a 304, the reused body's. Null or absent when unknown. */
129
+ rawBodySha256?: string | null;
130
+ /** EXTRACTOR_VERSION of the extractor that produced that Markdown. Null or absent when unknown: rows and cached bodies from before W2L recorded it. */
131
+ extractorVersion?: string | null;
132
+ transport?: {
133
+ etag: string | null;
134
+ lastModified: string | null;
135
+ representationKey: string;
136
+ reusedFrom?: string;
137
+ responseStatus?: number | null;
138
+ } | null;
139
+ outcome: ScrapeOutcome | null;
140
+ error: string | null;
141
+ }
142
+ export interface MonitorSnapshot {
143
+ id: string;
144
+ monitorId: string;
145
+ revision: number;
146
+ version: number;
147
+ observationId: string;
148
+ assessmentId: string;
149
+ fields: DocumentFields;
150
+ /** The extractor that produced these fields; null for snapshots committed before W2L recorded it. */
151
+ extractorVersion?: string | null;
152
+ createdAt: number;
153
+ }
154
+ export interface MonitorEvent {
155
+ id: string;
156
+ runId: string;
157
+ monitorId: string;
158
+ kind: 'initialized' | 'changed';
159
+ reason: MonitorChangeReason;
160
+ fromSnapshotId: string | null;
161
+ toSnapshotId: string;
162
+ /**
163
+ * Present when the extractor behind this snapshot differs from the one behind
164
+ * the previous snapshot: part of `changes` may then come from W2L, not the
165
+ * source. `from: null` means the previous snapshot predates recorded versions.
166
+ */
167
+ extractorChange?: {
168
+ from: string | null;
169
+ to: string | null;
170
+ };
171
+ changes: DocumentDiff[];
172
+ observedAt: number;
173
+ }
174
+ export interface MonitorView {
175
+ revision: MonitorRevision;
176
+ enabled: boolean;
177
+ controlEpoch: number;
178
+ nextRunAt: number;
179
+ lastCheckedAt: number | null;
180
+ lastVerifiedAt: number | null;
181
+ freshness: 'fresh' | 'stale';
182
+ baseline: MonitorSnapshot | null;
183
+ runs: MonitorRun[];
184
+ events: MonitorEvent[];
185
+ outbox: {
186
+ eventId: string;
187
+ state: 'pending' | 'acknowledged';
188
+ acknowledgedAt: number | null;
189
+ }[];
190
+ }
191
+ /** A bounded first-use sample. Preview never creates a run, baseline, or event. */
192
+ export interface MonitorPreview {
193
+ url: string;
194
+ finalUrl: string | null;
195
+ status: string;
196
+ assessment: DocumentAssessment;
197
+ sampleMarkdown: string | null;
198
+ capturedAt: number;
199
+ }
200
+ export interface MonitorRunDetail {
201
+ run: MonitorRun;
202
+ assessment: DocumentAssessment | null;
203
+ observation: MonitorObservation | null;
204
+ attempts: MonitorAttempt[];
205
+ }
206
+ export interface TransportRepresentation {
207
+ bodySha256?: string;
208
+ key: string;
209
+ url: string;
210
+ etag: string | null;
211
+ lastModified: string | null;
212
+ outcome: ScrapeOutcome;
213
+ /** EXTRACTOR_VERSION that produced `outcome.result.markdown`; absent on bodies cached before W2L recorded it. */
214
+ extractorVersion?: string;
215
+ storedAt: number;
216
+ }
217
+ //# sourceMappingURL=monitor.d.ts.map
@@ -0,0 +1,9 @@
1
+ import type { MonitorRevision } from './monitor.js';
2
+ export declare function parseMonitorRevision(input: unknown): MonitorRevision;
3
+ export declare function monitorIdentity(r: MonitorRevision): {
4
+ resourceKey: string;
5
+ viewKey: string;
6
+ workspaceId: string;
7
+ entityKey: string;
8
+ };
9
+ //# sourceMappingURL=monitorConfig.d.ts.map
@@ -0,0 +1,93 @@
1
+ /**
2
+ * Network egress policy.
3
+ *
4
+ * Contract: there is no per-request "allow private network" flag. Private-range
5
+ * access requires an explicit host/CIDR allowlist that only a server operator can
6
+ * set (config file or env at process start). API callers cannot widen it.
7
+ */
8
+ export type PolicyOrigin =
9
+ /** Set by the operator at process start. The only origin allowed to permit private ranges. */
10
+ 'operator'
11
+ /** Derived from an API request. May narrow the effective policy, never widen it. */
12
+ | 'request';
13
+ export interface NetworkPolicy {
14
+ origin: PolicyOrigin;
15
+ /**
16
+ * Hosts and CIDRs exempt from private-range blocking.
17
+ * Entries are literal hostnames, IPs, or CIDR blocks (e.g. '127.0.0.1/32').
18
+ * Ignored — and a violation — when origin is 'request'.
19
+ */
20
+ privateAllowlist: readonly string[];
21
+ maxRedirects: number;
22
+ maxBodyBytes: number;
23
+ /**
24
+ * Cap on a file saved as received (PDF, CSV, XLSX, ZIP, JSON, text), in
25
+ * place of maxBodyBytes: the operator's `W2L_MAX_FILE_BYTES`. Absent means
26
+ * DEFAULT_MAX_FILE_BYTES; a request may only lower it (`maxFileBytes`).
27
+ */
28
+ maxFileBytes?: number;
29
+ /** Cap on post-decompression size, to bound zip bombs. */
30
+ maxDecompressedBytes: number;
31
+ /** Per-host concurrent request ceiling. */
32
+ perHostConcurrency: number;
33
+ /** Minimum delay between requests to the same host. */
34
+ perHostMinDelayMs: number;
35
+ respectRobotsTxt: boolean;
36
+ /** How long one robots.txt lookup may take before the file counts as unreachable. Default 5000. */
37
+ robotsTimeoutMs?: number;
38
+ /**
39
+ * How long an unreachable robots.txt (a 5xx, a network error or a lookup
40
+ * timeout) stays a complete disallow for its origin before it is fetched
41
+ * again. Other robots.txt results are kept for the life of the process.
42
+ * Default 300000 (5 minutes).
43
+ */
44
+ robotsUnreachableTtlMs?: number;
45
+ /**
46
+ * The operator's forward proxy, read from the standard environment
47
+ * variables by local-mode entry points (`withEnvironmentProxy`). Hosted
48
+ * policies never carry one. A proxied host is resolved by the proxy, so only
49
+ * the literal host checks apply to it (`proxyFor`). Ignored unless origin
50
+ * is 'operator'.
51
+ */
52
+ egressProxy?: EgressProxy | null;
53
+ /**
54
+ * The operator's contact from `W2L_CONTACT` (`withOperatorContact`), which
55
+ * research mode declares in its User-Agent. Absent or null declares none.
56
+ */
57
+ contact?: string | null;
58
+ }
59
+ /**
60
+ * Where outbound requests go when the operator's environment names a proxy:
61
+ * one proxy per URL scheme, and NO_PROXY entries that go direct. Loopback
62
+ * always goes direct. `https` and `http` are equal or one of them is null,
63
+ * because the browser lane can route through only one proxy.
64
+ */
65
+ export interface EgressProxy {
66
+ source: 'environment';
67
+ /** From HTTPS_PROXY / https_proxy. Null sends https: URLs direct. */
68
+ https: ProxyServer | null;
69
+ /** From HTTP_PROXY / http_proxy. Null sends http: URLs direct. */
70
+ http: ProxyServer | null;
71
+ /** NO_PROXY / no_proxy entries, lower-cased. */
72
+ noProxy: readonly string[];
73
+ }
74
+ export interface ProxyServer {
75
+ /** `http://host:port` or `https://host:port`, without credentials. */
76
+ url: string;
77
+ /** `host:port`: the only form of the proxy that results record. */
78
+ endpoint: string;
79
+ /** Credentials from the variable's userinfo. Held in memory for the proxy, never recorded or logged. */
80
+ username?: string;
81
+ password?: string;
82
+ }
83
+ export declare const DEFAULT_NETWORK_POLICY: NetworkPolicy;
84
+ export declare const POLICY_VIOLATION: readonly ["private_address", "loopback_address", "link_local_address", "cloud_metadata_address", "unspecified_address", "unsupported_scheme", "malformed_url", "privilege_escalation"];
85
+ export type PolicyViolation = (typeof POLICY_VIOLATION)[number];
86
+ export interface PolicyDecision {
87
+ allowed: boolean;
88
+ violation: PolicyViolation | null;
89
+ /** The IP the hostname resolved to, pinned for the actual connection. */
90
+ pinnedAddress: string | null;
91
+ detail: string | null;
92
+ }
93
+ //# sourceMappingURL=policy.d.ts.map
@@ -0,0 +1,52 @@
1
+ /**
2
+ * The operator's forward proxy from the standard environment variables, for
3
+ * local mode only. Hosted mode never reads them: its SSRF guarantees depend
4
+ * on direct connections to DNS-pinned addresses.
5
+ *
6
+ * Semantics follow curl. HTTPS_PROXY serves https: URLs and HTTP_PROXY http:
7
+ * URLs (the lower-case names win). A NO_PROXY entry matches its host and
8
+ * every subdomain (a leading "." or "*." is ignored), `*` disables the proxy,
9
+ * and `host:port` limits an entry to one port. IP and CIDR entries match IP
10
+ * literal hosts only: nothing here resolves a name. Loopback always goes
11
+ * direct. `W2L_PROXY=off` ignores the variables.
12
+ *
13
+ * The proxy resolves the hosts it is asked for, so local mode trusts the
14
+ * operator's proxy for resolution: before a proxied request only the checks
15
+ * that need no DNS run (scheme, credentials, IP literals, metadata names).
16
+ * Direct requests keep the resolve-validate-pin checks.
17
+ */
18
+ import type { EgressProxy, NetworkPolicy, ProxyServer } from './policy.js';
19
+ export declare const PROXY_ENV_NAMES: readonly ["HTTPS_PROXY", "https_proxy", "HTTP_PROXY", "http_proxy", "NO_PROXY", "no_proxy"];
20
+ type Env = Readonly<Record<string, string | undefined>>;
21
+ export declare class ProxyConfigError extends Error {
22
+ constructor(message: string);
23
+ }
24
+ /** The proxy the environment names, or null when none applies. Error messages never repeat a variable's value. */
25
+ export declare function environmentProxy(env: Env): EgressProxy | null;
26
+ /** A local-mode policy that routes through the environment's proxy, when one is set. */
27
+ export declare function withEnvironmentProxy(policy: NetworkPolicy, env: Env): NetworkPolicy;
28
+ /** The operator proxy a URL leaves through under this policy; null means a direct connection. */
29
+ export declare function proxyFor(url: string | URL, policy: Pick<NetworkPolicy, 'origin' | 'egressProxy'>): ProxyServer | null;
30
+ export type NoProxyEntry = {
31
+ kind: 'all';
32
+ } | {
33
+ kind: 'host';
34
+ host: string;
35
+ port: number | null;
36
+ } | {
37
+ kind: 'ip';
38
+ address: string;
39
+ cidr: string;
40
+ port: number | null;
41
+ } | {
42
+ kind: 'cidr';
43
+ cidr: string;
44
+ };
45
+ /** One NO_PROXY entry, or null for a form W2L does not match (such as an inner wildcard). */
46
+ export declare function parseNoProxyEntry(raw: string): NoProxyEntry | null;
47
+ /** Startup line for local mode: which proxy outbound requests use and what goes direct. */
48
+ export declare function describeEgressProxy(proxy: EgressProxy): string;
49
+ /** Startup line for hosted mode when proxy variables are set: they are ignored. */
50
+ export declare function hostedProxyNotice(env: Env): string | null;
51
+ export {};
52
+ //# sourceMappingURL=proxy.d.ts.map
@@ -0,0 +1,74 @@
1
+ /** Operator-reviewed, data-only recipes. No script/evaluate/write operation. */
2
+ export interface RecipeLocator {
3
+ css: string;
4
+ }
5
+ export interface RecipeCondition {
6
+ target: RecipeLocator;
7
+ text: string;
8
+ }
9
+ export type RecipeStep = {
10
+ id: string;
11
+ op: 'navigate';
12
+ url: string;
13
+ } | {
14
+ id: string;
15
+ op: 'assertAccount';
16
+ } | {
17
+ id: string;
18
+ op: 'fill';
19
+ target: RecipeLocator;
20
+ value: string;
21
+ } | {
22
+ id: string;
23
+ op: 'query';
24
+ target: RecipeLocator;
25
+ postcondition: RecipeCondition;
26
+ request: {
27
+ method: 'GET' | 'POST';
28
+ path: string;
29
+ };
30
+ } | {
31
+ id: string;
32
+ op: 'extract';
33
+ field: string;
34
+ target: RecipeLocator;
35
+ } | {
36
+ id: string;
37
+ op: 'download';
38
+ target: RecipeLocator;
39
+ field: string;
40
+ };
41
+ export interface BackendRecipe {
42
+ id: string;
43
+ revision: number;
44
+ origin: string;
45
+ account: RecipeLocator;
46
+ login: RecipeLocator;
47
+ steps: RecipeStep[];
48
+ limits: {
49
+ timeoutMs: number;
50
+ maxSteps: number;
51
+ maxOutputBytes: number;
52
+ maxDownloadBytes: number;
53
+ };
54
+ }
55
+ export type RecipeState = 'running' | 'completed' | 'failed' | 'waiting_user' | 'effect_unknown' | 'cancelled';
56
+ export interface RecipeRun {
57
+ id: string;
58
+ recipeId: string;
59
+ revision: number;
60
+ recipeHash: string;
61
+ sessionRef: string;
62
+ workspaceId: string;
63
+ accountRef: string;
64
+ triggerKey: string;
65
+ grantEpoch: number;
66
+ state: RecipeState;
67
+ attempt: number;
68
+ createdAt: number;
69
+ updatedAt: number;
70
+ reason: string | null;
71
+ output: Record<string, string>;
72
+ }
73
+ export declare function validateRecipe(value: unknown): BackendRecipe;
74
+ //# sourceMappingURL=recipe.d.ts.map
@@ -0,0 +1,31 @@
1
+ /**
2
+ * A static check for regular expressions a caller supplies and W2L runs, on
3
+ * a backtracking engine, against text a web page controls: crawl
4
+ * `includePaths` / `excludePaths`, and JSON Schema `pattern`. Such a pattern
5
+ * can take time exponential or polynomial in the text's length (ReDoS).
6
+ * `unsafeRegexReason` refuses the shapes that cause it:
7
+ *
8
+ * - a repeated group with a repeated or optional part inside, unless each
9
+ * repetition is delimited by a character none of those parts can match:
10
+ * `(a+)+`, `(\w+\s?)*` and `(.*,)*` are refused,
11
+ * `[a-z0-9]+(?:-[a-z0-9]+)*` is not;
12
+ * - a repeated group whose alternatives can start with the same character:
13
+ * `(a|ab)*` and `(\d|\w)+` are refused, `(?:ab|cd)+` and `(?:[a-z]|-)+`
14
+ * are not;
15
+ * - three or more variable parts in a row that can take the same characters,
16
+ * counting the start of an unanchored pattern, which is tried at every
17
+ * position: `.*a.*b`, `^\d+\d+\d+x` and `^a?a?a?aaa` are refused,
18
+ * `^.*a.*b`, `.*\.pdf$` and `.*blog.*` are not.
19
+ *
20
+ * A pattern it accepts backtracks at most quadratically in the text's
21
+ * length, so a caller that runs it on the backtracking engine caps the text
22
+ * at REGEX_SUBJECT_MAX_LENGTH characters (a few milliseconds per match). The
23
+ * check reads JavaScript syntax and is conservative: what it cannot place (a
24
+ * backreference, a Unicode property) counts as matching any character. It
25
+ * assumes the pattern compiles; check that first.
26
+ */
27
+ /** The longest text a pattern unsafeRegexReason accepts should be matched against on a backtracking engine. */
28
+ export declare const REGEX_SUBJECT_MAX_LENGTH = 2048;
29
+ /** Why `pattern` can backtrack catastrophically, or null when it cannot. */
30
+ export declare function unsafeRegexReason(pattern: string): string | null;
31
+ //# sourceMappingURL=regexSafety.d.ts.map