@octocrawl/sdk 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +17 -0
  3. package/dist/index.cjs +1278 -0
  4. package/dist/index.js +1240 -0
  5. package/dist/types/cjs/client.d.ts +362 -0
  6. package/dist/types/cjs/contracts/access.d.ts +166 -0
  7. package/dist/types/cjs/contracts/actions.d.ts +191 -0
  8. package/dist/types/cjs/contracts/api.d.ts +891 -0
  9. package/dist/types/cjs/contracts/benchmark.d.ts +116 -0
  10. package/dist/types/cjs/contracts/checkpoint.d.ts +164 -0
  11. package/dist/types/cjs/contracts/compliance.d.ts +412 -0
  12. package/dist/types/cjs/contracts/crawl.d.ts +302 -0
  13. package/dist/types/cjs/contracts/delivery.d.ts +136 -0
  14. package/dist/types/cjs/contracts/evidenceRecord.d.ts +192 -0
  15. package/dist/types/cjs/contracts/execution.d.ts +197 -0
  16. package/dist/types/cjs/contracts/extractor.d.ts +379 -0
  17. package/dist/types/cjs/contracts/file.d.ts +117 -0
  18. package/dist/types/cjs/contracts/firecrawl.d.ts +258 -0
  19. package/dist/types/cjs/contracts/groundTruth.d.ts +77 -0
  20. package/dist/types/cjs/contracts/identityBundle.d.ts +70 -0
  21. package/dist/types/cjs/contracts/index.d.ts +30 -0
  22. package/dist/types/cjs/contracts/map.d.ts +180 -0
  23. package/dist/types/cjs/contracts/monitor.d.ts +217 -0
  24. package/dist/types/cjs/contracts/monitorConfig.d.ts +9 -0
  25. package/dist/types/cjs/contracts/policy.d.ts +93 -0
  26. package/dist/types/cjs/contracts/proxy.d.ts +52 -0
  27. package/dist/types/cjs/contracts/recipe.d.ts +74 -0
  28. package/dist/types/cjs/contracts/regexSafety.d.ts +31 -0
  29. package/dist/types/cjs/contracts/result.d.ts +503 -0
  30. package/dist/types/cjs/contracts/session.d.ts +51 -0
  31. package/dist/types/cjs/contracts/ssrf.d.ts +16 -0
  32. package/dist/types/cjs/contracts/status.d.ts +27 -0
  33. package/dist/types/cjs/contracts/structured.d.ts +185 -0
  34. package/dist/types/cjs/contracts/tableMarkdown.d.ts +58 -0
  35. package/dist/types/cjs/contracts/tokens.d.ts +47 -0
  36. package/dist/types/cjs/index.d.ts +9 -0
  37. package/dist/types/cjs/package.json +1 -0
  38. package/dist/types/cjs/version.d.ts +8 -0
  39. package/dist/types/cjs/watcher.d.ts +151 -0
  40. package/dist/types/esm/client.d.ts +362 -0
  41. package/dist/types/esm/contracts/access.d.ts +166 -0
  42. package/dist/types/esm/contracts/actions.d.ts +191 -0
  43. package/dist/types/esm/contracts/api.d.ts +891 -0
  44. package/dist/types/esm/contracts/benchmark.d.ts +116 -0
  45. package/dist/types/esm/contracts/checkpoint.d.ts +164 -0
  46. package/dist/types/esm/contracts/compliance.d.ts +412 -0
  47. package/dist/types/esm/contracts/crawl.d.ts +302 -0
  48. package/dist/types/esm/contracts/delivery.d.ts +136 -0
  49. package/dist/types/esm/contracts/evidenceRecord.d.ts +192 -0
  50. package/dist/types/esm/contracts/execution.d.ts +197 -0
  51. package/dist/types/esm/contracts/extractor.d.ts +379 -0
  52. package/dist/types/esm/contracts/file.d.ts +117 -0
  53. package/dist/types/esm/contracts/firecrawl.d.ts +258 -0
  54. package/dist/types/esm/contracts/groundTruth.d.ts +77 -0
  55. package/dist/types/esm/contracts/identityBundle.d.ts +70 -0
  56. package/dist/types/esm/contracts/index.d.ts +30 -0
  57. package/dist/types/esm/contracts/map.d.ts +180 -0
  58. package/dist/types/esm/contracts/monitor.d.ts +217 -0
  59. package/dist/types/esm/contracts/monitorConfig.d.ts +9 -0
  60. package/dist/types/esm/contracts/policy.d.ts +93 -0
  61. package/dist/types/esm/contracts/proxy.d.ts +52 -0
  62. package/dist/types/esm/contracts/recipe.d.ts +74 -0
  63. package/dist/types/esm/contracts/regexSafety.d.ts +31 -0
  64. package/dist/types/esm/contracts/result.d.ts +503 -0
  65. package/dist/types/esm/contracts/session.d.ts +51 -0
  66. package/dist/types/esm/contracts/ssrf.d.ts +16 -0
  67. package/dist/types/esm/contracts/status.d.ts +27 -0
  68. package/dist/types/esm/contracts/structured.d.ts +185 -0
  69. package/dist/types/esm/contracts/tableMarkdown.d.ts +58 -0
  70. package/dist/types/esm/contracts/tokens.d.ts +47 -0
  71. package/dist/types/esm/index.d.ts +9 -0
  72. package/dist/types/esm/version.d.ts +8 -0
  73. package/dist/types/esm/watcher.d.ts +151 -0
  74. package/package.json +40 -0
@@ -0,0 +1,362 @@
1
+ import type { ActiveCrawlList, CrawlAccepted, DeliveryDestination, DeliveryDestinationInput, DeliveryDetail, DeliveryQuery, DeliveryPage, DeliveryPageQuery, WebhookDelivery, CrawlError, CrawlPage, CrawlPageList, CrawlPageQuery, CrawlReport, CrawlStartRequest, BatchAccepted, BatchErrorsQuery, BatchErrorsResponse, BatchStartRequest, BatchStatusResponse, BatchHandoffRequest, BatchHandoffResponse, LoginImportRequest, LoginImportResponse, SavedLogin, CompactScrapeResponse, MapRecord, MapRequest, MapResponse, MonitorRevision, MonitorView, MonitorPreview, MonitorRun, MonitorRunDetail, ScrapeRecord, ScrapeRequest, ScrapeResponse } from './contracts/index.js';
2
+ import { RATE_LIMITED_CODE, type ApiErrorCode } from './contracts/index.js';
3
+ import { JobWatcher, type WatchOptions } from './watcher.js';
4
+ export interface W2LOptions {
5
+ baseUrl: string;
6
+ /**
7
+ * Bearer token for a server started with a token. Omitted, the
8
+ * W2L_API_TOKEN environment variable is used where there is one; '' sends
9
+ * no token.
10
+ */
11
+ token?: string;
12
+ fetch?: typeof fetch;
13
+ }
14
+ /** A scrape's options as a caller sends them: `handoff` as `true` or `{ waitMs }` (the API reads `true` as `{}`). */
15
+ export type ScrapeOptions = Omit<ScrapeRequest, 'url' | 'handoff'> & {
16
+ handoff?: boolean | {
17
+ waitMs?: number;
18
+ };
19
+ };
20
+ /** How long past a map's own deadline (its `timeout`, 60 000 ms by default) the SDK waits for the API's answer: Firecrawl's timeout + 5000 convention. */
21
+ export declare const MAP_ANSWER_MARGIN_MS = 5000;
22
+ /**
23
+ * Cancels the HTTP request and the current execution of synchronous scrape/runMonitor.
24
+ * Use cancelCrawl for an already-created background crawl, or cancelMonitorRun for explicit persisted run control.
25
+ */
26
+ export interface RequestOptions {
27
+ signal?: AbortSignal;
28
+ /**
29
+ * The `origin` a scrape, crawl or batch is recorded under when its options
30
+ * set none: a host built on the SDK (the MCP server) names its own client
31
+ * here. The default is `js-sdk@<SDK_VERSION>`.
32
+ */
33
+ origin?: string;
34
+ }
35
+ /** What the SDK records as `origin` unless the caller or the host says otherwise. */
36
+ export declare const SDK_ORIGIN = "js-sdk@0.3.0";
37
+ /**
38
+ * Polling for waitBatch and waitCrawl. A status request that fails with a
39
+ * network error, HTTP 408, 429 or 5xx is retried: after 1, 2, 4, 8, then
40
+ * 10 s, or after the response's Retry-After when it asks for a minute or
41
+ * less. Any other error ends the wait at once.
42
+ */
43
+ export interface WaitOptions extends RequestOptions {
44
+ /** Delay between status requests. Default 500 ms. */
45
+ pollIntervalMs?: number;
46
+ /**
47
+ * Stop waiting after this long, a status request in flight or a retry's
48
+ * wait included, and throw a WaitTimeoutError. Default: no limit. The task
49
+ * keeps running.
50
+ */
51
+ timeoutMs?: number;
52
+ /** Consecutive failed status requests retried before the last error is thrown. Default 5; 0 retries none. */
53
+ maxRetries?: number;
54
+ }
55
+ /**
56
+ * The task was still unfinished when the wait's timeoutMs ran out. `last` is
57
+ * the final status read, null when no status request answered in time;
58
+ * `cause` is the error of the last status request that failed, when no
59
+ * status was read after it.
60
+ */
61
+ export declare class WaitTimeoutError<T extends {
62
+ status: string;
63
+ } = {
64
+ status: string;
65
+ }> extends Error {
66
+ readonly taskId: string;
67
+ readonly last: T | null;
68
+ readonly timeoutMs: number;
69
+ readonly name = "WaitTimeoutError";
70
+ constructor(taskId: string, last: T | null, timeoutMs: number, options?: {
71
+ cause?: unknown;
72
+ });
73
+ }
74
+ /** What `appendToBatch` may send beside the URLs: what binds to them, and the attribution labels; the job's own options stay as they are. */
75
+ export type AppendToBatchOptions = Pick<BatchStartRequest, 'ignoreInvalidURLs' | 'idempotencyKey' | 'robotsOverrides' | 'origin' | 'integration'>;
76
+ /** How `batchScrapeChunked` splits a list and waits for each job (Python's `process_large_batch`: `chunk_size`, `poll_interval`, `timeout` are `chunkSize`, `pollIntervalMs`, `timeoutMs`). */
77
+ export interface ChunkedBatchOptions extends WaitOptions {
78
+ /** URLs per job, 1 to 1000. Default 100. */
79
+ chunkSize?: number;
80
+ /** Items per request while a job's items are listed, 1 to 50. Default 50. */
81
+ itemLimit?: number;
82
+ }
83
+ /** One job `batchScrapeChunked` ran: its id, how many URLs it was given, its final status and the entries it skipped. */
84
+ export interface ChunkedBatchJob {
85
+ taskId: string;
86
+ urls: number;
87
+ report: BatchStatusResponse;
88
+ invalidURLs?: string[];
89
+ }
90
+ /** `batchScrapeChunked`'s result: the jobs in submission order, every item of every job in the order the URLs were submitted, and every entry `ignoreInvalidURLs` skipped. */
91
+ export interface ChunkedBatchResult {
92
+ jobs: ChunkedBatchJob[];
93
+ items: CrawlPage[];
94
+ invalidURLs: string[];
95
+ }
96
+ /** crawlAndWait's result: the final status, every page and every error of the crawl's latest attempt. */
97
+ export interface CrawlCollected {
98
+ taskId: string;
99
+ report: CrawlReport;
100
+ pages: CrawlPage[];
101
+ errors: CrawlError[];
102
+ }
103
+ /** batchAndWait's result: the final status and every item, failed ones included. */
104
+ export interface BatchCollected {
105
+ taskId: string;
106
+ report: BatchStatusResponse;
107
+ items: CrawlPage[];
108
+ }
109
+ /**
110
+ * Caps on a listing that follows cursors (listCrawlPages, listBatchItems,
111
+ * getCrawlDocuments, getBatchDocuments). None by default: the listing reads to
112
+ * the end. With `maxResults` each page is requested no larger than what is
113
+ * still wanted, so the cursor the listing stops at continues exactly where it
114
+ * left off.
115
+ */
116
+ export interface PaginationLimits {
117
+ /** Pages read after the first; 0 reads the first page alone. */
118
+ maxPages?: number;
119
+ /** Items returned in all, at least 1. */
120
+ maxResults?: number;
121
+ /** Once this many milliseconds have passed since the first page was requested, no further page is requested. */
122
+ maxWaitMs?: number;
123
+ }
124
+ /** Why a bounded listing stopped: the last page (`end`), or the limit that stopped it with pages still unread. */
125
+ export type PaginationStop = 'end' | 'maxPages' | 'maxResults' | 'maxWait';
126
+ /** Where a bounded listing stopped: the cursor of the first page it did not read (null at the end) and what stopped it. */
127
+ export interface PaginationEnd {
128
+ nextCursor: string | null;
129
+ stoppedBy: PaginationStop;
130
+ }
131
+ /** The items a bounded listing read, with where it stopped; `hasMore` is whether a page remains. */
132
+ export interface PageCollection<T> extends PaginationEnd {
133
+ items: T[];
134
+ hasMore: boolean;
135
+ }
136
+ /** A crawl's status with its pages, as one answer: the latest attempt's pages unless `attemptId` is given. */
137
+ export interface CrawlDocuments extends PaginationEnd {
138
+ report: CrawlReport;
139
+ pages: CrawlPage[];
140
+ }
141
+ /** A batch's status with its items, as one answer. */
142
+ export interface BatchDocuments extends PaginationEnd {
143
+ report: BatchStatusResponse;
144
+ items: CrawlPage[];
145
+ }
146
+ /** A page listing's query (`limit`, `attemptId`, `debug`, `includeDuplicates`) with the caps on how far it follows cursors. */
147
+ export type PagedListOptions = Omit<CrawlPageQuery, 'cursor'> & PaginationLimits;
148
+ /**
149
+ * Split a URL list into lists of at most `chunkSize` (default 100), in order;
150
+ * an empty list gives none. Firecrawl's JS helper of the same name, exported
151
+ * here. `batchScrapeChunked` runs one batch per chunk.
152
+ */
153
+ export declare function chunkUrls(urls: readonly string[], chunkSize?: number): string[][];
154
+ /**
155
+ * The API answered with an error status. `code` is the API error code when the
156
+ * body carried one (`rate_limited` for HTTP 429); `body` is the parsed JSON
157
+ * body, or its text if it was not JSON. The SDK retries no 429 itself: a
158
+ * caller waits `retryAfterMs` and asks again.
159
+ */
160
+ export declare class W2LError extends Error {
161
+ readonly status: number;
162
+ readonly code: ApiErrorCode | typeof RATE_LIMITED_CODE | undefined;
163
+ readonly method: 'GET' | 'POST' | 'DELETE';
164
+ readonly path: string;
165
+ readonly body: unknown;
166
+ /** The response's Retry-After in milliseconds; null when it sent none or one that does not parse. */
167
+ readonly retryAfterMs: number | null;
168
+ /** The body's `agentHints` (the next honest step for a refused option, the wait for a rate limit); empty when it carried none. */
169
+ readonly agentHints: readonly string[];
170
+ readonly name = "W2LError";
171
+ constructor(message: string, status: number, code: ApiErrorCode | typeof RATE_LIMITED_CODE | undefined, method: 'GET' | 'POST' | 'DELETE', path: string, body: unknown,
172
+ /** The response's Retry-After in milliseconds; null when it sent none or one that does not parse. */
173
+ retryAfterMs?: number | null,
174
+ /** The body's `agentHints` (the next honest step for a refused option, the wait for a rate limit); empty when it carried none. */
175
+ agentHints?: readonly string[]);
176
+ }
177
+ export type CreateMonitorRequest = (Omit<MonitorRevision, 'createdAt'> & {
178
+ enabled?: boolean;
179
+ }) | {
180
+ preset: 'firecrawl-introduction';
181
+ enabled?: boolean;
182
+ };
183
+ export type ReviseMonitorRequest = Omit<MonitorRevision, 'monitorId' | 'createdAt'>;
184
+ export interface RunMonitorRequest {
185
+ /** Reusing a key replays the same logical run within this monitor. */
186
+ triggerKey?: string;
187
+ }
188
+ export declare class W2L {
189
+ private readonly baseUrl;
190
+ private readonly token;
191
+ private readonly fetchImpl;
192
+ /** The platform's fetch, not one passed in options, whose own limits are the caller's. */
193
+ private readonly platformFetch;
194
+ constructor(options: W2LOptions);
195
+ scrape(url: string, opts: ScrapeOptions & {
196
+ debug: false;
197
+ }, request?: RequestOptions): Promise<CompactScrapeResponse>;
198
+ scrape(url: string, opts?: ScrapeOptions, request?: RequestOptions): Promise<ScrapeResponse>;
199
+ /** The record of one scrape call, by the `scrapeId` its response carried (`metadata.scrapeId`); a W2LError with code `not_found` for an id the server has no record of. */
200
+ getScrape(id: string, request?: RequestOptions): Promise<ScrapeRecord>;
201
+ /**
202
+ * The URLs of a site from its sitemaps and its start page's links, without
203
+ * fetching each page (POST /v1/map). The API answers by the map's deadline
204
+ * with what it found; the SDK waits that long plus MAP_ANSWER_MARGIN_MS.
205
+ */
206
+ map(url: string, opts?: Omit<MapRequest, 'url'>, request?: RequestOptions): Promise<MapResponse>;
207
+ /** The record of one map, by the `id` its response carried; a W2LError with code `not_found` for an id the server has no record of. */
208
+ getMap(id: string, request?: RequestOptions): Promise<MapRecord>;
209
+ crawl(url: string, opts?: Omit<CrawlStartRequest, 'url'>, request?: RequestOptions): Promise<CrawlAccepted>;
210
+ /**
211
+ * Starts a batch: `{ taskId }`, plus `invalidURLs` (the entries skipped)
212
+ * when `ignoreInvalidURLs` was on; with `idempotencyKey` a retried start
213
+ * returns the first one's answer with `replayed: true`; with `appendToId`
214
+ * the URLs join that batch (see appendToBatch) and the answer carries
215
+ * `requested` and `appended`.
216
+ */
217
+ batchScrape(urls: readonly string[], opts?: Omit<BatchStartRequest, 'urls'>, request?: RequestOptions): Promise<BatchAccepted>;
218
+ /**
219
+ * Adds URLs to an existing batch (`appendToId`): the job keeps its mode,
220
+ * formats, includeLinks, maxConcurrency and page options, and its run picks
221
+ * the URLs up (a completed batch runs again for them). The answer carries
222
+ * `requested`, the job's URLs now, and `appended`. A cancelled or failed
223
+ * batch, a total over 1000 or a URL already in the batch is a W2LError.
224
+ */
225
+ appendToBatch(id: string, urls: readonly string[], opts?: AppendToBatchOptions, request?: RequestOptions): Promise<BatchAccepted>;
226
+ /**
227
+ * Runs a list of any length as batches of `chunkSize` URLs (default 100),
228
+ * one after another: each job is started, waited for (as waitBatch, with
229
+ * the WaitOptions) and listed before the next starts. The items are merged
230
+ * in the order the URLs were submitted; the jobs stay on the server as
231
+ * ordinary batches, each with its own task directory. A caller's
232
+ * `idempotencyKey` becomes `<key>:<chunkIndex>` per job, so a retry of the
233
+ * whole call replays the jobs that went through. A WaitTimeoutError or
234
+ * W2LError from any job ends the call, naming that job; the earlier jobs
235
+ * are complete and the later chunks were never sent.
236
+ */
237
+ batchScrapeChunked(urls: readonly string[], opts?: Omit<BatchStartRequest, 'urls' | 'appendToId'>, options?: ChunkedBatchOptions): Promise<ChunkedBatchResult>;
238
+ getBatch(id: string, request?: RequestOptions): Promise<BatchStatusResponse>;
239
+ /** The batch's failed, blocked, cancelled and budget-cut items across every attempt, in pages of up to 1000 (`limit`, `cursor`), with `robotsBlocked`, the URLs robots.txt refused. */
240
+ getBatchErrors(id: string, options?: BatchErrorsQuery, request?: RequestOptions): Promise<BatchErrorsResponse>;
241
+ getBatchItems(id: string, options?: CrawlPageQuery, request?: RequestOptions): Promise<CrawlPageList<CrawlPage>>;
242
+ /** Every item of a batch, page by page (at most 50 per request), or as many as the PaginationLimits allow; the generator's return value says where it stopped. */
243
+ listBatchItems(id: string, options?: PagedListOptions, request?: RequestOptions): AsyncGenerator<CrawlPage, PaginationEnd>;
244
+ /** The items listBatchItems would yield under the same options, collected, with where the listing stopped. */
245
+ collectBatchItems(id: string, options?: PagedListOptions, request?: RequestOptions): Promise<PageCollection<CrawlPage>>;
246
+ /** A batch's status and its items in one answer, every item unless the PaginationLimits stop the listing. */
247
+ getBatchDocuments(id: string, options?: PagedListOptions, request?: RequestOptions): Promise<BatchDocuments>;
248
+ /** Polls a batch until it completes, fails or is cancelled. Items come from listBatchItems. */
249
+ waitBatch(id: string, options?: WaitOptions): Promise<BatchStatusResponse>;
250
+ /** Starts a batch, waits for it (as waitBatch) and lists every item, failed ones included. */
251
+ batchAndWait(urls: readonly string[], opts?: Omit<BatchStartRequest, 'urls'>, wait?: WaitOptions): Promise<BatchCollected>;
252
+ /**
253
+ * Save the person's login to a site (a domain or a page URL) from the
254
+ * Chrome they use, as `octocrawl login import` does, on a server on their
255
+ * machine. Chrome asks them "Allow remote debugging?": the answer comes
256
+ * once they click Allow (within `approveTimeoutMs`, default 2 minutes).
257
+ * The saved login's cookies never leave the server: the answer names the
258
+ * domain, how many cookies and their hash.
259
+ */
260
+ importLogin(site: string, opts?: Omit<LoginImportRequest, 'site'>, request?: RequestOptions): Promise<LoginImportResponse>;
261
+ /** The person's saved logins, without their cookies. */
262
+ listLogins(request?: RequestOptions): Promise<{
263
+ logins: SavedLogin[];
264
+ }>;
265
+ /** Forget a saved login; a W2LError with code `not_found` when none was saved for the site. */
266
+ removeLogin(site: string, request?: RequestOptions): Promise<{
267
+ site: string;
268
+ removed: true;
269
+ }>;
270
+ /**
271
+ * Hands a finished batch's items that a check stopped (a captcha, a
272
+ * challenge, a login wall) to the person in their own Chrome, on a local
273
+ * server: each opens in a new tab, they get through it, and W2L reads the
274
+ * page there. Answers when every item is read or given up, so it waits for
275
+ * the person: `waitMs` is how long, per page (default 10 minutes). On
276
+ * Node the SDK waits for the answer as long as that takes (no 300 s limit
277
+ * on the response headers); `request.signal` ends the wait.
278
+ */
279
+ handOffBatch(id: string, body?: BatchHandoffRequest, request?: RequestOptions): Promise<BatchHandoffResponse>;
280
+ cancelBatch(id: string, request?: RequestOptions): Promise<BatchStatusResponse>;
281
+ /**
282
+ * Watches a crawl (`kind: 'crawl'`, the default) or a batch (`kind: 'batch'`)
283
+ * as it runs: `document` events with each page as it is recorded, `snapshot`
284
+ * events with the report, one `done` with the terminal report, or `error`.
285
+ * `transport: 'auto'` (default) tries the WebSocket route, then server-sent
286
+ * events, then polling (`pollIntervalMs`, default 2000, at least 250), each
287
+ * taking over from the last document seen; `timeoutMs` ends the watch with a
288
+ * `watcher_timeout` error while the job keeps running. `close()` stops
289
+ * watching only; cancelCrawl / cancelBatch stay explicit.
290
+ */
291
+ watcher(jobId: string, options?: WatchOptions): JobWatcher;
292
+ /** Starts a crawl and returns its watcher (as `crawl()` then `watcher(taskId, { kind: 'crawl' })`). */
293
+ crawlAndWatch(url: string, opts?: Omit<CrawlStartRequest, 'url'>, watch?: Omit<WatchOptions, 'kind'>, request?: RequestOptions): Promise<JobWatcher>;
294
+ /** Starts a batch and returns its watcher (as `batchScrape()` then `watcher(taskId, { kind: 'batch' })`). */
295
+ batchScrapeAndWatch(urls: readonly string[], opts?: Omit<BatchStartRequest, 'urls'>, watch?: Omit<WatchOptions, 'kind'>, request?: RequestOptions): Promise<JobWatcher>;
296
+ /** What a watcher needs of this client: the server, the token and fetch it was given, and the routes it polls, each carrying the bearer header. */
297
+ private watcherClient;
298
+ getCrawl(id: string, request?: RequestOptions): Promise<CrawlReport>;
299
+ /** The crawls the API process is running, with each one's start URL, status, pages so far and options; empty when nothing runs. */
300
+ getActiveCrawls(request?: RequestOptions): Promise<ActiveCrawlList>;
301
+ /** Polls a crawl until it completes, fails or is cancelled. Pages come from listCrawlPages. */
302
+ waitCrawl(id: string, options?: WaitOptions): Promise<CrawlReport>;
303
+ /** Starts a crawl, waits for it (as waitCrawl) and lists every page and every error of its latest attempt. */
304
+ crawlAndWait(url: string, opts?: Omit<CrawlStartRequest, 'url'>, wait?: WaitOptions): Promise<CrawlCollected>;
305
+ getCrawlPages(id: string, options?: CrawlPageQuery, request?: RequestOptions): Promise<CrawlPageList<CrawlPage>>;
306
+ /**
307
+ * A crawl's pages, page by page, or as many as the PaginationLimits allow;
308
+ * the generator's return value says where it stopped. The latest attempt's
309
+ * pages unless `attemptId` names another; a resume with `useCached` records
310
+ * the pages it reuses in its new attempt, so that attempt normally holds
311
+ * every page.
312
+ */
313
+ listCrawlPages(id: string, options?: PagedListOptions, request?: RequestOptions): AsyncGenerator<CrawlPage, PaginationEnd>;
314
+ /** The pages listCrawlPages would yield under the same options, collected, with where the listing stopped. */
315
+ collectCrawlPages(id: string, options?: PagedListOptions, request?: RequestOptions): Promise<PageCollection<CrawlPage>>;
316
+ /** A crawl's status and its pages in one answer, every page unless the PaginationLimits stop the listing. */
317
+ getCrawlDocuments(id: string, options?: PagedListOptions, request?: RequestOptions): Promise<CrawlDocuments>;
318
+ /**
319
+ * Follow a listing's cursors within its limits. Each page is requested no
320
+ * larger than the items still wanted, so a stop at `maxResults` leaves a
321
+ * cursor that continues exactly after the last item returned. A page's
322
+ * `hasMore` without a cursor is the API breaking its contract and throws.
323
+ */
324
+ private paginate;
325
+ getCrawlErrors(id: string, options?: CrawlPageQuery, request?: RequestOptions): Promise<CrawlPageList<CrawlError>>;
326
+ cancelCrawl(id: string, request?: RequestOptions): Promise<CrawlReport>;
327
+ /** Restarts a paused or failed crawl with the options it was started with; follow it with waitCrawl. */
328
+ resumeCrawl(id: string, request?: RequestOptions): Promise<CrawlAccepted>;
329
+ createMonitor(input: CreateMonitorRequest, request?: RequestOptions): Promise<MonitorRevision>;
330
+ previewMonitor(input: CreateMonitorRequest, request?: RequestOptions): Promise<MonitorPreview>;
331
+ reviseMonitor(id: string, input: ReviseMonitorRequest, request?: RequestOptions): Promise<MonitorRevision>;
332
+ listMonitors(request?: RequestOptions): Promise<MonitorView[]>;
333
+ getMonitor(id: string, request?: RequestOptions): Promise<MonitorView>;
334
+ getMonitorRun(id: string, runId: string, request?: RequestOptions): Promise<MonitorRunDetail>;
335
+ /** Durable run: returns after enqueue; client disconnect does not cancel it. */
336
+ enqueueMonitorRun(id: string, input?: RunMonitorRequest, request?: RequestOptions): Promise<MonitorRun>;
337
+ /** Waits for capture and assessment; baseline/events are included in the returned view. */
338
+ runMonitor(id: string, input?: RunMonitorRequest, request?: RequestOptions): Promise<MonitorView>;
339
+ pauseMonitor(id: string, request?: RequestOptions): Promise<MonitorView>;
340
+ resumeMonitor(id: string, request?: RequestOptions): Promise<MonitorView>;
341
+ cancelMonitorRun(id: string, runId: string, request?: RequestOptions): Promise<MonitorView>;
342
+ createDeliveryDestination(input: DeliveryDestinationInput, request?: RequestOptions): Promise<DeliveryDestination>;
343
+ /** The destinations of a Monitor (`monitorId`) or of a crawl or batch (`jobId`); every destination when neither is given. Header names only, never their values. */
344
+ listDeliveryDestinations(options?: {
345
+ monitorId?: string;
346
+ jobId?: string;
347
+ }, request?: RequestOptions): Promise<DeliveryDestination[]>;
348
+ pauseDeliveryDestination(id: string, request?: RequestOptions): Promise<DeliveryDestination>;
349
+ resumeDeliveryDestination(id: string, request?: RequestOptions): Promise<DeliveryDestination>;
350
+ /** The deliveries of a Monitor (`monitorId`) or of a crawl or batch (`jobId`, the task id), each with its payload. */
351
+ listDeliveries(options?: DeliveryQuery, request?: RequestOptions): Promise<WebhookDelivery[]>;
352
+ getDeliveriesPage(options?: DeliveryPageQuery, request?: RequestOptions): Promise<DeliveryPage>;
353
+ getDelivery(id: string, request?: RequestOptions): Promise<DeliveryDetail>;
354
+ retryDelivery(id: string, request?: RequestOptions): Promise<WebhookDelivery>;
355
+ private waitFor;
356
+ private headers;
357
+ /** `answerWithinMs`: how long the platform's fetch waits for the response headers, on Node instead of undici's 300 s. */
358
+ private post;
359
+ private getPageList;
360
+ private get;
361
+ }
362
+ //# sourceMappingURL=client.d.ts.map
@@ -0,0 +1,166 @@
1
+ /**
2
+ * BYO Proxy and Inherited Session: the user's own access, and the record that
3
+ * proves whose it was.
4
+ *
5
+ * WHY THIS EXISTS. When a site gates us, three answers are honest and one is
6
+ * not. Honest: run a real browser (more capability we legitimately have), use
7
+ * the user's own logged-in session, leave from the user's own network. Not
8
+ * honest: defeat the gate. `escalationForBlock` already encodes that ladder;
9
+ * this module is the configuration surface for its last two rungs.
10
+ *
11
+ * WHY IT IS NOT JUST CONFIG. Routing a fetch through a user's proxy or session
12
+ * moves real responsibility onto them — their IP is what the publisher sees,
13
+ * their account is what the ToS binds, their contract governs the egress. A
14
+ * product that quietly accepts a proxy URL and says nothing has taken their
15
+ * risk without their acknowledgement. So the transfer is recorded: every
16
+ * compliance record carries an `AccessFact` stating who owned the egress, who
17
+ * owned the session, and who attested to the right to use them. Signed, that
18
+ * is a artifact both sides can point at afterwards — which is the difference
19
+ * between transferring responsibility and merely assuming it transferred.
20
+ *
21
+ * WHAT MUST NEVER BE IN THE RECORD. Proxy passwords and cookie values are
22
+ * credentials. They are hashed, never stored: `proxyCredentialSha256` lets two
23
+ * records be recognized as using the same credential without disclosing it,
24
+ * which is what an auditor actually needs. A compliance record is meant to be
25
+ * handed to a publisher or a court; a record that leaks the user's session
26
+ * cookie would be a breach dressed as an attestation.
27
+ *
28
+ * WHAT THIS IS NOT. There is no fingerprint spoofing here, no CDP patching, no
29
+ * captcha solving, and no rotating pool. A proxy is an egress path the user
30
+ * already has the right to use; the honest-mode UA is sent through it
31
+ * unchanged, and the identity honesty check still applies.
32
+ *
33
+ * 换 IP ≠ 换身份. Locale, timezone, viewport, and Client Hints stay the
34
+ * mode's bundle. Retuning timezone to a proxy's geo is refused.
35
+ */
36
+ /** Who owned the network path a fetch actually left through. */
37
+ export type EgressOwner =
38
+ /** Our own network. We hold the responsibility. */
39
+ 'operator'
40
+ /** The user's proxy. Their IP, their contract, their responsibility. */
41
+ | 'user';
42
+ /** Whose authenticated state, if any, a fetch carried. */
43
+ export type SessionOwner =
44
+ /** No session; the fetch was anonymous. */
45
+ 'none'
46
+ /** The user's own logged-in state, supplied by them. */
47
+ | 'user';
48
+ /**
49
+ * A proxy the user brings. `url` carries scheme://host:port ONLY — credentials
50
+ * go in the separate fields so they are never accidentally logged, serialized
51
+ * into a trace, or embedded in an error message alongside the endpoint.
52
+ * `normalizeProxyConfig` enforces this rather than trusting the caller.
53
+ */
54
+ export interface ProxyConfig {
55
+ /** e.g. `http://proxy.example:8080`. Userinfo here is stripped, not honoured. */
56
+ url: string;
57
+ username?: string;
58
+ password?: string;
59
+ }
60
+ /** One cookie of an inherited session, in the shape Playwright accepts. */
61
+ export interface SessionCookie {
62
+ name: string;
63
+ value: string;
64
+ domain: string;
65
+ path: string;
66
+ expires?: number;
67
+ httpOnly?: boolean;
68
+ secure?: boolean;
69
+ sameSite?: 'Strict' | 'Lax' | 'None';
70
+ }
71
+ /**
72
+ * The user's own logged-in state. Either explicit cookies or an opaque
73
+ * Playwright storageState blob; both are credentials and both are hashed
74
+ * rather than recorded.
75
+ */
76
+ export interface SessionConfig {
77
+ cookies?: readonly SessionCookie[];
78
+ /** Serialized Playwright storageState JSON. */
79
+ storageState?: string;
80
+ }
81
+ /**
82
+ * The user's explicit acknowledgement that they hold the right to use the
83
+ * proxy and session they supplied, and that fetches made through them are
84
+ * theirs. Required — `normalizeAccessConfig` rejects a proxy or session
85
+ * without one, because an unattested transfer is not a transfer.
86
+ */
87
+ export interface AccessAttestation {
88
+ /** Identifier for the accepting principal: account id, email, org handle. */
89
+ principal: string;
90
+ /** ISO timestamp of acceptance. */
91
+ at: string;
92
+ /**
93
+ * What they accepted, verbatim. Stored in the record so the claim can be
94
+ * read later without trusting our summary of it.
95
+ */
96
+ statement: string;
97
+ }
98
+ export interface AccessConfig {
99
+ proxy?: ProxyConfig | null;
100
+ session?: SessionConfig | null;
101
+ attestation?: AccessAttestation | null;
102
+ }
103
+ /**
104
+ * The recorded, credential-free facts about whose access a fetch used. This is
105
+ * the field that makes the responsibility transfer provable; it is part of the
106
+ * canonical serialization, so it is covered by the record's contentHash and by
107
+ * any signature over it.
108
+ */
109
+ export interface AccessFact {
110
+ egressOwner: EgressOwner;
111
+ /**
112
+ * `scheme://host:port` of the proxy, with any userinfo removed. Null under
113
+ * operator egress. The endpoint is not a secret and an auditor needs it;
114
+ * the credential is a secret and appears only as a hash.
115
+ */
116
+ proxyEndpoint: string | null;
117
+ /**
118
+ * sha256 of the proxy credential. Recognizes reuse across records without
119
+ * disclosing the credential. Null when the proxy needs no credential.
120
+ */
121
+ proxyCredentialSha256: string | null;
122
+ sessionOwner: SessionOwner;
123
+ /** sha256 of the session material. Null when no session was supplied. */
124
+ sessionSha256: string | null;
125
+ /** Principal who accepted responsibility. Null under operator egress. */
126
+ attestedBy: string | null;
127
+ /** ISO timestamp of that acceptance. Null under operator egress. */
128
+ attestedAt: string | null;
129
+ /** The statement they accepted, verbatim. Null under operator egress. */
130
+ attestationStatement: string | null;
131
+ }
132
+ /**
133
+ * The default: we supplied the egress, no session, nobody else's
134
+ * responsibility. Stated explicitly rather than left absent — "we did not
135
+ * track this" and "this was ours" are different claims, and a record that
136
+ * cannot tell them apart is not much of a record.
137
+ */
138
+ export declare const OPERATOR_ACCESS_FACT: AccessFact;
139
+ /**
140
+ * What the caller must supply for a given escalation target. Returned to a
141
+ * blocked caller so the guidance is a machine-readable requirement, not a
142
+ * sentence in a log line.
143
+ */
144
+ export type AccessRequirement =
145
+ /** Nothing more is needed; we can run this lane ourselves. */
146
+ 'none'
147
+ /** The user's own logged-in session (cookies or storageState). */
148
+ | 'session'
149
+ /** The user's own proxy. */
150
+ | 'proxy';
151
+ export interface AccessGuidance {
152
+ requirement: AccessRequirement;
153
+ /** Why this is being asked for, in one sentence the caller can surface. */
154
+ rationale: string;
155
+ }
156
+ /**
157
+ * The access a lane needs from the user. Lanes we can run on our own return
158
+ * `none`; the two user-access lanes name what they need and why.
159
+ *
160
+ * Note what is NOT here: no lane's requirement is "solve the challenge" or
161
+ * "rotate to a fresh IP". `provider` returns `none` because the provider lane
162
+ * is our own contracted capacity, not the user's — its own gating lives in the
163
+ * provider adapter's robots check, not here.
164
+ */
165
+ export declare function accessGuidanceForLane(lane: string): AccessGuidance;
166
+ //# sourceMappingURL=access.d.ts.map