@octocrawl/sdk 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +17 -0
- package/dist/index.cjs +1278 -0
- package/dist/index.js +1240 -0
- package/dist/types/cjs/client.d.ts +362 -0
- package/dist/types/cjs/contracts/access.d.ts +166 -0
- package/dist/types/cjs/contracts/actions.d.ts +191 -0
- package/dist/types/cjs/contracts/api.d.ts +891 -0
- package/dist/types/cjs/contracts/benchmark.d.ts +116 -0
- package/dist/types/cjs/contracts/checkpoint.d.ts +164 -0
- package/dist/types/cjs/contracts/compliance.d.ts +412 -0
- package/dist/types/cjs/contracts/crawl.d.ts +302 -0
- package/dist/types/cjs/contracts/delivery.d.ts +136 -0
- package/dist/types/cjs/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/cjs/contracts/execution.d.ts +197 -0
- package/dist/types/cjs/contracts/extractor.d.ts +379 -0
- package/dist/types/cjs/contracts/file.d.ts +117 -0
- package/dist/types/cjs/contracts/firecrawl.d.ts +258 -0
- package/dist/types/cjs/contracts/groundTruth.d.ts +77 -0
- package/dist/types/cjs/contracts/identityBundle.d.ts +70 -0
- package/dist/types/cjs/contracts/index.d.ts +30 -0
- package/dist/types/cjs/contracts/map.d.ts +180 -0
- package/dist/types/cjs/contracts/monitor.d.ts +217 -0
- package/dist/types/cjs/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/cjs/contracts/policy.d.ts +93 -0
- package/dist/types/cjs/contracts/proxy.d.ts +52 -0
- package/dist/types/cjs/contracts/recipe.d.ts +74 -0
- package/dist/types/cjs/contracts/regexSafety.d.ts +31 -0
- package/dist/types/cjs/contracts/result.d.ts +503 -0
- package/dist/types/cjs/contracts/session.d.ts +51 -0
- package/dist/types/cjs/contracts/ssrf.d.ts +16 -0
- package/dist/types/cjs/contracts/status.d.ts +27 -0
- package/dist/types/cjs/contracts/structured.d.ts +185 -0
- package/dist/types/cjs/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/cjs/contracts/tokens.d.ts +47 -0
- package/dist/types/cjs/index.d.ts +9 -0
- package/dist/types/cjs/package.json +1 -0
- package/dist/types/cjs/version.d.ts +8 -0
- package/dist/types/cjs/watcher.d.ts +151 -0
- package/dist/types/esm/client.d.ts +362 -0
- package/dist/types/esm/contracts/access.d.ts +166 -0
- package/dist/types/esm/contracts/actions.d.ts +191 -0
- package/dist/types/esm/contracts/api.d.ts +891 -0
- package/dist/types/esm/contracts/benchmark.d.ts +116 -0
- package/dist/types/esm/contracts/checkpoint.d.ts +164 -0
- package/dist/types/esm/contracts/compliance.d.ts +412 -0
- package/dist/types/esm/contracts/crawl.d.ts +302 -0
- package/dist/types/esm/contracts/delivery.d.ts +136 -0
- package/dist/types/esm/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/esm/contracts/execution.d.ts +197 -0
- package/dist/types/esm/contracts/extractor.d.ts +379 -0
- package/dist/types/esm/contracts/file.d.ts +117 -0
- package/dist/types/esm/contracts/firecrawl.d.ts +258 -0
- package/dist/types/esm/contracts/groundTruth.d.ts +77 -0
- package/dist/types/esm/contracts/identityBundle.d.ts +70 -0
- package/dist/types/esm/contracts/index.d.ts +30 -0
- package/dist/types/esm/contracts/map.d.ts +180 -0
- package/dist/types/esm/contracts/monitor.d.ts +217 -0
- package/dist/types/esm/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/esm/contracts/policy.d.ts +93 -0
- package/dist/types/esm/contracts/proxy.d.ts +52 -0
- package/dist/types/esm/contracts/recipe.d.ts +74 -0
- package/dist/types/esm/contracts/regexSafety.d.ts +31 -0
- package/dist/types/esm/contracts/result.d.ts +503 -0
- package/dist/types/esm/contracts/session.d.ts +51 -0
- package/dist/types/esm/contracts/ssrf.d.ts +16 -0
- package/dist/types/esm/contracts/status.d.ts +27 -0
- package/dist/types/esm/contracts/structured.d.ts +185 -0
- package/dist/types/esm/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/esm/contracts/tokens.d.ts +47 -0
- package/dist/types/esm/index.d.ts +9 -0
- package/dist/types/esm/version.d.ts +8 -0
- package/dist/types/esm/watcher.d.ts +151 -0
- package/package.json +40 -0
|
@@ -0,0 +1,362 @@
|
|
|
1
|
+
import type { ActiveCrawlList, CrawlAccepted, DeliveryDestination, DeliveryDestinationInput, DeliveryDetail, DeliveryQuery, DeliveryPage, DeliveryPageQuery, WebhookDelivery, CrawlError, CrawlPage, CrawlPageList, CrawlPageQuery, CrawlReport, CrawlStartRequest, BatchAccepted, BatchErrorsQuery, BatchErrorsResponse, BatchStartRequest, BatchStatusResponse, BatchHandoffRequest, BatchHandoffResponse, LoginImportRequest, LoginImportResponse, SavedLogin, CompactScrapeResponse, MapRecord, MapRequest, MapResponse, MonitorRevision, MonitorView, MonitorPreview, MonitorRun, MonitorRunDetail, ScrapeRecord, ScrapeRequest, ScrapeResponse } from './contracts/index.js';
|
|
2
|
+
import { RATE_LIMITED_CODE, type ApiErrorCode } from './contracts/index.js';
|
|
3
|
+
import { JobWatcher, type WatchOptions } from './watcher.js';
|
|
4
|
+
export interface W2LOptions {
|
|
5
|
+
baseUrl: string;
|
|
6
|
+
/**
|
|
7
|
+
* Bearer token for a server started with a token. Omitted, the
|
|
8
|
+
* W2L_API_TOKEN environment variable is used where there is one; '' sends
|
|
9
|
+
* no token.
|
|
10
|
+
*/
|
|
11
|
+
token?: string;
|
|
12
|
+
fetch?: typeof fetch;
|
|
13
|
+
}
|
|
14
|
+
/** A scrape's options as a caller sends them: `handoff` as `true` or `{ waitMs }` (the API reads `true` as `{}`). */
|
|
15
|
+
export type ScrapeOptions = Omit<ScrapeRequest, 'url' | 'handoff'> & {
|
|
16
|
+
handoff?: boolean | {
|
|
17
|
+
waitMs?: number;
|
|
18
|
+
};
|
|
19
|
+
};
|
|
20
|
+
/** How long past a map's own deadline (its `timeout`, 60 000 ms by default) the SDK waits for the API's answer: Firecrawl's timeout + 5000 convention. */
|
|
21
|
+
export declare const MAP_ANSWER_MARGIN_MS = 5000;
|
|
22
|
+
/**
|
|
23
|
+
* Cancels the HTTP request and the current execution of synchronous scrape/runMonitor.
|
|
24
|
+
* Use cancelCrawl for an already-created background crawl, or cancelMonitorRun for explicit persisted run control.
|
|
25
|
+
*/
|
|
26
|
+
export interface RequestOptions {
|
|
27
|
+
signal?: AbortSignal;
|
|
28
|
+
/**
|
|
29
|
+
* The `origin` a scrape, crawl or batch is recorded under when its options
|
|
30
|
+
* set none: a host built on the SDK (the MCP server) names its own client
|
|
31
|
+
* here. The default is `js-sdk@<SDK_VERSION>`.
|
|
32
|
+
*/
|
|
33
|
+
origin?: string;
|
|
34
|
+
}
|
|
35
|
+
/** What the SDK records as `origin` unless the caller or the host says otherwise. */
|
|
36
|
+
export declare const SDK_ORIGIN = "js-sdk@0.3.0";
|
|
37
|
+
/**
|
|
38
|
+
* Polling for waitBatch and waitCrawl. A status request that fails with a
|
|
39
|
+
* network error, HTTP 408, 429 or 5xx is retried: after 1, 2, 4, 8, then
|
|
40
|
+
* 10 s, or after the response's Retry-After when it asks for a minute or
|
|
41
|
+
* less. Any other error ends the wait at once.
|
|
42
|
+
*/
|
|
43
|
+
export interface WaitOptions extends RequestOptions {
|
|
44
|
+
/** Delay between status requests. Default 500 ms. */
|
|
45
|
+
pollIntervalMs?: number;
|
|
46
|
+
/**
|
|
47
|
+
* Stop waiting after this long, a status request in flight or a retry's
|
|
48
|
+
* wait included, and throw a WaitTimeoutError. Default: no limit. The task
|
|
49
|
+
* keeps running.
|
|
50
|
+
*/
|
|
51
|
+
timeoutMs?: number;
|
|
52
|
+
/** Consecutive failed status requests retried before the last error is thrown. Default 5; 0 retries none. */
|
|
53
|
+
maxRetries?: number;
|
|
54
|
+
}
|
|
55
|
+
/**
|
|
56
|
+
* The task was still unfinished when the wait's timeoutMs ran out. `last` is
|
|
57
|
+
* the final status read, null when no status request answered in time;
|
|
58
|
+
* `cause` is the error of the last status request that failed, when no
|
|
59
|
+
* status was read after it.
|
|
60
|
+
*/
|
|
61
|
+
export declare class WaitTimeoutError<T extends {
|
|
62
|
+
status: string;
|
|
63
|
+
} = {
|
|
64
|
+
status: string;
|
|
65
|
+
}> extends Error {
|
|
66
|
+
readonly taskId: string;
|
|
67
|
+
readonly last: T | null;
|
|
68
|
+
readonly timeoutMs: number;
|
|
69
|
+
readonly name = "WaitTimeoutError";
|
|
70
|
+
constructor(taskId: string, last: T | null, timeoutMs: number, options?: {
|
|
71
|
+
cause?: unknown;
|
|
72
|
+
});
|
|
73
|
+
}
|
|
74
|
+
/** What `appendToBatch` may send beside the URLs: what binds to them, and the attribution labels; the job's own options stay as they are. */
|
|
75
|
+
export type AppendToBatchOptions = Pick<BatchStartRequest, 'ignoreInvalidURLs' | 'idempotencyKey' | 'robotsOverrides' | 'origin' | 'integration'>;
|
|
76
|
+
/** How `batchScrapeChunked` splits a list and waits for each job (Python's `process_large_batch`: `chunk_size`, `poll_interval`, `timeout` are `chunkSize`, `pollIntervalMs`, `timeoutMs`). */
|
|
77
|
+
export interface ChunkedBatchOptions extends WaitOptions {
|
|
78
|
+
/** URLs per job, 1 to 1000. Default 100. */
|
|
79
|
+
chunkSize?: number;
|
|
80
|
+
/** Items per request while a job's items are listed, 1 to 50. Default 50. */
|
|
81
|
+
itemLimit?: number;
|
|
82
|
+
}
|
|
83
|
+
/** One job `batchScrapeChunked` ran: its id, how many URLs it was given, its final status and the entries it skipped. */
|
|
84
|
+
export interface ChunkedBatchJob {
|
|
85
|
+
taskId: string;
|
|
86
|
+
urls: number;
|
|
87
|
+
report: BatchStatusResponse;
|
|
88
|
+
invalidURLs?: string[];
|
|
89
|
+
}
|
|
90
|
+
/** `batchScrapeChunked`'s result: the jobs in submission order, every item of every job in the order the URLs were submitted, and every entry `ignoreInvalidURLs` skipped. */
|
|
91
|
+
export interface ChunkedBatchResult {
|
|
92
|
+
jobs: ChunkedBatchJob[];
|
|
93
|
+
items: CrawlPage[];
|
|
94
|
+
invalidURLs: string[];
|
|
95
|
+
}
|
|
96
|
+
/** crawlAndWait's result: the final status, every page and every error of the crawl's latest attempt. */
|
|
97
|
+
export interface CrawlCollected {
|
|
98
|
+
taskId: string;
|
|
99
|
+
report: CrawlReport;
|
|
100
|
+
pages: CrawlPage[];
|
|
101
|
+
errors: CrawlError[];
|
|
102
|
+
}
|
|
103
|
+
/** batchAndWait's result: the final status and every item, failed ones included. */
|
|
104
|
+
export interface BatchCollected {
|
|
105
|
+
taskId: string;
|
|
106
|
+
report: BatchStatusResponse;
|
|
107
|
+
items: CrawlPage[];
|
|
108
|
+
}
|
|
109
|
+
/**
|
|
110
|
+
* Caps on a listing that follows cursors (listCrawlPages, listBatchItems,
|
|
111
|
+
* getCrawlDocuments, getBatchDocuments). None by default: the listing reads to
|
|
112
|
+
* the end. With `maxResults` each page is requested no larger than what is
|
|
113
|
+
* still wanted, so the cursor the listing stops at continues exactly where it
|
|
114
|
+
* left off.
|
|
115
|
+
*/
|
|
116
|
+
export interface PaginationLimits {
|
|
117
|
+
/** Pages read after the first; 0 reads the first page alone. */
|
|
118
|
+
maxPages?: number;
|
|
119
|
+
/** Items returned in all, at least 1. */
|
|
120
|
+
maxResults?: number;
|
|
121
|
+
/** Once this many milliseconds have passed since the first page was requested, no further page is requested. */
|
|
122
|
+
maxWaitMs?: number;
|
|
123
|
+
}
|
|
124
|
+
/** Why a bounded listing stopped: the last page (`end`), or the limit that stopped it with pages still unread. */
|
|
125
|
+
export type PaginationStop = 'end' | 'maxPages' | 'maxResults' | 'maxWait';
|
|
126
|
+
/** Where a bounded listing stopped: the cursor of the first page it did not read (null at the end) and what stopped it. */
|
|
127
|
+
export interface PaginationEnd {
|
|
128
|
+
nextCursor: string | null;
|
|
129
|
+
stoppedBy: PaginationStop;
|
|
130
|
+
}
|
|
131
|
+
/** The items a bounded listing read, with where it stopped; `hasMore` is whether a page remains. */
|
|
132
|
+
export interface PageCollection<T> extends PaginationEnd {
|
|
133
|
+
items: T[];
|
|
134
|
+
hasMore: boolean;
|
|
135
|
+
}
|
|
136
|
+
/** A crawl's status with its pages, as one answer: the latest attempt's pages unless `attemptId` is given. */
|
|
137
|
+
export interface CrawlDocuments extends PaginationEnd {
|
|
138
|
+
report: CrawlReport;
|
|
139
|
+
pages: CrawlPage[];
|
|
140
|
+
}
|
|
141
|
+
/** A batch's status with its items, as one answer. */
|
|
142
|
+
export interface BatchDocuments extends PaginationEnd {
|
|
143
|
+
report: BatchStatusResponse;
|
|
144
|
+
items: CrawlPage[];
|
|
145
|
+
}
|
|
146
|
+
/** A page listing's query (`limit`, `attemptId`, `debug`, `includeDuplicates`) with the caps on how far it follows cursors. */
|
|
147
|
+
export type PagedListOptions = Omit<CrawlPageQuery, 'cursor'> & PaginationLimits;
|
|
148
|
+
/**
|
|
149
|
+
* Split a URL list into lists of at most `chunkSize` (default 100), in order;
|
|
150
|
+
* an empty list gives none. Firecrawl's JS helper of the same name, exported
|
|
151
|
+
* here. `batchScrapeChunked` runs one batch per chunk.
|
|
152
|
+
*/
|
|
153
|
+
export declare function chunkUrls(urls: readonly string[], chunkSize?: number): string[][];
|
|
154
|
+
/**
|
|
155
|
+
* The API answered with an error status. `code` is the API error code when the
|
|
156
|
+
* body carried one (`rate_limited` for HTTP 429); `body` is the parsed JSON
|
|
157
|
+
* body, or its text if it was not JSON. The SDK retries no 429 itself: a
|
|
158
|
+
* caller waits `retryAfterMs` and asks again.
|
|
159
|
+
*/
|
|
160
|
+
export declare class W2LError extends Error {
|
|
161
|
+
readonly status: number;
|
|
162
|
+
readonly code: ApiErrorCode | typeof RATE_LIMITED_CODE | undefined;
|
|
163
|
+
readonly method: 'GET' | 'POST' | 'DELETE';
|
|
164
|
+
readonly path: string;
|
|
165
|
+
readonly body: unknown;
|
|
166
|
+
/** The response's Retry-After in milliseconds; null when it sent none or one that does not parse. */
|
|
167
|
+
readonly retryAfterMs: number | null;
|
|
168
|
+
/** The body's `agentHints` (the next honest step for a refused option, the wait for a rate limit); empty when it carried none. */
|
|
169
|
+
readonly agentHints: readonly string[];
|
|
170
|
+
readonly name = "W2LError";
|
|
171
|
+
constructor(message: string, status: number, code: ApiErrorCode | typeof RATE_LIMITED_CODE | undefined, method: 'GET' | 'POST' | 'DELETE', path: string, body: unknown,
|
|
172
|
+
/** The response's Retry-After in milliseconds; null when it sent none or one that does not parse. */
|
|
173
|
+
retryAfterMs?: number | null,
|
|
174
|
+
/** The body's `agentHints` (the next honest step for a refused option, the wait for a rate limit); empty when it carried none. */
|
|
175
|
+
agentHints?: readonly string[]);
|
|
176
|
+
}
|
|
177
|
+
export type CreateMonitorRequest = (Omit<MonitorRevision, 'createdAt'> & {
|
|
178
|
+
enabled?: boolean;
|
|
179
|
+
}) | {
|
|
180
|
+
preset: 'firecrawl-introduction';
|
|
181
|
+
enabled?: boolean;
|
|
182
|
+
};
|
|
183
|
+
export type ReviseMonitorRequest = Omit<MonitorRevision, 'monitorId' | 'createdAt'>;
|
|
184
|
+
export interface RunMonitorRequest {
|
|
185
|
+
/** Reusing a key replays the same logical run within this monitor. */
|
|
186
|
+
triggerKey?: string;
|
|
187
|
+
}
|
|
188
|
+
export declare class W2L {
|
|
189
|
+
private readonly baseUrl;
|
|
190
|
+
private readonly token;
|
|
191
|
+
private readonly fetchImpl;
|
|
192
|
+
/** The platform's fetch, not one passed in options, whose own limits are the caller's. */
|
|
193
|
+
private readonly platformFetch;
|
|
194
|
+
constructor(options: W2LOptions);
|
|
195
|
+
scrape(url: string, opts: ScrapeOptions & {
|
|
196
|
+
debug: false;
|
|
197
|
+
}, request?: RequestOptions): Promise<CompactScrapeResponse>;
|
|
198
|
+
scrape(url: string, opts?: ScrapeOptions, request?: RequestOptions): Promise<ScrapeResponse>;
|
|
199
|
+
/** The record of one scrape call, by the `scrapeId` its response carried (`metadata.scrapeId`); a W2LError with code `not_found` for an id the server has no record of. */
|
|
200
|
+
getScrape(id: string, request?: RequestOptions): Promise<ScrapeRecord>;
|
|
201
|
+
/**
|
|
202
|
+
* The URLs of a site from its sitemaps and its start page's links, without
|
|
203
|
+
* fetching each page (POST /v1/map). The API answers by the map's deadline
|
|
204
|
+
* with what it found; the SDK waits that long plus MAP_ANSWER_MARGIN_MS.
|
|
205
|
+
*/
|
|
206
|
+
map(url: string, opts?: Omit<MapRequest, 'url'>, request?: RequestOptions): Promise<MapResponse>;
|
|
207
|
+
/** The record of one map, by the `id` its response carried; a W2LError with code `not_found` for an id the server has no record of. */
|
|
208
|
+
getMap(id: string, request?: RequestOptions): Promise<MapRecord>;
|
|
209
|
+
crawl(url: string, opts?: Omit<CrawlStartRequest, 'url'>, request?: RequestOptions): Promise<CrawlAccepted>;
|
|
210
|
+
/**
|
|
211
|
+
* Starts a batch: `{ taskId }`, plus `invalidURLs` (the entries skipped)
|
|
212
|
+
* when `ignoreInvalidURLs` was on; with `idempotencyKey` a retried start
|
|
213
|
+
* returns the first one's answer with `replayed: true`; with `appendToId`
|
|
214
|
+
* the URLs join that batch (see appendToBatch) and the answer carries
|
|
215
|
+
* `requested` and `appended`.
|
|
216
|
+
*/
|
|
217
|
+
batchScrape(urls: readonly string[], opts?: Omit<BatchStartRequest, 'urls'>, request?: RequestOptions): Promise<BatchAccepted>;
|
|
218
|
+
/**
|
|
219
|
+
* Adds URLs to an existing batch (`appendToId`): the job keeps its mode,
|
|
220
|
+
* formats, includeLinks, maxConcurrency and page options, and its run picks
|
|
221
|
+
* the URLs up (a completed batch runs again for them). The answer carries
|
|
222
|
+
* `requested`, the job's URLs now, and `appended`. A cancelled or failed
|
|
223
|
+
* batch, a total over 1000 or a URL already in the batch is a W2LError.
|
|
224
|
+
*/
|
|
225
|
+
appendToBatch(id: string, urls: readonly string[], opts?: AppendToBatchOptions, request?: RequestOptions): Promise<BatchAccepted>;
|
|
226
|
+
/**
|
|
227
|
+
* Runs a list of any length as batches of `chunkSize` URLs (default 100),
|
|
228
|
+
* one after another: each job is started, waited for (as waitBatch, with
|
|
229
|
+
* the WaitOptions) and listed before the next starts. The items are merged
|
|
230
|
+
* in the order the URLs were submitted; the jobs stay on the server as
|
|
231
|
+
* ordinary batches, each with its own task directory. A caller's
|
|
232
|
+
* `idempotencyKey` becomes `<key>:<chunkIndex>` per job, so a retry of the
|
|
233
|
+
* whole call replays the jobs that went through. A WaitTimeoutError or
|
|
234
|
+
* W2LError from any job ends the call, naming that job; the earlier jobs
|
|
235
|
+
* are complete and the later chunks were never sent.
|
|
236
|
+
*/
|
|
237
|
+
batchScrapeChunked(urls: readonly string[], opts?: Omit<BatchStartRequest, 'urls' | 'appendToId'>, options?: ChunkedBatchOptions): Promise<ChunkedBatchResult>;
|
|
238
|
+
getBatch(id: string, request?: RequestOptions): Promise<BatchStatusResponse>;
|
|
239
|
+
/** The batch's failed, blocked, cancelled and budget-cut items across every attempt, in pages of up to 1000 (`limit`, `cursor`), with `robotsBlocked`, the URLs robots.txt refused. */
|
|
240
|
+
getBatchErrors(id: string, options?: BatchErrorsQuery, request?: RequestOptions): Promise<BatchErrorsResponse>;
|
|
241
|
+
getBatchItems(id: string, options?: CrawlPageQuery, request?: RequestOptions): Promise<CrawlPageList<CrawlPage>>;
|
|
242
|
+
/** Every item of a batch, page by page (at most 50 per request), or as many as the PaginationLimits allow; the generator's return value says where it stopped. */
|
|
243
|
+
listBatchItems(id: string, options?: PagedListOptions, request?: RequestOptions): AsyncGenerator<CrawlPage, PaginationEnd>;
|
|
244
|
+
/** The items listBatchItems would yield under the same options, collected, with where the listing stopped. */
|
|
245
|
+
collectBatchItems(id: string, options?: PagedListOptions, request?: RequestOptions): Promise<PageCollection<CrawlPage>>;
|
|
246
|
+
/** A batch's status and its items in one answer, every item unless the PaginationLimits stop the listing. */
|
|
247
|
+
getBatchDocuments(id: string, options?: PagedListOptions, request?: RequestOptions): Promise<BatchDocuments>;
|
|
248
|
+
/** Polls a batch until it completes, fails or is cancelled. Items come from listBatchItems. */
|
|
249
|
+
waitBatch(id: string, options?: WaitOptions): Promise<BatchStatusResponse>;
|
|
250
|
+
/** Starts a batch, waits for it (as waitBatch) and lists every item, failed ones included. */
|
|
251
|
+
batchAndWait(urls: readonly string[], opts?: Omit<BatchStartRequest, 'urls'>, wait?: WaitOptions): Promise<BatchCollected>;
|
|
252
|
+
/**
|
|
253
|
+
* Save the person's login to a site (a domain or a page URL) from the
|
|
254
|
+
* Chrome they use, as `octocrawl login import` does, on a server on their
|
|
255
|
+
* machine. Chrome asks them "Allow remote debugging?": the answer comes
|
|
256
|
+
* once they click Allow (within `approveTimeoutMs`, default 2 minutes).
|
|
257
|
+
* The saved login's cookies never leave the server: the answer names the
|
|
258
|
+
* domain, how many cookies and their hash.
|
|
259
|
+
*/
|
|
260
|
+
importLogin(site: string, opts?: Omit<LoginImportRequest, 'site'>, request?: RequestOptions): Promise<LoginImportResponse>;
|
|
261
|
+
/** The person's saved logins, without their cookies. */
|
|
262
|
+
listLogins(request?: RequestOptions): Promise<{
|
|
263
|
+
logins: SavedLogin[];
|
|
264
|
+
}>;
|
|
265
|
+
/** Forget a saved login; a W2LError with code `not_found` when none was saved for the site. */
|
|
266
|
+
removeLogin(site: string, request?: RequestOptions): Promise<{
|
|
267
|
+
site: string;
|
|
268
|
+
removed: true;
|
|
269
|
+
}>;
|
|
270
|
+
/**
|
|
271
|
+
* Hands a finished batch's items that a check stopped (a captcha, a
|
|
272
|
+
* challenge, a login wall) to the person in their own Chrome, on a local
|
|
273
|
+
* server: each opens in a new tab, they get through it, and W2L reads the
|
|
274
|
+
* page there. Answers when every item is read or given up, so it waits for
|
|
275
|
+
* the person: `waitMs` is how long, per page (default 10 minutes). On
|
|
276
|
+
* Node the SDK waits for the answer as long as that takes (no 300 s limit
|
|
277
|
+
* on the response headers); `request.signal` ends the wait.
|
|
278
|
+
*/
|
|
279
|
+
handOffBatch(id: string, body?: BatchHandoffRequest, request?: RequestOptions): Promise<BatchHandoffResponse>;
|
|
280
|
+
cancelBatch(id: string, request?: RequestOptions): Promise<BatchStatusResponse>;
|
|
281
|
+
/**
|
|
282
|
+
* Watches a crawl (`kind: 'crawl'`, the default) or a batch (`kind: 'batch'`)
|
|
283
|
+
* as it runs: `document` events with each page as it is recorded, `snapshot`
|
|
284
|
+
* events with the report, one `done` with the terminal report, or `error`.
|
|
285
|
+
* `transport: 'auto'` (default) tries the WebSocket route, then server-sent
|
|
286
|
+
* events, then polling (`pollIntervalMs`, default 2000, at least 250), each
|
|
287
|
+
* taking over from the last document seen; `timeoutMs` ends the watch with a
|
|
288
|
+
* `watcher_timeout` error while the job keeps running. `close()` stops
|
|
289
|
+
* watching only; cancelCrawl / cancelBatch stay explicit.
|
|
290
|
+
*/
|
|
291
|
+
watcher(jobId: string, options?: WatchOptions): JobWatcher;
|
|
292
|
+
/** Starts a crawl and returns its watcher (as `crawl()` then `watcher(taskId, { kind: 'crawl' })`). */
|
|
293
|
+
crawlAndWatch(url: string, opts?: Omit<CrawlStartRequest, 'url'>, watch?: Omit<WatchOptions, 'kind'>, request?: RequestOptions): Promise<JobWatcher>;
|
|
294
|
+
/** Starts a batch and returns its watcher (as `batchScrape()` then `watcher(taskId, { kind: 'batch' })`). */
|
|
295
|
+
batchScrapeAndWatch(urls: readonly string[], opts?: Omit<BatchStartRequest, 'urls'>, watch?: Omit<WatchOptions, 'kind'>, request?: RequestOptions): Promise<JobWatcher>;
|
|
296
|
+
/** What a watcher needs of this client: the server, the token and fetch it was given, and the routes it polls, each carrying the bearer header. */
|
|
297
|
+
private watcherClient;
|
|
298
|
+
getCrawl(id: string, request?: RequestOptions): Promise<CrawlReport>;
|
|
299
|
+
/** The crawls the API process is running, with each one's start URL, status, pages so far and options; empty when nothing runs. */
|
|
300
|
+
getActiveCrawls(request?: RequestOptions): Promise<ActiveCrawlList>;
|
|
301
|
+
/** Polls a crawl until it completes, fails or is cancelled. Pages come from listCrawlPages. */
|
|
302
|
+
waitCrawl(id: string, options?: WaitOptions): Promise<CrawlReport>;
|
|
303
|
+
/** Starts a crawl, waits for it (as waitCrawl) and lists every page and every error of its latest attempt. */
|
|
304
|
+
crawlAndWait(url: string, opts?: Omit<CrawlStartRequest, 'url'>, wait?: WaitOptions): Promise<CrawlCollected>;
|
|
305
|
+
getCrawlPages(id: string, options?: CrawlPageQuery, request?: RequestOptions): Promise<CrawlPageList<CrawlPage>>;
|
|
306
|
+
/**
|
|
307
|
+
* A crawl's pages, page by page, or as many as the PaginationLimits allow;
|
|
308
|
+
* the generator's return value says where it stopped. The latest attempt's
|
|
309
|
+
* pages unless `attemptId` names another; a resume with `useCached` records
|
|
310
|
+
* the pages it reuses in its new attempt, so that attempt normally holds
|
|
311
|
+
* every page.
|
|
312
|
+
*/
|
|
313
|
+
listCrawlPages(id: string, options?: PagedListOptions, request?: RequestOptions): AsyncGenerator<CrawlPage, PaginationEnd>;
|
|
314
|
+
/** The pages listCrawlPages would yield under the same options, collected, with where the listing stopped. */
|
|
315
|
+
collectCrawlPages(id: string, options?: PagedListOptions, request?: RequestOptions): Promise<PageCollection<CrawlPage>>;
|
|
316
|
+
/** A crawl's status and its pages in one answer, every page unless the PaginationLimits stop the listing. */
|
|
317
|
+
getCrawlDocuments(id: string, options?: PagedListOptions, request?: RequestOptions): Promise<CrawlDocuments>;
|
|
318
|
+
/**
|
|
319
|
+
* Follow a listing's cursors within its limits. Each page is requested no
|
|
320
|
+
* larger than the items still wanted, so a stop at `maxResults` leaves a
|
|
321
|
+
* cursor that continues exactly after the last item returned. A page's
|
|
322
|
+
* `hasMore` without a cursor is the API breaking its contract and throws.
|
|
323
|
+
*/
|
|
324
|
+
private paginate;
|
|
325
|
+
getCrawlErrors(id: string, options?: CrawlPageQuery, request?: RequestOptions): Promise<CrawlPageList<CrawlError>>;
|
|
326
|
+
cancelCrawl(id: string, request?: RequestOptions): Promise<CrawlReport>;
|
|
327
|
+
/** Restarts a paused or failed crawl with the options it was started with; follow it with waitCrawl. */
|
|
328
|
+
resumeCrawl(id: string, request?: RequestOptions): Promise<CrawlAccepted>;
|
|
329
|
+
createMonitor(input: CreateMonitorRequest, request?: RequestOptions): Promise<MonitorRevision>;
|
|
330
|
+
previewMonitor(input: CreateMonitorRequest, request?: RequestOptions): Promise<MonitorPreview>;
|
|
331
|
+
reviseMonitor(id: string, input: ReviseMonitorRequest, request?: RequestOptions): Promise<MonitorRevision>;
|
|
332
|
+
listMonitors(request?: RequestOptions): Promise<MonitorView[]>;
|
|
333
|
+
getMonitor(id: string, request?: RequestOptions): Promise<MonitorView>;
|
|
334
|
+
getMonitorRun(id: string, runId: string, request?: RequestOptions): Promise<MonitorRunDetail>;
|
|
335
|
+
/** Durable run: returns after enqueue; client disconnect does not cancel it. */
|
|
336
|
+
enqueueMonitorRun(id: string, input?: RunMonitorRequest, request?: RequestOptions): Promise<MonitorRun>;
|
|
337
|
+
/** Waits for capture and assessment; baseline/events are included in the returned view. */
|
|
338
|
+
runMonitor(id: string, input?: RunMonitorRequest, request?: RequestOptions): Promise<MonitorView>;
|
|
339
|
+
pauseMonitor(id: string, request?: RequestOptions): Promise<MonitorView>;
|
|
340
|
+
resumeMonitor(id: string, request?: RequestOptions): Promise<MonitorView>;
|
|
341
|
+
cancelMonitorRun(id: string, runId: string, request?: RequestOptions): Promise<MonitorView>;
|
|
342
|
+
createDeliveryDestination(input: DeliveryDestinationInput, request?: RequestOptions): Promise<DeliveryDestination>;
|
|
343
|
+
/** The destinations of a Monitor (`monitorId`) or of a crawl or batch (`jobId`); every destination when neither is given. Header names only, never their values. */
|
|
344
|
+
listDeliveryDestinations(options?: {
|
|
345
|
+
monitorId?: string;
|
|
346
|
+
jobId?: string;
|
|
347
|
+
}, request?: RequestOptions): Promise<DeliveryDestination[]>;
|
|
348
|
+
pauseDeliveryDestination(id: string, request?: RequestOptions): Promise<DeliveryDestination>;
|
|
349
|
+
resumeDeliveryDestination(id: string, request?: RequestOptions): Promise<DeliveryDestination>;
|
|
350
|
+
/** The deliveries of a Monitor (`monitorId`) or of a crawl or batch (`jobId`, the task id), each with its payload. */
|
|
351
|
+
listDeliveries(options?: DeliveryQuery, request?: RequestOptions): Promise<WebhookDelivery[]>;
|
|
352
|
+
getDeliveriesPage(options?: DeliveryPageQuery, request?: RequestOptions): Promise<DeliveryPage>;
|
|
353
|
+
getDelivery(id: string, request?: RequestOptions): Promise<DeliveryDetail>;
|
|
354
|
+
retryDelivery(id: string, request?: RequestOptions): Promise<WebhookDelivery>;
|
|
355
|
+
private waitFor;
|
|
356
|
+
private headers;
|
|
357
|
+
/** `answerWithinMs`: how long the platform's fetch waits for the response headers, on Node instead of undici's 300 s. */
|
|
358
|
+
private post;
|
|
359
|
+
private getPageList;
|
|
360
|
+
private get;
|
|
361
|
+
}
|
|
362
|
+
//# sourceMappingURL=client.d.ts.map
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* BYO Proxy and Inherited Session: the user's own access, and the record that
|
|
3
|
+
* proves whose it was.
|
|
4
|
+
*
|
|
5
|
+
* WHY THIS EXISTS. When a site gates us, three answers are honest and one is
|
|
6
|
+
* not. Honest: run a real browser (more capability we legitimately have), use
|
|
7
|
+
* the user's own logged-in session, leave from the user's own network. Not
|
|
8
|
+
* honest: defeat the gate. `escalationForBlock` already encodes that ladder;
|
|
9
|
+
* this module is the configuration surface for its last two rungs.
|
|
10
|
+
*
|
|
11
|
+
* WHY IT IS NOT JUST CONFIG. Routing a fetch through a user's proxy or session
|
|
12
|
+
* moves real responsibility onto them — their IP is what the publisher sees,
|
|
13
|
+
* their account is what the ToS binds, their contract governs the egress. A
|
|
14
|
+
* product that quietly accepts a proxy URL and says nothing has taken their
|
|
15
|
+
* risk without their acknowledgement. So the transfer is recorded: every
|
|
16
|
+
* compliance record carries an `AccessFact` stating who owned the egress, who
|
|
17
|
+
* owned the session, and who attested to the right to use them. Signed, that
|
|
18
|
+
* is a artifact both sides can point at afterwards — which is the difference
|
|
19
|
+
* between transferring responsibility and merely assuming it transferred.
|
|
20
|
+
*
|
|
21
|
+
* WHAT MUST NEVER BE IN THE RECORD. Proxy passwords and cookie values are
|
|
22
|
+
* credentials. They are hashed, never stored: `proxyCredentialSha256` lets two
|
|
23
|
+
* records be recognized as using the same credential without disclosing it,
|
|
24
|
+
* which is what an auditor actually needs. A compliance record is meant to be
|
|
25
|
+
* handed to a publisher or a court; a record that leaks the user's session
|
|
26
|
+
* cookie would be a breach dressed as an attestation.
|
|
27
|
+
*
|
|
28
|
+
* WHAT THIS IS NOT. There is no fingerprint spoofing here, no CDP patching, no
|
|
29
|
+
* captcha solving, and no rotating pool. A proxy is an egress path the user
|
|
30
|
+
* already has the right to use; the honest-mode UA is sent through it
|
|
31
|
+
* unchanged, and the identity honesty check still applies.
|
|
32
|
+
*
|
|
33
|
+
* 换 IP ≠ 换身份. Locale, timezone, viewport, and Client Hints stay the
|
|
34
|
+
* mode's bundle. Retuning timezone to a proxy's geo is refused.
|
|
35
|
+
*/
|
|
36
|
+
/** Who owned the network path a fetch actually left through. */
|
|
37
|
+
export type EgressOwner =
|
|
38
|
+
/** Our own network. We hold the responsibility. */
|
|
39
|
+
'operator'
|
|
40
|
+
/** The user's proxy. Their IP, their contract, their responsibility. */
|
|
41
|
+
| 'user';
|
|
42
|
+
/** Whose authenticated state, if any, a fetch carried. */
|
|
43
|
+
export type SessionOwner =
|
|
44
|
+
/** No session; the fetch was anonymous. */
|
|
45
|
+
'none'
|
|
46
|
+
/** The user's own logged-in state, supplied by them. */
|
|
47
|
+
| 'user';
|
|
48
|
+
/**
|
|
49
|
+
* A proxy the user brings. `url` carries scheme://host:port ONLY — credentials
|
|
50
|
+
* go in the separate fields so they are never accidentally logged, serialized
|
|
51
|
+
* into a trace, or embedded in an error message alongside the endpoint.
|
|
52
|
+
* `normalizeProxyConfig` enforces this rather than trusting the caller.
|
|
53
|
+
*/
|
|
54
|
+
export interface ProxyConfig {
|
|
55
|
+
/** e.g. `http://proxy.example:8080`. Userinfo here is stripped, not honoured. */
|
|
56
|
+
url: string;
|
|
57
|
+
username?: string;
|
|
58
|
+
password?: string;
|
|
59
|
+
}
|
|
60
|
+
/** One cookie of an inherited session, in the shape Playwright accepts. */
|
|
61
|
+
export interface SessionCookie {
|
|
62
|
+
name: string;
|
|
63
|
+
value: string;
|
|
64
|
+
domain: string;
|
|
65
|
+
path: string;
|
|
66
|
+
expires?: number;
|
|
67
|
+
httpOnly?: boolean;
|
|
68
|
+
secure?: boolean;
|
|
69
|
+
sameSite?: 'Strict' | 'Lax' | 'None';
|
|
70
|
+
}
|
|
71
|
+
/**
|
|
72
|
+
* The user's own logged-in state. Either explicit cookies or an opaque
|
|
73
|
+
* Playwright storageState blob; both are credentials and both are hashed
|
|
74
|
+
* rather than recorded.
|
|
75
|
+
*/
|
|
76
|
+
export interface SessionConfig {
|
|
77
|
+
cookies?: readonly SessionCookie[];
|
|
78
|
+
/** Serialized Playwright storageState JSON. */
|
|
79
|
+
storageState?: string;
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* The user's explicit acknowledgement that they hold the right to use the
|
|
83
|
+
* proxy and session they supplied, and that fetches made through them are
|
|
84
|
+
* theirs. Required — `normalizeAccessConfig` rejects a proxy or session
|
|
85
|
+
* without one, because an unattested transfer is not a transfer.
|
|
86
|
+
*/
|
|
87
|
+
export interface AccessAttestation {
|
|
88
|
+
/** Identifier for the accepting principal: account id, email, org handle. */
|
|
89
|
+
principal: string;
|
|
90
|
+
/** ISO timestamp of acceptance. */
|
|
91
|
+
at: string;
|
|
92
|
+
/**
|
|
93
|
+
* What they accepted, verbatim. Stored in the record so the claim can be
|
|
94
|
+
* read later without trusting our summary of it.
|
|
95
|
+
*/
|
|
96
|
+
statement: string;
|
|
97
|
+
}
|
|
98
|
+
export interface AccessConfig {
|
|
99
|
+
proxy?: ProxyConfig | null;
|
|
100
|
+
session?: SessionConfig | null;
|
|
101
|
+
attestation?: AccessAttestation | null;
|
|
102
|
+
}
|
|
103
|
+
/**
|
|
104
|
+
* The recorded, credential-free facts about whose access a fetch used. This is
|
|
105
|
+
* the field that makes the responsibility transfer provable; it is part of the
|
|
106
|
+
* canonical serialization, so it is covered by the record's contentHash and by
|
|
107
|
+
* any signature over it.
|
|
108
|
+
*/
|
|
109
|
+
export interface AccessFact {
|
|
110
|
+
egressOwner: EgressOwner;
|
|
111
|
+
/**
|
|
112
|
+
* `scheme://host:port` of the proxy, with any userinfo removed. Null under
|
|
113
|
+
* operator egress. The endpoint is not a secret and an auditor needs it;
|
|
114
|
+
* the credential is a secret and appears only as a hash.
|
|
115
|
+
*/
|
|
116
|
+
proxyEndpoint: string | null;
|
|
117
|
+
/**
|
|
118
|
+
* sha256 of the proxy credential. Recognizes reuse across records without
|
|
119
|
+
* disclosing the credential. Null when the proxy needs no credential.
|
|
120
|
+
*/
|
|
121
|
+
proxyCredentialSha256: string | null;
|
|
122
|
+
sessionOwner: SessionOwner;
|
|
123
|
+
/** sha256 of the session material. Null when no session was supplied. */
|
|
124
|
+
sessionSha256: string | null;
|
|
125
|
+
/** Principal who accepted responsibility. Null under operator egress. */
|
|
126
|
+
attestedBy: string | null;
|
|
127
|
+
/** ISO timestamp of that acceptance. Null under operator egress. */
|
|
128
|
+
attestedAt: string | null;
|
|
129
|
+
/** The statement they accepted, verbatim. Null under operator egress. */
|
|
130
|
+
attestationStatement: string | null;
|
|
131
|
+
}
|
|
132
|
+
/**
|
|
133
|
+
* The default: we supplied the egress, no session, nobody else's
|
|
134
|
+
* responsibility. Stated explicitly rather than left absent — "we did not
|
|
135
|
+
* track this" and "this was ours" are different claims, and a record that
|
|
136
|
+
* cannot tell them apart is not much of a record.
|
|
137
|
+
*/
|
|
138
|
+
export declare const OPERATOR_ACCESS_FACT: AccessFact;
|
|
139
|
+
/**
|
|
140
|
+
* What the caller must supply for a given escalation target. Returned to a
|
|
141
|
+
* blocked caller so the guidance is a machine-readable requirement, not a
|
|
142
|
+
* sentence in a log line.
|
|
143
|
+
*/
|
|
144
|
+
export type AccessRequirement =
|
|
145
|
+
/** Nothing more is needed; we can run this lane ourselves. */
|
|
146
|
+
'none'
|
|
147
|
+
/** The user's own logged-in session (cookies or storageState). */
|
|
148
|
+
| 'session'
|
|
149
|
+
/** The user's own proxy. */
|
|
150
|
+
| 'proxy';
|
|
151
|
+
export interface AccessGuidance {
|
|
152
|
+
requirement: AccessRequirement;
|
|
153
|
+
/** Why this is being asked for, in one sentence the caller can surface. */
|
|
154
|
+
rationale: string;
|
|
155
|
+
}
|
|
156
|
+
/**
|
|
157
|
+
* The access a lane needs from the user. Lanes we can run on our own return
|
|
158
|
+
* `none`; the two user-access lanes name what they need and why.
|
|
159
|
+
*
|
|
160
|
+
* Note what is NOT here: no lane's requirement is "solve the challenge" or
|
|
161
|
+
* "rotate to a fresh IP". `provider` returns `none` because the provider lane
|
|
162
|
+
* is our own contracted capacity, not the user's — its own gating lives in the
|
|
163
|
+
* provider adapter's robots check, not here.
|
|
164
|
+
*/
|
|
165
|
+
export declare function accessGuidanceForLane(lane: string): AccessGuidance;
|
|
166
|
+
//# sourceMappingURL=access.d.ts.map
|