@octocrawl/sdk 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +17 -0
- package/dist/index.cjs +1278 -0
- package/dist/index.js +1240 -0
- package/dist/types/cjs/client.d.ts +362 -0
- package/dist/types/cjs/contracts/access.d.ts +166 -0
- package/dist/types/cjs/contracts/actions.d.ts +191 -0
- package/dist/types/cjs/contracts/api.d.ts +891 -0
- package/dist/types/cjs/contracts/benchmark.d.ts +116 -0
- package/dist/types/cjs/contracts/checkpoint.d.ts +164 -0
- package/dist/types/cjs/contracts/compliance.d.ts +412 -0
- package/dist/types/cjs/contracts/crawl.d.ts +302 -0
- package/dist/types/cjs/contracts/delivery.d.ts +136 -0
- package/dist/types/cjs/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/cjs/contracts/execution.d.ts +197 -0
- package/dist/types/cjs/contracts/extractor.d.ts +379 -0
- package/dist/types/cjs/contracts/file.d.ts +117 -0
- package/dist/types/cjs/contracts/firecrawl.d.ts +258 -0
- package/dist/types/cjs/contracts/groundTruth.d.ts +77 -0
- package/dist/types/cjs/contracts/identityBundle.d.ts +70 -0
- package/dist/types/cjs/contracts/index.d.ts +30 -0
- package/dist/types/cjs/contracts/map.d.ts +180 -0
- package/dist/types/cjs/contracts/monitor.d.ts +217 -0
- package/dist/types/cjs/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/cjs/contracts/policy.d.ts +93 -0
- package/dist/types/cjs/contracts/proxy.d.ts +52 -0
- package/dist/types/cjs/contracts/recipe.d.ts +74 -0
- package/dist/types/cjs/contracts/regexSafety.d.ts +31 -0
- package/dist/types/cjs/contracts/result.d.ts +503 -0
- package/dist/types/cjs/contracts/session.d.ts +51 -0
- package/dist/types/cjs/contracts/ssrf.d.ts +16 -0
- package/dist/types/cjs/contracts/status.d.ts +27 -0
- package/dist/types/cjs/contracts/structured.d.ts +185 -0
- package/dist/types/cjs/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/cjs/contracts/tokens.d.ts +47 -0
- package/dist/types/cjs/index.d.ts +9 -0
- package/dist/types/cjs/package.json +1 -0
- package/dist/types/cjs/version.d.ts +8 -0
- package/dist/types/cjs/watcher.d.ts +151 -0
- package/dist/types/esm/client.d.ts +362 -0
- package/dist/types/esm/contracts/access.d.ts +166 -0
- package/dist/types/esm/contracts/actions.d.ts +191 -0
- package/dist/types/esm/contracts/api.d.ts +891 -0
- package/dist/types/esm/contracts/benchmark.d.ts +116 -0
- package/dist/types/esm/contracts/checkpoint.d.ts +164 -0
- package/dist/types/esm/contracts/compliance.d.ts +412 -0
- package/dist/types/esm/contracts/crawl.d.ts +302 -0
- package/dist/types/esm/contracts/delivery.d.ts +136 -0
- package/dist/types/esm/contracts/evidenceRecord.d.ts +192 -0
- package/dist/types/esm/contracts/execution.d.ts +197 -0
- package/dist/types/esm/contracts/extractor.d.ts +379 -0
- package/dist/types/esm/contracts/file.d.ts +117 -0
- package/dist/types/esm/contracts/firecrawl.d.ts +258 -0
- package/dist/types/esm/contracts/groundTruth.d.ts +77 -0
- package/dist/types/esm/contracts/identityBundle.d.ts +70 -0
- package/dist/types/esm/contracts/index.d.ts +30 -0
- package/dist/types/esm/contracts/map.d.ts +180 -0
- package/dist/types/esm/contracts/monitor.d.ts +217 -0
- package/dist/types/esm/contracts/monitorConfig.d.ts +9 -0
- package/dist/types/esm/contracts/policy.d.ts +93 -0
- package/dist/types/esm/contracts/proxy.d.ts +52 -0
- package/dist/types/esm/contracts/recipe.d.ts +74 -0
- package/dist/types/esm/contracts/regexSafety.d.ts +31 -0
- package/dist/types/esm/contracts/result.d.ts +503 -0
- package/dist/types/esm/contracts/session.d.ts +51 -0
- package/dist/types/esm/contracts/ssrf.d.ts +16 -0
- package/dist/types/esm/contracts/status.d.ts +27 -0
- package/dist/types/esm/contracts/structured.d.ts +185 -0
- package/dist/types/esm/contracts/tableMarkdown.d.ts +58 -0
- package/dist/types/esm/contracts/tokens.d.ts +47 -0
- package/dist/types/esm/index.d.ts +9 -0
- package/dist/types/esm/version.d.ts +8 -0
- package/dist/types/esm/watcher.d.ts +151 -0
- package/package.json +40 -0
package/dist/index.js
ADDED
|
@@ -0,0 +1,1240 @@
|
|
|
1
|
+
// packages/contracts/dist/extractor.js
|
|
2
|
+
var QuoteState;
|
|
3
|
+
(function(QuoteState2) {
|
|
4
|
+
QuoteState2["Present"] = "present";
|
|
5
|
+
QuoteState2["AbsentObserved"] = "absent_observed";
|
|
6
|
+
QuoteState2["Unobserved"] = "unobserved";
|
|
7
|
+
QuoteState2["Conflicting"] = "conflicting";
|
|
8
|
+
})(QuoteState || (QuoteState = {}));
|
|
9
|
+
|
|
10
|
+
// packages/contracts/dist/policy.js
|
|
11
|
+
var DEFAULT_NETWORK_POLICY = {
|
|
12
|
+
origin: "request",
|
|
13
|
+
privateAllowlist: [],
|
|
14
|
+
maxRedirects: 5,
|
|
15
|
+
maxBodyBytes: 10 * 1024 * 1024,
|
|
16
|
+
maxDecompressedBytes: 50 * 1024 * 1024,
|
|
17
|
+
perHostConcurrency: 2,
|
|
18
|
+
perHostMinDelayMs: 250,
|
|
19
|
+
respectRobotsTxt: true
|
|
20
|
+
};
|
|
21
|
+
|
|
22
|
+
// packages/contracts/dist/ssrf.js
|
|
23
|
+
var V4 = {
|
|
24
|
+
loopback: [ip4("127.0.0.0"), 8],
|
|
25
|
+
rfc1918a: [ip4("10.0.0.0"), 8],
|
|
26
|
+
rfc1918b: [ip4("172.16.0.0"), 12],
|
|
27
|
+
rfc1918c: [ip4("192.168.0.0"), 16],
|
|
28
|
+
linkLocal: [ip4("169.254.0.0"), 16],
|
|
29
|
+
cgnat: [ip4("100.64.0.0"), 10],
|
|
30
|
+
multicast: [ip4("224.0.0.0"), 4],
|
|
31
|
+
unspecified: ip4("0.0.0.0"),
|
|
32
|
+
metadata: ip4("169.254.169.254"),
|
|
33
|
+
ecsMetadata: ip4("169.254.170.2")
|
|
34
|
+
};
|
|
35
|
+
var V6 = {
|
|
36
|
+
unspecified: 0n,
|
|
37
|
+
loopback: 1n,
|
|
38
|
+
linkLocal: [0xfe80n << 112n, 10],
|
|
39
|
+
ula: [0xfcn << 120n, 7],
|
|
40
|
+
multicast: [0xffn << 120n, 8],
|
|
41
|
+
metadata: 0xfd00ec20000000000000000000000254n
|
|
42
|
+
};
|
|
43
|
+
function ip4(text) {
|
|
44
|
+
const parsed = parseV4(text);
|
|
45
|
+
if (parsed === null)
|
|
46
|
+
throw new Error(`bad fixture ip ${text}`);
|
|
47
|
+
return parsed;
|
|
48
|
+
}
|
|
49
|
+
function parseV4(text) {
|
|
50
|
+
const parts = text.split(".");
|
|
51
|
+
if (parts.length !== 4)
|
|
52
|
+
return null;
|
|
53
|
+
let n = 0;
|
|
54
|
+
for (const part of parts) {
|
|
55
|
+
if (!/^\d{1,3}$/.test(part))
|
|
56
|
+
return null;
|
|
57
|
+
const octet = Number(part);
|
|
58
|
+
if (octet > 255)
|
|
59
|
+
return null;
|
|
60
|
+
n = (n << 8) + octet;
|
|
61
|
+
}
|
|
62
|
+
return n >>> 0;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
// packages/contracts/dist/compliance.js
|
|
66
|
+
var RESEARCH_UA_COMMENT = "compatible; w2l-research/0.1; +https://github.com/77777R7/Octocrawl; research benchmark, one request per page";
|
|
67
|
+
var RESEARCH_USER_AGENT = `Mozilla/5.0 (${RESEARCH_UA_COMMENT})`;
|
|
68
|
+
var MAX_CONTACT_LENGTH = 200;
|
|
69
|
+
function contactIssue(contact) {
|
|
70
|
+
if (contact.length === 0 || contact.length > MAX_CONTACT_LENGTH)
|
|
71
|
+
return `must be 1 to ${MAX_CONTACT_LENGTH} characters`;
|
|
72
|
+
if (!/^[\x20-\x7e]+$/.test(contact))
|
|
73
|
+
return "must be printable ASCII";
|
|
74
|
+
if (/[()\\]/.test(contact))
|
|
75
|
+
return "must not contain parentheses or backslashes, which would end the User-Agent comment";
|
|
76
|
+
if (/(?:Chrome|Chromium)\/|HeadlessChrome/.test(contact))
|
|
77
|
+
return "must not name a browser product: research mode declares a bot";
|
|
78
|
+
return null;
|
|
79
|
+
}
|
|
80
|
+
var SEC_DECLARED_NAME = "W2L Research";
|
|
81
|
+
function isSecHost(host) {
|
|
82
|
+
const name = host.toLowerCase().replace(/\.$/, "");
|
|
83
|
+
return name === "sec.gov" || name.endsWith(".sec.gov");
|
|
84
|
+
}
|
|
85
|
+
function researchUserAgent(contact = null, host = null) {
|
|
86
|
+
if (contact === null)
|
|
87
|
+
return RESEARCH_USER_AGENT;
|
|
88
|
+
const issue = contactIssue(contact);
|
|
89
|
+
if (issue !== null)
|
|
90
|
+
throw new Error(`W2L_CONTACT ${issue}.`);
|
|
91
|
+
if (host !== null && isSecHost(host))
|
|
92
|
+
return `${SEC_DECLARED_NAME} ${contact}`;
|
|
93
|
+
return `Mozilla/5.0 (${RESEARCH_UA_COMMENT}; contact: ${contact})`;
|
|
94
|
+
}
|
|
95
|
+
var CHROME_MAJOR_FLOOR = 128;
|
|
96
|
+
function browserUserAgent(chromeMajor) {
|
|
97
|
+
return `Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/${chromeMajor}.0.0.0 Safari/537.36`;
|
|
98
|
+
}
|
|
99
|
+
function browserClientHints(chromeMajor) {
|
|
100
|
+
return {
|
|
101
|
+
"sec-ch-ua": `"Chromium";v="${chromeMajor}", "Google Chrome";v="${chromeMajor}", "Not;A=Brand";v="24"`,
|
|
102
|
+
"sec-ch-ua-mobile": "?0",
|
|
103
|
+
"sec-ch-ua-platform": '"macOS"'
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
function mobileBrowserUserAgent(chromeMajor) {
|
|
107
|
+
return `Mozilla/5.0 (Linux; Android 14; Pixel 7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/${chromeMajor}.0.0.0 Mobile Safari/537.36`;
|
|
108
|
+
}
|
|
109
|
+
function mobileBrowserClientHints(chromeMajor) {
|
|
110
|
+
return {
|
|
111
|
+
"sec-ch-ua": `"Chromium";v="${chromeMajor}", "Google Chrome";v="${chromeMajor}", "Not;A=Brand";v="24"`,
|
|
112
|
+
"sec-ch-ua-mobile": "?1",
|
|
113
|
+
"sec-ch-ua-platform": '"Android"'
|
|
114
|
+
};
|
|
115
|
+
}
|
|
116
|
+
var BROWSER_FINGERPRINT = {
|
|
117
|
+
locale: "en-US",
|
|
118
|
+
timezoneId: "America/Los_Angeles",
|
|
119
|
+
viewport: { width: 1280, height: 800 },
|
|
120
|
+
screen: { width: 1920, height: 1080 },
|
|
121
|
+
deviceScaleFactor: 2,
|
|
122
|
+
isMobile: false,
|
|
123
|
+
hasTouch: false
|
|
124
|
+
};
|
|
125
|
+
function modeIdentity(mode, chromeMajor = CHROME_MAJOR_FLOOR, contact = null, host = null, device = "desktop") {
|
|
126
|
+
const userAgent = device === "mobile" ? mobileBrowserUserAgent(chromeMajor) : browserUserAgent(chromeMajor);
|
|
127
|
+
const clientHints = device === "mobile" ? mobileBrowserClientHints(chromeMajor) : browserClientHints(chromeMajor);
|
|
128
|
+
switch (mode) {
|
|
129
|
+
case "research":
|
|
130
|
+
return { mode, userAgent: researchUserAgent(contact, host), clientHints: {}, respectsRobots: true, lane: "browser_local" };
|
|
131
|
+
case "standard":
|
|
132
|
+
return { mode, userAgent, clientHints, respectsRobots: true, lane: "browser_local", device };
|
|
133
|
+
case "authed":
|
|
134
|
+
return { mode, userAgent, clientHints, respectsRobots: true, lane: "browser_local_authed", device };
|
|
135
|
+
case "proxy":
|
|
136
|
+
return { mode, userAgent, clientHints, respectsRobots: true, lane: "browser_proxy", device };
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
var MODE_IDENTITIES = {
|
|
140
|
+
research: modeIdentity("research"),
|
|
141
|
+
standard: modeIdentity("standard"),
|
|
142
|
+
authed: modeIdentity("authed"),
|
|
143
|
+
proxy: modeIdentity("proxy")
|
|
144
|
+
};
|
|
145
|
+
|
|
146
|
+
// packages/contracts/dist/regexSafety.js
|
|
147
|
+
function set(ranges) {
|
|
148
|
+
const sorted = [...ranges].sort((a, b) => a[0] - b[0]);
|
|
149
|
+
const merged = [];
|
|
150
|
+
for (const [low, high] of sorted) {
|
|
151
|
+
const previous = merged[merged.length - 1];
|
|
152
|
+
if (previous !== void 0 && low <= previous[1] + 1)
|
|
153
|
+
previous[1] = Math.max(previous[1], high);
|
|
154
|
+
else
|
|
155
|
+
merged.push([low, high]);
|
|
156
|
+
}
|
|
157
|
+
return { kind: "set", ranges: merged };
|
|
158
|
+
}
|
|
159
|
+
var DIGIT = set([[48, 57]]);
|
|
160
|
+
var WORD = set([[48, 57], [65, 90], [95, 95], [97, 122]]);
|
|
161
|
+
var SPACE = set([[9, 13], [32, 32], [160, 160], [5760, 5760], [8192, 8202], [8232, 8233], [8239, 8239], [8287, 8287], [12288, 12288], [65279, 65279]]);
|
|
162
|
+
|
|
163
|
+
// packages/contracts/dist/file.js
|
|
164
|
+
var DEFAULT_MAX_FILE_BYTES = 50 * 1024 * 1024;
|
|
165
|
+
var MAX_FILE_BYTES_CEILING = 500 * 1024 * 1024;
|
|
166
|
+
|
|
167
|
+
// packages/contracts/dist/api.js
|
|
168
|
+
var DEFAULT_SCRAPE_TIMEOUT_MS = 3e5;
|
|
169
|
+
var WEBHOOK_EVENTS = ["started", "page", "completed", "failed", "cancelled"];
|
|
170
|
+
var MAX_WEBHOOK_HEADERS = 32;
|
|
171
|
+
var MAX_WEBHOOK_METADATA_ENTRIES = 32;
|
|
172
|
+
var MAX_WEBHOOK_METADATA_VALUE_LENGTH = 1e3;
|
|
173
|
+
var DEFAULT_MAP_TIMEOUT_MS = 6e4;
|
|
174
|
+
var WS_TOKEN_PROTOCOL_PREFIX = "w2l.token.";
|
|
175
|
+
var API_ERROR_CODES = ["invalid_json", "invalid_request", "unsupported_parameter", "unsupported_format", "unauthorized", "not_found", "conflict", "internal_error"];
|
|
176
|
+
function isApiErrorCode(value) {
|
|
177
|
+
return API_ERROR_CODES.includes(value);
|
|
178
|
+
}
|
|
179
|
+
var RATE_LIMITED_CODE = "rate_limited";
|
|
180
|
+
var PAGE_KEYS = ["onlyMainContent", "waitFor", "timeout", "maxFileBytes", "includeTags", "excludeTags", "headers", "mobile", "skipTlsVerification", "fastMode", "blockAds", "removeBase64Images", "parsers", "maxAge", "minAge", "storeInCache", "lockdown"];
|
|
181
|
+
var ATTRIBUTION_KEYS = ["origin", "integration"];
|
|
182
|
+
var SCRAPE_KEYS = ["url", "mode", "allowlistedDomains", "formats", "includeLinks", "debug", "robotsOverride", "actions", "handoff", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
183
|
+
var CRAWL_SCOPE_KEYS = ["regexOnFullURL", "ignoreQueryParameters", "deduplicateSimilarURLs", "crawlEntireDomain", "allowSubdomains", "allowExternalLinks"];
|
|
184
|
+
var CRAWL_KEYS = ["url", "mode", "maxPages", "maxDepth", "useCached", "allowlistedDomains", "formats", "includeLinks", "includePaths", "excludePaths", ...CRAWL_SCOPE_KEYS, "sitemap", "maxConcurrency", "idempotencyKey", "webhook", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
185
|
+
var BATCH_KEYS = ["urls", "mode", "formats", "includeLinks", "robotsOverrides", "maxConcurrency", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "appendToId", "webhook", "actions", ...PAGE_KEYS, ...ATTRIBUTION_KEYS];
|
|
186
|
+
var BATCH_APPEND_KEYS = ["urls", "appendToId", "ignoreInvalidURLs", "allowExternalLinks", "includeSubdomains", "idempotencyKey", "robotsOverrides", ...ATTRIBUTION_KEYS];
|
|
187
|
+
var MAP_SCOPE_KEYS = ["includeSubdomains", "ignoreQueryParameters", "regexOnFullURL", "crawlEntireDomain", "deduplicateSimilarURLs"];
|
|
188
|
+
var MAP_KEYS = ["url", "mode", "limit", "timeout", "search", "sitemap", ...MAP_SCOPE_KEYS, "includePaths", "excludePaths", ...ATTRIBUTION_KEYS];
|
|
189
|
+
var WEBHOOK_HEADERS_MESSAGE = `webhook.headers must be an object of at most ${MAX_WEBHOOK_HEADERS} string values`;
|
|
190
|
+
var WEBHOOK_METADATA_MESSAGE = `webhook.metadata must be an object of at most ${MAX_WEBHOOK_METADATA_ENTRIES} string values of at most ${MAX_WEBHOOK_METADATA_VALUE_LENGTH} characters`;
|
|
191
|
+
var WEBHOOK_EVENTS_MESSAGE = `webhook.events must be a non-empty array of ${WEBHOOK_EVENTS.join(", ")} without duplicates`;
|
|
192
|
+
var SCHEMA_STRUCTURE = ["type", "properties", "required", "items", "additionalProperties", "enum", "const", "$ref", "$defs", "definitions", "anyOf", "oneOf"];
|
|
193
|
+
var SCHEMA_NUMBERS = ["minimum", "maximum", "exclusiveMinimum", "exclusiveMaximum", "multipleOf"];
|
|
194
|
+
var SCHEMA_COUNTS = ["minLength", "maxLength", "minItems", "maxItems"];
|
|
195
|
+
var SCHEMA_ANNOTATIONS = ["title", "description", "$comment", "default", "examples", "deprecated", "readOnly", "writeOnly", "format"];
|
|
196
|
+
var SCHEMA_KEYS = /* @__PURE__ */ new Set([...SCHEMA_STRUCTURE, ...SCHEMA_NUMBERS, ...SCHEMA_COUNTS, "pattern", "uniqueItems", ...SCHEMA_ANNOTATIONS]);
|
|
197
|
+
var STRING_FORMATS = ["markdown", "links", "json", "html", "rawHtml", "images", "tables", "screenshot"];
|
|
198
|
+
var FORMAT_NAMES = [...STRING_FORMATS, "attributes", "list"];
|
|
199
|
+
var MIN_SCREENSHOT_VIEWPORT = { width: 320, height: 240 };
|
|
200
|
+
var SCREENSHOT_VIEWPORT_MESSAGE = `screenshot viewport must be {width, height} with integers within ${MIN_SCREENSHOT_VIEWPORT.width}..${BROWSER_FINGERPRINT.screen.width} by ${MIN_SCREENSHOT_VIEWPORT.height}..${BROWSER_FINGERPRINT.screen.height}`;
|
|
201
|
+
|
|
202
|
+
// packages/contracts/dist/firecrawl.js
|
|
203
|
+
var SHIM_ATTRIBUTION = ["origin", "integration"];
|
|
204
|
+
var SHIM_MAP_KEYS = ["url", "search", "sitemap", "ignoreSitemap", "sitemapOnly", "includeSubdomains", "ignoreQueryParameters", "limit", "timeout", ...SHIM_ATTRIBUTION];
|
|
205
|
+
|
|
206
|
+
// packages/contracts/dist/evidenceRecord.js
|
|
207
|
+
var keysOf = () => (keys) => keys;
|
|
208
|
+
var EVIDENCE_RECORD_KEYS = {
|
|
209
|
+
record: keysOf()(["schemaVersion", "requestedUrl", "finalUrl", "redirectChain", "fetchedAt", "httpStatus", "status", "reason", "lane", "robotsDecision", "rawSha256", "contentEncoding", "outputSha256", "extractor", "fieldEvidence", "artifacts", "proxy", "identity", "pageActions"]),
|
|
210
|
+
redirectChain: keysOf()(["urls", "complete"]),
|
|
211
|
+
robotsDecision: keysOf()(["decision", "robotsUrl", "robotsSha256", "unreachable", "crawlDelayMs", "userOverride"]),
|
|
212
|
+
outputSha256: keysOf()(["markdown", "json"]),
|
|
213
|
+
extractor: keysOf()(["name", "version", "commit"]),
|
|
214
|
+
fieldEvidence: keysOf()(["source", "locator"]),
|
|
215
|
+
artifact: keysOf()(["kind", "path", "sha256", "bytes", "contentType"]),
|
|
216
|
+
identity: keysOf()(["userAgent", "mode", "contact", "device", "requestHeaders"]),
|
|
217
|
+
pageActions: keysOf()(["steps", "scriptRan"]),
|
|
218
|
+
pageActionStep: keysOf()(["type", "outcome"]),
|
|
219
|
+
requestHeader: keysOf()(["name", "valueSha256"])
|
|
220
|
+
};
|
|
221
|
+
|
|
222
|
+
// packages/sdk/src/version.ts
|
|
223
|
+
var SDK_VERSION = "0.3.0";
|
|
224
|
+
|
|
225
|
+
// packages/sdk/src/watcher.ts
|
|
226
|
+
var DEFAULT_WATCH_POLL_INTERVAL_MS = 2e3;
|
|
227
|
+
var MIN_WATCH_POLL_INTERVAL_MS = 250;
|
|
228
|
+
var POLL_FAILURE_LIMIT = 5;
|
|
229
|
+
var FINISHED = /* @__PURE__ */ new Set(["completed", "failed", "cancelled"]);
|
|
230
|
+
var STREAM_FRAMES = /* @__PURE__ */ new Set(["catchup", "document", "snapshot", "done", "error"]);
|
|
231
|
+
function pageCursor(page) {
|
|
232
|
+
const json = JSON.stringify({ createdAt: page.createdAt, id: page.id });
|
|
233
|
+
const bytes = new TextEncoder().encode(json);
|
|
234
|
+
let binary = "";
|
|
235
|
+
for (const byte of bytes) binary += String.fromCharCode(byte);
|
|
236
|
+
return btoa(binary).replace(/\+/g, "-").replace(/\//g, "_").replace(/=+$/, "");
|
|
237
|
+
}
|
|
238
|
+
function parseSseBlock(block) {
|
|
239
|
+
let event;
|
|
240
|
+
let id;
|
|
241
|
+
const data = [];
|
|
242
|
+
for (const line of block.split(/\r?\n/)) {
|
|
243
|
+
if (line.length === 0 || line.startsWith(":")) continue;
|
|
244
|
+
const colon = line.indexOf(":");
|
|
245
|
+
const field = colon === -1 ? line : line.slice(0, colon);
|
|
246
|
+
const value = colon === -1 ? "" : line.slice(colon + 1).replace(/^ /, "");
|
|
247
|
+
if (field === "event") event = value;
|
|
248
|
+
else if (field === "data") data.push(value);
|
|
249
|
+
else if (field === "id") id = value;
|
|
250
|
+
}
|
|
251
|
+
if (event === void 0 || data.length === 0) return null;
|
|
252
|
+
return frameOf(event, data.join("\n"), id);
|
|
253
|
+
}
|
|
254
|
+
function frameOf(type, json, cursor) {
|
|
255
|
+
if (!STREAM_FRAMES.has(type)) return null;
|
|
256
|
+
let payload;
|
|
257
|
+
try {
|
|
258
|
+
payload = JSON.parse(json);
|
|
259
|
+
} catch {
|
|
260
|
+
return null;
|
|
261
|
+
}
|
|
262
|
+
if (payload === null || typeof payload !== "object") return null;
|
|
263
|
+
if (type === "error") return { type, error: payload };
|
|
264
|
+
if (type === "document") return cursor === void 0 ? null : { type, data: payload, cursor };
|
|
265
|
+
return { type, data: payload };
|
|
266
|
+
}
|
|
267
|
+
function parseSocketFrame(data) {
|
|
268
|
+
const text = typeof data === "string" ? data : data instanceof ArrayBuffer ? new TextDecoder().decode(data) : null;
|
|
269
|
+
if (text === null) return null;
|
|
270
|
+
let parsed;
|
|
271
|
+
try {
|
|
272
|
+
parsed = JSON.parse(text);
|
|
273
|
+
} catch {
|
|
274
|
+
return null;
|
|
275
|
+
}
|
|
276
|
+
if (parsed === null || typeof parsed !== "object") return null;
|
|
277
|
+
const frame = parsed;
|
|
278
|
+
if (typeof frame.type !== "string" || !STREAM_FRAMES.has(frame.type)) return null;
|
|
279
|
+
if (frame.type === "error") return frame.error !== null && typeof frame.error === "object" ? { type: "error", error: frame.error } : null;
|
|
280
|
+
if (frame.data === null || typeof frame.data !== "object") return null;
|
|
281
|
+
if (frame.type === "document") return typeof frame.cursor === "string" ? { type: "document", data: frame.data, cursor: frame.cursor } : null;
|
|
282
|
+
return { type: frame.type, data: frame.data };
|
|
283
|
+
}
|
|
284
|
+
function statusOf(error) {
|
|
285
|
+
return error !== null && typeof error === "object" && typeof error.status === "number" ? error.status : null;
|
|
286
|
+
}
|
|
287
|
+
function sleep(ms, signal) {
|
|
288
|
+
return new Promise((resolve, reject) => {
|
|
289
|
+
if (signal.aborted) {
|
|
290
|
+
reject(signal.reason);
|
|
291
|
+
return;
|
|
292
|
+
}
|
|
293
|
+
const timer = setTimeout(() => {
|
|
294
|
+
signal.removeEventListener("abort", abort);
|
|
295
|
+
resolve();
|
|
296
|
+
}, Math.max(0, ms));
|
|
297
|
+
const abort = () => {
|
|
298
|
+
clearTimeout(timer);
|
|
299
|
+
reject(signal.reason);
|
|
300
|
+
};
|
|
301
|
+
signal.addEventListener("abort", abort, { once: true });
|
|
302
|
+
});
|
|
303
|
+
}
|
|
304
|
+
var JobWatcher = class extends EventTarget {
|
|
305
|
+
jobId;
|
|
306
|
+
kind;
|
|
307
|
+
/** Every document delivered so far, each step once, in arrival order. */
|
|
308
|
+
data = [];
|
|
309
|
+
/** The job's status as last reported; null before the first report. */
|
|
310
|
+
status = null;
|
|
311
|
+
/** The transport delivering events; null before one is open and after the watch ends. */
|
|
312
|
+
transport = null;
|
|
313
|
+
client;
|
|
314
|
+
transportOption;
|
|
315
|
+
pollIntervalMs;
|
|
316
|
+
socketConstructor;
|
|
317
|
+
controller = new AbortController();
|
|
318
|
+
seen = /* @__PURE__ */ new Set();
|
|
319
|
+
log = [];
|
|
320
|
+
waiters = [];
|
|
321
|
+
/** The cursor of the last document delivered: where the next transport, or a poll, continues from. */
|
|
322
|
+
cursor;
|
|
323
|
+
lastSnapshot = "";
|
|
324
|
+
finished = false;
|
|
325
|
+
closed = false;
|
|
326
|
+
timer;
|
|
327
|
+
constructor(client, jobId, options = {}) {
|
|
328
|
+
super();
|
|
329
|
+
if (typeof jobId !== "string" || jobId.length === 0) throw new TypeError("jobId must be a non-empty string");
|
|
330
|
+
const kind = options.kind ?? "crawl";
|
|
331
|
+
if (kind !== "crawl" && kind !== "batch") throw new TypeError("kind must be 'crawl' or 'batch'");
|
|
332
|
+
const transport = options.transport ?? "auto";
|
|
333
|
+
if (!["auto", "websocket", "sse", "poll"].includes(transport)) throw new TypeError("transport must be 'auto', 'websocket', 'sse' or 'poll'");
|
|
334
|
+
const pollIntervalMs = options.pollIntervalMs ?? DEFAULT_WATCH_POLL_INTERVAL_MS;
|
|
335
|
+
if (!Number.isFinite(pollIntervalMs) || pollIntervalMs < MIN_WATCH_POLL_INTERVAL_MS) throw new TypeError(`pollIntervalMs must be at least ${MIN_WATCH_POLL_INTERVAL_MS}`);
|
|
336
|
+
if (options.timeoutMs !== void 0 && !(Number.isFinite(options.timeoutMs) && options.timeoutMs >= 0)) throw new TypeError("timeoutMs must be a finite number of milliseconds, 0 or more");
|
|
337
|
+
if (options.after !== void 0 && (typeof options.after !== "string" || options.after.length === 0)) throw new TypeError("after must be a non-empty cursor");
|
|
338
|
+
this.client = client;
|
|
339
|
+
this.jobId = jobId;
|
|
340
|
+
this.kind = kind;
|
|
341
|
+
this.transportOption = transport;
|
|
342
|
+
this.pollIntervalMs = pollIntervalMs;
|
|
343
|
+
this.socketConstructor = options.WebSocket === null ? void 0 : options.WebSocket ?? globalThis.WebSocket;
|
|
344
|
+
this.cursor = options.after;
|
|
345
|
+
if (options.signal !== void 0) {
|
|
346
|
+
if (options.signal.aborted) this.close();
|
|
347
|
+
else options.signal.addEventListener("abort", () => this.close(), { once: true });
|
|
348
|
+
}
|
|
349
|
+
if (options.timeoutMs !== void 0) {
|
|
350
|
+
const timeoutMs = options.timeoutMs;
|
|
351
|
+
this.timer = setTimeout(() => this.fail({ code: "watcher_timeout", message: `job ${jobId} did not finish within ${timeoutMs} ms (last status: ${this.status ?? "unknown"})` }), timeoutMs);
|
|
352
|
+
}
|
|
353
|
+
void this.run();
|
|
354
|
+
}
|
|
355
|
+
/** Stops watching: no further event, the job untouched. */
|
|
356
|
+
close() {
|
|
357
|
+
if (this.closed) return;
|
|
358
|
+
this.closed = true;
|
|
359
|
+
this.stop();
|
|
360
|
+
}
|
|
361
|
+
/** The events from the start, then as they arrive, until done, error or close. */
|
|
362
|
+
[Symbol.asyncIterator]() {
|
|
363
|
+
let index = 0;
|
|
364
|
+
return {
|
|
365
|
+
next: async () => {
|
|
366
|
+
for (; ; ) {
|
|
367
|
+
if (index < this.log.length) return { value: this.log[index++], done: false };
|
|
368
|
+
if (this.finished || this.closed) return { value: void 0, done: true };
|
|
369
|
+
await new Promise((resolve) => this.waiters.push(resolve));
|
|
370
|
+
}
|
|
371
|
+
},
|
|
372
|
+
return: async () => ({ value: void 0, done: true })
|
|
373
|
+
};
|
|
374
|
+
}
|
|
375
|
+
get stopped() {
|
|
376
|
+
return this.finished || this.closed;
|
|
377
|
+
}
|
|
378
|
+
async run() {
|
|
379
|
+
const order = this.transportOption === "auto" ? ["websocket", "sse", "poll"] : [this.transportOption];
|
|
380
|
+
for (const transport of order) {
|
|
381
|
+
if (this.stopped) return;
|
|
382
|
+
const outcome = transport === "websocket" ? await this.watchSocket() : transport === "sse" ? await this.watchEvents() : await this.poll();
|
|
383
|
+
if (outcome !== "fallback") return;
|
|
384
|
+
}
|
|
385
|
+
this.transport = null;
|
|
386
|
+
if (!this.stopped) this.fail({ code: "transport_unavailable", message: `no transport could watch ${this.kind} ${this.jobId} (tried ${order.join(", ")})` });
|
|
387
|
+
}
|
|
388
|
+
/** The stream route of this job, with the cursor to resume after when one is known. */
|
|
389
|
+
streamUrl(suffix) {
|
|
390
|
+
const url = new URL(`${this.client.baseUrl}/v1/${this.kind === "crawl" ? "crawl" : "batches"}/${encodeURIComponent(this.jobId)}/${suffix}`);
|
|
391
|
+
if (this.cursor !== void 0) url.searchParams.set("after", this.cursor);
|
|
392
|
+
if (suffix === "ws") url.protocol = url.protocol === "https:" ? "wss:" : "ws:";
|
|
393
|
+
return url.href;
|
|
394
|
+
}
|
|
395
|
+
/** The WebSocket route; the token, when there is one, as the `w2l.token.<token>` subprotocol, since the WebSocket API sets no header. */
|
|
396
|
+
watchSocket() {
|
|
397
|
+
const Socket = this.socketConstructor;
|
|
398
|
+
if (Socket === void 0) return Promise.resolve("fallback");
|
|
399
|
+
let socket;
|
|
400
|
+
try {
|
|
401
|
+
socket = new Socket(this.streamUrl("ws"), this.client.token === void 0 || this.client.token.length === 0 ? void 0 : [`${WS_TOKEN_PROTOCOL_PREFIX}${this.client.token}`]);
|
|
402
|
+
} catch {
|
|
403
|
+
return Promise.resolve("fallback");
|
|
404
|
+
}
|
|
405
|
+
return new Promise((resolve) => {
|
|
406
|
+
let settled = false;
|
|
407
|
+
const settle = (outcome) => {
|
|
408
|
+
if (settled) return;
|
|
409
|
+
settled = true;
|
|
410
|
+
this.controller.signal.removeEventListener("abort", onAbort);
|
|
411
|
+
resolve(outcome);
|
|
412
|
+
};
|
|
413
|
+
const onAbort = () => {
|
|
414
|
+
try {
|
|
415
|
+
socket.close(1e3, "closed");
|
|
416
|
+
} catch {
|
|
417
|
+
}
|
|
418
|
+
;
|
|
419
|
+
settle("closed");
|
|
420
|
+
};
|
|
421
|
+
this.controller.signal.addEventListener("abort", onAbort, { once: true });
|
|
422
|
+
let opened = false;
|
|
423
|
+
socket.addEventListener("open", () => {
|
|
424
|
+
opened = true;
|
|
425
|
+
if (!this.stopped) this.transport = "websocket";
|
|
426
|
+
});
|
|
427
|
+
socket.addEventListener("message", (event) => {
|
|
428
|
+
const frame = parseSocketFrame(event.data);
|
|
429
|
+
if (frame !== null) this.handle(frame);
|
|
430
|
+
if (this.stopped) {
|
|
431
|
+
try {
|
|
432
|
+
socket.close(1e3, "done");
|
|
433
|
+
} catch {
|
|
434
|
+
}
|
|
435
|
+
;
|
|
436
|
+
settle(this.finished ? "done" : "closed");
|
|
437
|
+
}
|
|
438
|
+
});
|
|
439
|
+
socket.addEventListener("close", (event) => {
|
|
440
|
+
if (this.finished) {
|
|
441
|
+
settle("done");
|
|
442
|
+
return;
|
|
443
|
+
}
|
|
444
|
+
if (this.closed) {
|
|
445
|
+
settle("closed");
|
|
446
|
+
return;
|
|
447
|
+
}
|
|
448
|
+
if (event.code === 4404) {
|
|
449
|
+
this.fail({ code: "not_found", message: `${this.kind} ${this.jobId} not found` });
|
|
450
|
+
settle("fatal");
|
|
451
|
+
return;
|
|
452
|
+
}
|
|
453
|
+
if (event.code === 4400) {
|
|
454
|
+
this.fail({ code: "invalid_request", message: event.reason ?? "cursor is not one this API issued" });
|
|
455
|
+
settle("fatal");
|
|
456
|
+
return;
|
|
457
|
+
}
|
|
458
|
+
settle("fallback");
|
|
459
|
+
});
|
|
460
|
+
socket.addEventListener("error", () => {
|
|
461
|
+
if (opened) return;
|
|
462
|
+
try {
|
|
463
|
+
socket.close();
|
|
464
|
+
} catch {
|
|
465
|
+
}
|
|
466
|
+
;
|
|
467
|
+
settle(this.closed ? "closed" : "fallback");
|
|
468
|
+
});
|
|
469
|
+
});
|
|
470
|
+
}
|
|
471
|
+
/** The server-sent events route: 404 leaves it to polling, 401/403 ends the watch, a stream that ends before done hands over from the last cursor. */
|
|
472
|
+
async watchEvents() {
|
|
473
|
+
const url = this.streamUrl("events");
|
|
474
|
+
let res;
|
|
475
|
+
try {
|
|
476
|
+
res = await this.client.fetch(url, { headers: this.client.headers({ accept: "text/event-stream" }), signal: this.controller.signal });
|
|
477
|
+
} catch {
|
|
478
|
+
return this.stopped ? "closed" : "fallback";
|
|
479
|
+
}
|
|
480
|
+
if (res.status === 404) return "fallback";
|
|
481
|
+
if (res.status === 401 || res.status === 403) {
|
|
482
|
+
this.fail({ code: "unauthorized", message: `GET ${new URL(url).pathname} answered ${res.status}` });
|
|
483
|
+
return "fatal";
|
|
484
|
+
}
|
|
485
|
+
if (!res.ok || res.body === null) return "fallback";
|
|
486
|
+
if (this.stopped) return "closed";
|
|
487
|
+
this.transport = "sse";
|
|
488
|
+
const reader = res.body.getReader();
|
|
489
|
+
const decoder = new TextDecoder();
|
|
490
|
+
let buffer = "";
|
|
491
|
+
const dispatch = (block) => {
|
|
492
|
+
const frame = parseSseBlock(block);
|
|
493
|
+
if (frame !== null) this.handle(frame);
|
|
494
|
+
};
|
|
495
|
+
try {
|
|
496
|
+
for (; ; ) {
|
|
497
|
+
const { value, done } = await reader.read();
|
|
498
|
+
if (done) break;
|
|
499
|
+
buffer += decoder.decode(value, { stream: true });
|
|
500
|
+
for (let boundary = buffer.search(/\r?\n\r?\n/); boundary !== -1; boundary = buffer.search(/\r?\n\r?\n/)) {
|
|
501
|
+
const block = buffer.slice(0, boundary);
|
|
502
|
+
buffer = buffer.slice(boundary).replace(/^\r?\n\r?\n/, "");
|
|
503
|
+
dispatch(block);
|
|
504
|
+
if (this.stopped) {
|
|
505
|
+
try {
|
|
506
|
+
await reader.cancel();
|
|
507
|
+
} catch {
|
|
508
|
+
}
|
|
509
|
+
;
|
|
510
|
+
return this.finished ? "done" : "closed";
|
|
511
|
+
}
|
|
512
|
+
}
|
|
513
|
+
}
|
|
514
|
+
if (buffer.trim().length > 0) dispatch(buffer);
|
|
515
|
+
} catch {
|
|
516
|
+
if (this.stopped) return this.finished ? "done" : "closed";
|
|
517
|
+
}
|
|
518
|
+
if (this.finished) return "done";
|
|
519
|
+
return this.closed ? "closed" : "fallback";
|
|
520
|
+
}
|
|
521
|
+
/** The status and listing routes every pollIntervalMs: the documents since the last cursor, a snapshot, and done once the status is terminal and the last pages are read. */
|
|
522
|
+
async poll() {
|
|
523
|
+
this.transport = "poll";
|
|
524
|
+
const signal = this.controller.signal;
|
|
525
|
+
const cursors = { pages: this.cursor, errors: this.cursor, items: this.cursor };
|
|
526
|
+
const drain = async (key, list) => {
|
|
527
|
+
for (; ; ) {
|
|
528
|
+
const page = await list(cursors[key]);
|
|
529
|
+
for (const item of page.items) this.document(item, pageCursor(item));
|
|
530
|
+
const last = page.items[page.items.length - 1];
|
|
531
|
+
cursors[key] = page.nextCursor ?? (last === void 0 ? cursors[key] : pageCursor(last));
|
|
532
|
+
if (!page.hasMore) return;
|
|
533
|
+
}
|
|
534
|
+
};
|
|
535
|
+
let failures = 0;
|
|
536
|
+
for (; ; ) {
|
|
537
|
+
if (this.stopped) return "closed";
|
|
538
|
+
try {
|
|
539
|
+
const report = this.kind === "crawl" ? await this.client.getCrawl(this.jobId, { signal }) : await this.client.getBatch(this.jobId, { signal });
|
|
540
|
+
if (this.kind === "batch") {
|
|
541
|
+
await drain("items", (cursor) => this.client.getBatchItems(this.jobId, { limit: 50, ...cursor === void 0 ? {} : { cursor } }, { signal }));
|
|
542
|
+
} else {
|
|
543
|
+
await drain("pages", (cursor) => this.client.getCrawlPages(this.jobId, { limit: 100, includeDuplicates: true, ...cursor === void 0 ? {} : { cursor } }, { signal }));
|
|
544
|
+
await drain("errors", (cursor) => this.client.getCrawlErrors(this.jobId, { limit: 100, ...cursor === void 0 ? {} : { cursor } }, { signal }));
|
|
545
|
+
}
|
|
546
|
+
if (this.stopped) return "closed";
|
|
547
|
+
if (FINISHED.has(report.status)) {
|
|
548
|
+
this.done(report);
|
|
549
|
+
return "done";
|
|
550
|
+
}
|
|
551
|
+
this.snapshot(report);
|
|
552
|
+
failures = 0;
|
|
553
|
+
} catch (error) {
|
|
554
|
+
if (this.stopped) return "closed";
|
|
555
|
+
const status = statusOf(error);
|
|
556
|
+
if (status === 404) {
|
|
557
|
+
this.fail({ code: "not_found", message: `${this.kind} ${this.jobId} not found` });
|
|
558
|
+
return "fatal";
|
|
559
|
+
}
|
|
560
|
+
if (status === 401 || status === 403) {
|
|
561
|
+
this.fail({ code: "unauthorized", message: `polling ${this.kind} ${this.jobId} answered ${status}` });
|
|
562
|
+
return "fatal";
|
|
563
|
+
}
|
|
564
|
+
if (++failures >= POLL_FAILURE_LIMIT) {
|
|
565
|
+
this.fail({ code: "poll_failed", message: `polling ${this.kind} ${this.jobId} failed ${failures} times in a row: ${error instanceof Error ? error.message : String(error)}` });
|
|
566
|
+
return "fatal";
|
|
567
|
+
}
|
|
568
|
+
}
|
|
569
|
+
try {
|
|
570
|
+
await sleep(this.pollIntervalMs, signal);
|
|
571
|
+
} catch {
|
|
572
|
+
return "closed";
|
|
573
|
+
}
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
handle(frame) {
|
|
577
|
+
if (this.stopped) return;
|
|
578
|
+
if (frame.type === "document") this.document(frame.data, frame.cursor);
|
|
579
|
+
else if (frame.type === "done") this.done(frame.data);
|
|
580
|
+
else if (frame.type === "error") this.fail(frame.error);
|
|
581
|
+
else this.snapshot(frame.data);
|
|
582
|
+
}
|
|
583
|
+
document(page, cursor) {
|
|
584
|
+
if (this.stopped || this.seen.has(page.id)) return;
|
|
585
|
+
this.seen.add(page.id);
|
|
586
|
+
this.data.push(page);
|
|
587
|
+
this.cursor = cursor;
|
|
588
|
+
this.emit({ type: "document", data: page });
|
|
589
|
+
}
|
|
590
|
+
snapshot(report) {
|
|
591
|
+
if (this.stopped) return;
|
|
592
|
+
this.status = report.status;
|
|
593
|
+
const encoded = JSON.stringify(report);
|
|
594
|
+
if (encoded === this.lastSnapshot) return;
|
|
595
|
+
this.lastSnapshot = encoded;
|
|
596
|
+
this.emit({ type: "snapshot", data: report });
|
|
597
|
+
}
|
|
598
|
+
done(report) {
|
|
599
|
+
if (this.stopped) return;
|
|
600
|
+
this.status = report.status;
|
|
601
|
+
this.finished = true;
|
|
602
|
+
this.emit({ type: "done", data: report });
|
|
603
|
+
this.stop();
|
|
604
|
+
}
|
|
605
|
+
fail(error) {
|
|
606
|
+
if (this.stopped) return;
|
|
607
|
+
this.finished = true;
|
|
608
|
+
this.emit({ type: "error", error });
|
|
609
|
+
this.stop();
|
|
610
|
+
}
|
|
611
|
+
emit(event) {
|
|
612
|
+
this.log.push(event);
|
|
613
|
+
this.dispatchEvent(new CustomEvent(event.type, { detail: event.type === "error" ? event.error : event.data }));
|
|
614
|
+
this.wake();
|
|
615
|
+
}
|
|
616
|
+
stop() {
|
|
617
|
+
if (this.timer !== void 0) {
|
|
618
|
+
clearTimeout(this.timer);
|
|
619
|
+
this.timer = void 0;
|
|
620
|
+
}
|
|
621
|
+
this.transport = this.finished ? this.transport : null;
|
|
622
|
+
this.controller.abort();
|
|
623
|
+
this.wake();
|
|
624
|
+
}
|
|
625
|
+
wake() {
|
|
626
|
+
const waiting = this.waiters.splice(0);
|
|
627
|
+
for (const resume of waiting) resume();
|
|
628
|
+
}
|
|
629
|
+
};
|
|
630
|
+
|
|
631
|
+
// packages/sdk/src/client.ts
|
|
632
|
+
function environmentToken() {
|
|
633
|
+
try {
|
|
634
|
+
const token = globalThis.process?.env?.["W2L_API_TOKEN"];
|
|
635
|
+
return token === void 0 || token.length === 0 ? void 0 : token;
|
|
636
|
+
} catch {
|
|
637
|
+
return void 0;
|
|
638
|
+
}
|
|
639
|
+
}
|
|
640
|
+
var SCRAPE_ANSWER_MARGIN_MS = 3e4;
|
|
641
|
+
var MAP_ANSWER_MARGIN_MS = 5e3;
|
|
642
|
+
function headersWait(ms) {
|
|
643
|
+
if (globalThis.process?.versions?.undici === void 0) return void 0;
|
|
644
|
+
return {
|
|
645
|
+
dispatch(options, handler) {
|
|
646
|
+
const global = globalThis;
|
|
647
|
+
const dispatcher = global[/* @__PURE__ */ Symbol.for("undici.globalDispatcher.2")] ?? global[/* @__PURE__ */ Symbol.for("undici.globalDispatcher.1")];
|
|
648
|
+
if (dispatcher === void 0) throw new Error("no global fetch dispatcher");
|
|
649
|
+
return dispatcher.dispatch({ ...options, headersTimeout: ms }, handler);
|
|
650
|
+
}
|
|
651
|
+
};
|
|
652
|
+
}
|
|
653
|
+
var SDK_ORIGIN = `js-sdk@${SDK_VERSION}`;
|
|
654
|
+
var WaitTimeoutError = class extends Error {
|
|
655
|
+
constructor(taskId, last, timeoutMs, options) {
|
|
656
|
+
super(last === null ? `task ${taskId} status not read within ${timeoutMs} ms` : `task ${taskId} still ${last.status} after ${timeoutMs} ms`, options);
|
|
657
|
+
this.taskId = taskId;
|
|
658
|
+
this.last = last;
|
|
659
|
+
this.timeoutMs = timeoutMs;
|
|
660
|
+
}
|
|
661
|
+
taskId;
|
|
662
|
+
last;
|
|
663
|
+
timeoutMs;
|
|
664
|
+
name = "WaitTimeoutError";
|
|
665
|
+
};
|
|
666
|
+
var CRAWL_PAGE_MAX_LIMIT = 1e3;
|
|
667
|
+
var BATCH_ITEM_MAX_LIMIT = 50;
|
|
668
|
+
var BATCH_MAX_URLS = 1e3;
|
|
669
|
+
function chunkUrls(urls, chunkSize = 100) {
|
|
670
|
+
if (!Number.isInteger(chunkSize) || chunkSize < 1 || chunkSize > BATCH_MAX_URLS) throw new RangeError(`chunkSize must be an integer between 1 and ${BATCH_MAX_URLS}`);
|
|
671
|
+
const chunks = [];
|
|
672
|
+
for (let start = 0; start < urls.length; start += chunkSize) chunks.push(urls.slice(start, start + chunkSize));
|
|
673
|
+
return chunks;
|
|
674
|
+
}
|
|
675
|
+
function hrefOf(url) {
|
|
676
|
+
try {
|
|
677
|
+
return new URL(url).href;
|
|
678
|
+
} catch {
|
|
679
|
+
return url;
|
|
680
|
+
}
|
|
681
|
+
}
|
|
682
|
+
function checkPaginationLimits(options) {
|
|
683
|
+
if (options.maxPages !== void 0 && !(Number.isInteger(options.maxPages) && options.maxPages >= 0)) throw new RangeError("maxPages must be an integer, 0 or more");
|
|
684
|
+
if (options.maxResults !== void 0 && !(Number.isInteger(options.maxResults) && options.maxResults >= 1)) throw new RangeError("maxResults must be an integer, 1 or more");
|
|
685
|
+
if (options.maxWaitMs !== void 0 && !(Number.isFinite(options.maxWaitMs) && options.maxWaitMs >= 0)) throw new RangeError("maxWaitMs must be a finite number of milliseconds, 0 or more");
|
|
686
|
+
}
|
|
687
|
+
var W2LError = class extends Error {
|
|
688
|
+
constructor(message, status, code, method, path, body, retryAfterMs2 = null, agentHints = []) {
|
|
689
|
+
super(message);
|
|
690
|
+
this.status = status;
|
|
691
|
+
this.code = code;
|
|
692
|
+
this.method = method;
|
|
693
|
+
this.path = path;
|
|
694
|
+
this.body = body;
|
|
695
|
+
this.retryAfterMs = retryAfterMs2;
|
|
696
|
+
this.agentHints = agentHints;
|
|
697
|
+
}
|
|
698
|
+
status;
|
|
699
|
+
code;
|
|
700
|
+
method;
|
|
701
|
+
path;
|
|
702
|
+
body;
|
|
703
|
+
retryAfterMs;
|
|
704
|
+
agentHints;
|
|
705
|
+
name = "W2LError";
|
|
706
|
+
};
|
|
707
|
+
async function responseError(method, path, res, message) {
|
|
708
|
+
const text = await res.text();
|
|
709
|
+
let body = text;
|
|
710
|
+
try {
|
|
711
|
+
body = JSON.parse(text);
|
|
712
|
+
} catch {
|
|
713
|
+
}
|
|
714
|
+
const fields = body !== null && typeof body === "object" ? body : {};
|
|
715
|
+
const code = isApiErrorCode(fields.code) || fields.code === RATE_LIMITED_CODE ? fields.code : void 0;
|
|
716
|
+
const agentHints = Array.isArray(fields.agentHints) ? fields.agentHints.filter((hint) => typeof hint === "string") : [];
|
|
717
|
+
return new W2LError(message ?? `${method} ${path} failed: ${res.status} ${text}`, res.status, code, method, path, body, retryAfterMs(res.headers.get("retry-after")), agentHints);
|
|
718
|
+
}
|
|
719
|
+
function retryAfterMs(value) {
|
|
720
|
+
if (value === null) return null;
|
|
721
|
+
const trimmed = value.trim();
|
|
722
|
+
if (/^\d+$/.test(trimmed)) return Number(trimmed) * 1e3;
|
|
723
|
+
const at = /^[+-]?[\d.]+$/.test(trimmed) ? Number.NaN : Date.parse(trimmed);
|
|
724
|
+
return Number.isFinite(at) ? Math.max(0, at - Date.now()) : null;
|
|
725
|
+
}
|
|
726
|
+
var MAX_RETRY_AFTER_MS = 6e4;
|
|
727
|
+
function retryDelayMs(error, failures) {
|
|
728
|
+
if (error instanceof W2LError) {
|
|
729
|
+
if (error.status !== 408 && error.status !== 429 && error.status < 500) return null;
|
|
730
|
+
if (error.retryAfterMs !== null) return error.retryAfterMs <= MAX_RETRY_AFTER_MS ? error.retryAfterMs : null;
|
|
731
|
+
} else if (!(error instanceof TypeError)) return null;
|
|
732
|
+
return Math.min(1e4, 1e3 * 2 ** (failures - 1));
|
|
733
|
+
}
|
|
734
|
+
function checkWaitOptions(options) {
|
|
735
|
+
for (const name of ["pollIntervalMs", "timeoutMs"]) {
|
|
736
|
+
const value = options[name];
|
|
737
|
+
if (value !== void 0 && !(Number.isFinite(value) && value >= 0)) throw new RangeError(`${name} must be a finite number of milliseconds, 0 or more`);
|
|
738
|
+
}
|
|
739
|
+
if (options.maxRetries !== void 0 && !(Number.isInteger(options.maxRetries) && options.maxRetries >= 0)) throw new RangeError("maxRetries must be an integer, 0 or more");
|
|
740
|
+
}
|
|
741
|
+
var FINISHED2 = ["completed", "failed", "cancelled"];
|
|
742
|
+
var W2L = class {
|
|
743
|
+
baseUrl;
|
|
744
|
+
token;
|
|
745
|
+
fetchImpl;
|
|
746
|
+
/** The platform's fetch, not one passed in options, whose own limits are the caller's. */
|
|
747
|
+
platformFetch;
|
|
748
|
+
constructor(options) {
|
|
749
|
+
this.baseUrl = options.baseUrl.replace(/\/$/, "");
|
|
750
|
+
this.token = options.token ?? environmentToken();
|
|
751
|
+
this.fetchImpl = options.fetch ?? fetch;
|
|
752
|
+
this.platformFetch = options.fetch === void 0;
|
|
753
|
+
}
|
|
754
|
+
async scrape(url, opts = {}, request = {}) {
|
|
755
|
+
const deadlineMs = Number.isInteger(opts.timeout) ? Math.min(Math.max(opts.timeout, 0), DEFAULT_SCRAPE_TIMEOUT_MS) : DEFAULT_SCRAPE_TIMEOUT_MS;
|
|
756
|
+
const handedOver = opts.handoff !== void 0 && opts.handoff !== false;
|
|
757
|
+
return this.post("/v1/scrape", { ...opts, url, origin: originOf(opts, request) }, 200, request, handedOver ? 0 : deadlineMs + SCRAPE_ANSWER_MARGIN_MS);
|
|
758
|
+
}
|
|
759
|
+
/** The record of one scrape call, by the `scrapeId` its response carried (`metadata.scrapeId`); a W2LError with code `not_found` for an id the server has no record of. */
|
|
760
|
+
async getScrape(id, request = {}) {
|
|
761
|
+
return this.get(`/v1/scrapes/${encodeURIComponent(id)}`, request, `scrape not found: ${id}`);
|
|
762
|
+
}
|
|
763
|
+
/**
|
|
764
|
+
* The URLs of a site from its sitemaps and its start page's links, without
|
|
765
|
+
* fetching each page (POST /v1/map). The API answers by the map's deadline
|
|
766
|
+
* with what it found; the SDK waits that long plus MAP_ANSWER_MARGIN_MS.
|
|
767
|
+
*/
|
|
768
|
+
async map(url, opts = {}, request = {}) {
|
|
769
|
+
const deadlineMs = Number.isInteger(opts.timeout) ? Math.max(opts.timeout, 0) : DEFAULT_MAP_TIMEOUT_MS;
|
|
770
|
+
return this.post("/v1/map", { ...opts, url, origin: originOf(opts, request) }, 200, request, deadlineMs + MAP_ANSWER_MARGIN_MS);
|
|
771
|
+
}
|
|
772
|
+
/** The record of one map, by the `id` its response carried; a W2LError with code `not_found` for an id the server has no record of. */
|
|
773
|
+
async getMap(id, request = {}) {
|
|
774
|
+
return this.get(`/v1/maps/${encodeURIComponent(id)}`, request, `map not found: ${id}`);
|
|
775
|
+
}
|
|
776
|
+
async crawl(url, opts = {}, request = {}) {
|
|
777
|
+
return this.post("/v1/crawl", { ...opts, url, origin: originOf(opts, request) }, 202, request);
|
|
778
|
+
}
|
|
779
|
+
/**
|
|
780
|
+
* Starts a batch: `{ taskId }`, plus `invalidURLs` (the entries skipped)
|
|
781
|
+
* when `ignoreInvalidURLs` was on; with `idempotencyKey` a retried start
|
|
782
|
+
* returns the first one's answer with `replayed: true`; with `appendToId`
|
|
783
|
+
* the URLs join that batch (see appendToBatch) and the answer carries
|
|
784
|
+
* `requested` and `appended`.
|
|
785
|
+
*/
|
|
786
|
+
async batchScrape(urls, opts = {}, request = {}) {
|
|
787
|
+
return this.post("/v1/batches", { ...opts, urls, origin: originOf(opts, request) }, 202, request);
|
|
788
|
+
}
|
|
789
|
+
/**
|
|
790
|
+
* Adds URLs to an existing batch (`appendToId`): the job keeps its mode,
|
|
791
|
+
* formats, includeLinks, maxConcurrency and page options, and its run picks
|
|
792
|
+
* the URLs up (a completed batch runs again for them). The answer carries
|
|
793
|
+
* `requested`, the job's URLs now, and `appended`. A cancelled or failed
|
|
794
|
+
* batch, a total over 1000 or a URL already in the batch is a W2LError.
|
|
795
|
+
*/
|
|
796
|
+
async appendToBatch(id, urls, opts = {}, request = {}) {
|
|
797
|
+
return this.batchScrape(urls, { ...opts, appendToId: id }, request);
|
|
798
|
+
}
|
|
799
|
+
/**
|
|
800
|
+
* Runs a list of any length as batches of `chunkSize` URLs (default 100),
|
|
801
|
+
* one after another: each job is started, waited for (as waitBatch, with
|
|
802
|
+
* the WaitOptions) and listed before the next starts. The items are merged
|
|
803
|
+
* in the order the URLs were submitted; the jobs stay on the server as
|
|
804
|
+
* ordinary batches, each with its own task directory. A caller's
|
|
805
|
+
* `idempotencyKey` becomes `<key>:<chunkIndex>` per job, so a retry of the
|
|
806
|
+
* whole call replays the jobs that went through. A WaitTimeoutError or
|
|
807
|
+
* W2LError from any job ends the call, naming that job; the earlier jobs
|
|
808
|
+
* are complete and the later chunks were never sent.
|
|
809
|
+
*/
|
|
810
|
+
async batchScrapeChunked(urls, opts = {}, options = {}) {
|
|
811
|
+
if (opts.appendToId !== void 0) throw new TypeError("batchScrapeChunked cannot append; use appendToBatch");
|
|
812
|
+
const { chunkSize = 100, itemLimit = BATCH_ITEM_MAX_LIMIT, ...wait } = options;
|
|
813
|
+
checkWaitOptions(wait);
|
|
814
|
+
if (!Number.isInteger(itemLimit) || itemLimit < 1 || itemLimit > BATCH_ITEM_MAX_LIMIT) throw new RangeError(`itemLimit must be an integer between 1 and ${BATCH_ITEM_MAX_LIMIT}`);
|
|
815
|
+
const chunks = chunkUrls(urls, chunkSize);
|
|
816
|
+
const jobs = [];
|
|
817
|
+
const items = [];
|
|
818
|
+
const invalidURLs = [];
|
|
819
|
+
for (const [index, chunk] of chunks.entries()) {
|
|
820
|
+
const accepted = await this.batchScrape(chunk, { ...opts, ...opts.idempotencyKey === void 0 ? {} : { idempotencyKey: `${opts.idempotencyKey}:${index}` } }, wait);
|
|
821
|
+
const report = await this.waitBatch(accepted.taskId, wait);
|
|
822
|
+
const order = new Map(chunk.map((url, position) => [hrefOf(url), position]));
|
|
823
|
+
const listed = [];
|
|
824
|
+
for await (const item of this.listBatchItems(accepted.taskId, { limit: itemLimit }, wait)) listed.push(item);
|
|
825
|
+
items.push(...listed.sort((a, b) => (order.get(a.url) ?? Number.MAX_SAFE_INTEGER) - (order.get(b.url) ?? Number.MAX_SAFE_INTEGER)));
|
|
826
|
+
jobs.push({ taskId: accepted.taskId, urls: chunk.length, report, ...accepted.invalidURLs === void 0 ? {} : { invalidURLs: accepted.invalidURLs } });
|
|
827
|
+
if (accepted.invalidURLs !== void 0) invalidURLs.push(...accepted.invalidURLs);
|
|
828
|
+
}
|
|
829
|
+
return { jobs, items, invalidURLs };
|
|
830
|
+
}
|
|
831
|
+
async getBatch(id, request = {}) {
|
|
832
|
+
return this.get(`/v1/batches/${encodeURIComponent(id)}`, request, `batch not found: ${id}`);
|
|
833
|
+
}
|
|
834
|
+
/** The batch's failed, blocked, cancelled and budget-cut items across every attempt, in pages of up to 1000 (`limit`, `cursor`), with `robotsBlocked`, the URLs robots.txt refused. */
|
|
835
|
+
async getBatchErrors(id, options = {}, request = {}) {
|
|
836
|
+
const params = new URLSearchParams();
|
|
837
|
+
if (options.cursor !== void 0) params.set("cursor", options.cursor);
|
|
838
|
+
if (options.limit !== void 0) params.set("limit", String(options.limit));
|
|
839
|
+
const suffix = params.size === 0 ? "" : `?${params.toString()}`;
|
|
840
|
+
return this.get(`/v1/batches/${encodeURIComponent(id)}/errors${suffix}`, request, `batch not found: ${id}`);
|
|
841
|
+
}
|
|
842
|
+
async getBatchItems(id, options = {}, request = {}) {
|
|
843
|
+
return this.getPageList(`/v1/batches/${encodeURIComponent(id)}/items`, options, request);
|
|
844
|
+
}
|
|
845
|
+
/** Every item of a batch, page by page (at most 50 per request), or as many as the PaginationLimits allow; the generator's return value says where it stopped. */
|
|
846
|
+
listBatchItems(id, options = {}, request = {}) {
|
|
847
|
+
return this.paginate((query) => this.getBatchItems(id, query, request), options, request, BATCH_ITEM_MAX_LIMIT, "batch items");
|
|
848
|
+
}
|
|
849
|
+
/** The items listBatchItems would yield under the same options, collected, with where the listing stopped. */
|
|
850
|
+
async collectBatchItems(id, options = {}, request = {}) {
|
|
851
|
+
return collect(this.listBatchItems(id, options, request));
|
|
852
|
+
}
|
|
853
|
+
/** A batch's status and its items in one answer, every item unless the PaginationLimits stop the listing. */
|
|
854
|
+
async getBatchDocuments(id, options = {}, request = {}) {
|
|
855
|
+
const report = await this.getBatch(id, request);
|
|
856
|
+
const { items, nextCursor, stoppedBy } = await this.collectBatchItems(id, options, request);
|
|
857
|
+
return { report, items, nextCursor, stoppedBy };
|
|
858
|
+
}
|
|
859
|
+
/** Polls a batch until it completes, fails or is cancelled. Items come from listBatchItems. */
|
|
860
|
+
async waitBatch(id, options = {}) {
|
|
861
|
+
return this.waitFor(id, (request) => this.getBatch(id, request), options);
|
|
862
|
+
}
|
|
863
|
+
/** Starts a batch, waits for it (as waitBatch) and lists every item, failed ones included. */
|
|
864
|
+
async batchAndWait(urls, opts = {}, wait = {}) {
|
|
865
|
+
checkWaitOptions(wait);
|
|
866
|
+
const { taskId } = await this.batchScrape(urls, opts, wait);
|
|
867
|
+
const report = await this.waitBatch(taskId, wait);
|
|
868
|
+
const items = [];
|
|
869
|
+
for await (const item of this.listBatchItems(taskId, { limit: 50 }, wait)) items.push(item);
|
|
870
|
+
return { taskId, report, items };
|
|
871
|
+
}
|
|
872
|
+
/**
|
|
873
|
+
* Save the person's login to a site (a domain or a page URL) from the
|
|
874
|
+
* Chrome they use, as `octocrawl login import` does, on a server on their
|
|
875
|
+
* machine. Chrome asks them "Allow remote debugging?": the answer comes
|
|
876
|
+
* once they click Allow (within `approveTimeoutMs`, default 2 minutes).
|
|
877
|
+
* The saved login's cookies never leave the server: the answer names the
|
|
878
|
+
* domain, how many cookies and their hash.
|
|
879
|
+
*/
|
|
880
|
+
async importLogin(site, opts = {}, request = {}) {
|
|
881
|
+
return this.post("/v1/logins/import", { ...opts, site }, 200, request, (opts.approveTimeoutMs ?? 12e4) + 3e4);
|
|
882
|
+
}
|
|
883
|
+
/** The person's saved logins, without their cookies. */
|
|
884
|
+
async listLogins(request = {}) {
|
|
885
|
+
return this.get("/v1/logins", request);
|
|
886
|
+
}
|
|
887
|
+
/** Forget a saved login; a W2LError with code `not_found` when none was saved for the site. */
|
|
888
|
+
async removeLogin(site, request = {}) {
|
|
889
|
+
const path = `/v1/logins/${encodeURIComponent(site)}`;
|
|
890
|
+
const res = await this.fetchImpl(`${this.baseUrl}${path}`, { method: "DELETE", headers: this.headers(), signal: request.signal });
|
|
891
|
+
if (!res.ok) throw await responseError("DELETE", path, res, res.status === 404 ? `no login saved for ${site}` : void 0);
|
|
892
|
+
return await res.json();
|
|
893
|
+
}
|
|
894
|
+
/**
|
|
895
|
+
* Hands a finished batch's items that a check stopped (a captcha, a
|
|
896
|
+
* challenge, a login wall) to the person in their own Chrome, on a local
|
|
897
|
+
* server: each opens in a new tab, they get through it, and W2L reads the
|
|
898
|
+
* page there. Answers when every item is read or given up, so it waits for
|
|
899
|
+
* the person: `waitMs` is how long, per page (default 10 minutes). On
|
|
900
|
+
* Node the SDK waits for the answer as long as that takes (no 300 s limit
|
|
901
|
+
* on the response headers); `request.signal` ends the wait.
|
|
902
|
+
*/
|
|
903
|
+
async handOffBatch(id, body = {}, request = {}) {
|
|
904
|
+
return this.post(`/v1/batches/${encodeURIComponent(id)}/handoff`, body, 200, request, 0);
|
|
905
|
+
}
|
|
906
|
+
async cancelBatch(id, request = {}) {
|
|
907
|
+
return this.post(`/v1/batches/${encodeURIComponent(id)}/cancel`, void 0, 200, request);
|
|
908
|
+
}
|
|
909
|
+
/**
|
|
910
|
+
* Watches a crawl (`kind: 'crawl'`, the default) or a batch (`kind: 'batch'`)
|
|
911
|
+
* as it runs: `document` events with each page as it is recorded, `snapshot`
|
|
912
|
+
* events with the report, one `done` with the terminal report, or `error`.
|
|
913
|
+
* `transport: 'auto'` (default) tries the WebSocket route, then server-sent
|
|
914
|
+
* events, then polling (`pollIntervalMs`, default 2000, at least 250), each
|
|
915
|
+
* taking over from the last document seen; `timeoutMs` ends the watch with a
|
|
916
|
+
* `watcher_timeout` error while the job keeps running. `close()` stops
|
|
917
|
+
* watching only; cancelCrawl / cancelBatch stay explicit.
|
|
918
|
+
*/
|
|
919
|
+
watcher(jobId, options = {}) {
|
|
920
|
+
return new JobWatcher(this.watcherClient(), jobId, options);
|
|
921
|
+
}
|
|
922
|
+
/** Starts a crawl and returns its watcher (as `crawl()` then `watcher(taskId, { kind: 'crawl' })`). */
|
|
923
|
+
async crawlAndWatch(url, opts = {}, watch = {}, request = {}) {
|
|
924
|
+
const { taskId } = await this.crawl(url, opts, request);
|
|
925
|
+
return this.watcher(taskId, { ...watch, kind: "crawl" });
|
|
926
|
+
}
|
|
927
|
+
/** Starts a batch and returns its watcher (as `batchScrape()` then `watcher(taskId, { kind: 'batch' })`). */
|
|
928
|
+
async batchScrapeAndWatch(urls, opts = {}, watch = {}, request = {}) {
|
|
929
|
+
const { taskId } = await this.batchScrape(urls, opts, request);
|
|
930
|
+
return this.watcher(taskId, { ...watch, kind: "batch" });
|
|
931
|
+
}
|
|
932
|
+
/** What a watcher needs of this client: the server, the token and fetch it was given, and the routes it polls, each carrying the bearer header. */
|
|
933
|
+
watcherClient() {
|
|
934
|
+
return {
|
|
935
|
+
baseUrl: this.baseUrl,
|
|
936
|
+
token: this.token,
|
|
937
|
+
fetch: this.fetchImpl,
|
|
938
|
+
headers: (extra) => this.headers(extra),
|
|
939
|
+
getCrawl: (id, request) => this.getCrawl(id, request),
|
|
940
|
+
getBatch: (id, request) => this.getBatch(id, request),
|
|
941
|
+
getCrawlPages: (id, options, request) => this.getCrawlPages(id, options, request),
|
|
942
|
+
getCrawlErrors: (id, options, request) => this.getCrawlErrors(id, options, request),
|
|
943
|
+
getBatchItems: (id, options, request) => this.getBatchItems(id, options, request)
|
|
944
|
+
};
|
|
945
|
+
}
|
|
946
|
+
async getCrawl(id, request = {}) {
|
|
947
|
+
return this.get(`/v1/crawl/${encodeURIComponent(id)}`, request, `crawl not found: ${id}`);
|
|
948
|
+
}
|
|
949
|
+
/** The crawls the API process is running, with each one's start URL, status, pages so far and options; empty when nothing runs. */
|
|
950
|
+
async getActiveCrawls(request = {}) {
|
|
951
|
+
return this.get("/v1/crawl/active", request);
|
|
952
|
+
}
|
|
953
|
+
/** Polls a crawl until it completes, fails or is cancelled. Pages come from listCrawlPages. */
|
|
954
|
+
async waitCrawl(id, options = {}) {
|
|
955
|
+
return this.waitFor(id, (request) => this.getCrawl(id, request), options);
|
|
956
|
+
}
|
|
957
|
+
/** Starts a crawl, waits for it (as waitCrawl) and lists every page and every error of its latest attempt. */
|
|
958
|
+
async crawlAndWait(url, opts = {}, wait = {}) {
|
|
959
|
+
checkWaitOptions(wait);
|
|
960
|
+
const { taskId } = await this.crawl(url, opts, wait);
|
|
961
|
+
const report = await this.waitCrawl(taskId, wait);
|
|
962
|
+
const pages = [];
|
|
963
|
+
for await (const page of this.listCrawlPages(taskId, { limit: 100 }, wait)) pages.push(page);
|
|
964
|
+
const errors = [];
|
|
965
|
+
let cursor;
|
|
966
|
+
do {
|
|
967
|
+
const page = await this.getCrawlErrors(taskId, { limit: 100, cursor }, wait);
|
|
968
|
+
errors.push(...page.items);
|
|
969
|
+
cursor = page.hasMore ? page.nextCursor ?? void 0 : void 0;
|
|
970
|
+
if (page.hasMore && cursor === void 0) throw new Error("crawl errors response omitted nextCursor");
|
|
971
|
+
} while (cursor !== void 0);
|
|
972
|
+
return { taskId, report, pages, errors };
|
|
973
|
+
}
|
|
974
|
+
async getCrawlPages(id, options = {}, request = {}) {
|
|
975
|
+
return this.getPageList(`/v1/crawl/${encodeURIComponent(id)}/pages`, options, request);
|
|
976
|
+
}
|
|
977
|
+
/**
|
|
978
|
+
* A crawl's pages, page by page, or as many as the PaginationLimits allow;
|
|
979
|
+
* the generator's return value says where it stopped. The latest attempt's
|
|
980
|
+
* pages unless `attemptId` names another; a resume with `useCached` records
|
|
981
|
+
* the pages it reuses in its new attempt, so that attempt normally holds
|
|
982
|
+
* every page.
|
|
983
|
+
*/
|
|
984
|
+
listCrawlPages(id, options = {}, request = {}) {
|
|
985
|
+
return this.paginate((query) => this.getCrawlPages(id, query, request), options, request, CRAWL_PAGE_MAX_LIMIT, "crawl pages");
|
|
986
|
+
}
|
|
987
|
+
/** The pages listCrawlPages would yield under the same options, collected, with where the listing stopped. */
|
|
988
|
+
async collectCrawlPages(id, options = {}, request = {}) {
|
|
989
|
+
return collect(this.listCrawlPages(id, options, request));
|
|
990
|
+
}
|
|
991
|
+
/** A crawl's status and its pages in one answer, every page unless the PaginationLimits stop the listing. */
|
|
992
|
+
async getCrawlDocuments(id, options = {}, request = {}) {
|
|
993
|
+
const report = await this.getCrawl(id, request);
|
|
994
|
+
const { items, nextCursor, stoppedBy } = await this.collectCrawlPages(id, options, request);
|
|
995
|
+
return { report, pages: items, nextCursor, stoppedBy };
|
|
996
|
+
}
|
|
997
|
+
/**
|
|
998
|
+
* Follow a listing's cursors within its limits. Each page is requested no
|
|
999
|
+
* larger than the items still wanted, so a stop at `maxResults` leaves a
|
|
1000
|
+
* cursor that continues exactly after the last item returned. A page's
|
|
1001
|
+
* `hasMore` without a cursor is the API breaking its contract and throws.
|
|
1002
|
+
*/
|
|
1003
|
+
async *paginate(fetchPage, options, request, maxLimit, what) {
|
|
1004
|
+
checkPaginationLimits(options);
|
|
1005
|
+
const { maxPages, maxResults, maxWaitMs, ...query } = options;
|
|
1006
|
+
const startedAt = Date.now();
|
|
1007
|
+
let cursor;
|
|
1008
|
+
let pagesAfterFirst = 0;
|
|
1009
|
+
let returned = 0;
|
|
1010
|
+
for (; ; ) {
|
|
1011
|
+
const remaining = maxResults === void 0 ? void 0 : maxResults - returned;
|
|
1012
|
+
const limit = remaining === void 0 ? query.limit : Math.min(remaining, query.limit ?? maxLimit);
|
|
1013
|
+
const page = await fetchPage({ ...query, ...limit === void 0 ? {} : { limit }, ...cursor === void 0 ? {} : { cursor } });
|
|
1014
|
+
for (const item of page.items) {
|
|
1015
|
+
if (remaining !== void 0 && returned >= maxResults) break;
|
|
1016
|
+
request.signal?.throwIfAborted();
|
|
1017
|
+
returned++;
|
|
1018
|
+
yield item;
|
|
1019
|
+
}
|
|
1020
|
+
const next = page.hasMore ? page.nextCursor ?? void 0 : void 0;
|
|
1021
|
+
if (page.hasMore && next === void 0) throw new Error(`${what} response omitted nextCursor`);
|
|
1022
|
+
if (next === void 0) return { nextCursor: null, stoppedBy: "end" };
|
|
1023
|
+
if (maxResults !== void 0 && returned >= maxResults) return { nextCursor: next, stoppedBy: "maxResults" };
|
|
1024
|
+
if (maxPages !== void 0 && pagesAfterFirst >= maxPages) return { nextCursor: next, stoppedBy: "maxPages" };
|
|
1025
|
+
if (maxWaitMs !== void 0 && Date.now() - startedAt >= maxWaitMs) return { nextCursor: next, stoppedBy: "maxWait" };
|
|
1026
|
+
cursor = next;
|
|
1027
|
+
pagesAfterFirst++;
|
|
1028
|
+
}
|
|
1029
|
+
}
|
|
1030
|
+
async getCrawlErrors(id, options = {}, request = {}) {
|
|
1031
|
+
return this.getPageList(`/v1/crawl/${encodeURIComponent(id)}/errors`, options, request);
|
|
1032
|
+
}
|
|
1033
|
+
async cancelCrawl(id, request = {}) {
|
|
1034
|
+
return this.post(`/v1/crawl/${encodeURIComponent(id)}/cancel`, void 0, 200, request);
|
|
1035
|
+
}
|
|
1036
|
+
/** Restarts a paused or failed crawl with the options it was started with; follow it with waitCrawl. */
|
|
1037
|
+
async resumeCrawl(id, request = {}) {
|
|
1038
|
+
return this.post(`/v1/crawl/${encodeURIComponent(id)}/resume`, void 0, 202, request);
|
|
1039
|
+
}
|
|
1040
|
+
async createMonitor(input, request = {}) {
|
|
1041
|
+
return this.post("/v1/monitors", input, 201, request);
|
|
1042
|
+
}
|
|
1043
|
+
async previewMonitor(input, request = {}) {
|
|
1044
|
+
return this.post("/v1/monitors/preview", input, 200, request);
|
|
1045
|
+
}
|
|
1046
|
+
async reviseMonitor(id, input, request = {}) {
|
|
1047
|
+
return this.post(`/v1/monitors/${encodeURIComponent(id)}/revisions`, input, 201, request);
|
|
1048
|
+
}
|
|
1049
|
+
async listMonitors(request = {}) {
|
|
1050
|
+
return this.get("/v1/monitors", request);
|
|
1051
|
+
}
|
|
1052
|
+
async getMonitor(id, request = {}) {
|
|
1053
|
+
return this.get(`/v1/monitors/${encodeURIComponent(id)}`, request);
|
|
1054
|
+
}
|
|
1055
|
+
async getMonitorRun(id, runId, request = {}) {
|
|
1056
|
+
return this.get(`/v1/monitors/${encodeURIComponent(id)}/runs/${encodeURIComponent(runId)}`, request);
|
|
1057
|
+
}
|
|
1058
|
+
/** Durable run: returns after enqueue; client disconnect does not cancel it. */
|
|
1059
|
+
async enqueueMonitorRun(id, input = {}, request = {}) {
|
|
1060
|
+
return this.post(`/v1/monitors/${encodeURIComponent(id)}/runs`, input, 202, request);
|
|
1061
|
+
}
|
|
1062
|
+
/** Waits for capture and assessment; baseline/events are included in the returned view. */
|
|
1063
|
+
async runMonitor(id, input = {}, request = {}) {
|
|
1064
|
+
return this.post(`/v1/monitors/${encodeURIComponent(id)}/run`, input, 200, request);
|
|
1065
|
+
}
|
|
1066
|
+
async pauseMonitor(id, request = {}) {
|
|
1067
|
+
return this.post(`/v1/monitors/${encodeURIComponent(id)}/pause`, void 0, 200, request);
|
|
1068
|
+
}
|
|
1069
|
+
async resumeMonitor(id, request = {}) {
|
|
1070
|
+
return this.post(`/v1/monitors/${encodeURIComponent(id)}/resume`, void 0, 200, request);
|
|
1071
|
+
}
|
|
1072
|
+
async cancelMonitorRun(id, runId, request = {}) {
|
|
1073
|
+
return this.post(`/v1/monitors/${encodeURIComponent(id)}/runs/${encodeURIComponent(runId)}/cancel`, void 0, 200, request);
|
|
1074
|
+
}
|
|
1075
|
+
async createDeliveryDestination(input, request = {}) {
|
|
1076
|
+
return this.post("/v1/delivery/destinations", input, 201, request);
|
|
1077
|
+
}
|
|
1078
|
+
/** The destinations of a Monitor (`monitorId`) or of a crawl or batch (`jobId`); every destination when neither is given. Header names only, never their values. */
|
|
1079
|
+
async listDeliveryDestinations(options = {}, request = {}) {
|
|
1080
|
+
const params = new URLSearchParams();
|
|
1081
|
+
if (options.monitorId !== void 0) params.set("monitorId", options.monitorId);
|
|
1082
|
+
if (options.jobId !== void 0) params.set("jobId", options.jobId);
|
|
1083
|
+
return this.get(`/v1/delivery/destinations${params.size === 0 ? "" : `?${params}`}`, request);
|
|
1084
|
+
}
|
|
1085
|
+
async pauseDeliveryDestination(id, request = {}) {
|
|
1086
|
+
return this.post(`/v1/delivery/destinations/${encodeURIComponent(id)}/pause`, void 0, 200, request);
|
|
1087
|
+
}
|
|
1088
|
+
async resumeDeliveryDestination(id, request = {}) {
|
|
1089
|
+
return this.post(`/v1/delivery/destinations/${encodeURIComponent(id)}/resume`, void 0, 200, request);
|
|
1090
|
+
}
|
|
1091
|
+
/** The deliveries of a Monitor (`monitorId`) or of a crawl or batch (`jobId`, the task id), each with its payload. */
|
|
1092
|
+
async listDeliveries(options = {}, request = {}) {
|
|
1093
|
+
const params = new URLSearchParams();
|
|
1094
|
+
if (options.monitorId !== void 0) params.set("monitorId", options.monitorId);
|
|
1095
|
+
if (options.jobId !== void 0) params.set("jobId", options.jobId);
|
|
1096
|
+
if (options.destinationId !== void 0) params.set("destinationId", options.destinationId);
|
|
1097
|
+
if (options.state !== void 0) params.set("state", options.state);
|
|
1098
|
+
return this.get(`/v1/deliveries${params.size === 0 ? "" : `?${params}`}`, request);
|
|
1099
|
+
}
|
|
1100
|
+
async getDeliveriesPage(options = {}, request = {}) {
|
|
1101
|
+
const params = new URLSearchParams();
|
|
1102
|
+
if (options.monitorId !== void 0) params.set("monitorId", options.monitorId);
|
|
1103
|
+
if (options.jobId !== void 0) params.set("jobId", options.jobId);
|
|
1104
|
+
if (options.destinationId !== void 0) params.set("destinationId", options.destinationId);
|
|
1105
|
+
if (options.state !== void 0) params.set("state", options.state);
|
|
1106
|
+
if (options.cursor !== void 0) params.set("cursor", options.cursor);
|
|
1107
|
+
if (options.limit !== void 0) params.set("limit", String(options.limit));
|
|
1108
|
+
return this.get(`/v1/deliveries/page${params.size ? `?${params}` : ""}`, request);
|
|
1109
|
+
}
|
|
1110
|
+
async getDelivery(id, request = {}) {
|
|
1111
|
+
return this.get(`/v1/deliveries/${encodeURIComponent(id)}`, request);
|
|
1112
|
+
}
|
|
1113
|
+
async retryDelivery(id, request = {}) {
|
|
1114
|
+
return this.post(`/v1/deliveries/${encodeURIComponent(id)}/retry`, void 0, 200, request);
|
|
1115
|
+
}
|
|
1116
|
+
async waitFor(id, poll, options) {
|
|
1117
|
+
checkWaitOptions(options);
|
|
1118
|
+
const deadline = options.timeoutMs === void 0 ? void 0 : Date.now() + options.timeoutMs;
|
|
1119
|
+
const expiry = new AbortController();
|
|
1120
|
+
let timer;
|
|
1121
|
+
const arm = () => {
|
|
1122
|
+
timer = setTimeout(() => {
|
|
1123
|
+
if (Date.now() >= deadline) expiry.abort(new DOMException("wait timed out", "TimeoutError"));
|
|
1124
|
+
else arm();
|
|
1125
|
+
}, Math.min(Math.max(0, deadline - Date.now()), 2147483647));
|
|
1126
|
+
};
|
|
1127
|
+
if (deadline !== void 0) arm();
|
|
1128
|
+
const signal = options.signal === void 0 ? expiry.signal : AbortSignal.any([options.signal, expiry.signal]);
|
|
1129
|
+
let last = null;
|
|
1130
|
+
let failure = void 0;
|
|
1131
|
+
let failures = 0;
|
|
1132
|
+
const timedOut = () => new WaitTimeoutError(id, last, options.timeoutMs, failure === void 0 ? void 0 : { cause: failure });
|
|
1133
|
+
try {
|
|
1134
|
+
for (; ; ) {
|
|
1135
|
+
options.signal?.throwIfAborted();
|
|
1136
|
+
let pause = options.pollIntervalMs ?? 500;
|
|
1137
|
+
try {
|
|
1138
|
+
const report = await poll({ signal });
|
|
1139
|
+
if (FINISHED2.includes(report.status)) return report;
|
|
1140
|
+
last = report;
|
|
1141
|
+
failure = void 0;
|
|
1142
|
+
failures = 0;
|
|
1143
|
+
} catch (error) {
|
|
1144
|
+
options.signal?.throwIfAborted();
|
|
1145
|
+
if (expiry.signal.aborted) throw timedOut();
|
|
1146
|
+
const delay = failures < (options.maxRetries ?? 5) ? retryDelayMs(error, failures + 1) : null;
|
|
1147
|
+
if (delay === null) throw error;
|
|
1148
|
+
failure = error;
|
|
1149
|
+
failures++;
|
|
1150
|
+
pause = delay;
|
|
1151
|
+
}
|
|
1152
|
+
if (deadline !== void 0 && Date.now() >= deadline) throw timedOut();
|
|
1153
|
+
try {
|
|
1154
|
+
await sleep2(Math.min(pause, deadline === void 0 ? Infinity : deadline - Date.now()), signal);
|
|
1155
|
+
} catch (error) {
|
|
1156
|
+
options.signal?.throwIfAborted();
|
|
1157
|
+
if (expiry.signal.aborted) throw timedOut();
|
|
1158
|
+
throw error;
|
|
1159
|
+
}
|
|
1160
|
+
}
|
|
1161
|
+
} finally {
|
|
1162
|
+
if (timer !== void 0) clearTimeout(timer);
|
|
1163
|
+
}
|
|
1164
|
+
}
|
|
1165
|
+
headers(extra = {}) {
|
|
1166
|
+
return this.token === void 0 || this.token.length === 0 ? extra : { ...extra, authorization: `Bearer ${this.token}` };
|
|
1167
|
+
}
|
|
1168
|
+
/** `answerWithinMs`: how long the platform's fetch waits for the response headers, on Node instead of undici's 300 s. */
|
|
1169
|
+
async post(path, body, ok = 200, request = {}, answerWithinMs) {
|
|
1170
|
+
const dispatcher = answerWithinMs === void 0 || !this.platformFetch ? void 0 : headersWait(answerWithinMs);
|
|
1171
|
+
const init = {
|
|
1172
|
+
method: "POST",
|
|
1173
|
+
signal: request.signal,
|
|
1174
|
+
headers: this.headers({ "content-type": "application/json" }),
|
|
1175
|
+
...body === void 0 ? {} : { body: JSON.stringify(body) },
|
|
1176
|
+
...dispatcher === void 0 ? {} : { dispatcher }
|
|
1177
|
+
};
|
|
1178
|
+
const res = await this.fetchImpl(`${this.baseUrl}${path}`, init);
|
|
1179
|
+
if (res.status !== ok) throw await responseError("POST", path, res);
|
|
1180
|
+
return await res.json();
|
|
1181
|
+
}
|
|
1182
|
+
async getPageList(path, options, request) {
|
|
1183
|
+
const params = new URLSearchParams();
|
|
1184
|
+
if (options.cursor !== void 0) params.set("cursor", options.cursor);
|
|
1185
|
+
if (options.limit !== void 0) params.set("limit", String(options.limit));
|
|
1186
|
+
if (options.attemptId !== void 0) params.set("attemptId", options.attemptId);
|
|
1187
|
+
if (options.debug !== void 0) params.set("debug", String(options.debug));
|
|
1188
|
+
if (options.includeDuplicates !== void 0) params.set("includeDuplicates", String(options.includeDuplicates));
|
|
1189
|
+
const suffix = params.size === 0 ? "" : `?${params.toString()}`;
|
|
1190
|
+
return this.get(`${path}${suffix}`, request, `crawl not found: ${path}`);
|
|
1191
|
+
}
|
|
1192
|
+
async get(path, request, notFound) {
|
|
1193
|
+
const res = await this.fetchImpl(`${this.baseUrl}${path}`, { headers: this.headers(), signal: request.signal });
|
|
1194
|
+
if (res.status === 404 && notFound !== void 0) throw await responseError("GET", path, res, notFound);
|
|
1195
|
+
if (!res.ok) throw await responseError("GET", path, res);
|
|
1196
|
+
return await res.json();
|
|
1197
|
+
}
|
|
1198
|
+
};
|
|
1199
|
+
async function collect(listing) {
|
|
1200
|
+
const items = [];
|
|
1201
|
+
for (; ; ) {
|
|
1202
|
+
const next = await listing.next();
|
|
1203
|
+
if (next.done) return { items, nextCursor: next.value.nextCursor, hasMore: next.value.nextCursor !== null, stoppedBy: next.value.stoppedBy };
|
|
1204
|
+
items.push(next.value);
|
|
1205
|
+
}
|
|
1206
|
+
}
|
|
1207
|
+
function originOf(opts, request) {
|
|
1208
|
+
return opts.origin ?? request.origin ?? SDK_ORIGIN;
|
|
1209
|
+
}
|
|
1210
|
+
function sleep2(ms, signal) {
|
|
1211
|
+
return new Promise((resolve, reject) => {
|
|
1212
|
+
if (signal.aborted) {
|
|
1213
|
+
reject(signal.reason);
|
|
1214
|
+
return;
|
|
1215
|
+
}
|
|
1216
|
+
const timer = setTimeout(() => {
|
|
1217
|
+
signal.removeEventListener("abort", abort);
|
|
1218
|
+
resolve();
|
|
1219
|
+
}, Math.max(0, ms));
|
|
1220
|
+
const abort = () => {
|
|
1221
|
+
clearTimeout(timer);
|
|
1222
|
+
reject(signal.reason);
|
|
1223
|
+
};
|
|
1224
|
+
signal.addEventListener("abort", abort, { once: true });
|
|
1225
|
+
});
|
|
1226
|
+
}
|
|
1227
|
+
export {
|
|
1228
|
+
DEFAULT_WATCH_POLL_INTERVAL_MS,
|
|
1229
|
+
JobWatcher,
|
|
1230
|
+
MAP_ANSWER_MARGIN_MS,
|
|
1231
|
+
MIN_WATCH_POLL_INTERVAL_MS,
|
|
1232
|
+
SDK_ORIGIN,
|
|
1233
|
+
SDK_VERSION,
|
|
1234
|
+
W2L,
|
|
1235
|
+
W2LError,
|
|
1236
|
+
WaitTimeoutError,
|
|
1237
|
+
chunkUrls,
|
|
1238
|
+
pageCursor,
|
|
1239
|
+
parseSseBlock
|
|
1240
|
+
};
|