@crawlora-org/youtube 0.0.0-stage → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Crawlora
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md CHANGED
@@ -1,3 +1,35 @@
1
- # Temporary Holding Version
1
+ # @crawlora-org/youtube
2
2
 
3
- This version is a temporary placeholder for this package. An operational version to replace this has been submitted for review and is awaiting a staged release.
3
+ JavaScript and TypeScript client for Crawlora's hosted YouTube API.
4
+ It calls Crawlora's service; it does not run a browser or scrape YouTube locally. A Crawlora account and `CRAWLORA_API_KEY` are required, and API use is billed under your Crawlora account. Crawlora is independent from and not endorsed by YouTube or its owners.
5
+
6
+ ## Install
7
+
8
+ ```sh
9
+ npm install @crawlora-org/youtube
10
+ ```
11
+
12
+ ## Use
13
+
14
+ ```js
15
+ import { YouTubeClient } from "@crawlora-org/youtube";
16
+
17
+ const client = new YouTubeClient({ apiKey: process.env.CRAWLORA_API_KEY });
18
+ const result = await client.search({ q: "science explainers", type: "video" });
19
+ console.log(result);
20
+ ```
21
+
22
+ The client also exports `Client` as an alias for `YouTubeClient`. Operation
23
+ methods are available directly in camelCase and through the `youtube`
24
+ group. See the full method and parameter list in the [online reference](https://github.com/Crawlora-org/crawlora-youtube/blob/main/docs/usage.md).
25
+
26
+ Methods return promises and can be awaited. See the [runnable example](https://github.com/Crawlora-org/crawlora-youtube/blob/main/examples/javascript.mjs) for contract-backed examples and text transcript output where supported.
27
+
28
+ The import snippet above is for a project where this npm package is installed. The checked-in repository example instead imports `../javascript/src/index.js` so it runs directly from the repository root; see the [source-checkout instructions](https://github.com/Crawlora-org/crawlora-youtube#run-examples-from-a-source-checkout).
29
+
30
+ ## Configuration
31
+
32
+ Pass your key through `apiKey` or set `CRAWLORA_API_KEY` and read it from the
33
+ environment. Keep credentials out of source control and logs. Requests are
34
+ made to Crawlora's hosted API; response data and availability follow that
35
+ service's current contract.
package/package.json CHANGED
@@ -1,6 +1,34 @@
1
1
  {
2
2
  "name": "@crawlora-org/youtube",
3
- "version": "0.0.0-stage",
4
- "stub": true,
5
- "description": "Temporary package placeholder for staged publishing"
6
- }
3
+ "version": "0.1.0",
4
+ "description": "Typed JavaScript client for the Crawlora YouTube API",
5
+ "homepage": "https://github.com/Crawlora-org/crawlora-youtube#readme",
6
+ "bugs": { "url": "https://github.com/Crawlora-org/crawlora-youtube/issues" },
7
+ "keywords": ["crawlora", "api-client", "web-scraping", "youtube", "typescript"],
8
+ "type": "module",
9
+ "main": "./src/index.js",
10
+ "types": "./src/index.d.ts",
11
+ "exports": {
12
+ ".": {
13
+ "types": "./src/index.d.ts",
14
+ "import": "./src/index.js"
15
+ }
16
+ },
17
+ "files": ["src", "README.md", "LICENSE"],
18
+ "engines": { "node": ">=18" },
19
+ "license": "MIT",
20
+ "repository": {
21
+ "type": "git",
22
+ "url": "https://github.com/Crawlora-org/crawlora-youtube",
23
+ "directory": "javascript"
24
+ },
25
+ "scripts": {
26
+ "test": "node --test test/*.test.js",
27
+ "typecheck": "tsc --noEmit",
28
+ "pack:check": "npm pack --dry-run --ignore-scripts",
29
+ "generate": "python3 ../scripts/generate.py"
30
+ },
31
+ "devDependencies": {
32
+ "typescript": "5.8.3"
33
+ }
34
+ }
package/src/client.js ADDED
@@ -0,0 +1,582 @@
1
+ import { operations, groups } from "./operations.js";
2
+
3
+ const DEFAULT_BASE_URL = "https://api.crawlora.net/api/v1";
4
+ export const VERSION = "1.46.0-sdk.1";
5
+ const DEFAULT_USER_AGENT = `crawlora-js-sdk/${VERSION}`;
6
+
7
+ export class CrawloraError extends Error {
8
+ constructor(message, { status = 0, code, body, headers = {}, response, cause, retryable, requestId } = {}) {
9
+ super(message);
10
+ this.name = "CrawloraError";
11
+ this.status = status;
12
+ this.code = code;
13
+ this.body = body;
14
+ this.headers = headers;
15
+ this.response = response;
16
+ this.cause = cause;
17
+ this.retryable = retryable;
18
+ this.requestId = requestId;
19
+ }
20
+ }
21
+
22
+ // Response status 4xx: the request was rejected by the API (bad params, auth,
23
+ // not found). Usually not retryable.
24
+ export class CrawloraClientError extends CrawloraError {
25
+ constructor(message, options) {
26
+ super(message, options);
27
+ this.name = "CrawloraClientError";
28
+ }
29
+ }
30
+
31
+ // Response status 5xx: the API failed to handle a valid request. Retryable.
32
+ export class CrawloraServerError extends CrawloraError {
33
+ constructor(message, options) {
34
+ super(message, options);
35
+ this.name = "CrawloraServerError";
36
+ }
37
+ }
38
+
39
+ // Transport failure, timeout, or abort before a response was received.
40
+ export class CrawloraNetworkError extends CrawloraError {
41
+ constructor(message, options) {
42
+ super(message, options);
43
+ this.name = "CrawloraNetworkError";
44
+ }
45
+ }
46
+
47
+ function apiErrorClass(status) {
48
+ return status >= 500 ? CrawloraServerError : CrawloraClientError;
49
+ }
50
+
51
+ /**
52
+ * Client for the Crawlora API.
53
+ *
54
+ * Call operations via grouped helpers (`client.bing.search({ q })`) or
55
+ * dynamically (`client.request("bing-search", { q })`). Supports configurable
56
+ * retries, an `onRetry` hook, opt-in `requestId` and `idempotencyKeys`,
57
+ * `beforeRequest`/`afterResponse` middleware, client-side `rateLimit` /
58
+ * `maxConcurrency`, pagination (`paginate`/`paginateItems`), and
59
+ * `responseType: "stream"`.
60
+ */
61
+ export class CrawloraClient {
62
+ constructor(options = {}) {
63
+ // Precedence: explicit option > environment variable > default.
64
+ this.apiKey = options.apiKey || envVar("CRAWLORA_API_KEY") || "";
65
+ this.jwtToken = options.jwtToken || "";
66
+ this.baseUrl = (options.baseUrl || envVar("CRAWLORA_BASE_URL") || DEFAULT_BASE_URL).replace(/\/+$/, "");
67
+ this.timeout = options.timeout ?? 30000;
68
+ this.retries = normalizeNonNegativeInteger(options.retries ?? 0);
69
+ this.retryDelay = normalizeNonNegativeNumber(options.retryDelay ?? 250);
70
+ this.maxRetryDelay = normalizeNonNegativeNumber(options.maxRetryDelay ?? 30000);
71
+ this.retryStatuses = options.retryStatuses ? new Set(options.retryStatuses) : null;
72
+ this.isRetryable = typeof options.isRetryable === "function" ? options.isRetryable : null;
73
+ this.onRetry = typeof options.onRetry === "function" ? options.onRetry : null;
74
+ this.requestId = options.requestId === true;
75
+ this.idempotencyKeys = options.idempotencyKeys === true;
76
+ this.logger = typeof options.logger === "function" ? options.logger : null;
77
+ this.beforeRequest = asHookList(options.beforeRequest);
78
+ this.afterResponse = asHookList(options.afterResponse);
79
+ this.limiter = (options.rateLimit > 0 || options.maxConcurrency > 0)
80
+ ? new RateLimiter(options.rateLimit || 0, options.maxConcurrency || 0)
81
+ : null;
82
+ this.headers = { ...(options.headers || {}) };
83
+ this.userAgent = options.userAgent === false ? "" : (options.userAgent || DEFAULT_USER_AGENT);
84
+ this.fetch = options.fetch || globalThis.fetch;
85
+
86
+ if (typeof this.fetch !== "function") {
87
+ throw new TypeError("CrawloraClient requires a fetch implementation");
88
+ }
89
+
90
+ for (const [groupName, groupOperations] of Object.entries(groups)) {
91
+ this[groupName] = {};
92
+ for (const [methodName, operationId] of Object.entries(groupOperations)) {
93
+ this[groupName][methodName] = (params = {}, requestOptions = {}) =>
94
+ this.request(operationId, params, requestOptions);
95
+ }
96
+ }
97
+ }
98
+
99
+ operation(operationId, params = {}, options = {}) {
100
+ return this.request(operationId, params, options);
101
+ }
102
+
103
+ async request(operationId, params = {}, options = {}) {
104
+ const operation = operations[operationId];
105
+ if (!operation) {
106
+ throw new TypeError(`Unknown Crawlora operation: ${operationId}`);
107
+ }
108
+ params = params ?? {};
109
+ options = options ?? {};
110
+ const responseType = validateResponseType(options.responseType || "auto");
111
+ this.#log({ event: "request", operation: operationId });
112
+ const maxRetries = options.retries ?? this.retries;
113
+ const isRetryable = typeof options.isRetryable === "function" ? options.isRetryable : null;
114
+ const idempotencyKey = (this.idempotencyKeys && (operation.method === "POST" || operation.method === "PATCH"))
115
+ ? generateId() : undefined;
116
+
117
+ let attempt = 0;
118
+ for (;;) {
119
+ try {
120
+ const run = () => this.#send(operation, params, options, responseType, idempotencyKey);
121
+ return await (this.limiter ? this.limiter.run(run, options.signal) : run());
122
+ } catch (error) {
123
+ const retryable = isRetryable ? !!isRetryable(error.status, error) : this.#isRetryable(error.status, error);
124
+ if (!(error instanceof CrawloraError) || error.retryable === false || attempt >= maxRetries || !retryable) {
125
+ throw error;
126
+ }
127
+ attempt++;
128
+ const delay = this.#retryDelayFor(error, attempt);
129
+ this.#log({ event: "retry", operation: operationId, attempt, status: error.status, delay });
130
+ if (this.onRetry) this.onRetry(attempt, error, delay);
131
+ await sleep(delay, options.signal);
132
+ }
133
+ }
134
+ }
135
+
136
+ #isRetryable(status, error) {
137
+ if (this.isRetryable) return !!this.isRetryable(status, error);
138
+ if (this.retryStatuses) return status === 0 || this.retryStatuses.has(status);
139
+ return shouldRetry(status);
140
+ }
141
+
142
+ #retryDelayFor(error, attempt) {
143
+ const retryAfter = parseRetryAfter(error.headers, this.maxRetryDelay);
144
+ return retryAfter ?? retryDelay(this.retryDelay, attempt);
145
+ }
146
+
147
+ #log(event) {
148
+ if (this.logger) this.logger(event);
149
+ }
150
+
151
+ async #send(operation, params, options, responseType, idempotencyKey) {
152
+ const { url, body, bodyHeaders } = buildRequest(operation, this.baseUrl, params);
153
+ const headers = mergeHeaders(
154
+ this.headers,
155
+ authHeaders(operation.security, this.apiKey, this.jwtToken),
156
+ userAgentHeader(this.userAgent),
157
+ bodyHeaders,
158
+ options.headers
159
+ );
160
+ const requestId = this.requestId ? ensureRequestId(headers) : (headerValue(headers, "x-request-id") || undefined);
161
+ if (idempotencyKey && !headerValue(headers, "idempotency-key")) {
162
+ headers["Idempotency-Key"] = idempotencyKey;
163
+ }
164
+
165
+ let requestUrl = url;
166
+ let requestHeaders = headers;
167
+ if (this.beforeRequest.length) {
168
+ const ctx = { operationId: operation.id, method: operation.method, url: requestUrl, headers: requestHeaders };
169
+ for (const hook of this.beforeRequest) await hook(ctx);
170
+ requestUrl = ctx.url;
171
+ requestHeaders = ctx.headers;
172
+ }
173
+
174
+ const controller = new AbortController();
175
+ const timeoutMs = options.timeout ?? this.timeout;
176
+ let timedOut = false;
177
+ const timeout = timeoutMs > 0 ? setTimeout(() => {
178
+ timedOut = true;
179
+ controller.abort();
180
+ }, timeoutMs) : undefined;
181
+ const signal = composeSignal(controller, options.signal);
182
+ let response;
183
+ try {
184
+ response = await this.fetch(requestUrl, {
185
+ method: operation.method,
186
+ headers: requestHeaders,
187
+ body,
188
+ signal
189
+ });
190
+ } catch (error) {
191
+ if (options.signal?.aborted) {
192
+ throw new CrawloraNetworkError("Crawlora request aborted", { cause: error, retryable: false, requestId });
193
+ }
194
+ if (timedOut) {
195
+ throw new CrawloraNetworkError("Crawlora request timed out", { cause: error, retryable: false, requestId });
196
+ }
197
+ throw new CrawloraNetworkError("Crawlora transport error", { cause: error, requestId });
198
+ } finally {
199
+ if (timeout) clearTimeout(timeout);
200
+ }
201
+
202
+ const headersObject = responseHeaders(response);
203
+ // Streaming success returns the raw Response; the caller reads response.body.
204
+ if (responseType === "stream" && response.ok) {
205
+ return response;
206
+ }
207
+ let parsed = await parseResponse(response, responseType === "stream" ? "auto" : responseType, headersObject, requestId);
208
+ if (!response.ok) {
209
+ const code = parsed && typeof parsed === "object" ? parsed.code : undefined;
210
+ const message = parsed && typeof parsed === "object" && parsed.msg ? parsed.msg : response.statusText;
211
+ const ApiError = apiErrorClass(response.status);
212
+ throw new ApiError(message || `Crawlora request failed with status ${response.status}`, {
213
+ status: response.status,
214
+ code,
215
+ body: parsed,
216
+ headers: headersObject,
217
+ response,
218
+ requestId
219
+ });
220
+ }
221
+ if (this.afterResponse.length) {
222
+ for (const hook of this.afterResponse) {
223
+ const result = await hook(operation.id, response.status, headersObject, parsed);
224
+ if (result !== undefined) parsed = result;
225
+ }
226
+ }
227
+ return parsed;
228
+ }
229
+
230
+ // Async iterator over pages of a paginated operation. Numeric mode advances
231
+ // the page/offset query parameter and stops on an empty page; cursor mode
232
+ // (cursorParam + nextCursor) sends the cursor and stops when nextCursor is falsy.
233
+ // for await (const page of client.paginate("ebay-seller-feedback", { seller })) { ... }
234
+ async *paginate(operationId, params = {}, options = {}) {
235
+ const operation = operations[operationId];
236
+ if (!operation) {
237
+ throw new TypeError(`Unknown Crawlora operation: ${operationId}`);
238
+ }
239
+ const maxPages = options.maxPages ?? Infinity;
240
+
241
+ if (options.cursorParam || options.nextCursor) {
242
+ if (!(options.cursorParam && options.nextCursor)) {
243
+ throw new TypeError("cursor pagination requires both cursorParam and nextCursor");
244
+ }
245
+ if (!operation.queryParams.some((parameter) => parameter.name === options.cursorParam)) {
246
+ throw new TypeError(`cursorParam ${options.cursorParam} is not a query parameter of ${operationId}`);
247
+ }
248
+ let cursor = options.start;
249
+ for (let i = 0; i < maxPages; i++) {
250
+ const pageParams = { ...params };
251
+ if (cursor !== undefined && cursor !== null) pageParams[options.cursorParam] = cursor;
252
+ const response = await this.request(operationId, pageParams, options);
253
+ yield response;
254
+ cursor = options.nextCursor(response);
255
+ if (!cursor) break;
256
+ }
257
+ return;
258
+ }
259
+
260
+ const pageParam = options.pageParam || detectPageParam(operation);
261
+ if (!pageParam) {
262
+ throw new TypeError(`Operation ${operationId} has no page or offset query parameter to paginate`);
263
+ }
264
+ const step = options.step ?? 1;
265
+ let pageValue = options.start ?? (pageParam === "offset" ? 0 : 1);
266
+ for (let i = 0; i < maxPages; i++) {
267
+ const response = await this.request(operationId, { ...params, [pageParam]: pageValue }, options);
268
+ yield response;
269
+ if (pageIsEmpty(response)) break;
270
+ pageValue += step;
271
+ }
272
+ }
273
+
274
+ // Async iterator over individual items across pages. `items` extracts the list
275
+ // from a page (default: the Crawlora `data` array).
276
+ async *paginateItems(operationId, params = {}, options = {}) {
277
+ const extract = options.items || defaultItems;
278
+ for await (const page of this.paginate(operationId, params, options)) {
279
+ for (const item of extract(page)) yield item;
280
+ }
281
+ }
282
+ }
283
+
284
+ const PAGE_PARAM_NAMES = ["page", "offset"];
285
+
286
+ function defaultItems(response) {
287
+ if (response && typeof response === "object" && Array.isArray(response.data)) return response.data;
288
+ if (Array.isArray(response)) return response;
289
+ return [];
290
+ }
291
+
292
+ function detectPageParam(operation) {
293
+ for (const name of PAGE_PARAM_NAMES) {
294
+ if (operation.queryParams.some((parameter) => parameter.name === name)) return name;
295
+ }
296
+ return undefined;
297
+ }
298
+
299
+ function pageIsEmpty(response) {
300
+ if (response === undefined || response === null) return true;
301
+ let data = response;
302
+ if (typeof response === "object" && !Array.isArray(response) && "data" in response) {
303
+ data = response.data;
304
+ }
305
+ if (data === undefined || data === null) return true;
306
+ if (Array.isArray(data)) return data.length === 0;
307
+ if (typeof data === "object") return Object.keys(data).length === 0;
308
+ return !data;
309
+ }
310
+
311
+ function buildRequest(operation, baseUrl, params) {
312
+ validateRequiredParams(operation, params);
313
+ validateEnumParams(operation, params);
314
+ let path = operation.path;
315
+ for (const name of operation.pathParams) {
316
+ const value = params[name];
317
+ if (value === undefined || value === null || value === "") {
318
+ throw new TypeError(`Missing required path parameter: ${name}`);
319
+ }
320
+ path = path.replace(`{${name}}`, encodeURIComponent(String(value)));
321
+ }
322
+
323
+ const query = new URLSearchParams();
324
+ for (const parameter of operation.queryParams) {
325
+ const value = params[parameter.name];
326
+ if (value === undefined || value === null || value === "") {
327
+ continue;
328
+ }
329
+ if (Array.isArray(value)) {
330
+ for (const item of value) query.append(parameter.name, String(item));
331
+ } else {
332
+ query.append(parameter.name, String(value));
333
+ }
334
+ }
335
+
336
+ const suffix = query.toString();
337
+ const url = `${baseUrl}${path}${suffix ? `?${suffix}` : ""}`;
338
+ const bodyHeaders = {};
339
+ let body;
340
+
341
+ if (operation.formParams.length > 0) {
342
+ const form = new FormData();
343
+ for (const parameter of operation.formParams) {
344
+ const value = params[parameter.name];
345
+ if (value !== undefined && value !== null) form.append(parameter.name, value);
346
+ }
347
+ body = form;
348
+ } else if (operation.bodyParam) {
349
+ const value = params[operation.bodyParam] ?? params.body;
350
+ if (value !== undefined) {
351
+ body = JSON.stringify(value);
352
+ bodyHeaders["content-type"] = "application/json";
353
+ }
354
+ }
355
+
356
+ return { url, body, bodyHeaders };
357
+ }
358
+
359
+ function validateRequiredParams(operation, params) {
360
+ for (const parameter of [...operation.pathParams.map((name) => ({ name, in: "path", required: true })), ...operation.queryParams, ...operation.formParams]) {
361
+ if (parameter.required && isMissing(params[parameter.name])) {
362
+ throw new TypeError(`Missing required ${parameter.in || "request"} parameter: ${parameter.name}`);
363
+ }
364
+ }
365
+ if (operation.bodyRequired && isMissing(params[operation.bodyParam]) && isMissing(params.body)) {
366
+ throw new TypeError(`Missing required body parameter: ${operation.bodyParam}`);
367
+ }
368
+ }
369
+
370
+ function validateEnumParams(operation, params) {
371
+ for (const parameter of [...operation.queryParams, ...operation.formParams]) {
372
+ if (!parameter.enum?.length || isMissing(params[parameter.name])) continue;
373
+ const values = Array.isArray(params[parameter.name]) ? params[parameter.name] : [params[parameter.name]];
374
+ for (const value of values) {
375
+ if (!parameter.enum.includes(String(value))) {
376
+ throw new TypeError(`invalid ${parameter.in || "request"} parameter ${parameter.name}: expected one of ${parameter.enum.join(", ")}`);
377
+ }
378
+ }
379
+ }
380
+ }
381
+
382
+ function isMissing(value) {
383
+ return value === undefined || value === null || value === "" || (Array.isArray(value) && value.length === 0);
384
+ }
385
+
386
+ function authHeaders(security, apiKey, jwtToken) {
387
+ const headers = {};
388
+ if (security.includes("ApiKeyAuth") && apiKey) {
389
+ headers["x-api-key"] = apiKey;
390
+ }
391
+ if (security.includes("JWTAuth") && jwtToken) {
392
+ headers.Authorization = /^(Token|Bearer)\s+/i.test(jwtToken) ? jwtToken : `Token ${jwtToken}`;
393
+ }
394
+ return headers;
395
+ }
396
+
397
+ function userAgentHeader(userAgent) {
398
+ if (!userAgent || typeof process === "undefined" || !process.versions?.node) {
399
+ return {};
400
+ }
401
+ return { "user-agent": userAgent };
402
+ }
403
+
404
+ function mergeHeaders(...sources) {
405
+ const headers = {};
406
+ const names = new Map();
407
+ for (const source of sources) {
408
+ for (const [name, value] of Object.entries(source || {})) {
409
+ if (value === undefined || value === null) continue;
410
+ const lower = name.toLowerCase();
411
+ const existing = names.get(lower);
412
+ if (existing && existing !== name) delete headers[existing];
413
+ headers[name] = String(value);
414
+ names.set(lower, name);
415
+ }
416
+ }
417
+ return headers;
418
+ }
419
+
420
+ function composeSignal(controller, signal) {
421
+ if (!signal) return controller.signal;
422
+ if (signal.aborted) controller.abort();
423
+ signal.addEventListener("abort", () => controller.abort(), { once: true });
424
+ return controller.signal;
425
+ }
426
+
427
+ function validateResponseType(responseType) {
428
+ if (responseType === "auto" || responseType === "json" || responseType === "text" || responseType === "stream") {
429
+ return responseType;
430
+ }
431
+ throw new TypeError("Invalid responseType: expected one of auto, json, text, stream");
432
+ }
433
+
434
+ async function parseResponse(response, responseType, headers, requestId) {
435
+ if (responseType === "text") return response.text();
436
+ const contentType = response.headers.get("content-type") || "";
437
+ if (responseType === "json" || contentType.toLowerCase().includes("application/json")) {
438
+ const text = await response.text();
439
+ try {
440
+ return text ? JSON.parse(text) : null;
441
+ } catch (error) {
442
+ throw new CrawloraError("Crawlora JSON parse error", {
443
+ status: response.status,
444
+ body: text,
445
+ headers,
446
+ response,
447
+ cause: error,
448
+ requestId
449
+ });
450
+ }
451
+ }
452
+ return response.text();
453
+ }
454
+
455
+ function shouldRetry(status) {
456
+ return status === 0 || status === 408 || status === 409 || status === 425 || status === 429 || status >= 500;
457
+ }
458
+
459
+ function retryDelay(baseDelay, attempt) {
460
+ if (!baseDelay || baseDelay <= 0) return 0;
461
+ const delay = baseDelay * 2 ** Math.max(0, attempt - 1);
462
+ const jitter = Math.floor(Math.random() * Math.max(1, baseDelay / 2));
463
+ return delay + jitter;
464
+ }
465
+
466
+ function parseRetryAfter(headers, cap = 30000) {
467
+ const value = headerValue(headers, "retry-after");
468
+ if (!value) return undefined;
469
+ const seconds = Number(value);
470
+ if (Number.isFinite(seconds) && seconds > 0) {
471
+ return Math.min(seconds * 1000, cap);
472
+ }
473
+ const date = Date.parse(value);
474
+ if (Number.isFinite(date)) {
475
+ const delay = date - Date.now();
476
+ if (delay > 0) return Math.min(delay, cap);
477
+ }
478
+ return undefined;
479
+ }
480
+
481
+ function asHookList(value) {
482
+ if (!value) return [];
483
+ return typeof value === "function" ? [value] : Array.from(value);
484
+ }
485
+
486
+ // Optional client-side throttle: caps concurrency and spaces requests to a
487
+ // maximum rate (requests per second).
488
+ class RateLimiter {
489
+ constructor(rps, concurrency) {
490
+ this.interval = rps > 0 ? 1000 / rps : 0;
491
+ this.limit = concurrency > 0 ? concurrency : Infinity;
492
+ this.active = 0;
493
+ this.waiters = [];
494
+ this.nextAllowed = 0;
495
+ }
496
+
497
+ async run(fn, signal) {
498
+ await this.#acquire();
499
+ try {
500
+ await this.#rateWait(signal);
501
+ return await fn();
502
+ } finally {
503
+ const next = this.waiters.shift();
504
+ if (next) {
505
+ next(); // transfer the slot to the next waiter (active unchanged)
506
+ } else {
507
+ this.active--;
508
+ }
509
+ }
510
+ }
511
+
512
+ #acquire() {
513
+ if (this.active < this.limit) {
514
+ this.active++;
515
+ return Promise.resolve();
516
+ }
517
+ // Wait for a slot to be transferred to us (active stays at the limit).
518
+ return new Promise((resolve) => this.waiters.push(resolve));
519
+ }
520
+
521
+ async #rateWait(signal) {
522
+ if (!this.interval) return;
523
+ const now = Date.now();
524
+ const wait = Math.max(0, this.nextAllowed - now);
525
+ this.nextAllowed = Math.max(now, this.nextAllowed) + this.interval;
526
+ if (wait > 0) await sleep(wait, signal);
527
+ }
528
+ }
529
+
530
+ function envVar(name) {
531
+ return (typeof process !== "undefined" && process.env && process.env[name]) || undefined;
532
+ }
533
+
534
+ function generateId() {
535
+ return (typeof crypto !== "undefined" && crypto.randomUUID) ? crypto.randomUUID() : `id-${Date.now()}-${Math.random().toString(16).slice(2)}`;
536
+ }
537
+
538
+ function ensureRequestId(headers) {
539
+ const existing = headerValue(headers, "x-request-id");
540
+ if (existing) return existing;
541
+ const id = generateId();
542
+ headers["x-request-id"] = id;
543
+ return id;
544
+ }
545
+
546
+ function headerValue(headers, name) {
547
+ for (const [key, value] of Object.entries(headers || {})) {
548
+ if (key.toLowerCase() === name.toLowerCase()) return value;
549
+ }
550
+ return "";
551
+ }
552
+
553
+ function responseHeaders(response) {
554
+ return Object.fromEntries(response.headers.entries());
555
+ }
556
+
557
+ function normalizeNonNegativeInteger(value) {
558
+ const number = Number(value);
559
+ if (!Number.isFinite(number) || number <= 0) return 0;
560
+ return Math.trunc(number);
561
+ }
562
+
563
+ function normalizeNonNegativeNumber(value) {
564
+ const number = Number(value);
565
+ if (!Number.isFinite(number) || number <= 0) return 0;
566
+ return number;
567
+ }
568
+
569
+ function sleep(ms, signal) {
570
+ if (!ms || ms <= 0) return Promise.resolve();
571
+ return new Promise((resolve, reject) => {
572
+ if (signal?.aborted) {
573
+ reject(new CrawloraError("Crawlora request aborted", { cause: signal.reason, retryable: false }));
574
+ return;
575
+ }
576
+ const timer = setTimeout(resolve, ms);
577
+ signal?.addEventListener("abort", () => {
578
+ clearTimeout(timer);
579
+ reject(new CrawloraError("Crawlora request aborted", { cause: signal.reason, retryable: false }));
580
+ }, { once: true });
581
+ });
582
+ }