scoutline 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/README.md +23 -20
  2. package/dist/capabilities/quota.d.ts +1 -1
  3. package/dist/capabilities/quota.d.ts.map +1 -1
  4. package/dist/commands/crawl.js +1 -1
  5. package/dist/commands/doctor.js +1 -1
  6. package/dist/commands/map.js +1 -1
  7. package/dist/commands/read.js +1 -1
  8. package/dist/commands/research.js +5 -5
  9. package/dist/commands/research.js.map +1 -1
  10. package/dist/commands/search.js +1 -1
  11. package/dist/index.js +1 -1
  12. package/dist/lib/async-job-state.d.ts +119 -0
  13. package/dist/lib/async-job-state.d.ts.map +1 -0
  14. package/dist/lib/{research-state.js → async-job-state.js} +43 -34
  15. package/dist/lib/async-job-state.js.map +1 -0
  16. package/dist/lib/cache.d.ts +16 -8
  17. package/dist/lib/cache.d.ts.map +1 -1
  18. package/dist/lib/cache.js +33 -8
  19. package/dist/lib/cache.js.map +1 -1
  20. package/dist/lib/execution.d.ts.map +1 -1
  21. package/dist/lib/execution.js +8 -4
  22. package/dist/lib/execution.js.map +1 -1
  23. package/dist/lib/redact.d.ts +1 -1
  24. package/dist/lib/redact.d.ts.map +1 -1
  25. package/dist/lib/redact.js +9 -1
  26. package/dist/lib/redact.js.map +1 -1
  27. package/dist/providers/exa/adapter.d.ts +2 -2
  28. package/dist/providers/exa/adapter.d.ts.map +1 -1
  29. package/dist/providers/exa/adapter.js +5 -3
  30. package/dist/providers/exa/adapter.js.map +1 -1
  31. package/dist/providers/firecrawl/adapter.d.ts +53 -0
  32. package/dist/providers/firecrawl/adapter.d.ts.map +1 -0
  33. package/dist/providers/firecrawl/adapter.js +962 -0
  34. package/dist/providers/firecrawl/adapter.js.map +1 -0
  35. package/dist/providers/firecrawl/client.d.ts +182 -0
  36. package/dist/providers/firecrawl/client.d.ts.map +1 -0
  37. package/dist/providers/firecrawl/client.js +426 -0
  38. package/dist/providers/firecrawl/client.js.map +1 -0
  39. package/dist/providers/firecrawl/credentials.d.ts +38 -0
  40. package/dist/providers/firecrawl/credentials.d.ts.map +1 -0
  41. package/dist/providers/firecrawl/credentials.js +60 -0
  42. package/dist/providers/firecrawl/credentials.js.map +1 -0
  43. package/dist/providers/firecrawl/diagnostics.d.ts +36 -0
  44. package/dist/providers/firecrawl/diagnostics.d.ts.map +1 -0
  45. package/dist/providers/firecrawl/diagnostics.js +65 -0
  46. package/dist/providers/firecrawl/diagnostics.js.map +1 -0
  47. package/dist/providers/firecrawl/quota.d.ts +46 -0
  48. package/dist/providers/firecrawl/quota.d.ts.map +1 -0
  49. package/dist/providers/firecrawl/quota.js +120 -0
  50. package/dist/providers/firecrawl/quota.js.map +1 -0
  51. package/dist/providers/registry.d.ts.map +1 -1
  52. package/dist/providers/registry.js +2 -0
  53. package/dist/providers/registry.js.map +1 -1
  54. package/dist/providers/tavily/adapter.d.ts +2 -2
  55. package/dist/providers/tavily/adapter.d.ts.map +1 -1
  56. package/dist/providers/tavily/adapter.js +5 -3
  57. package/dist/providers/tavily/adapter.js.map +1 -1
  58. package/dist/providers/types.d.ts +1 -1
  59. package/dist/providers/types.d.ts.map +1 -1
  60. package/dist/providers/types.js +1 -1
  61. package/dist/providers/types.js.map +1 -1
  62. package/package.json +1 -1
  63. package/dist/lib/research-state.d.ts +0 -104
  64. package/dist/lib/research-state.d.ts.map +0 -1
  65. package/dist/lib/research-state.js.map +0 -1
  66. package/dist/providers/minimax/sdk-client.d.ts +0 -29
  67. package/dist/providers/minimax/sdk-client.d.ts.map +0 -1
  68. package/dist/providers/minimax/sdk-client.js +0 -50
  69. package/dist/providers/minimax/sdk-client.js.map +0 -1
@@ -0,0 +1,962 @@
1
+ /**
2
+ * Firecrawl Provider Adapter (firecrawl tech-plan §1, D3–D6).
3
+ *
4
+ * Implements the Firecrawl Provider Descriptor with Search, Reader, Map,
5
+ * async Crawl, Quota, and Diagnostics capabilities on top of the
6
+ * direct-HTTP transport (`./client.ts`). The Adapter owns credentials,
7
+ * transport lifecycle, Provider field mapping, and failure normalization;
8
+ * shared execution owns cache and retry policy.
9
+ *
10
+ * Boundary rules (ARCHITECTURE.md §2):
11
+ * - May import capability types, normalized errors, Provider identity
12
+ * types, and the Adapter-local credential and transport Modules.
13
+ * - Must NOT import command presentation, output mode, or another
14
+ * Provider's Adapter.
15
+ *
16
+ * Field mapping (tech-plan D4/D5/D6; investigation §3.1–§3.4):
17
+ * Search data[].title -> title
18
+ * Search data[].url -> url
19
+ * Search data[].markdown -> summary (high content-size; else description)
20
+ * Search data[].description -> summary (medium content-size fallback)
21
+ *
22
+ * Scrape data.markdown|text -> content
23
+ * Scrape data.metadata.sourceURL -> finalUrl
24
+ * Scrape data.metadata.title -> title (better than Tavily's null)
25
+ *
26
+ * Map links[] -> urls
27
+ */
28
+ import crypto from "node:crypto";
29
+ import * as fs from "node:fs/promises";
30
+ import path from "node:path";
31
+ import { decodeReaderFetchResult } from "../../capabilities/reader.js";
32
+ import { decodeMapResult } from "../../capabilities/map.js";
33
+ import { decodeCrawlResult } from "../../capabilities/crawl.js";
34
+ import { computeAsyncJobStateHash, createProductionAsyncJobStateFile, } from "../../lib/async-job-state.js";
35
+ import { asyncJobStateDir } from "../../lib/cache.js";
36
+ import { ApiError, AuthError, ConfigurationError, NetworkError, QuotaError, TimeoutError, UnsupportedOptionError, ValidationError, } from "../../lib/errors.js";
37
+ import { createFirecrawlCrawl, fetchFirecrawlCrawlNext, fetchFirecrawlMap, fetchFirecrawlScrape, fetchFirecrawlSearch, listActiveFirecrawlCrawls, pollFirecrawlCrawl, } from "./client.js";
38
+ import { isFirecrawlConfigured, requireFirecrawlApiKey } from "./credentials.js";
39
+ import { createFirecrawlQuotaCapability } from "./quota.js";
40
+ import { createFirecrawlDiagnosticsCapability } from "./diagnostics.js";
41
+ // ---------------------------------------------------------------------------
42
+ // Credential + shape helpers
43
+ // ---------------------------------------------------------------------------
44
+ function credentialFingerprint(apiKey) {
45
+ return crypto.createHash("sha256").update(apiKey).digest("hex");
46
+ }
47
+ function resolveApiKey(env) {
48
+ return requireFirecrawlApiKey(env);
49
+ }
50
+ function isPlainObject(value) {
51
+ return typeof value === "object" && value !== null && !Array.isArray(value);
52
+ }
53
+ function assertHttpUrl(url) {
54
+ if (typeof url !== "string" || url.length === 0) {
55
+ throw new ValidationError("Firecrawl reader URL must be a non-empty string");
56
+ }
57
+ if (!/^https?:\/\//.test(url)) {
58
+ throw new ValidationError("URL must start with http:// or https://");
59
+ }
60
+ }
61
+ // ---------------------------------------------------------------------------
62
+ // Search control mapping (SearchControls → Firecrawl-native API params)
63
+ // ---------------------------------------------------------------------------
64
+ /**
65
+ * Map `--recency` to Firecrawl's Google-style `tbs` time filter
66
+ * (investigation §3.2). `noLimit` and absent recency omit the param.
67
+ */
68
+ function mapRecencyToTbs(recency) {
69
+ switch (recency) {
70
+ case "oneDay":
71
+ return "qdr:d";
72
+ case "oneWeek":
73
+ return "qdr:w";
74
+ case "oneMonth":
75
+ return "qdr:m";
76
+ case "oneYear":
77
+ return "qdr:y";
78
+ case "noLimit":
79
+ return undefined;
80
+ default:
81
+ return undefined;
82
+ }
83
+ }
84
+ /**
85
+ * Map Provider-neutral `SearchControls` to Firecrawl-native search params.
86
+ *
87
+ * domain -> includeDomains: [domain]
88
+ * recency -> tbs (qdr:*)
89
+ * contentSize -> scrapeOptions.formats:["markdown"] (high only)
90
+ * topic -> sources: news → [{type:"news"}], else [{type:"web"}]
91
+ *
92
+ * `location` is rejected in `validate` (Z.AI-specific). `sources` always
93
+ * carries a default of `[{type:"web"}]` per the investigation's
94
+ * recommended default.
95
+ */
96
+ function mapSearchControls(controls) {
97
+ const topic = controls?.topic;
98
+ const sources = [{ type: topic === "news" ? "news" : "web" }];
99
+ const params = { sources };
100
+ if (controls?.domain) {
101
+ params.includeDomains = [controls.domain];
102
+ }
103
+ if (controls?.recency) {
104
+ const tbs = mapRecencyToTbs(controls.recency);
105
+ if (tbs)
106
+ params.tbs = tbs;
107
+ }
108
+ if (controls?.contentSize === "high") {
109
+ params.scrapeOptions = { formats: ["markdown"] };
110
+ }
111
+ return params;
112
+ }
113
+ // ---------------------------------------------------------------------------
114
+ // Response normalization
115
+ // ---------------------------------------------------------------------------
116
+ /**
117
+ * Normalize a raw Firecrawl search response into `SearchSource[]`.
118
+ *
119
+ * Firecrawl nests the result list under a source-type key inside `data`
120
+ * (`data.web` for a web search, `data.news` for a news-topic search) rather
121
+ * than as a flat `data` array. A flat `data` array is also accepted for
122
+ * forward-compat.
123
+ *
124
+ * data.web[].title -> title
125
+ * data.web[].url -> url
126
+ * data.web[].markdown -> summary (high content-size — richer)
127
+ * data.web[].description -> summary (medium fallback)
128
+ * data.web[].source|category -> source
129
+ *
130
+ * Any malformed shape is a retryable `ApiError` 500.
131
+ */
132
+ function normalizeFirecrawlSearchResults(raw) {
133
+ if (!isPlainObject(raw)) {
134
+ throw new ApiError("Firecrawl search returned a malformed response", 500);
135
+ }
136
+ const data = raw.data;
137
+ let results;
138
+ if (Array.isArray(data)) {
139
+ results = data;
140
+ }
141
+ else if (isPlainObject(data)) {
142
+ // Results are keyed by source type (web for the default, news for a
143
+ // news-topic search). Collect from web, falling back to news.
144
+ const web = data.web;
145
+ results =
146
+ Array.isArray(web) && web.length > 0 ? web : Array.isArray(data.news) ? data.news : undefined;
147
+ }
148
+ if (!Array.isArray(results)) {
149
+ throw new ApiError("Firecrawl search returned a malformed response", 500);
150
+ }
151
+ const out = [];
152
+ for (const entry of results) {
153
+ if (!isPlainObject(entry)) {
154
+ throw new ApiError("Firecrawl search returned a malformed response", 500);
155
+ }
156
+ const title = entry.title;
157
+ const url = entry.url;
158
+ if (typeof title !== "string" || typeof url !== "string") {
159
+ throw new ApiError("Firecrawl search returned a malformed response", 500);
160
+ }
161
+ // Prefer the scraped markdown (high content-size) over the bare
162
+ // description (medium). An absent description is an empty summary.
163
+ const markdown = typeof entry.markdown === "string" ? entry.markdown : undefined;
164
+ const description = typeof entry.description === "string" ? entry.description : undefined;
165
+ const summary = markdown ?? description ?? "";
166
+ const source = typeof entry.source === "string"
167
+ ? entry.source
168
+ : typeof entry.category === "string"
169
+ ? entry.category
170
+ : undefined;
171
+ const result = { title, url, summary };
172
+ if (source !== undefined)
173
+ result.source = source;
174
+ out.push(result);
175
+ }
176
+ return out;
177
+ }
178
+ /**
179
+ * Normalize a raw Firecrawl scrape response into a `ReaderFetchResult`.
180
+ *
181
+ * data.markdown|text -> content (per requested format)
182
+ * data.metadata.sourceURL -> finalUrl
183
+ * data.metadata.title -> title (null when absent/blank)
184
+ *
185
+ * Firecrawl returns a genuine title (better than Tavily's null). Any
186
+ * malformed shape is a retryable `ApiError` 500.
187
+ */
188
+ function normalizeFirecrawlScrapeResult(raw, request) {
189
+ if (!isPlainObject(raw)) {
190
+ throw new ApiError("Firecrawl scrape returned a malformed response", 500);
191
+ }
192
+ const data = raw.data;
193
+ if (!isPlainObject(data)) {
194
+ throw new ApiError("Firecrawl scrape returned a malformed response", 500);
195
+ }
196
+ const contentFormat = request.format ?? "markdown";
197
+ const contentField = contentFormat === "text" ? "text" : "markdown";
198
+ const content = data[contentField];
199
+ if (typeof content !== "string" || content.length === 0) {
200
+ throw new ApiError("Firecrawl scrape returned a malformed response", 500);
201
+ }
202
+ const metadata = isPlainObject(data.metadata) ? data.metadata : {};
203
+ const sourceURL = typeof metadata.sourceURL === "string" && metadata.sourceURL.length > 0
204
+ ? metadata.sourceURL
205
+ : undefined;
206
+ const finalUrl = sourceURL ?? request.url;
207
+ const rawTitle = typeof metadata.title === "string" ? metadata.title : undefined;
208
+ const title = rawTitle !== undefined && rawTitle.trim().length > 0 ? rawTitle : null;
209
+ return {
210
+ schemaVersion: 1,
211
+ url: request.url,
212
+ finalUrl,
213
+ title,
214
+ content,
215
+ contentFormat,
216
+ };
217
+ }
218
+ /**
219
+ * Normalize a raw Firecrawl map response into a `MapResult`.
220
+ *
221
+ * links[].url -> urls (links are objects {url, title}, not strings)
222
+ * baseUrl -> the request URL
223
+ * totalUrls -> links array length
224
+ *
225
+ * A bare-string `links[]` entry is also accepted for forward-compat. Any
226
+ * malformed shape is a retryable `ApiError` 500.
227
+ */
228
+ function normalizeFirecrawlMapResult(raw, request) {
229
+ if (!isPlainObject(raw)) {
230
+ throw new ApiError("Firecrawl map returned a malformed response", 500);
231
+ }
232
+ const links = raw.links;
233
+ if (!Array.isArray(links)) {
234
+ throw new ApiError("Firecrawl map returned a malformed response", 500);
235
+ }
236
+ const urls = [];
237
+ for (const entry of links) {
238
+ let url;
239
+ if (typeof entry === "string") {
240
+ url = entry;
241
+ }
242
+ else if (isPlainObject(entry)) {
243
+ url = entry.url;
244
+ }
245
+ if (typeof url !== "string" || url.length === 0) {
246
+ throw new ApiError("Firecrawl map returned a malformed response", 500);
247
+ }
248
+ urls.push(url);
249
+ }
250
+ return {
251
+ schemaVersion: 1,
252
+ baseUrl: request.url,
253
+ urls,
254
+ totalUrls: urls.length,
255
+ };
256
+ }
257
+ // ---------------------------------------------------------------------------
258
+ // Failure normalization: stable public codes, no raw payloads (NFR-006)
259
+ // ---------------------------------------------------------------------------
260
+ /**
261
+ * Resolve a stable HTTP-style status code for retry classification.
262
+ * Explicit typed errors carry their own status; the fallback default is
263
+ * 500 (transient). Mirrors the Tavily helper.
264
+ */
265
+ function inferStatusCode(known) {
266
+ if (typeof known === "number" && Number.isFinite(known))
267
+ return known;
268
+ return 500;
269
+ }
270
+ /**
271
+ * Status-keyed outward message for rewrapped Firecrawl ApiErrors. Curated
272
+ * constants only — a raw Provider body embedded upstream never survives.
273
+ */
274
+ function firecrawlApiErrorMessage(statusCode) {
275
+ if (statusCode === 429)
276
+ return "Firecrawl rate limit exceeded";
277
+ return "Firecrawl request failed";
278
+ }
279
+ /**
280
+ * Normalize a Provider failure with sanitized messages. Raw response
281
+ * bodies never cross the adapter boundary. Mirrors `normalizeTavilyError`.
282
+ */
283
+ function normalizeFirecrawlError(error) {
284
+ // QuotaError pass-through — terminal retry guarantee preserved.
285
+ if (error instanceof QuotaError)
286
+ return error;
287
+ // Configuration/option/validation errors carry clean, human-authored
288
+ // messages and are safe to surface verbatim.
289
+ if (error instanceof ValidationError ||
290
+ error instanceof UnsupportedOptionError ||
291
+ error instanceof ConfigurationError) {
292
+ return error;
293
+ }
294
+ // Re-wrap typed transport errors with sanitized messages so a raw
295
+ // Provider response body embedded upstream never survives.
296
+ if (error instanceof AuthError) {
297
+ return new AuthError("Firecrawl authentication failed", "FIRECRAWL_API_KEY");
298
+ }
299
+ if (error instanceof NetworkError) {
300
+ return new NetworkError("Firecrawl network error");
301
+ }
302
+ if (error instanceof TimeoutError) {
303
+ return new TimeoutError(error.durationMs, "Try again or increase timeout with FIRECRAWL_TIMEOUT env var");
304
+ }
305
+ if (error instanceof ApiError) {
306
+ const statusCode = inferStatusCode(error.statusCode);
307
+ return new ApiError(firecrawlApiErrorMessage(statusCode), statusCode);
308
+ }
309
+ const message = error instanceof Error ? error.message : String(error);
310
+ const lower = message.toLowerCase();
311
+ if (lower.includes("401") ||
312
+ lower.includes("403") ||
313
+ lower.includes("unauthorized") ||
314
+ lower.includes("forbidden")) {
315
+ return new AuthError("Firecrawl authentication failed");
316
+ }
317
+ if (lower.includes("timeout") || lower.includes("timed out") || lower.includes("etimedout")) {
318
+ return new TimeoutError(30000);
319
+ }
320
+ if (lower.includes("econnrefused") ||
321
+ lower.includes("econnreset") ||
322
+ lower.includes("network") ||
323
+ lower.includes("enotfound") ||
324
+ lower.includes("fetch failed")) {
325
+ return new NetworkError("Firecrawl network error");
326
+ }
327
+ if (lower.includes("429") || lower.includes("rate limit")) {
328
+ return new ApiError("Firecrawl rate limit exceeded", 429);
329
+ }
330
+ return new ApiError("Firecrawl request failed", 500);
331
+ }
332
+ // ---------------------------------------------------------------------------
333
+ // Reader validation helpers
334
+ // ---------------------------------------------------------------------------
335
+ /** Z.AI-only reader options that Firecrawl does not accept. */
336
+ const UNSUPPORTED_READER_OPTIONS = [
337
+ "withLinksSummary",
338
+ "noGfm",
339
+ "keepImgDataUrl",
340
+ "withImagesSummary",
341
+ ];
342
+ function assertNoUnsupportedReaderOptions(request) {
343
+ for (const key of UNSUPPORTED_READER_OPTIONS) {
344
+ // Only reject when the user explicitly enabled the option (`true`).
345
+ // The read command handler sets boolean options to `false` (not
346
+ // `undefined`) when the flag is absent, so `!== undefined` would
347
+ // over-reject. `false` means "user didn't pass the flag" → accept.
348
+ if (request[key] === true) {
349
+ throw new UnsupportedOptionError("firecrawl", "reader", key);
350
+ }
351
+ }
352
+ }
353
+ function createFirecrawlSearchCapability(options) {
354
+ const { env, transport } = options;
355
+ const capability = {
356
+ validate(request) {
357
+ if (!request || typeof request.query !== "string" || request.query.trim() === "") {
358
+ throw new ValidationError("Search query must contain at least one non-whitespace character");
359
+ }
360
+ // Firecrawl supports domain, recency, contentSize, and topic.
361
+ // location is Z.AI-specific and rejected before any transport call.
362
+ if (request.controls?.location !== undefined) {
363
+ throw new UnsupportedOptionError("firecrawl", "search", "location");
364
+ }
365
+ // Firecrawl sources are web/news only; --topic finance has no native
366
+ // source mapping — reject explicitly rather than silently returning
367
+ // generic web results (supported topics: general, news).
368
+ if (request.controls?.topic === "finance") {
369
+ throw new ValidationError("Firecrawl search does not support --topic finance (supported: general, news)");
370
+ }
371
+ },
372
+ cacheIdentity(request) {
373
+ const apiKey = resolveApiKey(env);
374
+ const identityRequest = {
375
+ query: request.query,
376
+ };
377
+ if (request.controls) {
378
+ identityRequest.controls = request.controls;
379
+ }
380
+ return {
381
+ provider: "firecrawl",
382
+ capability: "search",
383
+ credentialFingerprint: credentialFingerprint(apiKey),
384
+ request: identityRequest,
385
+ };
386
+ },
387
+ async invoke(request) {
388
+ capability.validate(request);
389
+ const apiKey = resolveApiKey(env);
390
+ try {
391
+ const params = mapSearchControls(request.controls);
392
+ const raw = await fetchFirecrawlSearch(apiKey, request.query, params, transport);
393
+ return normalizeFirecrawlSearchResults(raw);
394
+ }
395
+ catch (error) {
396
+ throw normalizeFirecrawlError(error);
397
+ }
398
+ },
399
+ };
400
+ return capability;
401
+ }
402
+ function createFirecrawlReaderCapability(options) {
403
+ const { env, transport } = options;
404
+ const fetch = {
405
+ kind: "reader-fetch",
406
+ validate(request) {
407
+ assertHttpUrl(request.url);
408
+ assertNoUnsupportedReaderOptions(request);
409
+ },
410
+ cacheIdentity(request) {
411
+ const apiKey = resolveApiKey(env);
412
+ return {
413
+ provider: "firecrawl",
414
+ capability: "reader",
415
+ operation: "reader-fetch",
416
+ credentialFingerprint: credentialFingerprint(apiKey),
417
+ request,
418
+ legacyCandidates: [],
419
+ };
420
+ },
421
+ decodeCached(value) {
422
+ return decodeReaderFetchResult(value);
423
+ },
424
+ async invoke(request) {
425
+ fetch.validate(request);
426
+ const apiKey = resolveApiKey(env);
427
+ try {
428
+ const formats = request.format === "text" ? ["text"] : ["markdown"];
429
+ // retainImages is set only when --no-images is passed (read.ts),
430
+ // so `=== false` is exactly the "strip images" case. Inverted to
431
+ // Firecrawl's removeBase64Images.
432
+ const params = {
433
+ formats,
434
+ proxy: "basic",
435
+ ...(request.retainImages === false ? { removeBase64Images: true } : {}),
436
+ };
437
+ const raw = await fetchFirecrawlScrape(apiKey, request.url, params, transport);
438
+ return normalizeFirecrawlScrapeResult(raw, request);
439
+ }
440
+ catch (error) {
441
+ throw normalizeFirecrawlError(error);
442
+ }
443
+ },
444
+ };
445
+ return { fetch };
446
+ }
447
+ // ---------------------------------------------------------------------------
448
+ // Map Capability
449
+ // ---------------------------------------------------------------------------
450
+ /**
451
+ * Map Provider-neutral `MapRequest` to Firecrawl-native map params.
452
+ *
453
+ * limit -> limit
454
+ * instructions -> search
455
+ *
456
+ * `depth`, `breadth`, `selectPaths`, and `excludePaths` have NO native
457
+ * Firecrawl /v2/map equivalent (map is discovery-only) and are omitted
458
+ * (tech-plan D6).
459
+ */
460
+ function mapMapControls(request) {
461
+ const params = {};
462
+ if (request.limit !== undefined)
463
+ params.limit = request.limit;
464
+ if (request.instructions !== undefined)
465
+ params.search = request.instructions;
466
+ return params;
467
+ }
468
+ function createFirecrawlMapCapability(options) {
469
+ const { env, transport } = options;
470
+ const fetch = {
471
+ kind: "map-fetch",
472
+ validate(request) {
473
+ assertHttpUrl(request.url);
474
+ if (request.limit !== undefined && request.limit <= 0) {
475
+ throw new ValidationError("Map limit must be greater than 0");
476
+ }
477
+ // Firecrawl /v2/map supports only url, limit, and search(instructions).
478
+ // depth/breadth/selectPaths/excludePaths have no native map equivalent
479
+ // (Tavily map supports them) — reject rather than silently discard.
480
+ if (request.depth !== undefined) {
481
+ throw new UnsupportedOptionError("firecrawl", "map", "depth");
482
+ }
483
+ if (request.breadth !== undefined) {
484
+ throw new UnsupportedOptionError("firecrawl", "map", "breadth");
485
+ }
486
+ if (request.selectPaths !== undefined) {
487
+ throw new UnsupportedOptionError("firecrawl", "map", "selectPaths");
488
+ }
489
+ if (request.excludePaths !== undefined) {
490
+ throw new UnsupportedOptionError("firecrawl", "map", "excludePaths");
491
+ }
492
+ },
493
+ cacheIdentity(request) {
494
+ const apiKey = resolveApiKey(env);
495
+ return {
496
+ provider: "firecrawl",
497
+ capability: "map",
498
+ credentialFingerprint: credentialFingerprint(apiKey),
499
+ request,
500
+ };
501
+ },
502
+ decodeCached(value) {
503
+ return decodeMapResult(value);
504
+ },
505
+ async invoke(request) {
506
+ fetch.validate(request);
507
+ const apiKey = resolveApiKey(env);
508
+ try {
509
+ const params = mapMapControls(request);
510
+ const raw = await fetchFirecrawlMap(apiKey, request.url, params, transport);
511
+ return normalizeFirecrawlMapResult(raw, request);
512
+ }
513
+ catch (error) {
514
+ throw normalizeFirecrawlError(error);
515
+ }
516
+ },
517
+ };
518
+ return { fetch };
519
+ }
520
+ // ---------------------------------------------------------------------------
521
+ // Crawl Capability — async create→poll→resume (tech-plan D2)
522
+ // ---------------------------------------------------------------------------
523
+ const DEFAULT_CRAWL_POLL_INTERVAL_MS = 2000;
524
+ /** Safety bound on `next`-cursor traversal for very large crawls. */
525
+ const MAX_CRAWL_NEXT_ITERATIONS = 500;
526
+ /** Reclaim-on-miss staleness guard — don't adopt jobs older than this. */
527
+ const CRAWL_RECLAIM_STALE_MS = 24 * 60 * 60 * 1000;
528
+ /** Split a comma-separated path-pattern string into a trimmed array. */
529
+ function splitPathPatterns(value) {
530
+ if (value === undefined || value.trim() === "")
531
+ return undefined;
532
+ return value
533
+ .split(",")
534
+ .map((s) => s.trim())
535
+ .filter((s) => s.length > 0);
536
+ }
537
+ function isEexistError(err) {
538
+ return (typeof err === "object" &&
539
+ err !== null &&
540
+ "code" in err &&
541
+ err.code === "EEXIST");
542
+ }
543
+ function resolveCrawlPollIntervalMs(env) {
544
+ const raw = env?.FIRECRAWL_CRAWL_POLL_INTERVAL_MS;
545
+ const parsed = parseInt(raw ?? "", 10);
546
+ return Number.isFinite(parsed) && parsed >= 0 ? parsed : DEFAULT_CRAWL_POLL_INTERVAL_MS;
547
+ }
548
+ /**
549
+ * Abortable sleep built from the injected timers. When `signal` aborts
550
+ * (the command handler's `--timeout`), the pending timer is cleared and
551
+ * the promise rejects with a `TimeoutError` so the poll loop unwinds
552
+ * promptly. Mirrors the research poll loop's `makeSleep`.
553
+ */
554
+ function makeCrawlSleep(deps, signal) {
555
+ const setT = deps?.setTimeout ?? setTimeout;
556
+ const clearT = deps?.clearTimeout ?? clearTimeout;
557
+ return (ms) => new Promise((resolve, reject) => {
558
+ if (signal?.aborted) {
559
+ reject(new TimeoutError(0, "Crawl polling aborted"));
560
+ return;
561
+ }
562
+ if (ms <= 0) {
563
+ setImmediate(() => {
564
+ if (signal?.aborted) {
565
+ reject(new TimeoutError(0, "Crawl polling aborted"));
566
+ return;
567
+ }
568
+ resolve();
569
+ });
570
+ return;
571
+ }
572
+ const onAbort = () => {
573
+ clearT(id);
574
+ reject(new TimeoutError(0, "Crawl polling aborted"));
575
+ };
576
+ const id = setT(() => {
577
+ signal?.removeEventListener("abort", onAbort);
578
+ resolve();
579
+ }, ms);
580
+ signal?.addEventListener("abort", onAbort);
581
+ });
582
+ }
583
+ /**
584
+ * Map a Provider-neutral `CrawlRequest` into Firecrawl-native /v2/crawl
585
+ * body fields. `breadth` has no Firecrawl equivalent (rejected in
586
+ * `validate`); `proxy` is pinned to `"basic"` (D9 cost-safety) and nests
587
+ * under `scrapeOptions` (crawl nests scrape fields; `/scrape` takes
588
+ * `proxy` top-level). `format` nests under `scrapeOptions.formats`.
589
+ */
590
+ function mapCrawlControls(request) {
591
+ const contentFormat = request.format ?? "markdown";
592
+ const params = { scrapeOptions: { formats: [contentFormat], proxy: "basic" } };
593
+ if (request.depth !== undefined)
594
+ params.maxDepth = request.depth;
595
+ if (request.limit !== undefined)
596
+ params.limit = request.limit;
597
+ const includePaths = splitPathPatterns(request.selectPaths);
598
+ if (includePaths !== undefined)
599
+ params.includePaths = includePaths;
600
+ const excludePaths = splitPathPatterns(request.excludePaths);
601
+ if (excludePaths !== undefined)
602
+ params.excludePaths = excludePaths;
603
+ return params;
604
+ }
605
+ /**
606
+ * Normalize a batch of Firecrawl crawl page objects into `CrawlPage[]`.
607
+ *
608
+ * data[].metadata.sourceURL -> url
609
+ * data[].markdown|text -> content
610
+ *
611
+ * Any malformed entry is a retryable `ApiError` 500.
612
+ */
613
+ function normalizeCrawlPages(data, request) {
614
+ const contentFormat = request.format ?? "markdown";
615
+ const contentField = contentFormat === "text" ? "text" : "markdown";
616
+ const pages = [];
617
+ for (const entry of data) {
618
+ if (!isPlainObject(entry)) {
619
+ throw new ApiError("Firecrawl crawl returned a malformed response", 500);
620
+ }
621
+ const metadata = isPlainObject(entry.metadata) ? entry.metadata : {};
622
+ const url = metadata.sourceURL;
623
+ const content = entry[contentField];
624
+ if (typeof url !== "string" ||
625
+ url.length === 0 ||
626
+ typeof content !== "string" ||
627
+ content.length === 0) {
628
+ throw new ApiError("Firecrawl crawl returned a malformed response", 500);
629
+ }
630
+ pages.push({ url, content, contentFormat });
631
+ }
632
+ return pages;
633
+ }
634
+ /**
635
+ * Collect the full crawl result from a completed poll, following the
636
+ * pagination cursor `next` to exhaustion for large sets (each batch
637
+ * distinct — no dedup needed; tech-plan D2 / G0 #1). Bounded by a
638
+ * max-iteration guard against a runaway cursor.
639
+ */
640
+ async function collectCrawlResult(poll, apiKey, request, transport) {
641
+ const pages = [];
642
+ if (poll.data)
643
+ pages.push(...normalizeCrawlPages(poll.data, request));
644
+ let next = poll.next;
645
+ let guard = 0;
646
+ while (next !== undefined) {
647
+ guard += 1;
648
+ if (guard > MAX_CRAWL_NEXT_ITERATIONS) {
649
+ throw new ApiError("Firecrawl crawl pagination exceeded the safety limit", 500);
650
+ }
651
+ const page = await fetchFirecrawlCrawlNext(apiKey, next, transport);
652
+ if (page.data)
653
+ pages.push(...normalizeCrawlPages(page.data, request));
654
+ next = page.next;
655
+ }
656
+ return { schemaVersion: 1, baseUrl: request.url, pages, totalPages: pages.length };
657
+ }
658
+ /**
659
+ * Unordered array equality for path-filter matching. `undefined` expected
660
+ * means "no constraint requested" (always compatible).
661
+ */
662
+ function stringArrayEqualUnordered(a, expected) {
663
+ if (expected === undefined)
664
+ return true;
665
+ if (!Array.isArray(a))
666
+ return false;
667
+ return JSON.stringify([...a].sort()) === JSON.stringify([...expected].sort());
668
+ }
669
+ /**
670
+ * Best-effort compatibility check between an active-job `options` blob and
671
+ * the params this request would send. The server echoes the request
672
+ * options; verifying the cost-bearing fields (limit, maxDepth, path
673
+ * filters, scrapeOptions.formats) is a strong signal it is the same job.
674
+ * Missing or differently-shaped options → not compatible (safer to create
675
+ * fresh than to mis-adopt a different crawl and return the wrong pages).
676
+ */
677
+ function crawlOptionsCompatible(options, params) {
678
+ if (!isPlainObject(options))
679
+ return false;
680
+ if (params.maxDepth !== undefined && options.maxDepth !== params.maxDepth)
681
+ return false;
682
+ if (params.limit !== undefined && options.limit !== params.limit)
683
+ return false;
684
+ const expectedFormats = params.scrapeOptions?.formats;
685
+ const so = options.scrapeOptions;
686
+ if (expectedFormats !== undefined && isPlainObject(so) && Array.isArray(so.formats)) {
687
+ const got = JSON.stringify([...so.formats].sort());
688
+ const want = JSON.stringify([...expectedFormats].sort());
689
+ if (got !== want)
690
+ return false;
691
+ }
692
+ // Path filters determine which pages are crawled (cost AND result
693
+ // content) — a mismatched filter would adopt a differently-scoped crawl.
694
+ if (!stringArrayEqualUnordered(options.includePaths, params.includePaths))
695
+ return false;
696
+ if (!stringArrayEqualUnordered(options.excludePaths, params.excludePaths))
697
+ return false;
698
+ return true;
699
+ }
700
+ /**
701
+ * Reclaim-on-miss: find an in-flight job matching this request by `url`
702
+ * (with a `created_at` recency guard against adopting stale jobs, and a
703
+ * best-effort options check when the server echoes them). A missing or
704
+ * unparseable `created_at` is treated as stale (skipped) so a zombie entry
705
+ * is never adopted. Returns the job id, or `undefined` when no match exists.
706
+ */
707
+ function matchActiveCrawl(active, request, params) {
708
+ const now = Date.now();
709
+ for (const entry of active) {
710
+ if (entry.url !== request.url)
711
+ continue;
712
+ const ts = entry.created_at !== undefined ? Date.parse(entry.created_at) : NaN;
713
+ if (!Number.isFinite(ts) || now - ts > CRAWL_RECLAIM_STALE_MS)
714
+ continue;
715
+ if (entry.options !== undefined && !crawlOptionsCompatible(entry.options, params))
716
+ continue;
717
+ return entry.id;
718
+ }
719
+ return undefined;
720
+ }
721
+ /**
722
+ * Persist a crawl job id in the state file (atomic `wx` create). On EEXIST
723
+ * (a concurrent invocation already persisted a job for this request), read
724
+ * and return its id instead — the concurrent job is the one to poll.
725
+ */
726
+ async function persistCrawlId(stateFile, identityHash, id) {
727
+ const state = {
728
+ requestId: id,
729
+ identityHash,
730
+ createdAt: new Date().toISOString(),
731
+ status: "pending",
732
+ };
733
+ try {
734
+ await stateFile.write(identityHash, state);
735
+ return id;
736
+ }
737
+ catch (err) {
738
+ if (isEexistError(err)) {
739
+ const existing = await stateFile.read(identityHash);
740
+ return existing !== null ? existing.requestId : id;
741
+ }
742
+ throw err;
743
+ }
744
+ }
745
+ // ---------------------------------------------------------------------------
746
+ // Concurrent-create lock (cost-safety — review C3)
747
+ // ---------------------------------------------------------------------------
748
+ /** How long to wait for a contended crawl create-lock before giving up. */
749
+ const CRAWL_LOCK_TIMEOUT_MS = 30000;
750
+ /** A lock older than this is treated as stale (holder died) and broken. */
751
+ const CRAWL_LOCK_STALE_MS = 10 * 60 * 1000;
752
+ /**
753
+ * Serialize the listActive→create→persist critical section per request so two
754
+ * concurrent identical crawls can't both POST (and charge) a job — the second
755
+ * waits, then reclaims the first's job from `/active` instead of re-POSTing.
756
+ * Uses an exclusive `wx`-create lockfile sibling to the state file; a stale
757
+ * lock (holder crashed) is broken after {@link CRAWL_LOCK_STALE_MS}.
758
+ *
759
+ * When `stateDir` is undefined (in-memory test mode), the lock is a no-op —
760
+ * tests are single-process and need no cross-process serialization.
761
+ */
762
+ async function withCrawlLock(stateDir, identityHash, fn, deps) {
763
+ if (stateDir === undefined)
764
+ return fn();
765
+ const setT = deps?.setTimeout ?? setTimeout;
766
+ const sleep = (ms) => new Promise((r) => setT(() => r(), ms));
767
+ await fs.mkdir(stateDir, { recursive: true }).catch(() => { });
768
+ const lockPath = path.join(stateDir, `${identityHash}.lock`);
769
+ const deadline = Date.now() + CRAWL_LOCK_TIMEOUT_MS;
770
+ for (;;) {
771
+ try {
772
+ const handle = await fs.open(lockPath, "wx");
773
+ try {
774
+ return await fn();
775
+ }
776
+ finally {
777
+ await handle.close().catch(() => { });
778
+ await fs.unlink(lockPath).catch(() => { });
779
+ }
780
+ }
781
+ catch (err) {
782
+ if (!isEexistError(err))
783
+ throw err;
784
+ if (Date.now() > deadline) {
785
+ throw new ApiError("Firecrawl crawl create-lock timed out", 500);
786
+ }
787
+ // Break a stale lock (the holder died without releasing).
788
+ const stat = await fs.stat(lockPath).catch(() => null);
789
+ if (stat && Date.now() - stat.mtimeMs > CRAWL_LOCK_STALE_MS) {
790
+ await fs.unlink(lockPath).catch(() => { });
791
+ continue;
792
+ }
793
+ await sleep(500);
794
+ }
795
+ }
796
+ }
797
+ /**
798
+ * Reclaim an in-flight job whose create-POST response was lost, or create a
799
+ * fresh one. On a state-file miss, GET /v2/crawl/active and adopt a matching
800
+ * job; else POST /v2/crawl and persist the id. The whole listActive→create→
801
+ * persist sequence runs under {@link withCrawlLock} so concurrent identical
802
+ * invocations serialize (the second reclaims the first's job instead of
803
+ * re-POSTing). The create POST is zero-retry at the shared-execution layer.
804
+ */
805
+ async function reclaimOrCreateCrawlJob(apiKey, request, identityHash, params, stateFile, transport, stateDir) {
806
+ return withCrawlLock(stateDir, identityHash, async () => {
807
+ const active = await listActiveFirecrawlCrawls(apiKey, transport);
808
+ const matched = matchActiveCrawl(active, request, params);
809
+ if (matched !== undefined) {
810
+ return persistCrawlId(stateFile, identityHash, matched);
811
+ }
812
+ const created = await createFirecrawlCrawl(apiKey, request.url, params, transport);
813
+ return persistCrawlId(stateFile, identityHash, created.id);
814
+ }, transport);
815
+ }
816
+ function createFirecrawlCrawlCapability(options) {
817
+ const { env, transport, stateFile, stateDir } = options;
818
+ const fetch = {
819
+ kind: "crawl-fetch",
820
+ validate(request) {
821
+ assertHttpUrl(request.url);
822
+ // breadth has no Firecrawl equivalent (D9) — reject, don't ignore.
823
+ if (request.breadth !== undefined) {
824
+ throw new UnsupportedOptionError("firecrawl", "crawl", "breadth");
825
+ }
826
+ if (request.depth !== undefined) {
827
+ if (!Number.isInteger(request.depth) || request.depth < 1 || request.depth > 5) {
828
+ throw new ValidationError("Crawl depth must be an integer between 1 and 5");
829
+ }
830
+ }
831
+ if (request.limit !== undefined && request.limit <= 0) {
832
+ throw new ValidationError("Crawl limit must be greater than 0");
833
+ }
834
+ },
835
+ cacheIdentity(request) {
836
+ const apiKey = resolveApiKey(env);
837
+ return {
838
+ provider: "firecrawl",
839
+ capability: "crawl",
840
+ credentialFingerprint: credentialFingerprint(apiKey),
841
+ request,
842
+ };
843
+ },
844
+ decodeCached(value) {
845
+ return decodeCrawlResult(value);
846
+ },
847
+ async invoke(request, signal) {
848
+ fetch.validate(request);
849
+ const apiKey = resolveApiKey(env);
850
+ const credFingerprint = credentialFingerprint(apiKey);
851
+ const identityHash = computeAsyncJobStateHash({
852
+ provider: "firecrawl",
853
+ capability: "crawl",
854
+ credentialFingerprint: credFingerprint,
855
+ request,
856
+ });
857
+ const params = mapCrawlControls(request);
858
+ const pollIntervalMs = resolveCrawlPollIntervalMs(env);
859
+ const sleep = makeCrawlSleep(transport, signal);
860
+ try {
861
+ // 1. Resume: an in-flight job for this request was already persisted
862
+ // (Ctrl-C / crash mid-poll). Poll it instead of creating a second
863
+ // one (double-charge prevention).
864
+ const existing = await stateFile.read(identityHash);
865
+ let id;
866
+ if (existing !== null) {
867
+ id = existing.requestId;
868
+ }
869
+ else {
870
+ // 2. Reclaim-on-miss or create (zero-retry create; wx-flag write).
871
+ id = await reclaimOrCreateCrawlJob(apiKey, request, identityHash, params, stateFile, transport, stateDir);
872
+ }
873
+ // 3. Poll status to terminal, then collect the result once. We do
874
+ // NOT accumulate data[] across pre-completion polls — the
875
+ // completed response carries the full set (G0 #1).
876
+ let recreated = false;
877
+ for (;;) {
878
+ if (signal?.aborted) {
879
+ throw new TimeoutError(0, "Crawl polling aborted");
880
+ }
881
+ const poll = await pollFirecrawlCrawl(apiKey, id, transport);
882
+ if (poll.status === "completed") {
883
+ // Collect FIRST, remove only on success. If collection throws
884
+ // (a malformed page, a pagination/network blip), the state file
885
+ // stays so a re-run re-polls this already-completed (and paid-
886
+ // for) job instead of POSTing a second, doubly-charging one.
887
+ const result = await collectCrawlResult(poll, apiKey, request, transport);
888
+ await stateFile.remove(identityHash);
889
+ return result;
890
+ }
891
+ if (poll.status === "failed") {
892
+ await stateFile.remove(identityHash);
893
+ throw new ApiError("Firecrawl crawl job failed", 500);
894
+ }
895
+ if (poll.status === "not_found") {
896
+ // Server-side job disappeared — drop state and create fresh,
897
+ // but at most once. A pathological server that 404s every
898
+ // freshly-created job would otherwise loop and re-POST
899
+ // indefinitely (each POST charges credits).
900
+ if (recreated) {
901
+ throw new ApiError("Firecrawl crawl job could not be found after recreate", 500);
902
+ }
903
+ recreated = true;
904
+ await stateFile.remove(identityHash);
905
+ id = await reclaimOrCreateCrawlJob(apiKey, request, identityHash, params, stateFile, transport, stateDir);
906
+ continue;
907
+ }
908
+ // scraping: sleep and poll again. The state file already holds
909
+ // the id (the load-bearing field for resume).
910
+ await sleep(pollIntervalMs);
911
+ }
912
+ }
913
+ catch (error) {
914
+ throw normalizeFirecrawlError(error);
915
+ }
916
+ },
917
+ };
918
+ return { fetch };
919
+ }
920
+ /**
921
+ * Build the Firecrawl Provider Descriptor. Advertises the full Firecrawl
922
+ * capability set. `create()` wires all six capabilities — Search, Reader,
923
+ * Map, async Crawl (create→poll→resume + reclaim-on-miss), Quota
924
+ * (`unit:"credits"`), and Diagnostics (single-scrape probe). Construction
925
+ * is side-effect-free; the transport is invoked per Capability call.
926
+ */
927
+ export function createFirecrawlDescriptor(dependencies) {
928
+ const transport = dependencies?.transport;
929
+ const crawlStateDir = dependencies?.crawlStateDir ?? asyncJobStateDir("crawl");
930
+ const crawlStateFile = dependencies?.crawlStateFile ?? createProductionAsyncJobStateFile(crawlStateDir);
931
+ return {
932
+ id: "firecrawl",
933
+ isConfigured(env) {
934
+ return isFirecrawlConfigured(env);
935
+ },
936
+ capabilities() {
937
+ return new Set([
938
+ "search",
939
+ "reader",
940
+ "crawl",
941
+ "map",
942
+ "quota",
943
+ "diagnostics",
944
+ ]);
945
+ },
946
+ create(context) {
947
+ const search = createFirecrawlSearchCapability({ env: context.env, transport });
948
+ const reader = createFirecrawlReaderCapability({ env: context.env, transport });
949
+ const crawl = createFirecrawlCrawlCapability({
950
+ env: context.env,
951
+ transport,
952
+ stateFile: crawlStateFile,
953
+ stateDir: crawlStateDir,
954
+ });
955
+ const map = createFirecrawlMapCapability({ env: context.env, transport });
956
+ const quota = createFirecrawlQuotaCapability({ env: context.env, transport });
957
+ const diagnostics = createFirecrawlDiagnosticsCapability({ env: context.env, transport });
958
+ return { id: "firecrawl", search, reader, crawl, map, quota, diagnostics };
959
+ },
960
+ };
961
+ }
962
+ //# sourceMappingURL=adapter.js.map