scoutline 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +23 -20
- package/dist/capabilities/quota.d.ts +1 -1
- package/dist/capabilities/quota.d.ts.map +1 -1
- package/dist/commands/crawl.js +1 -1
- package/dist/commands/doctor.js +1 -1
- package/dist/commands/map.js +1 -1
- package/dist/commands/read.js +1 -1
- package/dist/commands/research.js +5 -5
- package/dist/commands/research.js.map +1 -1
- package/dist/commands/search.js +1 -1
- package/dist/index.js +1 -1
- package/dist/lib/async-job-state.d.ts +119 -0
- package/dist/lib/async-job-state.d.ts.map +1 -0
- package/dist/lib/{research-state.js → async-job-state.js} +43 -34
- package/dist/lib/async-job-state.js.map +1 -0
- package/dist/lib/cache.d.ts +16 -8
- package/dist/lib/cache.d.ts.map +1 -1
- package/dist/lib/cache.js +33 -8
- package/dist/lib/cache.js.map +1 -1
- package/dist/lib/execution.d.ts.map +1 -1
- package/dist/lib/execution.js +8 -4
- package/dist/lib/execution.js.map +1 -1
- package/dist/lib/redact.d.ts +1 -1
- package/dist/lib/redact.d.ts.map +1 -1
- package/dist/lib/redact.js +9 -1
- package/dist/lib/redact.js.map +1 -1
- package/dist/providers/exa/adapter.d.ts +2 -2
- package/dist/providers/exa/adapter.d.ts.map +1 -1
- package/dist/providers/exa/adapter.js +5 -3
- package/dist/providers/exa/adapter.js.map +1 -1
- package/dist/providers/firecrawl/adapter.d.ts +53 -0
- package/dist/providers/firecrawl/adapter.d.ts.map +1 -0
- package/dist/providers/firecrawl/adapter.js +962 -0
- package/dist/providers/firecrawl/adapter.js.map +1 -0
- package/dist/providers/firecrawl/client.d.ts +182 -0
- package/dist/providers/firecrawl/client.d.ts.map +1 -0
- package/dist/providers/firecrawl/client.js +426 -0
- package/dist/providers/firecrawl/client.js.map +1 -0
- package/dist/providers/firecrawl/credentials.d.ts +38 -0
- package/dist/providers/firecrawl/credentials.d.ts.map +1 -0
- package/dist/providers/firecrawl/credentials.js +60 -0
- package/dist/providers/firecrawl/credentials.js.map +1 -0
- package/dist/providers/firecrawl/diagnostics.d.ts +36 -0
- package/dist/providers/firecrawl/diagnostics.d.ts.map +1 -0
- package/dist/providers/firecrawl/diagnostics.js +65 -0
- package/dist/providers/firecrawl/diagnostics.js.map +1 -0
- package/dist/providers/firecrawl/quota.d.ts +46 -0
- package/dist/providers/firecrawl/quota.d.ts.map +1 -0
- package/dist/providers/firecrawl/quota.js +120 -0
- package/dist/providers/firecrawl/quota.js.map +1 -0
- package/dist/providers/registry.d.ts.map +1 -1
- package/dist/providers/registry.js +2 -0
- package/dist/providers/registry.js.map +1 -1
- package/dist/providers/tavily/adapter.d.ts +2 -2
- package/dist/providers/tavily/adapter.d.ts.map +1 -1
- package/dist/providers/tavily/adapter.js +5 -3
- package/dist/providers/tavily/adapter.js.map +1 -1
- package/dist/providers/types.d.ts +1 -1
- package/dist/providers/types.d.ts.map +1 -1
- package/dist/providers/types.js +1 -1
- package/dist/providers/types.js.map +1 -1
- package/package.json +1 -1
- package/dist/lib/research-state.d.ts +0 -104
- package/dist/lib/research-state.d.ts.map +0 -1
- package/dist/lib/research-state.js.map +0 -1
- package/dist/providers/minimax/sdk-client.d.ts +0 -29
- package/dist/providers/minimax/sdk-client.d.ts.map +0 -1
- package/dist/providers/minimax/sdk-client.js +0 -50
- package/dist/providers/minimax/sdk-client.js.map +0 -1
|
@@ -0,0 +1,962 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Firecrawl Provider Adapter (firecrawl tech-plan §1, D3–D6).
|
|
3
|
+
*
|
|
4
|
+
* Implements the Firecrawl Provider Descriptor with Search, Reader, Map,
|
|
5
|
+
* async Crawl, Quota, and Diagnostics capabilities on top of the
|
|
6
|
+
* direct-HTTP transport (`./client.ts`). The Adapter owns credentials,
|
|
7
|
+
* transport lifecycle, Provider field mapping, and failure normalization;
|
|
8
|
+
* shared execution owns cache and retry policy.
|
|
9
|
+
*
|
|
10
|
+
* Boundary rules (ARCHITECTURE.md §2):
|
|
11
|
+
* - May import capability types, normalized errors, Provider identity
|
|
12
|
+
* types, and the Adapter-local credential and transport Modules.
|
|
13
|
+
* - Must NOT import command presentation, output mode, or another
|
|
14
|
+
* Provider's Adapter.
|
|
15
|
+
*
|
|
16
|
+
* Field mapping (tech-plan D4/D5/D6; investigation §3.1–§3.4):
|
|
17
|
+
* Search data[].title -> title
|
|
18
|
+
* Search data[].url -> url
|
|
19
|
+
* Search data[].markdown -> summary (high content-size; else description)
|
|
20
|
+
* Search data[].description -> summary (medium content-size fallback)
|
|
21
|
+
*
|
|
22
|
+
* Scrape data.markdown|text -> content
|
|
23
|
+
* Scrape data.metadata.sourceURL -> finalUrl
|
|
24
|
+
* Scrape data.metadata.title -> title (better than Tavily's null)
|
|
25
|
+
*
|
|
26
|
+
* Map links[] -> urls
|
|
27
|
+
*/
|
|
28
|
+
import crypto from "node:crypto";
|
|
29
|
+
import * as fs from "node:fs/promises";
|
|
30
|
+
import path from "node:path";
|
|
31
|
+
import { decodeReaderFetchResult } from "../../capabilities/reader.js";
|
|
32
|
+
import { decodeMapResult } from "../../capabilities/map.js";
|
|
33
|
+
import { decodeCrawlResult } from "../../capabilities/crawl.js";
|
|
34
|
+
import { computeAsyncJobStateHash, createProductionAsyncJobStateFile, } from "../../lib/async-job-state.js";
|
|
35
|
+
import { asyncJobStateDir } from "../../lib/cache.js";
|
|
36
|
+
import { ApiError, AuthError, ConfigurationError, NetworkError, QuotaError, TimeoutError, UnsupportedOptionError, ValidationError, } from "../../lib/errors.js";
|
|
37
|
+
import { createFirecrawlCrawl, fetchFirecrawlCrawlNext, fetchFirecrawlMap, fetchFirecrawlScrape, fetchFirecrawlSearch, listActiveFirecrawlCrawls, pollFirecrawlCrawl, } from "./client.js";
|
|
38
|
+
import { isFirecrawlConfigured, requireFirecrawlApiKey } from "./credentials.js";
|
|
39
|
+
import { createFirecrawlQuotaCapability } from "./quota.js";
|
|
40
|
+
import { createFirecrawlDiagnosticsCapability } from "./diagnostics.js";
|
|
41
|
+
// ---------------------------------------------------------------------------
|
|
42
|
+
// Credential + shape helpers
|
|
43
|
+
// ---------------------------------------------------------------------------
|
|
44
|
+
function credentialFingerprint(apiKey) {
|
|
45
|
+
return crypto.createHash("sha256").update(apiKey).digest("hex");
|
|
46
|
+
}
|
|
47
|
+
function resolveApiKey(env) {
|
|
48
|
+
return requireFirecrawlApiKey(env);
|
|
49
|
+
}
|
|
50
|
+
function isPlainObject(value) {
|
|
51
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
52
|
+
}
|
|
53
|
+
function assertHttpUrl(url) {
|
|
54
|
+
if (typeof url !== "string" || url.length === 0) {
|
|
55
|
+
throw new ValidationError("Firecrawl reader URL must be a non-empty string");
|
|
56
|
+
}
|
|
57
|
+
if (!/^https?:\/\//.test(url)) {
|
|
58
|
+
throw new ValidationError("URL must start with http:// or https://");
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
// ---------------------------------------------------------------------------
|
|
62
|
+
// Search control mapping (SearchControls → Firecrawl-native API params)
|
|
63
|
+
// ---------------------------------------------------------------------------
|
|
64
|
+
/**
|
|
65
|
+
* Map `--recency` to Firecrawl's Google-style `tbs` time filter
|
|
66
|
+
* (investigation §3.2). `noLimit` and absent recency omit the param.
|
|
67
|
+
*/
|
|
68
|
+
function mapRecencyToTbs(recency) {
|
|
69
|
+
switch (recency) {
|
|
70
|
+
case "oneDay":
|
|
71
|
+
return "qdr:d";
|
|
72
|
+
case "oneWeek":
|
|
73
|
+
return "qdr:w";
|
|
74
|
+
case "oneMonth":
|
|
75
|
+
return "qdr:m";
|
|
76
|
+
case "oneYear":
|
|
77
|
+
return "qdr:y";
|
|
78
|
+
case "noLimit":
|
|
79
|
+
return undefined;
|
|
80
|
+
default:
|
|
81
|
+
return undefined;
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
/**
|
|
85
|
+
* Map Provider-neutral `SearchControls` to Firecrawl-native search params.
|
|
86
|
+
*
|
|
87
|
+
* domain -> includeDomains: [domain]
|
|
88
|
+
* recency -> tbs (qdr:*)
|
|
89
|
+
* contentSize -> scrapeOptions.formats:["markdown"] (high only)
|
|
90
|
+
* topic -> sources: news → [{type:"news"}], else [{type:"web"}]
|
|
91
|
+
*
|
|
92
|
+
* `location` is rejected in `validate` (Z.AI-specific). `sources` always
|
|
93
|
+
* carries a default of `[{type:"web"}]` per the investigation's
|
|
94
|
+
* recommended default.
|
|
95
|
+
*/
|
|
96
|
+
function mapSearchControls(controls) {
|
|
97
|
+
const topic = controls?.topic;
|
|
98
|
+
const sources = [{ type: topic === "news" ? "news" : "web" }];
|
|
99
|
+
const params = { sources };
|
|
100
|
+
if (controls?.domain) {
|
|
101
|
+
params.includeDomains = [controls.domain];
|
|
102
|
+
}
|
|
103
|
+
if (controls?.recency) {
|
|
104
|
+
const tbs = mapRecencyToTbs(controls.recency);
|
|
105
|
+
if (tbs)
|
|
106
|
+
params.tbs = tbs;
|
|
107
|
+
}
|
|
108
|
+
if (controls?.contentSize === "high") {
|
|
109
|
+
params.scrapeOptions = { formats: ["markdown"] };
|
|
110
|
+
}
|
|
111
|
+
return params;
|
|
112
|
+
}
|
|
113
|
+
// ---------------------------------------------------------------------------
|
|
114
|
+
// Response normalization
|
|
115
|
+
// ---------------------------------------------------------------------------
|
|
116
|
+
/**
|
|
117
|
+
* Normalize a raw Firecrawl search response into `SearchSource[]`.
|
|
118
|
+
*
|
|
119
|
+
* Firecrawl nests the result list under a source-type key inside `data`
|
|
120
|
+
* (`data.web` for a web search, `data.news` for a news-topic search) rather
|
|
121
|
+
* than as a flat `data` array. A flat `data` array is also accepted for
|
|
122
|
+
* forward-compat.
|
|
123
|
+
*
|
|
124
|
+
* data.web[].title -> title
|
|
125
|
+
* data.web[].url -> url
|
|
126
|
+
* data.web[].markdown -> summary (high content-size — richer)
|
|
127
|
+
* data.web[].description -> summary (medium fallback)
|
|
128
|
+
* data.web[].source|category -> source
|
|
129
|
+
*
|
|
130
|
+
* Any malformed shape is a retryable `ApiError` 500.
|
|
131
|
+
*/
|
|
132
|
+
function normalizeFirecrawlSearchResults(raw) {
|
|
133
|
+
if (!isPlainObject(raw)) {
|
|
134
|
+
throw new ApiError("Firecrawl search returned a malformed response", 500);
|
|
135
|
+
}
|
|
136
|
+
const data = raw.data;
|
|
137
|
+
let results;
|
|
138
|
+
if (Array.isArray(data)) {
|
|
139
|
+
results = data;
|
|
140
|
+
}
|
|
141
|
+
else if (isPlainObject(data)) {
|
|
142
|
+
// Results are keyed by source type (web for the default, news for a
|
|
143
|
+
// news-topic search). Collect from web, falling back to news.
|
|
144
|
+
const web = data.web;
|
|
145
|
+
results =
|
|
146
|
+
Array.isArray(web) && web.length > 0 ? web : Array.isArray(data.news) ? data.news : undefined;
|
|
147
|
+
}
|
|
148
|
+
if (!Array.isArray(results)) {
|
|
149
|
+
throw new ApiError("Firecrawl search returned a malformed response", 500);
|
|
150
|
+
}
|
|
151
|
+
const out = [];
|
|
152
|
+
for (const entry of results) {
|
|
153
|
+
if (!isPlainObject(entry)) {
|
|
154
|
+
throw new ApiError("Firecrawl search returned a malformed response", 500);
|
|
155
|
+
}
|
|
156
|
+
const title = entry.title;
|
|
157
|
+
const url = entry.url;
|
|
158
|
+
if (typeof title !== "string" || typeof url !== "string") {
|
|
159
|
+
throw new ApiError("Firecrawl search returned a malformed response", 500);
|
|
160
|
+
}
|
|
161
|
+
// Prefer the scraped markdown (high content-size) over the bare
|
|
162
|
+
// description (medium). An absent description is an empty summary.
|
|
163
|
+
const markdown = typeof entry.markdown === "string" ? entry.markdown : undefined;
|
|
164
|
+
const description = typeof entry.description === "string" ? entry.description : undefined;
|
|
165
|
+
const summary = markdown ?? description ?? "";
|
|
166
|
+
const source = typeof entry.source === "string"
|
|
167
|
+
? entry.source
|
|
168
|
+
: typeof entry.category === "string"
|
|
169
|
+
? entry.category
|
|
170
|
+
: undefined;
|
|
171
|
+
const result = { title, url, summary };
|
|
172
|
+
if (source !== undefined)
|
|
173
|
+
result.source = source;
|
|
174
|
+
out.push(result);
|
|
175
|
+
}
|
|
176
|
+
return out;
|
|
177
|
+
}
|
|
178
|
+
/**
|
|
179
|
+
* Normalize a raw Firecrawl scrape response into a `ReaderFetchResult`.
|
|
180
|
+
*
|
|
181
|
+
* data.markdown|text -> content (per requested format)
|
|
182
|
+
* data.metadata.sourceURL -> finalUrl
|
|
183
|
+
* data.metadata.title -> title (null when absent/blank)
|
|
184
|
+
*
|
|
185
|
+
* Firecrawl returns a genuine title (better than Tavily's null). Any
|
|
186
|
+
* malformed shape is a retryable `ApiError` 500.
|
|
187
|
+
*/
|
|
188
|
+
function normalizeFirecrawlScrapeResult(raw, request) {
|
|
189
|
+
if (!isPlainObject(raw)) {
|
|
190
|
+
throw new ApiError("Firecrawl scrape returned a malformed response", 500);
|
|
191
|
+
}
|
|
192
|
+
const data = raw.data;
|
|
193
|
+
if (!isPlainObject(data)) {
|
|
194
|
+
throw new ApiError("Firecrawl scrape returned a malformed response", 500);
|
|
195
|
+
}
|
|
196
|
+
const contentFormat = request.format ?? "markdown";
|
|
197
|
+
const contentField = contentFormat === "text" ? "text" : "markdown";
|
|
198
|
+
const content = data[contentField];
|
|
199
|
+
if (typeof content !== "string" || content.length === 0) {
|
|
200
|
+
throw new ApiError("Firecrawl scrape returned a malformed response", 500);
|
|
201
|
+
}
|
|
202
|
+
const metadata = isPlainObject(data.metadata) ? data.metadata : {};
|
|
203
|
+
const sourceURL = typeof metadata.sourceURL === "string" && metadata.sourceURL.length > 0
|
|
204
|
+
? metadata.sourceURL
|
|
205
|
+
: undefined;
|
|
206
|
+
const finalUrl = sourceURL ?? request.url;
|
|
207
|
+
const rawTitle = typeof metadata.title === "string" ? metadata.title : undefined;
|
|
208
|
+
const title = rawTitle !== undefined && rawTitle.trim().length > 0 ? rawTitle : null;
|
|
209
|
+
return {
|
|
210
|
+
schemaVersion: 1,
|
|
211
|
+
url: request.url,
|
|
212
|
+
finalUrl,
|
|
213
|
+
title,
|
|
214
|
+
content,
|
|
215
|
+
contentFormat,
|
|
216
|
+
};
|
|
217
|
+
}
|
|
218
|
+
/**
|
|
219
|
+
* Normalize a raw Firecrawl map response into a `MapResult`.
|
|
220
|
+
*
|
|
221
|
+
* links[].url -> urls (links are objects {url, title}, not strings)
|
|
222
|
+
* baseUrl -> the request URL
|
|
223
|
+
* totalUrls -> links array length
|
|
224
|
+
*
|
|
225
|
+
* A bare-string `links[]` entry is also accepted for forward-compat. Any
|
|
226
|
+
* malformed shape is a retryable `ApiError` 500.
|
|
227
|
+
*/
|
|
228
|
+
function normalizeFirecrawlMapResult(raw, request) {
|
|
229
|
+
if (!isPlainObject(raw)) {
|
|
230
|
+
throw new ApiError("Firecrawl map returned a malformed response", 500);
|
|
231
|
+
}
|
|
232
|
+
const links = raw.links;
|
|
233
|
+
if (!Array.isArray(links)) {
|
|
234
|
+
throw new ApiError("Firecrawl map returned a malformed response", 500);
|
|
235
|
+
}
|
|
236
|
+
const urls = [];
|
|
237
|
+
for (const entry of links) {
|
|
238
|
+
let url;
|
|
239
|
+
if (typeof entry === "string") {
|
|
240
|
+
url = entry;
|
|
241
|
+
}
|
|
242
|
+
else if (isPlainObject(entry)) {
|
|
243
|
+
url = entry.url;
|
|
244
|
+
}
|
|
245
|
+
if (typeof url !== "string" || url.length === 0) {
|
|
246
|
+
throw new ApiError("Firecrawl map returned a malformed response", 500);
|
|
247
|
+
}
|
|
248
|
+
urls.push(url);
|
|
249
|
+
}
|
|
250
|
+
return {
|
|
251
|
+
schemaVersion: 1,
|
|
252
|
+
baseUrl: request.url,
|
|
253
|
+
urls,
|
|
254
|
+
totalUrls: urls.length,
|
|
255
|
+
};
|
|
256
|
+
}
|
|
257
|
+
// ---------------------------------------------------------------------------
|
|
258
|
+
// Failure normalization: stable public codes, no raw payloads (NFR-006)
|
|
259
|
+
// ---------------------------------------------------------------------------
|
|
260
|
+
/**
|
|
261
|
+
* Resolve a stable HTTP-style status code for retry classification.
|
|
262
|
+
* Explicit typed errors carry their own status; the fallback default is
|
|
263
|
+
* 500 (transient). Mirrors the Tavily helper.
|
|
264
|
+
*/
|
|
265
|
+
function inferStatusCode(known) {
|
|
266
|
+
if (typeof known === "number" && Number.isFinite(known))
|
|
267
|
+
return known;
|
|
268
|
+
return 500;
|
|
269
|
+
}
|
|
270
|
+
/**
|
|
271
|
+
* Status-keyed outward message for rewrapped Firecrawl ApiErrors. Curated
|
|
272
|
+
* constants only — a raw Provider body embedded upstream never survives.
|
|
273
|
+
*/
|
|
274
|
+
function firecrawlApiErrorMessage(statusCode) {
|
|
275
|
+
if (statusCode === 429)
|
|
276
|
+
return "Firecrawl rate limit exceeded";
|
|
277
|
+
return "Firecrawl request failed";
|
|
278
|
+
}
|
|
279
|
+
/**
|
|
280
|
+
* Normalize a Provider failure with sanitized messages. Raw response
|
|
281
|
+
* bodies never cross the adapter boundary. Mirrors `normalizeTavilyError`.
|
|
282
|
+
*/
|
|
283
|
+
function normalizeFirecrawlError(error) {
|
|
284
|
+
// QuotaError pass-through — terminal retry guarantee preserved.
|
|
285
|
+
if (error instanceof QuotaError)
|
|
286
|
+
return error;
|
|
287
|
+
// Configuration/option/validation errors carry clean, human-authored
|
|
288
|
+
// messages and are safe to surface verbatim.
|
|
289
|
+
if (error instanceof ValidationError ||
|
|
290
|
+
error instanceof UnsupportedOptionError ||
|
|
291
|
+
error instanceof ConfigurationError) {
|
|
292
|
+
return error;
|
|
293
|
+
}
|
|
294
|
+
// Re-wrap typed transport errors with sanitized messages so a raw
|
|
295
|
+
// Provider response body embedded upstream never survives.
|
|
296
|
+
if (error instanceof AuthError) {
|
|
297
|
+
return new AuthError("Firecrawl authentication failed", "FIRECRAWL_API_KEY");
|
|
298
|
+
}
|
|
299
|
+
if (error instanceof NetworkError) {
|
|
300
|
+
return new NetworkError("Firecrawl network error");
|
|
301
|
+
}
|
|
302
|
+
if (error instanceof TimeoutError) {
|
|
303
|
+
return new TimeoutError(error.durationMs, "Try again or increase timeout with FIRECRAWL_TIMEOUT env var");
|
|
304
|
+
}
|
|
305
|
+
if (error instanceof ApiError) {
|
|
306
|
+
const statusCode = inferStatusCode(error.statusCode);
|
|
307
|
+
return new ApiError(firecrawlApiErrorMessage(statusCode), statusCode);
|
|
308
|
+
}
|
|
309
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
310
|
+
const lower = message.toLowerCase();
|
|
311
|
+
if (lower.includes("401") ||
|
|
312
|
+
lower.includes("403") ||
|
|
313
|
+
lower.includes("unauthorized") ||
|
|
314
|
+
lower.includes("forbidden")) {
|
|
315
|
+
return new AuthError("Firecrawl authentication failed");
|
|
316
|
+
}
|
|
317
|
+
if (lower.includes("timeout") || lower.includes("timed out") || lower.includes("etimedout")) {
|
|
318
|
+
return new TimeoutError(30000);
|
|
319
|
+
}
|
|
320
|
+
if (lower.includes("econnrefused") ||
|
|
321
|
+
lower.includes("econnreset") ||
|
|
322
|
+
lower.includes("network") ||
|
|
323
|
+
lower.includes("enotfound") ||
|
|
324
|
+
lower.includes("fetch failed")) {
|
|
325
|
+
return new NetworkError("Firecrawl network error");
|
|
326
|
+
}
|
|
327
|
+
if (lower.includes("429") || lower.includes("rate limit")) {
|
|
328
|
+
return new ApiError("Firecrawl rate limit exceeded", 429);
|
|
329
|
+
}
|
|
330
|
+
return new ApiError("Firecrawl request failed", 500);
|
|
331
|
+
}
|
|
332
|
+
// ---------------------------------------------------------------------------
|
|
333
|
+
// Reader validation helpers
|
|
334
|
+
// ---------------------------------------------------------------------------
|
|
335
|
+
/** Z.AI-only reader options that Firecrawl does not accept. */
|
|
336
|
+
const UNSUPPORTED_READER_OPTIONS = [
|
|
337
|
+
"withLinksSummary",
|
|
338
|
+
"noGfm",
|
|
339
|
+
"keepImgDataUrl",
|
|
340
|
+
"withImagesSummary",
|
|
341
|
+
];
|
|
342
|
+
function assertNoUnsupportedReaderOptions(request) {
|
|
343
|
+
for (const key of UNSUPPORTED_READER_OPTIONS) {
|
|
344
|
+
// Only reject when the user explicitly enabled the option (`true`).
|
|
345
|
+
// The read command handler sets boolean options to `false` (not
|
|
346
|
+
// `undefined`) when the flag is absent, so `!== undefined` would
|
|
347
|
+
// over-reject. `false` means "user didn't pass the flag" → accept.
|
|
348
|
+
if (request[key] === true) {
|
|
349
|
+
throw new UnsupportedOptionError("firecrawl", "reader", key);
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
}
|
|
353
|
+
function createFirecrawlSearchCapability(options) {
|
|
354
|
+
const { env, transport } = options;
|
|
355
|
+
const capability = {
|
|
356
|
+
validate(request) {
|
|
357
|
+
if (!request || typeof request.query !== "string" || request.query.trim() === "") {
|
|
358
|
+
throw new ValidationError("Search query must contain at least one non-whitespace character");
|
|
359
|
+
}
|
|
360
|
+
// Firecrawl supports domain, recency, contentSize, and topic.
|
|
361
|
+
// location is Z.AI-specific and rejected before any transport call.
|
|
362
|
+
if (request.controls?.location !== undefined) {
|
|
363
|
+
throw new UnsupportedOptionError("firecrawl", "search", "location");
|
|
364
|
+
}
|
|
365
|
+
// Firecrawl sources are web/news only; --topic finance has no native
|
|
366
|
+
// source mapping — reject explicitly rather than silently returning
|
|
367
|
+
// generic web results (supported topics: general, news).
|
|
368
|
+
if (request.controls?.topic === "finance") {
|
|
369
|
+
throw new ValidationError("Firecrawl search does not support --topic finance (supported: general, news)");
|
|
370
|
+
}
|
|
371
|
+
},
|
|
372
|
+
cacheIdentity(request) {
|
|
373
|
+
const apiKey = resolveApiKey(env);
|
|
374
|
+
const identityRequest = {
|
|
375
|
+
query: request.query,
|
|
376
|
+
};
|
|
377
|
+
if (request.controls) {
|
|
378
|
+
identityRequest.controls = request.controls;
|
|
379
|
+
}
|
|
380
|
+
return {
|
|
381
|
+
provider: "firecrawl",
|
|
382
|
+
capability: "search",
|
|
383
|
+
credentialFingerprint: credentialFingerprint(apiKey),
|
|
384
|
+
request: identityRequest,
|
|
385
|
+
};
|
|
386
|
+
},
|
|
387
|
+
async invoke(request) {
|
|
388
|
+
capability.validate(request);
|
|
389
|
+
const apiKey = resolveApiKey(env);
|
|
390
|
+
try {
|
|
391
|
+
const params = mapSearchControls(request.controls);
|
|
392
|
+
const raw = await fetchFirecrawlSearch(apiKey, request.query, params, transport);
|
|
393
|
+
return normalizeFirecrawlSearchResults(raw);
|
|
394
|
+
}
|
|
395
|
+
catch (error) {
|
|
396
|
+
throw normalizeFirecrawlError(error);
|
|
397
|
+
}
|
|
398
|
+
},
|
|
399
|
+
};
|
|
400
|
+
return capability;
|
|
401
|
+
}
|
|
402
|
+
function createFirecrawlReaderCapability(options) {
|
|
403
|
+
const { env, transport } = options;
|
|
404
|
+
const fetch = {
|
|
405
|
+
kind: "reader-fetch",
|
|
406
|
+
validate(request) {
|
|
407
|
+
assertHttpUrl(request.url);
|
|
408
|
+
assertNoUnsupportedReaderOptions(request);
|
|
409
|
+
},
|
|
410
|
+
cacheIdentity(request) {
|
|
411
|
+
const apiKey = resolveApiKey(env);
|
|
412
|
+
return {
|
|
413
|
+
provider: "firecrawl",
|
|
414
|
+
capability: "reader",
|
|
415
|
+
operation: "reader-fetch",
|
|
416
|
+
credentialFingerprint: credentialFingerprint(apiKey),
|
|
417
|
+
request,
|
|
418
|
+
legacyCandidates: [],
|
|
419
|
+
};
|
|
420
|
+
},
|
|
421
|
+
decodeCached(value) {
|
|
422
|
+
return decodeReaderFetchResult(value);
|
|
423
|
+
},
|
|
424
|
+
async invoke(request) {
|
|
425
|
+
fetch.validate(request);
|
|
426
|
+
const apiKey = resolveApiKey(env);
|
|
427
|
+
try {
|
|
428
|
+
const formats = request.format === "text" ? ["text"] : ["markdown"];
|
|
429
|
+
// retainImages is set only when --no-images is passed (read.ts),
|
|
430
|
+
// so `=== false` is exactly the "strip images" case. Inverted to
|
|
431
|
+
// Firecrawl's removeBase64Images.
|
|
432
|
+
const params = {
|
|
433
|
+
formats,
|
|
434
|
+
proxy: "basic",
|
|
435
|
+
...(request.retainImages === false ? { removeBase64Images: true } : {}),
|
|
436
|
+
};
|
|
437
|
+
const raw = await fetchFirecrawlScrape(apiKey, request.url, params, transport);
|
|
438
|
+
return normalizeFirecrawlScrapeResult(raw, request);
|
|
439
|
+
}
|
|
440
|
+
catch (error) {
|
|
441
|
+
throw normalizeFirecrawlError(error);
|
|
442
|
+
}
|
|
443
|
+
},
|
|
444
|
+
};
|
|
445
|
+
return { fetch };
|
|
446
|
+
}
|
|
447
|
+
// ---------------------------------------------------------------------------
|
|
448
|
+
// Map Capability
|
|
449
|
+
// ---------------------------------------------------------------------------
|
|
450
|
+
/**
|
|
451
|
+
* Map Provider-neutral `MapRequest` to Firecrawl-native map params.
|
|
452
|
+
*
|
|
453
|
+
* limit -> limit
|
|
454
|
+
* instructions -> search
|
|
455
|
+
*
|
|
456
|
+
* `depth`, `breadth`, `selectPaths`, and `excludePaths` have NO native
|
|
457
|
+
* Firecrawl /v2/map equivalent (map is discovery-only) and are omitted
|
|
458
|
+
* (tech-plan D6).
|
|
459
|
+
*/
|
|
460
|
+
function mapMapControls(request) {
|
|
461
|
+
const params = {};
|
|
462
|
+
if (request.limit !== undefined)
|
|
463
|
+
params.limit = request.limit;
|
|
464
|
+
if (request.instructions !== undefined)
|
|
465
|
+
params.search = request.instructions;
|
|
466
|
+
return params;
|
|
467
|
+
}
|
|
468
|
+
function createFirecrawlMapCapability(options) {
|
|
469
|
+
const { env, transport } = options;
|
|
470
|
+
const fetch = {
|
|
471
|
+
kind: "map-fetch",
|
|
472
|
+
validate(request) {
|
|
473
|
+
assertHttpUrl(request.url);
|
|
474
|
+
if (request.limit !== undefined && request.limit <= 0) {
|
|
475
|
+
throw new ValidationError("Map limit must be greater than 0");
|
|
476
|
+
}
|
|
477
|
+
// Firecrawl /v2/map supports only url, limit, and search(instructions).
|
|
478
|
+
// depth/breadth/selectPaths/excludePaths have no native map equivalent
|
|
479
|
+
// (Tavily map supports them) — reject rather than silently discard.
|
|
480
|
+
if (request.depth !== undefined) {
|
|
481
|
+
throw new UnsupportedOptionError("firecrawl", "map", "depth");
|
|
482
|
+
}
|
|
483
|
+
if (request.breadth !== undefined) {
|
|
484
|
+
throw new UnsupportedOptionError("firecrawl", "map", "breadth");
|
|
485
|
+
}
|
|
486
|
+
if (request.selectPaths !== undefined) {
|
|
487
|
+
throw new UnsupportedOptionError("firecrawl", "map", "selectPaths");
|
|
488
|
+
}
|
|
489
|
+
if (request.excludePaths !== undefined) {
|
|
490
|
+
throw new UnsupportedOptionError("firecrawl", "map", "excludePaths");
|
|
491
|
+
}
|
|
492
|
+
},
|
|
493
|
+
cacheIdentity(request) {
|
|
494
|
+
const apiKey = resolveApiKey(env);
|
|
495
|
+
return {
|
|
496
|
+
provider: "firecrawl",
|
|
497
|
+
capability: "map",
|
|
498
|
+
credentialFingerprint: credentialFingerprint(apiKey),
|
|
499
|
+
request,
|
|
500
|
+
};
|
|
501
|
+
},
|
|
502
|
+
decodeCached(value) {
|
|
503
|
+
return decodeMapResult(value);
|
|
504
|
+
},
|
|
505
|
+
async invoke(request) {
|
|
506
|
+
fetch.validate(request);
|
|
507
|
+
const apiKey = resolveApiKey(env);
|
|
508
|
+
try {
|
|
509
|
+
const params = mapMapControls(request);
|
|
510
|
+
const raw = await fetchFirecrawlMap(apiKey, request.url, params, transport);
|
|
511
|
+
return normalizeFirecrawlMapResult(raw, request);
|
|
512
|
+
}
|
|
513
|
+
catch (error) {
|
|
514
|
+
throw normalizeFirecrawlError(error);
|
|
515
|
+
}
|
|
516
|
+
},
|
|
517
|
+
};
|
|
518
|
+
return { fetch };
|
|
519
|
+
}
|
|
520
|
+
// ---------------------------------------------------------------------------
|
|
521
|
+
// Crawl Capability — async create→poll→resume (tech-plan D2)
|
|
522
|
+
// ---------------------------------------------------------------------------
|
|
523
|
+
const DEFAULT_CRAWL_POLL_INTERVAL_MS = 2000;
|
|
524
|
+
/** Safety bound on `next`-cursor traversal for very large crawls. */
|
|
525
|
+
const MAX_CRAWL_NEXT_ITERATIONS = 500;
|
|
526
|
+
/** Reclaim-on-miss staleness guard — don't adopt jobs older than this. */
|
|
527
|
+
const CRAWL_RECLAIM_STALE_MS = 24 * 60 * 60 * 1000;
|
|
528
|
+
/** Split a comma-separated path-pattern string into a trimmed array. */
|
|
529
|
+
function splitPathPatterns(value) {
|
|
530
|
+
if (value === undefined || value.trim() === "")
|
|
531
|
+
return undefined;
|
|
532
|
+
return value
|
|
533
|
+
.split(",")
|
|
534
|
+
.map((s) => s.trim())
|
|
535
|
+
.filter((s) => s.length > 0);
|
|
536
|
+
}
|
|
537
|
+
function isEexistError(err) {
|
|
538
|
+
return (typeof err === "object" &&
|
|
539
|
+
err !== null &&
|
|
540
|
+
"code" in err &&
|
|
541
|
+
err.code === "EEXIST");
|
|
542
|
+
}
|
|
543
|
+
function resolveCrawlPollIntervalMs(env) {
|
|
544
|
+
const raw = env?.FIRECRAWL_CRAWL_POLL_INTERVAL_MS;
|
|
545
|
+
const parsed = parseInt(raw ?? "", 10);
|
|
546
|
+
return Number.isFinite(parsed) && parsed >= 0 ? parsed : DEFAULT_CRAWL_POLL_INTERVAL_MS;
|
|
547
|
+
}
|
|
548
|
+
/**
|
|
549
|
+
* Abortable sleep built from the injected timers. When `signal` aborts
|
|
550
|
+
* (the command handler's `--timeout`), the pending timer is cleared and
|
|
551
|
+
* the promise rejects with a `TimeoutError` so the poll loop unwinds
|
|
552
|
+
* promptly. Mirrors the research poll loop's `makeSleep`.
|
|
553
|
+
*/
|
|
554
|
+
function makeCrawlSleep(deps, signal) {
|
|
555
|
+
const setT = deps?.setTimeout ?? setTimeout;
|
|
556
|
+
const clearT = deps?.clearTimeout ?? clearTimeout;
|
|
557
|
+
return (ms) => new Promise((resolve, reject) => {
|
|
558
|
+
if (signal?.aborted) {
|
|
559
|
+
reject(new TimeoutError(0, "Crawl polling aborted"));
|
|
560
|
+
return;
|
|
561
|
+
}
|
|
562
|
+
if (ms <= 0) {
|
|
563
|
+
setImmediate(() => {
|
|
564
|
+
if (signal?.aborted) {
|
|
565
|
+
reject(new TimeoutError(0, "Crawl polling aborted"));
|
|
566
|
+
return;
|
|
567
|
+
}
|
|
568
|
+
resolve();
|
|
569
|
+
});
|
|
570
|
+
return;
|
|
571
|
+
}
|
|
572
|
+
const onAbort = () => {
|
|
573
|
+
clearT(id);
|
|
574
|
+
reject(new TimeoutError(0, "Crawl polling aborted"));
|
|
575
|
+
};
|
|
576
|
+
const id = setT(() => {
|
|
577
|
+
signal?.removeEventListener("abort", onAbort);
|
|
578
|
+
resolve();
|
|
579
|
+
}, ms);
|
|
580
|
+
signal?.addEventListener("abort", onAbort);
|
|
581
|
+
});
|
|
582
|
+
}
|
|
583
|
+
/**
|
|
584
|
+
* Map a Provider-neutral `CrawlRequest` into Firecrawl-native /v2/crawl
|
|
585
|
+
* body fields. `breadth` has no Firecrawl equivalent (rejected in
|
|
586
|
+
* `validate`); `proxy` is pinned to `"basic"` (D9 cost-safety) and nests
|
|
587
|
+
* under `scrapeOptions` (crawl nests scrape fields; `/scrape` takes
|
|
588
|
+
* `proxy` top-level). `format` nests under `scrapeOptions.formats`.
|
|
589
|
+
*/
|
|
590
|
+
function mapCrawlControls(request) {
|
|
591
|
+
const contentFormat = request.format ?? "markdown";
|
|
592
|
+
const params = { scrapeOptions: { formats: [contentFormat], proxy: "basic" } };
|
|
593
|
+
if (request.depth !== undefined)
|
|
594
|
+
params.maxDepth = request.depth;
|
|
595
|
+
if (request.limit !== undefined)
|
|
596
|
+
params.limit = request.limit;
|
|
597
|
+
const includePaths = splitPathPatterns(request.selectPaths);
|
|
598
|
+
if (includePaths !== undefined)
|
|
599
|
+
params.includePaths = includePaths;
|
|
600
|
+
const excludePaths = splitPathPatterns(request.excludePaths);
|
|
601
|
+
if (excludePaths !== undefined)
|
|
602
|
+
params.excludePaths = excludePaths;
|
|
603
|
+
return params;
|
|
604
|
+
}
|
|
605
|
+
/**
|
|
606
|
+
* Normalize a batch of Firecrawl crawl page objects into `CrawlPage[]`.
|
|
607
|
+
*
|
|
608
|
+
* data[].metadata.sourceURL -> url
|
|
609
|
+
* data[].markdown|text -> content
|
|
610
|
+
*
|
|
611
|
+
* Any malformed entry is a retryable `ApiError` 500.
|
|
612
|
+
*/
|
|
613
|
+
function normalizeCrawlPages(data, request) {
|
|
614
|
+
const contentFormat = request.format ?? "markdown";
|
|
615
|
+
const contentField = contentFormat === "text" ? "text" : "markdown";
|
|
616
|
+
const pages = [];
|
|
617
|
+
for (const entry of data) {
|
|
618
|
+
if (!isPlainObject(entry)) {
|
|
619
|
+
throw new ApiError("Firecrawl crawl returned a malformed response", 500);
|
|
620
|
+
}
|
|
621
|
+
const metadata = isPlainObject(entry.metadata) ? entry.metadata : {};
|
|
622
|
+
const url = metadata.sourceURL;
|
|
623
|
+
const content = entry[contentField];
|
|
624
|
+
if (typeof url !== "string" ||
|
|
625
|
+
url.length === 0 ||
|
|
626
|
+
typeof content !== "string" ||
|
|
627
|
+
content.length === 0) {
|
|
628
|
+
throw new ApiError("Firecrawl crawl returned a malformed response", 500);
|
|
629
|
+
}
|
|
630
|
+
pages.push({ url, content, contentFormat });
|
|
631
|
+
}
|
|
632
|
+
return pages;
|
|
633
|
+
}
|
|
634
|
+
/**
|
|
635
|
+
* Collect the full crawl result from a completed poll, following the
|
|
636
|
+
* pagination cursor `next` to exhaustion for large sets (each batch
|
|
637
|
+
* distinct — no dedup needed; tech-plan D2 / G0 #1). Bounded by a
|
|
638
|
+
* max-iteration guard against a runaway cursor.
|
|
639
|
+
*/
|
|
640
|
+
async function collectCrawlResult(poll, apiKey, request, transport) {
|
|
641
|
+
const pages = [];
|
|
642
|
+
if (poll.data)
|
|
643
|
+
pages.push(...normalizeCrawlPages(poll.data, request));
|
|
644
|
+
let next = poll.next;
|
|
645
|
+
let guard = 0;
|
|
646
|
+
while (next !== undefined) {
|
|
647
|
+
guard += 1;
|
|
648
|
+
if (guard > MAX_CRAWL_NEXT_ITERATIONS) {
|
|
649
|
+
throw new ApiError("Firecrawl crawl pagination exceeded the safety limit", 500);
|
|
650
|
+
}
|
|
651
|
+
const page = await fetchFirecrawlCrawlNext(apiKey, next, transport);
|
|
652
|
+
if (page.data)
|
|
653
|
+
pages.push(...normalizeCrawlPages(page.data, request));
|
|
654
|
+
next = page.next;
|
|
655
|
+
}
|
|
656
|
+
return { schemaVersion: 1, baseUrl: request.url, pages, totalPages: pages.length };
|
|
657
|
+
}
|
|
658
|
+
/**
|
|
659
|
+
* Unordered array equality for path-filter matching. `undefined` expected
|
|
660
|
+
* means "no constraint requested" (always compatible).
|
|
661
|
+
*/
|
|
662
|
+
function stringArrayEqualUnordered(a, expected) {
|
|
663
|
+
if (expected === undefined)
|
|
664
|
+
return true;
|
|
665
|
+
if (!Array.isArray(a))
|
|
666
|
+
return false;
|
|
667
|
+
return JSON.stringify([...a].sort()) === JSON.stringify([...expected].sort());
|
|
668
|
+
}
|
|
669
|
+
/**
|
|
670
|
+
* Best-effort compatibility check between an active-job `options` blob and
|
|
671
|
+
* the params this request would send. The server echoes the request
|
|
672
|
+
* options; verifying the cost-bearing fields (limit, maxDepth, path
|
|
673
|
+
* filters, scrapeOptions.formats) is a strong signal it is the same job.
|
|
674
|
+
* Missing or differently-shaped options → not compatible (safer to create
|
|
675
|
+
* fresh than to mis-adopt a different crawl and return the wrong pages).
|
|
676
|
+
*/
|
|
677
|
+
function crawlOptionsCompatible(options, params) {
|
|
678
|
+
if (!isPlainObject(options))
|
|
679
|
+
return false;
|
|
680
|
+
if (params.maxDepth !== undefined && options.maxDepth !== params.maxDepth)
|
|
681
|
+
return false;
|
|
682
|
+
if (params.limit !== undefined && options.limit !== params.limit)
|
|
683
|
+
return false;
|
|
684
|
+
const expectedFormats = params.scrapeOptions?.formats;
|
|
685
|
+
const so = options.scrapeOptions;
|
|
686
|
+
if (expectedFormats !== undefined && isPlainObject(so) && Array.isArray(so.formats)) {
|
|
687
|
+
const got = JSON.stringify([...so.formats].sort());
|
|
688
|
+
const want = JSON.stringify([...expectedFormats].sort());
|
|
689
|
+
if (got !== want)
|
|
690
|
+
return false;
|
|
691
|
+
}
|
|
692
|
+
// Path filters determine which pages are crawled (cost AND result
|
|
693
|
+
// content) — a mismatched filter would adopt a differently-scoped crawl.
|
|
694
|
+
if (!stringArrayEqualUnordered(options.includePaths, params.includePaths))
|
|
695
|
+
return false;
|
|
696
|
+
if (!stringArrayEqualUnordered(options.excludePaths, params.excludePaths))
|
|
697
|
+
return false;
|
|
698
|
+
return true;
|
|
699
|
+
}
|
|
700
|
+
/**
|
|
701
|
+
* Reclaim-on-miss: find an in-flight job matching this request by `url`
|
|
702
|
+
* (with a `created_at` recency guard against adopting stale jobs, and a
|
|
703
|
+
* best-effort options check when the server echoes them). A missing or
|
|
704
|
+
* unparseable `created_at` is treated as stale (skipped) so a zombie entry
|
|
705
|
+
* is never adopted. Returns the job id, or `undefined` when no match exists.
|
|
706
|
+
*/
|
|
707
|
+
function matchActiveCrawl(active, request, params) {
|
|
708
|
+
const now = Date.now();
|
|
709
|
+
for (const entry of active) {
|
|
710
|
+
if (entry.url !== request.url)
|
|
711
|
+
continue;
|
|
712
|
+
const ts = entry.created_at !== undefined ? Date.parse(entry.created_at) : NaN;
|
|
713
|
+
if (!Number.isFinite(ts) || now - ts > CRAWL_RECLAIM_STALE_MS)
|
|
714
|
+
continue;
|
|
715
|
+
if (entry.options !== undefined && !crawlOptionsCompatible(entry.options, params))
|
|
716
|
+
continue;
|
|
717
|
+
return entry.id;
|
|
718
|
+
}
|
|
719
|
+
return undefined;
|
|
720
|
+
}
|
|
721
|
+
/**
|
|
722
|
+
* Persist a crawl job id in the state file (atomic `wx` create). On EEXIST
|
|
723
|
+
* (a concurrent invocation already persisted a job for this request), read
|
|
724
|
+
* and return its id instead — the concurrent job is the one to poll.
|
|
725
|
+
*/
|
|
726
|
+
async function persistCrawlId(stateFile, identityHash, id) {
|
|
727
|
+
const state = {
|
|
728
|
+
requestId: id,
|
|
729
|
+
identityHash,
|
|
730
|
+
createdAt: new Date().toISOString(),
|
|
731
|
+
status: "pending",
|
|
732
|
+
};
|
|
733
|
+
try {
|
|
734
|
+
await stateFile.write(identityHash, state);
|
|
735
|
+
return id;
|
|
736
|
+
}
|
|
737
|
+
catch (err) {
|
|
738
|
+
if (isEexistError(err)) {
|
|
739
|
+
const existing = await stateFile.read(identityHash);
|
|
740
|
+
return existing !== null ? existing.requestId : id;
|
|
741
|
+
}
|
|
742
|
+
throw err;
|
|
743
|
+
}
|
|
744
|
+
}
|
|
745
|
+
// ---------------------------------------------------------------------------
|
|
746
|
+
// Concurrent-create lock (cost-safety — review C3)
|
|
747
|
+
// ---------------------------------------------------------------------------
|
|
748
|
+
/** How long to wait for a contended crawl create-lock before giving up. */
|
|
749
|
+
const CRAWL_LOCK_TIMEOUT_MS = 30000;
|
|
750
|
+
/** A lock older than this is treated as stale (holder died) and broken. */
|
|
751
|
+
const CRAWL_LOCK_STALE_MS = 10 * 60 * 1000;
|
|
752
|
+
/**
|
|
753
|
+
* Serialize the listActive→create→persist critical section per request so two
|
|
754
|
+
* concurrent identical crawls can't both POST (and charge) a job — the second
|
|
755
|
+
* waits, then reclaims the first's job from `/active` instead of re-POSTing.
|
|
756
|
+
* Uses an exclusive `wx`-create lockfile sibling to the state file; a stale
|
|
757
|
+
* lock (holder crashed) is broken after {@link CRAWL_LOCK_STALE_MS}.
|
|
758
|
+
*
|
|
759
|
+
* When `stateDir` is undefined (in-memory test mode), the lock is a no-op —
|
|
760
|
+
* tests are single-process and need no cross-process serialization.
|
|
761
|
+
*/
|
|
762
|
+
async function withCrawlLock(stateDir, identityHash, fn, deps) {
|
|
763
|
+
if (stateDir === undefined)
|
|
764
|
+
return fn();
|
|
765
|
+
const setT = deps?.setTimeout ?? setTimeout;
|
|
766
|
+
const sleep = (ms) => new Promise((r) => setT(() => r(), ms));
|
|
767
|
+
await fs.mkdir(stateDir, { recursive: true }).catch(() => { });
|
|
768
|
+
const lockPath = path.join(stateDir, `${identityHash}.lock`);
|
|
769
|
+
const deadline = Date.now() + CRAWL_LOCK_TIMEOUT_MS;
|
|
770
|
+
for (;;) {
|
|
771
|
+
try {
|
|
772
|
+
const handle = await fs.open(lockPath, "wx");
|
|
773
|
+
try {
|
|
774
|
+
return await fn();
|
|
775
|
+
}
|
|
776
|
+
finally {
|
|
777
|
+
await handle.close().catch(() => { });
|
|
778
|
+
await fs.unlink(lockPath).catch(() => { });
|
|
779
|
+
}
|
|
780
|
+
}
|
|
781
|
+
catch (err) {
|
|
782
|
+
if (!isEexistError(err))
|
|
783
|
+
throw err;
|
|
784
|
+
if (Date.now() > deadline) {
|
|
785
|
+
throw new ApiError("Firecrawl crawl create-lock timed out", 500);
|
|
786
|
+
}
|
|
787
|
+
// Break a stale lock (the holder died without releasing).
|
|
788
|
+
const stat = await fs.stat(lockPath).catch(() => null);
|
|
789
|
+
if (stat && Date.now() - stat.mtimeMs > CRAWL_LOCK_STALE_MS) {
|
|
790
|
+
await fs.unlink(lockPath).catch(() => { });
|
|
791
|
+
continue;
|
|
792
|
+
}
|
|
793
|
+
await sleep(500);
|
|
794
|
+
}
|
|
795
|
+
}
|
|
796
|
+
}
|
|
797
|
+
/**
|
|
798
|
+
* Reclaim an in-flight job whose create-POST response was lost, or create a
|
|
799
|
+
* fresh one. On a state-file miss, GET /v2/crawl/active and adopt a matching
|
|
800
|
+
* job; else POST /v2/crawl and persist the id. The whole listActive→create→
|
|
801
|
+
* persist sequence runs under {@link withCrawlLock} so concurrent identical
|
|
802
|
+
* invocations serialize (the second reclaims the first's job instead of
|
|
803
|
+
* re-POSTing). The create POST is zero-retry at the shared-execution layer.
|
|
804
|
+
*/
|
|
805
|
+
async function reclaimOrCreateCrawlJob(apiKey, request, identityHash, params, stateFile, transport, stateDir) {
|
|
806
|
+
return withCrawlLock(stateDir, identityHash, async () => {
|
|
807
|
+
const active = await listActiveFirecrawlCrawls(apiKey, transport);
|
|
808
|
+
const matched = matchActiveCrawl(active, request, params);
|
|
809
|
+
if (matched !== undefined) {
|
|
810
|
+
return persistCrawlId(stateFile, identityHash, matched);
|
|
811
|
+
}
|
|
812
|
+
const created = await createFirecrawlCrawl(apiKey, request.url, params, transport);
|
|
813
|
+
return persistCrawlId(stateFile, identityHash, created.id);
|
|
814
|
+
}, transport);
|
|
815
|
+
}
|
|
816
|
+
function createFirecrawlCrawlCapability(options) {
|
|
817
|
+
const { env, transport, stateFile, stateDir } = options;
|
|
818
|
+
const fetch = {
|
|
819
|
+
kind: "crawl-fetch",
|
|
820
|
+
validate(request) {
|
|
821
|
+
assertHttpUrl(request.url);
|
|
822
|
+
// breadth has no Firecrawl equivalent (D9) — reject, don't ignore.
|
|
823
|
+
if (request.breadth !== undefined) {
|
|
824
|
+
throw new UnsupportedOptionError("firecrawl", "crawl", "breadth");
|
|
825
|
+
}
|
|
826
|
+
if (request.depth !== undefined) {
|
|
827
|
+
if (!Number.isInteger(request.depth) || request.depth < 1 || request.depth > 5) {
|
|
828
|
+
throw new ValidationError("Crawl depth must be an integer between 1 and 5");
|
|
829
|
+
}
|
|
830
|
+
}
|
|
831
|
+
if (request.limit !== undefined && request.limit <= 0) {
|
|
832
|
+
throw new ValidationError("Crawl limit must be greater than 0");
|
|
833
|
+
}
|
|
834
|
+
},
|
|
835
|
+
cacheIdentity(request) {
|
|
836
|
+
const apiKey = resolveApiKey(env);
|
|
837
|
+
return {
|
|
838
|
+
provider: "firecrawl",
|
|
839
|
+
capability: "crawl",
|
|
840
|
+
credentialFingerprint: credentialFingerprint(apiKey),
|
|
841
|
+
request,
|
|
842
|
+
};
|
|
843
|
+
},
|
|
844
|
+
decodeCached(value) {
|
|
845
|
+
return decodeCrawlResult(value);
|
|
846
|
+
},
|
|
847
|
+
async invoke(request, signal) {
|
|
848
|
+
fetch.validate(request);
|
|
849
|
+
const apiKey = resolveApiKey(env);
|
|
850
|
+
const credFingerprint = credentialFingerprint(apiKey);
|
|
851
|
+
const identityHash = computeAsyncJobStateHash({
|
|
852
|
+
provider: "firecrawl",
|
|
853
|
+
capability: "crawl",
|
|
854
|
+
credentialFingerprint: credFingerprint,
|
|
855
|
+
request,
|
|
856
|
+
});
|
|
857
|
+
const params = mapCrawlControls(request);
|
|
858
|
+
const pollIntervalMs = resolveCrawlPollIntervalMs(env);
|
|
859
|
+
const sleep = makeCrawlSleep(transport, signal);
|
|
860
|
+
try {
|
|
861
|
+
// 1. Resume: an in-flight job for this request was already persisted
|
|
862
|
+
// (Ctrl-C / crash mid-poll). Poll it instead of creating a second
|
|
863
|
+
// one (double-charge prevention).
|
|
864
|
+
const existing = await stateFile.read(identityHash);
|
|
865
|
+
let id;
|
|
866
|
+
if (existing !== null) {
|
|
867
|
+
id = existing.requestId;
|
|
868
|
+
}
|
|
869
|
+
else {
|
|
870
|
+
// 2. Reclaim-on-miss or create (zero-retry create; wx-flag write).
|
|
871
|
+
id = await reclaimOrCreateCrawlJob(apiKey, request, identityHash, params, stateFile, transport, stateDir);
|
|
872
|
+
}
|
|
873
|
+
// 3. Poll status to terminal, then collect the result once. We do
|
|
874
|
+
// NOT accumulate data[] across pre-completion polls — the
|
|
875
|
+
// completed response carries the full set (G0 #1).
|
|
876
|
+
let recreated = false;
|
|
877
|
+
for (;;) {
|
|
878
|
+
if (signal?.aborted) {
|
|
879
|
+
throw new TimeoutError(0, "Crawl polling aborted");
|
|
880
|
+
}
|
|
881
|
+
const poll = await pollFirecrawlCrawl(apiKey, id, transport);
|
|
882
|
+
if (poll.status === "completed") {
|
|
883
|
+
// Collect FIRST, remove only on success. If collection throws
|
|
884
|
+
// (a malformed page, a pagination/network blip), the state file
|
|
885
|
+
// stays so a re-run re-polls this already-completed (and paid-
|
|
886
|
+
// for) job instead of POSTing a second, doubly-charging one.
|
|
887
|
+
const result = await collectCrawlResult(poll, apiKey, request, transport);
|
|
888
|
+
await stateFile.remove(identityHash);
|
|
889
|
+
return result;
|
|
890
|
+
}
|
|
891
|
+
if (poll.status === "failed") {
|
|
892
|
+
await stateFile.remove(identityHash);
|
|
893
|
+
throw new ApiError("Firecrawl crawl job failed", 500);
|
|
894
|
+
}
|
|
895
|
+
if (poll.status === "not_found") {
|
|
896
|
+
// Server-side job disappeared — drop state and create fresh,
|
|
897
|
+
// but at most once. A pathological server that 404s every
|
|
898
|
+
// freshly-created job would otherwise loop and re-POST
|
|
899
|
+
// indefinitely (each POST charges credits).
|
|
900
|
+
if (recreated) {
|
|
901
|
+
throw new ApiError("Firecrawl crawl job could not be found after recreate", 500);
|
|
902
|
+
}
|
|
903
|
+
recreated = true;
|
|
904
|
+
await stateFile.remove(identityHash);
|
|
905
|
+
id = await reclaimOrCreateCrawlJob(apiKey, request, identityHash, params, stateFile, transport, stateDir);
|
|
906
|
+
continue;
|
|
907
|
+
}
|
|
908
|
+
// scraping: sleep and poll again. The state file already holds
|
|
909
|
+
// the id (the load-bearing field for resume).
|
|
910
|
+
await sleep(pollIntervalMs);
|
|
911
|
+
}
|
|
912
|
+
}
|
|
913
|
+
catch (error) {
|
|
914
|
+
throw normalizeFirecrawlError(error);
|
|
915
|
+
}
|
|
916
|
+
},
|
|
917
|
+
};
|
|
918
|
+
return { fetch };
|
|
919
|
+
}
|
|
920
|
+
/**
|
|
921
|
+
* Build the Firecrawl Provider Descriptor. Advertises the full Firecrawl
|
|
922
|
+
* capability set. `create()` wires all six capabilities — Search, Reader,
|
|
923
|
+
* Map, async Crawl (create→poll→resume + reclaim-on-miss), Quota
|
|
924
|
+
* (`unit:"credits"`), and Diagnostics (single-scrape probe). Construction
|
|
925
|
+
* is side-effect-free; the transport is invoked per Capability call.
|
|
926
|
+
*/
|
|
927
|
+
export function createFirecrawlDescriptor(dependencies) {
|
|
928
|
+
const transport = dependencies?.transport;
|
|
929
|
+
const crawlStateDir = dependencies?.crawlStateDir ?? asyncJobStateDir("crawl");
|
|
930
|
+
const crawlStateFile = dependencies?.crawlStateFile ?? createProductionAsyncJobStateFile(crawlStateDir);
|
|
931
|
+
return {
|
|
932
|
+
id: "firecrawl",
|
|
933
|
+
isConfigured(env) {
|
|
934
|
+
return isFirecrawlConfigured(env);
|
|
935
|
+
},
|
|
936
|
+
capabilities() {
|
|
937
|
+
return new Set([
|
|
938
|
+
"search",
|
|
939
|
+
"reader",
|
|
940
|
+
"crawl",
|
|
941
|
+
"map",
|
|
942
|
+
"quota",
|
|
943
|
+
"diagnostics",
|
|
944
|
+
]);
|
|
945
|
+
},
|
|
946
|
+
create(context) {
|
|
947
|
+
const search = createFirecrawlSearchCapability({ env: context.env, transport });
|
|
948
|
+
const reader = createFirecrawlReaderCapability({ env: context.env, transport });
|
|
949
|
+
const crawl = createFirecrawlCrawlCapability({
|
|
950
|
+
env: context.env,
|
|
951
|
+
transport,
|
|
952
|
+
stateFile: crawlStateFile,
|
|
953
|
+
stateDir: crawlStateDir,
|
|
954
|
+
});
|
|
955
|
+
const map = createFirecrawlMapCapability({ env: context.env, transport });
|
|
956
|
+
const quota = createFirecrawlQuotaCapability({ env: context.env, transport });
|
|
957
|
+
const diagnostics = createFirecrawlDiagnosticsCapability({ env: context.env, transport });
|
|
958
|
+
return { id: "firecrawl", search, reader, crawl, map, quota, diagnostics };
|
|
959
|
+
},
|
|
960
|
+
};
|
|
961
|
+
}
|
|
962
|
+
//# sourceMappingURL=adapter.js.map
|