scoutline 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/README.md +11 -3
  2. package/dist/capabilities/quota.d.ts +11 -0
  3. package/dist/capabilities/quota.d.ts.map +1 -1
  4. package/dist/capabilities/quota.js.map +1 -1
  5. package/dist/capabilities/search.d.ts +11 -1
  6. package/dist/capabilities/search.d.ts.map +1 -1
  7. package/dist/commands/crawl.js +1 -1
  8. package/dist/commands/doctor.d.ts.map +1 -1
  9. package/dist/commands/doctor.js +13 -10
  10. package/dist/commands/doctor.js.map +1 -1
  11. package/dist/commands/map.js +1 -1
  12. package/dist/commands/quota.d.ts +23 -0
  13. package/dist/commands/quota.d.ts.map +1 -1
  14. package/dist/commands/quota.js +21 -4
  15. package/dist/commands/quota.js.map +1 -1
  16. package/dist/commands/read.js +1 -1
  17. package/dist/commands/repo.js +1 -1
  18. package/dist/commands/research.js +1 -1
  19. package/dist/commands/search.d.ts +2 -1
  20. package/dist/commands/search.d.ts.map +1 -1
  21. package/dist/commands/search.js +21 -12
  22. package/dist/commands/search.js.map +1 -1
  23. package/dist/index.d.ts +7 -0
  24. package/dist/index.d.ts.map +1 -1
  25. package/dist/index.js +35 -4
  26. package/dist/index.js.map +1 -1
  27. package/dist/lib/redact.d.ts +2 -2
  28. package/dist/lib/redact.d.ts.map +1 -1
  29. package/dist/lib/redact.js +8 -2
  30. package/dist/lib/redact.js.map +1 -1
  31. package/dist/providers/brave/adapter.d.ts +42 -0
  32. package/dist/providers/brave/adapter.d.ts.map +1 -0
  33. package/dist/providers/brave/adapter.js +504 -0
  34. package/dist/providers/brave/adapter.js.map +1 -0
  35. package/dist/providers/brave/client.d.ts +161 -0
  36. package/dist/providers/brave/client.d.ts.map +1 -0
  37. package/dist/providers/brave/client.js +297 -0
  38. package/dist/providers/brave/client.js.map +1 -0
  39. package/dist/providers/brave/credentials.d.ts +33 -0
  40. package/dist/providers/brave/credentials.d.ts.map +1 -0
  41. package/dist/providers/brave/credentials.js +55 -0
  42. package/dist/providers/brave/credentials.js.map +1 -0
  43. package/dist/providers/brave/diagnostics.d.ts +46 -0
  44. package/dist/providers/brave/diagnostics.d.ts.map +1 -0
  45. package/dist/providers/brave/diagnostics.js +76 -0
  46. package/dist/providers/brave/diagnostics.js.map +1 -0
  47. package/dist/providers/brave/quota.d.ts +86 -0
  48. package/dist/providers/brave/quota.d.ts.map +1 -0
  49. package/dist/providers/brave/quota.js +247 -0
  50. package/dist/providers/brave/quota.js.map +1 -0
  51. package/dist/providers/exa/adapter.d.ts +58 -0
  52. package/dist/providers/exa/adapter.d.ts.map +1 -0
  53. package/dist/providers/exa/adapter.js +892 -0
  54. package/dist/providers/exa/adapter.js.map +1 -0
  55. package/dist/providers/exa/client.d.ts +143 -0
  56. package/dist/providers/exa/client.d.ts.map +1 -0
  57. package/dist/providers/exa/client.js +328 -0
  58. package/dist/providers/exa/client.js.map +1 -0
  59. package/dist/providers/exa/credentials.d.ts +34 -0
  60. package/dist/providers/exa/credentials.d.ts.map +1 -0
  61. package/dist/providers/exa/credentials.js +56 -0
  62. package/dist/providers/exa/credentials.js.map +1 -0
  63. package/dist/providers/exa/diagnostics.d.ts +42 -0
  64. package/dist/providers/exa/diagnostics.d.ts.map +1 -0
  65. package/dist/providers/exa/diagnostics.js +69 -0
  66. package/dist/providers/exa/diagnostics.js.map +1 -0
  67. package/dist/providers/minimax/adapter.js +1 -1
  68. package/dist/providers/minimax/adapter.js.map +1 -1
  69. package/dist/providers/registry.d.ts.map +1 -1
  70. package/dist/providers/registry.js +4 -0
  71. package/dist/providers/registry.js.map +1 -1
  72. package/dist/providers/tavily/adapter.d.ts.map +1 -1
  73. package/dist/providers/tavily/adapter.js +5 -0
  74. package/dist/providers/tavily/adapter.js.map +1 -1
  75. package/dist/providers/types.d.ts +1 -1
  76. package/dist/providers/types.d.ts.map +1 -1
  77. package/dist/providers/types.js +1 -1
  78. package/dist/providers/types.js.map +1 -1
  79. package/dist/providers/zai/adapter.d.ts.map +1 -1
  80. package/dist/providers/zai/adapter.js +7 -1
  81. package/dist/providers/zai/adapter.js.map +1 -1
  82. package/package.json +1 -1
@@ -0,0 +1,892 @@
1
+ /**
2
+ * Exa Provider Adapter (tech-plan §7, Control Mapping, Field Normalization,
3
+ * Failure Normalization).
4
+ *
5
+ * Implements the Exa Provider Descriptor with Search, Reader, Research,
6
+ * and Diagnostics capabilities on top of the direct-HTTP transport
7
+ * (`./client.ts`). The Adapter owns credentials, transport lifecycle,
8
+ * Provider field mapping, and failure normalization; shared execution
9
+ * owns cache and retry policy.
10
+ *
11
+ * Boundary rules (ARCHITECTURE.md §2):
12
+ * - May import capability types, normalized errors, Provider identity
13
+ * types, and the Adapter-local credential and transport Modules.
14
+ * - Must NOT import command presentation, output mode, or another
15
+ * Provider's Adapter.
16
+ *
17
+ * Field mapping (tech-plan §7 Exa mapping):
18
+ * Search results[].title -> title
19
+ * Search results[].url -> url
20
+ * Search results[].highlights[] -> summary (join with " ")
21
+ * Search results[].author -> source
22
+ * Search results[].publishedDate -> date
23
+ * Search results[].score -> (dropped)
24
+ *
25
+ * Control mapping (SearchControls → Exa-native API params):
26
+ * domain -> includeDomains: [domain]
27
+ * recency -> startPublishedDate (oneDay→now-1d ISO, etc.)
28
+ * contentSize -> type (medium/omitted→"auto", high→"deep")
29
+ * topic -> category (general→omit, news→"news",
30
+ * finance→"financial report")
31
+ * location -> REJECTED (UnsupportedOptionError)
32
+ */
33
+ import crypto from "node:crypto";
34
+ import { decodeReaderFetchResult } from "../../capabilities/reader.js";
35
+ import { decodeResearchResult } from "../../capabilities/research.js";
36
+ import { computeResearchStateHash, createProductionResearchStateFile, } from "../../lib/research-state.js";
37
+ import { ApiError, AuthError, ConfigurationError, NetworkError, QuotaError, TimeoutError, UnsupportedOptionError, ValidationError, } from "../../lib/errors.js";
38
+ import { requireExaApiKey, isExaConfigured } from "./credentials.js";
39
+ import { fetchExaSearch, fetchExaContents, createExaAgentRun, pollExaAgentRun, } from "./client.js";
40
+ import { createExaDiagnosticsCapability } from "./diagnostics.js";
41
+ // ---------------------------------------------------------------------------
42
+ // Provider-owned credential fingerprint
43
+ // ---------------------------------------------------------------------------
44
+ function credentialFingerprint(apiKey) {
45
+ return crypto.createHash("sha256").update(apiKey).digest("hex");
46
+ }
47
+ function resolveApiKey(env) {
48
+ return requireExaApiKey(env);
49
+ }
50
+ // ---------------------------------------------------------------------------
51
+ // Helpers
52
+ // ---------------------------------------------------------------------------
53
+ function isPlainObject(value) {
54
+ return typeof value === "object" && value !== null && !Array.isArray(value);
55
+ }
56
+ // ---------------------------------------------------------------------------
57
+ // Control mapping (SearchControls → Exa-native API params)
58
+ // ---------------------------------------------------------------------------
59
+ /**
60
+ * Map a recency filter to an Exa `startPublishedDate` (ISO 8601). Exa
61
+ * accepts a date lower bound; `noLimit` omits the filter. The cutoff is
62
+ * computed relative to `now` so the Adapter can produce a deterministic
63
+ * value in tests.
64
+ */
65
+ function mapRecencyToStartPublishedDate(recency, now) {
66
+ const MS_PER_DAY = 24 * 60 * 60 * 1000;
67
+ switch (recency) {
68
+ case "oneDay":
69
+ return new Date(now.getTime() - 1 * MS_PER_DAY).toISOString();
70
+ case "oneWeek":
71
+ return new Date(now.getTime() - 7 * MS_PER_DAY).toISOString();
72
+ case "oneMonth":
73
+ return new Date(now.getTime() - 30 * MS_PER_DAY).toISOString();
74
+ case "oneYear":
75
+ return new Date(now.getTime() - 365 * MS_PER_DAY).toISOString();
76
+ case "noLimit":
77
+ return undefined;
78
+ default:
79
+ return undefined;
80
+ }
81
+ }
82
+ /**
83
+ * Map a topic hint to an Exa `category`. Exa has no `general` category;
84
+ * general is omitted so the search is unscoped. The Tavily adapter
85
+ * passes `topic` natively; Exa remaps to `category`.
86
+ */
87
+ function mapTopicToCategory(topic) {
88
+ switch (topic) {
89
+ case "news":
90
+ return "news";
91
+ case "finance":
92
+ return "financial report";
93
+ case "general":
94
+ return undefined;
95
+ default:
96
+ return undefined;
97
+ }
98
+ }
99
+ /**
100
+ * Map Provider-neutral `SearchControls` to Exa-native search params.
101
+ * `now` defaults to the current time; tests pass a fixed date so the
102
+ * `startPublishedDate` cutoff is deterministic.
103
+ */
104
+ function mapSearchControls(controls, now = new Date()) {
105
+ if (!controls)
106
+ return undefined;
107
+ const params = {};
108
+ if (controls.domain) {
109
+ params.includeDomains = [controls.domain];
110
+ }
111
+ if (controls.recency) {
112
+ const start = mapRecencyToStartPublishedDate(controls.recency, now);
113
+ if (start)
114
+ params.startPublishedDate = start;
115
+ }
116
+ if (controls.contentSize) {
117
+ params.type = controls.contentSize === "high" ? "deep" : "auto";
118
+ }
119
+ if (controls.topic) {
120
+ const category = mapTopicToCategory(controls.topic);
121
+ if (category)
122
+ params.category = category;
123
+ }
124
+ return params;
125
+ }
126
+ // ---------------------------------------------------------------------------
127
+ // Response normalization
128
+ // ---------------------------------------------------------------------------
129
+ /**
130
+ * Normalize a raw Exa search response into `SearchSource[]`.
131
+ *
132
+ * results[].title -> title
133
+ * results[].url -> url
134
+ * results[].highlights[] -> summary (join with " "; empty→"")
135
+ * results[].author -> source
136
+ * results[].publishedDate -> date
137
+ * results[].score -> (dropped)
138
+ *
139
+ * `title` and `url` must be strings; any malformed shape is a retryable
140
+ * `ApiError` 500. `highlights` is optional — when missing or empty,
141
+ * `summary` is the empty string (the result is still returned with its
142
+ * title/url). `author` and `publishedDate` are optional; when present
143
+ * they must be strings.
144
+ */
145
+ function normalizeExaSearchResults(raw) {
146
+ if (!isPlainObject(raw)) {
147
+ throw new ApiError("Exa search returned a malformed response", 500);
148
+ }
149
+ const results = raw.results;
150
+ if (!Array.isArray(results)) {
151
+ throw new ApiError("Exa search returned a malformed response", 500);
152
+ }
153
+ const out = [];
154
+ for (const entry of results) {
155
+ if (!isPlainObject(entry)) {
156
+ throw new ApiError("Exa search returned a malformed response", 500);
157
+ }
158
+ const title = entry.title;
159
+ const url = entry.url;
160
+ if (typeof title !== "string" || typeof url !== "string") {
161
+ throw new ApiError("Exa search returned a malformed response", 500);
162
+ }
163
+ // highlights is an optional string array; join with " ".
164
+ const highlights = entry.highlights;
165
+ let summary;
166
+ if (highlights === undefined) {
167
+ summary = "";
168
+ }
169
+ else if (Array.isArray(highlights)) {
170
+ summary = highlights.filter((h) => typeof h === "string").join(" ");
171
+ }
172
+ else {
173
+ throw new ApiError("Exa search returned a malformed response", 500);
174
+ }
175
+ const source = { title, url, summary };
176
+ if (typeof entry.author === "string") {
177
+ source.source = entry.author;
178
+ }
179
+ if (typeof entry.publishedDate === "string") {
180
+ source.date = entry.publishedDate;
181
+ }
182
+ out.push(source);
183
+ }
184
+ return out;
185
+ }
186
+ // ---------------------------------------------------------------------------
187
+ // Failure normalization: stable public codes, no raw payloads (NFR-006)
188
+ // ---------------------------------------------------------------------------
189
+ /**
190
+ * Resolve a stable HTTP-style status code for retry classification.
191
+ * Explicit terminal client errors (400, 404, 410, 422) map to their
192
+ * real codes; transient failures map to a representative status in the
193
+ * retryable set (429 or any 5xx 500..599 inclusive). Unknown failures
194
+ * default to 500 (transient). When the caller already carries a numeric
195
+ * status (a typed ApiError), that status is honoured directly.
196
+ */
197
+ function inferStatusCode(lower, known) {
198
+ if (typeof known === "number" && Number.isFinite(known))
199
+ return known;
200
+ if (lower.includes("404") || lower.includes("not found"))
201
+ return 404;
202
+ if (lower.includes("400") || lower.includes("bad request"))
203
+ return 400;
204
+ if (lower.includes("410") || lower.includes("gone"))
205
+ return 410;
206
+ if (lower.includes("422") || lower.includes("unprocessable"))
207
+ return 422;
208
+ if (lower.includes("500") || lower.includes("internal"))
209
+ return 500;
210
+ if (lower.includes("502") || lower.includes("bad gateway"))
211
+ return 502;
212
+ if (lower.includes("503") || lower.includes("service unavailable"))
213
+ return 503;
214
+ if (lower.includes("504") || lower.includes("gateway timeout"))
215
+ return 504;
216
+ return 500;
217
+ }
218
+ /**
219
+ * Status-keyed outward message for rewrapped Exa ApiErrors. The rewrap
220
+ * does not echo upstream `error.message` — a future change embedding a
221
+ * raw Provider body in an ApiError message would leak through
222
+ * normalization, the cache, and stdout. Curated constants only.
223
+ */
224
+ function exaApiErrorMessage(statusCode) {
225
+ if (statusCode === 429)
226
+ return "Exa rate limit exceeded";
227
+ return "Exa request failed";
228
+ }
229
+ /**
230
+ * Normalize a Provider failure with sanitized messages. Raw response
231
+ * bodies never cross the adapter boundary. Mirrors the Tavily adapter's
232
+ * `normalizeTavilyError` pattern.
233
+ */
234
+ function normalizeExaError(error) {
235
+ // QuotaError pass-through — terminal retry guarantee preserved.
236
+ if (error instanceof QuotaError)
237
+ return error;
238
+ // Configuration/option/validation errors carry clean, human-authored
239
+ // messages and are safe to surface verbatim.
240
+ if (error instanceof ValidationError ||
241
+ error instanceof UnsupportedOptionError ||
242
+ error instanceof ConfigurationError) {
243
+ return error;
244
+ }
245
+ // Re-wrap typed transport errors with sanitized messages so a raw
246
+ // Provider response body embedded upstream never survives. Code +
247
+ // statusCode (retry signal) are preserved.
248
+ if (error instanceof AuthError) {
249
+ return new AuthError("Exa authentication failed", "EXA_API_KEY");
250
+ }
251
+ if (error instanceof NetworkError) {
252
+ return new NetworkError("Exa network error");
253
+ }
254
+ if (error instanceof TimeoutError) {
255
+ return new TimeoutError(error.durationMs, "Try again or increase timeout with EXA_TIMEOUT env var");
256
+ }
257
+ if (error instanceof ApiError) {
258
+ const statusCode = inferStatusCode("", error.statusCode);
259
+ return new ApiError(exaApiErrorMessage(statusCode), statusCode);
260
+ }
261
+ const message = error instanceof Error ? error.message : String(error);
262
+ const lower = message.toLowerCase();
263
+ if (lower.includes("401") ||
264
+ lower.includes("403") ||
265
+ lower.includes("unauthorized") ||
266
+ lower.includes("forbidden")) {
267
+ return new AuthError("Exa authentication failed");
268
+ }
269
+ if (lower.includes("timeout") || lower.includes("timed out") || lower.includes("etimedout")) {
270
+ return new TimeoutError(30000);
271
+ }
272
+ if (lower.includes("econnrefused") ||
273
+ lower.includes("econnreset") ||
274
+ lower.includes("network") ||
275
+ lower.includes("enotfound") ||
276
+ lower.includes("fetch failed")) {
277
+ return new NetworkError("Exa network error");
278
+ }
279
+ if (lower.includes("429") || lower.includes("rate limit")) {
280
+ return new ApiError("Exa rate limit exceeded", 429);
281
+ }
282
+ return new ApiError("Exa request failed", inferStatusCode(lower));
283
+ }
284
+ function createExaSearchCapability(options) {
285
+ const { env, transport } = options;
286
+ const capability = {
287
+ validate(request) {
288
+ if (!request || typeof request.query !== "string" || request.query.trim() === "") {
289
+ throw new ValidationError("Search query must contain at least one non-whitespace character");
290
+ }
291
+ // Exa supports domain, recency, contentSize, and topic natively.
292
+ // location is Z.AI-specific and rejected before any transport call.
293
+ if (request.controls?.location !== undefined) {
294
+ throw new UnsupportedOptionError("exa", "search", "location");
295
+ }
296
+ },
297
+ cacheIdentity(request) {
298
+ const apiKey = resolveApiKey(env);
299
+ const identityRequest = {
300
+ query: request.query,
301
+ };
302
+ if (request.controls) {
303
+ identityRequest.controls = request.controls;
304
+ }
305
+ return {
306
+ provider: "exa",
307
+ capability: "search",
308
+ credentialFingerprint: credentialFingerprint(apiKey),
309
+ request: identityRequest,
310
+ // Exa never probes legacy keys — no legacyCandidates.
311
+ };
312
+ },
313
+ async invoke(request) {
314
+ capability.validate(request);
315
+ const apiKey = resolveApiKey(env);
316
+ try {
317
+ const params = mapSearchControls(request.controls);
318
+ const raw = await fetchExaSearch(apiKey, request.query, params, transport);
319
+ return normalizeExaSearchResults(raw);
320
+ }
321
+ catch (error) {
322
+ throw normalizeExaError(error);
323
+ }
324
+ },
325
+ };
326
+ return capability;
327
+ }
328
+ // ---------------------------------------------------------------------------
329
+ // Reader validation helpers
330
+ // ---------------------------------------------------------------------------
331
+ /** Options the Exa Reader does NOT accept (Z.AI-only). `retainImages`
332
+ * is accepted but silently ignored (Exa has no equivalent param); the
333
+ * read command handler sends it as `true` by default. */
334
+ const UNSUPPORTED_READER_OPTIONS = [
335
+ "withLinksSummary",
336
+ "noGfm",
337
+ "keepImgDataUrl",
338
+ "withImagesSummary",
339
+ ];
340
+ function assertHttpUrl(url) {
341
+ if (typeof url !== "string" || url.length === 0) {
342
+ throw new ValidationError("Exa reader URL must be a non-empty string");
343
+ }
344
+ if (!/^https?:\/\//.test(url)) {
345
+ throw new ValidationError("URL must start with http:// or https://");
346
+ }
347
+ }
348
+ function assertNoUnsupportedReaderOptions(request) {
349
+ for (const key of UNSUPPORTED_READER_OPTIONS) {
350
+ if (request[key] === true) {
351
+ throw new UnsupportedOptionError("exa", "reader", key);
352
+ }
353
+ }
354
+ }
355
+ // ---------------------------------------------------------------------------
356
+ // Markdown stripping (format: text — best-effort)
357
+ // ---------------------------------------------------------------------------
358
+ /**
359
+ * Best-effort markdown-to-text conversion. Exa always returns text
360
+ * content; when the caller requests `format: "text"`, this strips
361
+ * common markdown markers so the output is closer to plain text.
362
+ * Code blocks and tables degrade (their content is kept but
363
+ * formatting is lost); that is an acceptable edge case for a "rough
364
+ * text" mode. This is adapter-local — no shared stripper exists.
365
+ */
366
+ function stripMarkdown(input) {
367
+ let result = input;
368
+ // Remove fenced code blocks (keep the inner text).
369
+ result = result.replace(/```[\s\S]*?```/g, (block) => block.replace(/```\w*\n?/g, "").replace(/```$/g, ""));
370
+ // Remove inline code backticks.
371
+ result = result.replace(/`([^`]+)`/g, "$1");
372
+ // Images: ![alt](url) → alt.
373
+ result = result.replace(/!\[([^\]]*)\]\([^)]+\)/g, "$1");
374
+ // Links: [text](url) → text.
375
+ result = result.replace(/\[([^\]]*)\]\([^)]+\)/g, "$1");
376
+ // Headers: leading # markers.
377
+ result = result.replace(/^#{1,6}\s+/gm, "");
378
+ // Emphasis: **bold**, __bold__, *italic*, _italic_, ~~strike~~.
379
+ result = result.replace(/\*\*(.+?)\*\*/g, "$1");
380
+ result = result.replace(/__(.+?)__/g, "$1");
381
+ result = result.replace(/\*(.+?)\*/g, "$1");
382
+ result = result.replace(/_(.+?)_/g, "$1");
383
+ result = result.replace(/~~(.+?)~~/g, "$1");
384
+ // Blockquotes: leading > markers.
385
+ result = result.replace(/^>\s+/gm, "");
386
+ // Horizontal rules: ---, ***, ___.
387
+ result = result.replace(/^[-*_]{3,}\s*$/gm, "");
388
+ // List markers: -, *, +, 1.
389
+ result = result.replace(/^[\s]*[-*+]\s+/gm, "");
390
+ result = result.replace(/^[\s]*\d+\.\s+/gm, "");
391
+ return result.trim();
392
+ }
393
+ // ---------------------------------------------------------------------------
394
+ // Per-URL status total function (the load-bearing mechanism)
395
+ // ---------------------------------------------------------------------------
396
+ /**
397
+ * Known error-tag → HTTP status mapping. When `error.httpStatusCode` is
398
+ * present on the response, it is preferred over this table for retry
399
+ * classification. Unknown tags default to 500 (retryable).
400
+ */
401
+ const CONTENTS_ERROR_STATUS = {
402
+ CRAWL_NOT_FOUND: 404,
403
+ UNSUPPORTED_URL: 400,
404
+ SOURCE_NOT_AVAILABLE: 403,
405
+ CRAWL_TIMEOUT: 504,
406
+ CRAWL_LIVECRAWL_TIMEOUT: 504,
407
+ CRAWL_UNKNOWN_ERROR: 500,
408
+ };
409
+ /**
410
+ * Normalize a raw Exa `/contents` response into a `ReaderFetchResult`.
411
+ *
412
+ * **Critical mechanism — per-URL status inspection (total function).**
413
+ * `/contents` returns HTTP 200 even when an individual URL fails. The
414
+ * adapter MUST resolve the status entry whose `statuses[].id` matches
415
+ * the requested URL, then apply a total mapping. Match by `id`, never
416
+ * assume `results[0]` is the requested URL.
417
+ *
418
+ * Field mapping:
419
+ * results[].text -> content (the entry whose id/status matches)
420
+ * results[].url -> finalUrl
421
+ * results[].title -> title (coerce blank → null)
422
+ * request.url -> url
423
+ * request.format -> contentFormat (after any text stripping)
424
+ *
425
+ * On `format: "text"`, the content is run through {@link stripMarkdown}
426
+ * (best-effort) and `contentFormat` is set to `"text"`.
427
+ */
428
+ function normalizeExaContentsResult(raw, request) {
429
+ if (!isPlainObject(raw)) {
430
+ throw new ApiError("Exa contents returned a malformed response", 500);
431
+ }
432
+ // Step 1: find the status entry matching the requested URL. Never
433
+ // assume results[0] is the match — the API returns HTTP 200 even on
434
+ // per-URL failure. For a single-URL fetch, accept the sole entry
435
+ // even if its id doesn't exactly match (Exa may normalize URLs).
436
+ const statuses = raw.statuses;
437
+ if (!Array.isArray(statuses)) {
438
+ throw new ApiError("Exa contents returned a malformed response", 500);
439
+ }
440
+ let statusEntry = statuses.find((s) => isPlainObject(s) && typeof s.id === "string" && s.id === request.url);
441
+ // Single-URL fallback: if no exact id match but exactly one status
442
+ // entry exists, accept it. This mirrors the results[] leniency and
443
+ // guards against URL normalization differences.
444
+ if (!statusEntry && statuses.length === 1 && isPlainObject(statuses[0])) {
445
+ statusEntry = statuses[0];
446
+ }
447
+ if (!statusEntry) {
448
+ throw new ApiError("Exa contents returned a malformed response", 500);
449
+ }
450
+ const statusValue = statusEntry.status;
451
+ // Step 2: error path — map the tag + httpStatusCode to a sanitized ApiError.
452
+ if (statusValue !== "success") {
453
+ const errorObj = isPlainObject(statusEntry.error) ? statusEntry.error : {};
454
+ const tag = typeof errorObj.tag === "string" ? errorObj.tag : undefined;
455
+ const httpStatusCode = typeof errorObj.httpStatusCode === "number" ? errorObj.httpStatusCode : undefined;
456
+ const statusCode = httpStatusCode ?? CONTENTS_ERROR_STATUS[tag ?? ""] ?? 500;
457
+ throw new ApiError("Exa contents request failed", statusCode);
458
+ }
459
+ // Step 3: success path — find the matching result in results[].
460
+ const results = raw.results;
461
+ if (!Array.isArray(results)) {
462
+ throw new ApiError("Exa contents returned a malformed response", 500);
463
+ }
464
+ // For a single-URL fetch, the result entry's id or url should match.
465
+ const result = results.find((r) => isPlainObject(r) &&
466
+ ((typeof r.id === "string" && r.id === request.url) ||
467
+ (typeof r.url === "string" && r.url === request.url)));
468
+ // Fall back to the first result if no URL match (single-URL fetch).
469
+ const entry = result ?? (results.length > 0 && isPlainObject(results[0]) ? results[0] : null);
470
+ if (!entry) {
471
+ throw new ApiError("Exa contents returned a malformed response", 500);
472
+ }
473
+ // Step 4: field mapping.
474
+ const content = entry.text;
475
+ if (typeof content !== "string" || content.length === 0) {
476
+ throw new ApiError("Exa contents returned a malformed response", 500);
477
+ }
478
+ const finalUrl = typeof entry.url === "string" && entry.url.length > 0 ? entry.url : request.url;
479
+ const rawTitle = typeof entry.title === "string" ? entry.title.trim() : "";
480
+ const title = rawTitle.length > 0 ? rawTitle : null;
481
+ const requestedFormat = request.format ?? "markdown";
482
+ if (requestedFormat === "text") {
483
+ return {
484
+ schemaVersion: 1,
485
+ url: request.url,
486
+ finalUrl,
487
+ title,
488
+ content: stripMarkdown(content),
489
+ contentFormat: "text",
490
+ };
491
+ }
492
+ return {
493
+ schemaVersion: 1,
494
+ url: request.url,
495
+ finalUrl,
496
+ title,
497
+ content,
498
+ contentFormat: "markdown",
499
+ };
500
+ }
501
+ /**
502
+ * Map a `ReaderFetchRequest` to Exa-native contents params. The CLI
503
+ * `--timeout` is in seconds; Exa's `livecrawlTimeout` is in
504
+ * milliseconds. The conversion (`* 1000`) is validated here so a direct
505
+ * pass-through doesn't send 20ms instead of 20s.
506
+ */
507
+ function mapReaderControls(request) {
508
+ if (typeof request.timeout !== "number" ||
509
+ !Number.isFinite(request.timeout) ||
510
+ request.timeout <= 0) {
511
+ return undefined;
512
+ }
513
+ const livecrawlTimeout = Math.round(request.timeout * 1000);
514
+ if (!Number.isFinite(livecrawlTimeout) || livecrawlTimeout <= 0) {
515
+ return undefined;
516
+ }
517
+ return { livecrawlTimeout };
518
+ }
519
+ function createExaReaderCapability(options) {
520
+ const { env, transport } = options;
521
+ const fetchOp = {
522
+ kind: "reader-fetch",
523
+ validate(request) {
524
+ assertHttpUrl(request.url);
525
+ assertNoUnsupportedReaderOptions(request);
526
+ },
527
+ cacheIdentity(request) {
528
+ const apiKey = resolveApiKey(env);
529
+ return {
530
+ provider: "exa",
531
+ capability: "reader",
532
+ operation: "reader-fetch",
533
+ credentialFingerprint: credentialFingerprint(apiKey),
534
+ request,
535
+ legacyCandidates: [],
536
+ };
537
+ },
538
+ decodeCached(value) {
539
+ return decodeReaderFetchResult(value);
540
+ },
541
+ async invoke(request) {
542
+ fetchOp.validate(request);
543
+ const apiKey = resolveApiKey(env);
544
+ try {
545
+ const params = mapReaderControls(request);
546
+ const raw = await fetchExaContents(apiKey, request.url, params, transport);
547
+ return normalizeExaContentsResult(raw, request);
548
+ }
549
+ catch (error) {
550
+ throw normalizeExaError(error);
551
+ }
552
+ },
553
+ };
554
+ return { fetch: fetchOp };
555
+ }
556
+ // ---------------------------------------------------------------------------
557
+ // Research Capability (the hardest mechanism — tech-plan §3, §7)
558
+ // ---------------------------------------------------------------------------
559
+ const DEFAULT_RESEARCH_POLL_INTERVAL_MS = 5000;
560
+ function resolvePollIntervalMs(env) {
561
+ const raw = env?.EXA_RESEARCH_POLL_INTERVAL_MS;
562
+ const parsed = parseInt(raw ?? "", 10);
563
+ return Number.isFinite(parsed) && parsed >= 0 ? parsed : DEFAULT_RESEARCH_POLL_INTERVAL_MS;
564
+ }
565
+ /**
566
+ * Build an abortable `sleep(ms)` from the injected timers. Copied from
567
+ * the Tavily adapter — same mechanism, different transport type.
568
+ */
569
+ function makeSleep(deps, signal) {
570
+ const setT = deps?.setTimeout ?? setTimeout;
571
+ const clearT = deps?.clearTimeout ?? clearTimeout;
572
+ return (ms) => new Promise((resolve, reject) => {
573
+ if (signal?.aborted) {
574
+ reject(new TimeoutError(0, "Research polling aborted"));
575
+ return;
576
+ }
577
+ if (ms <= 0) {
578
+ setImmediate(() => {
579
+ if (signal?.aborted) {
580
+ reject(new TimeoutError(0, "Research polling aborted"));
581
+ return;
582
+ }
583
+ resolve();
584
+ });
585
+ return;
586
+ }
587
+ const onAbort = () => {
588
+ clearT(id);
589
+ reject(new TimeoutError(0, "Research polling aborted"));
590
+ };
591
+ const id = setT(() => {
592
+ signal?.removeEventListener("abort", onAbort);
593
+ resolve();
594
+ }, ms);
595
+ signal?.addEventListener("abort", onAbort);
596
+ });
597
+ }
598
+ function isEexistError(err) {
599
+ return (typeof err === "object" &&
600
+ err !== null &&
601
+ "code" in err &&
602
+ err.code === "EEXIST");
603
+ }
604
+ /**
605
+ * Map `model` → Exa Agent `effort`. The result echoes the REQUESTED
606
+ * model (not the effort string) so the contract is identical across
607
+ * Tavily and Exa. Exa accepts effort values `low|medium|high|xhigh|
608
+ * auto`; only `low` (mini), `high` (pro), and `auto` are reachable
609
+ * from the Normal command.
610
+ */
611
+ function mapModelToEffort(model) {
612
+ switch (model) {
613
+ case "mini":
614
+ return "low";
615
+ case "pro":
616
+ return "high";
617
+ case "auto":
618
+ default:
619
+ return "auto";
620
+ }
621
+ }
622
+ /**
623
+ * Validate a `ResearchRequest` for Exa. Exa supports `query` and
624
+ * `model` natively; `outputLength`, `citationFormat`, and `domain` are
625
+ * concepts the Agent lacks and are rejected before transport.
626
+ *
627
+ * **OD1 note:** `domain` is rejected for now. It MAY be revalidatable
628
+ * against the pinned Agent's internal search-tool config — track as a
629
+ * follow-up if the config accepts `includeDomains`.
630
+ */
631
+ function assertNoUnsupportedResearchOptions(request) {
632
+ if (request.outputLength !== undefined) {
633
+ throw new UnsupportedOptionError("exa", "research", "outputLength");
634
+ }
635
+ if (request.citationFormat !== undefined) {
636
+ throw new UnsupportedOptionError("exa", "research", "citationFormat");
637
+ }
638
+ if (request.domain !== undefined) {
639
+ throw new UnsupportedOptionError("exa", "research", "domain");
640
+ }
641
+ }
642
+ /**
643
+ * Normalize a completed Exa Agent run's `output` into a
644
+ * `ResearchResult`.
645
+ *
646
+ * output.text -> report
647
+ * output.grounding[].citations[] -> sources[] (flatten {title, url};
648
+ * drop incomplete)
649
+ * output.structured -> ignored (distinct capability)
650
+ * request.model -> model (echoed, NOT effort string)
651
+ */
652
+ function normalizeExaResearchResult(poll, request) {
653
+ const output = poll.output;
654
+ if (!isPlainObject(output)) {
655
+ throw new ApiError("Exa research returned a malformed response", 500);
656
+ }
657
+ const text = output.text;
658
+ if (typeof text !== "string") {
659
+ throw new ApiError("Exa research returned a malformed response", 500);
660
+ }
661
+ const sources = [];
662
+ const grounding = output.grounding;
663
+ if (Array.isArray(grounding)) {
664
+ for (const entry of grounding) {
665
+ if (!isPlainObject(entry))
666
+ continue;
667
+ const citations = entry.citations;
668
+ if (!Array.isArray(citations))
669
+ continue;
670
+ for (const citation of citations) {
671
+ if (!isPlainObject(citation))
672
+ continue;
673
+ if (typeof citation.title === "string" && typeof citation.url === "string") {
674
+ sources.push({ title: citation.title, url: citation.url });
675
+ }
676
+ }
677
+ }
678
+ }
679
+ return {
680
+ schemaVersion: 1,
681
+ query: request.query,
682
+ model: request.model ?? "auto",
683
+ report: text,
684
+ sources,
685
+ };
686
+ }
687
+ /**
688
+ * True when an error from the poll GET is safe to retry. The poll is
689
+ * idempotent — retrying it never creates a new run or charges the
690
+ * account. Only transient failures (429, 5xx, network, timeout) qualify;
691
+ * auth/quota/validation errors are terminal and propagate immediately.
692
+ */
693
+ function isTransientPollError(err) {
694
+ if (err instanceof ApiError && typeof err.statusCode === "number") {
695
+ return err.statusCode === 429 || (err.statusCode >= 500 && err.statusCode <= 599);
696
+ }
697
+ if (err instanceof NetworkError || err instanceof TimeoutError)
698
+ return true;
699
+ return false;
700
+ }
701
+ /**
702
+ * POST /agent/runs to create a task, then persist its run ID in the
703
+ * state file atomically. On EEXIST (a concurrent invocation already
704
+ * created a task for this request), read the existing state file and
705
+ * return its run ID instead — the concurrent task is polled, not
706
+ * duplicated.
707
+ *
708
+ * **OD1 limitation:** the POST happens before the `wx` write, so two
709
+ * callers that both read "absent" can both POST before either writes.
710
+ * This reduces (Ctrl-C+retry) but does NOT eliminate duplicate runs.
711
+ * Same pre-existing Tavily issue Exa inherits.
712
+ */
713
+ async function createResearchTask(apiKey, request, identityHash, stateFile, transport) {
714
+ const agentParams = {
715
+ query: request.query,
716
+ effort: mapModelToEffort(request.model),
717
+ };
718
+ const created = await createExaAgentRun(apiKey, agentParams, transport);
719
+ const runId = created.id;
720
+ const state = {
721
+ requestId: runId,
722
+ identityHash,
723
+ createdAt: new Date().toISOString(),
724
+ status: "pending",
725
+ };
726
+ try {
727
+ await stateFile.write(identityHash, state);
728
+ }
729
+ catch (err) {
730
+ if (isEexistError(err)) {
731
+ const existing = await stateFile.read(identityHash);
732
+ if (existing !== null) {
733
+ return existing.requestId;
734
+ }
735
+ }
736
+ else {
737
+ throw err;
738
+ }
739
+ }
740
+ return runId;
741
+ }
742
+ function createExaResearchCapability(options) {
743
+ const { env, transport, researchStateFile } = options;
744
+ const run = {
745
+ kind: "research-fetch",
746
+ validate(request) {
747
+ if (!request || typeof request.query !== "string" || request.query.trim() === "") {
748
+ throw new ValidationError("Research query must contain at least one non-whitespace character");
749
+ }
750
+ assertNoUnsupportedResearchOptions(request);
751
+ },
752
+ cacheIdentity(request) {
753
+ const apiKey = resolveApiKey(env);
754
+ return {
755
+ provider: "exa",
756
+ capability: "research",
757
+ credentialFingerprint: credentialFingerprint(apiKey),
758
+ request,
759
+ };
760
+ },
761
+ decodeCached(value) {
762
+ return decodeResearchResult(value);
763
+ },
764
+ async invoke(request, signal) {
765
+ run.validate(request);
766
+ const apiKey = resolveApiKey(env);
767
+ const credFingerprint = credentialFingerprint(apiKey);
768
+ const identityHash = computeResearchStateHash({
769
+ provider: "exa",
770
+ capability: "research",
771
+ credentialFingerprint: credFingerprint,
772
+ request,
773
+ });
774
+ const pollIntervalMs = resolvePollIntervalMs(transport?.env);
775
+ const sleep = makeSleep(transport, signal);
776
+ try {
777
+ // 1. Check for an in-flight task (resume after Ctrl-C / crash).
778
+ const existingState = await researchStateFile.read(identityHash);
779
+ let runId;
780
+ if (existingState !== null) {
781
+ runId = existingState.requestId;
782
+ }
783
+ else {
784
+ // 2. No in-flight task: POST to create one. NO retry — a
785
+ // transient POST failure is terminal (double-charge
786
+ // prevention on a usage-based endpoint).
787
+ runId = await createResearchTask(apiKey, request, identityHash, researchStateFile, transport);
788
+ }
789
+ // 3. Poll loop until terminal status.
790
+ // The zero-retry policy wraps the whole invoke() and protects
791
+ // the POST (create). The GET (poll) is idempotent and safe to
792
+ // retry — a transient 429/5xx/network error on poll MUST NOT
793
+ // terminate a paid research run that is still active
794
+ // server-side. So we catch transient poll errors and retry
795
+ // the GET (bounded by MAX_POLL_RETRIES) before propagating.
796
+ const MAX_POLL_RETRIES = 3;
797
+ let consecutivePollFailures = 0;
798
+ for (;;) {
799
+ if (signal?.aborted) {
800
+ throw new TimeoutError(0, "Research polling aborted");
801
+ }
802
+ let poll;
803
+ try {
804
+ poll = await pollExaAgentRun(apiKey, runId, transport);
805
+ consecutivePollFailures = 0;
806
+ }
807
+ catch (pollErr) {
808
+ if (isTransientPollError(pollErr) && consecutivePollFailures < MAX_POLL_RETRIES) {
809
+ consecutivePollFailures++;
810
+ await sleep(pollIntervalMs);
811
+ continue;
812
+ }
813
+ throw pollErr;
814
+ }
815
+ if (poll.status === "completed") {
816
+ await researchStateFile.remove(identityHash);
817
+ return normalizeExaResearchResult(poll, request);
818
+ }
819
+ if (poll.status === "failed") {
820
+ await researchStateFile.remove(identityHash);
821
+ throw new ApiError("Exa research task failed", 500);
822
+ }
823
+ if (poll.status === "cancelled") {
824
+ // Exa-specific: cancelled is terminal (treated as failure).
825
+ await researchStateFile.remove(identityHash);
826
+ throw new ApiError("Exa research task was cancelled", 500);
827
+ }
828
+ if (poll.status === "not_found") {
829
+ // 404 — server-side run expired. Delete stale state and
830
+ // create a fresh run.
831
+ await researchStateFile.remove(identityHash);
832
+ runId = await createResearchTask(apiKey, request, identityHash, researchStateFile, transport);
833
+ continue;
834
+ }
835
+ // queued or running: sleep and poll again.
836
+ await sleep(pollIntervalMs);
837
+ }
838
+ }
839
+ catch (error) {
840
+ throw normalizeExaError(error);
841
+ }
842
+ },
843
+ };
844
+ return { run };
845
+ }
846
+ // ---------------------------------------------------------------------------
847
+ // Descriptor factory
848
+ // ---------------------------------------------------------------------------
849
+ /**
850
+ * Build the Exa Provider Descriptor. The descriptor advertises the Exa
851
+ * capability set (search, reader, research, diagnostics) and constructs
852
+ * an Adapter whose Capabilities own credentials, transport, Provider
853
+ * field mapping, and failure normalization. Construction is
854
+ * side-effect-free; the transport is invoked per Capability call. Tests
855
+ * pass `transport` (typically a fake-fetch wrapper); production uses
856
+ * the no-argument factory which resolves to the global `fetch` and
857
+ * timers inside the transport Module.
858
+ */
859
+ export function createExaDescriptor(dependencies) {
860
+ const transport = dependencies?.transport;
861
+ const researchStateFile = dependencies?.researchStateFile ?? createProductionResearchStateFile();
862
+ return {
863
+ id: "exa",
864
+ isConfigured(env) {
865
+ return isExaConfigured(env);
866
+ },
867
+ capabilities() {
868
+ return new Set(["search", "reader", "research", "diagnostics"]);
869
+ },
870
+ create(context) {
871
+ const search = createExaSearchCapability({
872
+ env: context.env,
873
+ transport,
874
+ });
875
+ const reader = createExaReaderCapability({
876
+ env: context.env,
877
+ transport,
878
+ });
879
+ const research = createExaResearchCapability({
880
+ env: context.env,
881
+ transport,
882
+ researchStateFile,
883
+ });
884
+ const diagnostics = createExaDiagnosticsCapability({
885
+ env: context.env,
886
+ transport,
887
+ });
888
+ return { id: "exa", search, reader, research, diagnostics };
889
+ },
890
+ };
891
+ }
892
+ //# sourceMappingURL=adapter.js.map