mcp-castor 2026.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +487 -0
  2. package/bin/castor.js +706 -0
  3. package/index.js +206 -0
  4. package/package.json +97 -0
  5. package/skills/canary-test-staging/SKILL.md +24 -0
  6. package/skills/evo-mutation-rollback/SKILL.md +29 -0
  7. package/skills/hypothesis-generation/SKILL.md +26 -0
  8. package/skills/traceback-condensing/SKILL.md +26 -0
  9. package/src/castor_runner.js +469 -0
  10. package/src/config.js +1204 -0
  11. package/src/env.js +10 -0
  12. package/src/evo_engine.js +214 -0
  13. package/src/harness/core/events.js +75 -0
  14. package/src/harness/core/kernel.js +209 -0
  15. package/src/harness/evo/evaluator.js +156 -0
  16. package/src/harness/evo/evo_operator.js +550 -0
  17. package/src/harness/evo/lineage_dag.js +383 -0
  18. package/src/harness/evo/trace_repair.js +173 -0
  19. package/src/harness/evo/watchdog.js +72 -0
  20. package/src/harness/loop_detector.js +135 -0
  21. package/src/harness/runner.js +1216 -0
  22. package/src/harness/services/ast_service.js +1813 -0
  23. package/src/harness/services/event_logger.js +275 -0
  24. package/src/harness/services/mcp_bridge.js +408 -0
  25. package/src/harness/services/provider_vllm.js +728 -0
  26. package/src/harness/services/sandbox_fs.js +1238 -0
  27. package/src/harness/services/searxng_lifecycle.js +254 -0
  28. package/src/harness/services/shell_executor.js +264 -0
  29. package/src/harness/services/shell_validator.js +506 -0
  30. package/src/harness/services/web_service.js +828 -0
  31. package/src/platform.js +344 -0
  32. package/src/repetition_detector.js +139 -0
  33. package/src/semaphore.js +373 -0
  34. package/src/server_lifecycle.js +781 -0
  35. package/src/skills.js +400 -0
  36. package/src/state_pruner.js +392 -0
  37. package/src/task_registry.js +1357 -0
  38. package/src/telemetry.js +638 -0
  39. package/src/tools.js +997 -0
  40. package/src/wsl_bridge.js +629 -0
  41. package/src/wsl_env.js +171 -0
  42. package/stream_proxy.js +453 -0
@@ -0,0 +1,828 @@
1
+ /**
2
+ * Castor Web & Research Service
3
+ *
4
+ * Provides:
5
+ * - Live web search via DuckDuckGo without API keys (web_search)
6
+ * - Autonomous webpage fetching & extraction via Mozilla Readability + Turndown (web_fetch)
7
+ * - Safe handling of JSON, plaintext, HTML articles, and general webpages
8
+ * - Native in-process V8 execution with zero external subprocess/stdio overhead
9
+ */
10
+
11
+ import { Readability } from "@mozilla/readability";
12
+ import { JSDOM } from "jsdom";
13
+ import TurndownService from "turndown";
14
+ import { PDFParse } from "pdf-parse";
15
+ import { search, SafeSearchType } from "duck-duck-scrape";
16
+ import { getSearchConfig, MAX_FETCH_CHARS } from "../../config.js";
17
+ import {
18
+ ensureSearxngRunning,
19
+ markSearxngActive,
20
+ registerShutdown,
21
+ } from "./searxng_lifecycle.js";
22
+ import { createRequire } from "node:module";
23
+
24
+ // Single source of truth for the harness version: read from package.json
25
+ // (same createRequire idiom as index.js and mcp_bridge.js).
26
+ const require = createRequire(import.meta.url);
27
+ const PKG_VERSION = require("../../../package.json").version;
28
+
29
+ const DEFAULT_USER_AGENT =
30
+ `Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36 Castor/${PKG_VERSION}`;
31
+
32
+ const DEFAULT_FETCH_TIMEOUT_MS = 20_000;
33
+ const DEFAULT_SEARCH_TIMEOUT_MS = 15_000;
34
+
35
+ const NOISE_JSON_KEYS = new Set([
36
+ "avatar_url",
37
+ "gravatar_id",
38
+ "node_id",
39
+ "followers_url",
40
+ "following_url",
41
+ "gists_url",
42
+ "starred_url",
43
+ "subscriptions_url",
44
+ "organizations_url",
45
+ "repos_url",
46
+ "events_url",
47
+ "received_events_url",
48
+ "site_admin",
49
+ "user_view_type",
50
+ "reactions",
51
+ ]);
52
+
53
+ function pruneJsonPayload(val, depth = 0) {
54
+ if (depth > 8) return val;
55
+ if (Array.isArray(val)) {
56
+ const maxItems = 30;
57
+ const pruned = val.slice(0, maxItems).map((item) => pruneJsonPayload(item, depth + 1));
58
+ if (val.length > maxItems) {
59
+ pruned.push({ _notice: `...[${val.length - maxItems} additional items omitted for context efficiency]...` });
60
+ }
61
+ return pruned;
62
+ }
63
+ if (val && typeof val === "object") {
64
+ const res = {};
65
+ for (const [k, v] of Object.entries(val)) {
66
+ if (NOISE_JSON_KEYS.has(k)) continue;
67
+ res[k] = pruneJsonPayload(v, depth + 1);
68
+ }
69
+ return res;
70
+ }
71
+ return val;
72
+ }
73
+
74
+ // Leaky-bucket serialization queue for DuckDuckGo unauthenticated requests
75
+ let lastDdgRequestTime = 0;
76
+ const DDG_MIN_INTERVAL_MS = 1500;
77
+ let ddgQueue = Promise.resolve();
78
+
79
+ async function throttleDdgRequest() {
80
+ const current = ddgQueue;
81
+ let resolveNext;
82
+ ddgQueue = new Promise((resolve) => {
83
+ resolveNext = resolve;
84
+ });
85
+ await current;
86
+ try {
87
+ const now = Date.now();
88
+ const elapsed = now - lastDdgRequestTime;
89
+ if (elapsed < DDG_MIN_INTERVAL_MS) {
90
+ await new Promise((r) => setTimeout(r, DDG_MIN_INTERVAL_MS - elapsed));
91
+ }
92
+ lastDdgRequestTime = Date.now();
93
+ } finally {
94
+ resolveNext();
95
+ }
96
+ }
97
+
98
+ function isDocQuery(q) {
99
+ return /\b(docs?|documentation|api|library|package|sdk|crate|module|import|framework|reference|guide)\b/i.test(q);
100
+ }
101
+
102
+ /**
103
+ * Removes common tracking query parameters (utm_*, gclid, fbclid, mc_eid,
104
+ * msclkid, ref, source) from a URL string, preserving all other parameters.
105
+ *
106
+ * @param {string} urlStr
107
+ * @returns {string}
108
+ */
109
+ function stripTrackingParams(urlStr) {
110
+ if (!urlStr) return urlStr;
111
+ try {
112
+ const u = new URL(urlStr);
113
+ for (const key of [...u.searchParams.keys()]) {
114
+ if (
115
+ key.toLowerCase().startsWith("utm_") ||
116
+ ["gclid", "fbclid", "mc_eid", "msclkid", "ref", "source"].includes(key.toLowerCase())
117
+ ) {
118
+ u.searchParams.delete(key);
119
+ }
120
+ }
121
+ return u.toString();
122
+ } catch {
123
+ return urlStr;
124
+ }
125
+ }
126
+
127
+ /**
128
+ * Normalizes a URL for deduplication: strips tracking parameters, lowercases
129
+ * the scheme and host, drops the fragment, and removes a trailing slash.
130
+ *
131
+ * @param {string} urlStr
132
+ * @returns {string}
133
+ */
134
+ function normalizeUrl(urlStr) {
135
+ const stripped = stripTrackingParams(urlStr);
136
+ try {
137
+ const u = new URL(stripped);
138
+ u.hash = "";
139
+ u.hostname = u.hostname.toLowerCase();
140
+ u.protocol = u.protocol.toLowerCase();
141
+ let s = u.toString();
142
+ if (s.endsWith("/")) s = s.slice(0, -1);
143
+ return s;
144
+ } catch {
145
+ return stripped;
146
+ }
147
+ }
148
+
149
+ /**
150
+ * Strips inline HTML tags and decodes common HTML entities in a string.
151
+ *
152
+ * @param {string} s
153
+ * @returns {string}
154
+ */
155
+ function sanitizeSnippet(s) {
156
+ if (!s) return s;
157
+ return s
158
+ .replace(/<[^>]+>/g, "")
159
+ .replace(/&#x27;/gi, "'")
160
+ .replace(/&#39;/g, "'")
161
+ .replace(/&quot;/g, '"')
162
+ .replace(/&lt;/g, "<")
163
+ .replace(/&gt;/g, ">")
164
+ .replace(/&amp;/g, "&");
165
+ }
166
+
167
+ // Media types that must never be decoded as UTF-8 text.
168
+ const BINARY_MEDIA_PREFIXES = ["image/", "audio/", "video/"];
169
+ const BINARY_MEDIA_TYPES = new Set([
170
+ "application/zip",
171
+ "application/octet-stream",
172
+ "application/gzip",
173
+ "application/x-gzip",
174
+ "application/x-tar",
175
+ "application/x-7z-compressed",
176
+ "application/x-rar-compressed",
177
+ "application/wasm",
178
+ "application/x-bzip2",
179
+ "application/x-xz",
180
+ "font/woff",
181
+ "font/woff2",
182
+ "font/ttf",
183
+ "font/otf",
184
+ ]);
185
+
186
+ const PDF_MAGIC = Buffer.from("%PDF-");
187
+
188
+ /**
189
+ * Returns true when the leading bytes of a buffer contain a NUL byte,
190
+ * a strong signal of binary (non-text) content.
191
+ *
192
+ * @param {Buffer} buf
193
+ * @returns {boolean}
194
+ */
195
+ function hasNulByte(buf) {
196
+ const limit = Math.min(buf.length, 8192);
197
+ for (let i = 0; i < limit; i++) {
198
+ if (buf[i] === 0) return true;
199
+ }
200
+ return false;
201
+ }
202
+
203
+ /**
204
+ * Classifies an HTTP response body into a handling category before any text
205
+ * decoding, using the Content-Type header and leading magic bytes.
206
+ *
207
+ * @param {string} contentType Lowercased Content-Type header value.
208
+ * @param {Buffer} buf Full response body.
209
+ * @returns {"json"|"text"|"html"|"pdf"|"binary"}
210
+ */
211
+ function classifyResponse(contentType, buf) {
212
+ const base = contentType.split(";")[0].trim().toLowerCase();
213
+
214
+ // PDF: explicit type or %PDF- magic (catches mislabeled/octet-stream PDFs).
215
+ if (base === "application/pdf" || buf.subarray(0, 5).equals(PDF_MAGIC)) {
216
+ return "pdf";
217
+ }
218
+ if (base === "application/json" || base.endsWith("+json")) {
219
+ return "json";
220
+ }
221
+ if (base === "text/html" || base === "application/xhtml+xml") {
222
+ return "html";
223
+ }
224
+ if (base.startsWith("text/")) {
225
+ return "text";
226
+ }
227
+ if (
228
+ BINARY_MEDIA_TYPES.has(base) ||
229
+ BINARY_MEDIA_PREFIXES.some((p) => base.startsWith(p))
230
+ ) {
231
+ return "binary";
232
+ }
233
+ // Unknown/absent Content-Type: sniff the leading bytes.
234
+ if (hasNulByte(buf)) {
235
+ return "binary";
236
+ }
237
+ return "html";
238
+ }
239
+
240
+ /**
241
+ * Extracts plain text from a PDF buffer using pdf-parse (pdfjs-dist based).
242
+ *
243
+ * @param {Buffer} buf Raw PDF bytes.
244
+ * @returns {Promise<string>} Concatenated per-page text.
245
+ */
246
+ async function extractPdfText(buf) {
247
+ const parser = new PDFParse({ data: new Uint8Array(buf) });
248
+ try {
249
+ const result = await parser.getText();
250
+ return result.pages.map((p) => p.text).join("\n\n").trim();
251
+ } finally {
252
+ await parser.destroy();
253
+ }
254
+ }
255
+
256
+ /**
257
+ * Truncates text to at most `maxChars` characters, cutting at a clean
258
+ * boundary (a paragraph `\n\n`, else a line `\n`) when one exists within the
259
+ * last 500 characters of the window, and closing any open markdown code
260
+ * fence. Appends a concise marker reporting the shown and total character
261
+ * counts. Returns the input unchanged when it already fits.
262
+ *
263
+ * @param {string} text
264
+ * @param {number} maxChars
265
+ * @returns {string}
266
+ */
267
+ function smartTruncate(text, maxChars) {
268
+ if (text.length <= maxChars) return text;
269
+
270
+ const window = text.slice(0, maxChars);
271
+ const searchStart = Math.max(0, maxChars - 500);
272
+
273
+ let cut = window.lastIndexOf("\n\n");
274
+ if (cut < searchStart) cut = window.lastIndexOf("\n");
275
+ if (cut < searchStart) cut = maxChars;
276
+
277
+ let out = window.slice(0, cut);
278
+
279
+ // Close an open markdown code fence so the document stays well-formed.
280
+ const fenceCount = (out.match(/^```/gm) || []).length;
281
+ if (fenceCount % 2 === 1) {
282
+ out = out.replace(/\s+$/, "") + "\n```";
283
+ }
284
+
285
+ return out + `\n\n[Content truncated: showing ${out.length} of ${text.length} chars]`;
286
+ }
287
+
288
+ export class WebService {
289
+ constructor(options = {}) {
290
+ this.userAgent = options.userAgent || DEFAULT_USER_AGENT;
291
+ this.fetchTimeoutMs = options.fetchTimeoutMs || DEFAULT_FETCH_TIMEOUT_MS;
292
+ this.searchTimeoutMs = options.searchTimeoutMs || DEFAULT_SEARCH_TIMEOUT_MS;
293
+ this._initTurndown();
294
+ }
295
+
296
+ _initTurndown() {
297
+ this.turndown = new TurndownService({
298
+ headingStyle: "atx",
299
+ hr: "---",
300
+ bulletListMarker: "-",
301
+ codeBlockStyle: "fenced",
302
+ });
303
+
304
+ // Remove script, style, noscript, svg, canvas, iframe from markdown conversion
305
+ this.turndown.remove(["script", "style", "noscript", "svg", "canvas", "iframe"]);
306
+ }
307
+
308
+ /**
309
+ * Search via Brave Search API.
310
+ */
311
+ async _searchBrave(query, limit, apiKey) {
312
+ if (!apiKey) throw new Error("MissingBraveApiKey");
313
+ const u = new URL("https://api.search.brave.com/res/v1/web/search");
314
+ u.searchParams.set("q", query);
315
+ u.searchParams.set("count", String(limit));
316
+ const res = await globalThis.fetch(u.toString(), {
317
+ headers: {
318
+ "X-Subscription-Token": apiKey,
319
+ Accept: "application/json",
320
+ "User-Agent": this.userAgent,
321
+ },
322
+ signal: AbortSignal.timeout(this.searchTimeoutMs),
323
+ });
324
+ if (!res.ok) {
325
+ throw new Error(`BraveSearchError: HTTP ${res.status} ${res.statusText}`);
326
+ }
327
+ const data = await res.json();
328
+ const items = data.web?.results || [];
329
+ return items.slice(0, limit).map((r) => ({
330
+ title: sanitizeSnippet(r.title || ""),
331
+ url: stripTrackingParams(r.url || ""),
332
+ snippet: sanitizeSnippet(r.description || ""),
333
+ }));
334
+ }
335
+
336
+ /**
337
+ * Search via Tavily Search API.
338
+ */
339
+ async _searchTavily(query, limit, apiKey) {
340
+ if (!apiKey) throw new Error("MissingTavilyApiKey");
341
+ const res = await globalThis.fetch("https://api.tavily.com/search", {
342
+ method: "POST",
343
+ headers: {
344
+ "Content-Type": "application/json",
345
+ "User-Agent": this.userAgent,
346
+ },
347
+ body: JSON.stringify({
348
+ api_key: apiKey,
349
+ query,
350
+ max_results: limit,
351
+ include_raw_content: false,
352
+ }),
353
+ signal: AbortSignal.timeout(this.searchTimeoutMs),
354
+ });
355
+ if (!res.ok) {
356
+ throw new Error(`TavilySearchError: HTTP ${res.status} ${res.statusText}`);
357
+ }
358
+ const data = await res.json();
359
+ const items = data.results || [];
360
+ return items.slice(0, limit).map((r) => ({
361
+ title: r.title || "",
362
+ url: stripTrackingParams(r.url || ""),
363
+ snippet: r.content || "",
364
+ }));
365
+ }
366
+
367
+ /**
368
+ * Search via Upstash Context7 API (Library & Framework Documentation).
369
+ */
370
+ async _searchContext7(query, limit, apiKey) {
371
+ if (!apiKey) throw new Error("MissingContext7ApiKey");
372
+ const u = new URL("https://context7.com/api/v3/search");
373
+ u.searchParams.set("query", query);
374
+ u.searchParams.set("type", "json");
375
+ const res = await globalThis.fetch(u.toString(), {
376
+ headers: {
377
+ Authorization: `Bearer ${apiKey}`,
378
+ "User-Agent": this.userAgent,
379
+ },
380
+ signal: AbortSignal.timeout(this.searchTimeoutMs),
381
+ });
382
+ if (!res.ok) {
383
+ throw new Error(`Context7SearchError: HTTP ${res.status} ${res.statusText}`);
384
+ }
385
+ const data = await res.json();
386
+ const items = data.results || data.snippets || (Array.isArray(data) ? data : []);
387
+ return items.slice(0, limit).map((r) => ({
388
+ title: r.title || r.library || "Context7 Documentation",
389
+ url: stripTrackingParams(r.url || r.source || "https://context7.com"),
390
+ snippet: r.content || r.snippet || r.text || "",
391
+ }));
392
+ }
393
+
394
+ /**
395
+ * Search via SearXNG JSON instance.
396
+ * For local (loopback) instances, ensures the Docker container is running
397
+ * before issuing the request, and marks activity for the idle-stop watchdog.
398
+ */
399
+ async _searchSearxng(query, limit, baseUrl) {
400
+ if (!baseUrl) throw new Error("MissingSearxngUrl");
401
+ await ensureSearxngRunning(baseUrl);
402
+ const u = new URL(baseUrl.replace(/\/+$/, "") + "/search");
403
+ u.searchParams.set("q", query);
404
+ u.searchParams.set("format", "json");
405
+ const res = await globalThis.fetch(u.toString(), {
406
+ headers: {
407
+ "User-Agent": this.userAgent,
408
+ },
409
+ signal: AbortSignal.timeout(this.searchTimeoutMs),
410
+ });
411
+ if (!res.ok) {
412
+ throw new Error(`SearxngSearchError: HTTP ${res.status} ${res.statusText}`);
413
+ }
414
+ const data = await res.json();
415
+ markSearxngActive();
416
+ const items = data.results || [];
417
+ return items.slice(0, limit).map((r) => ({
418
+ title: r.title || "",
419
+ url: stripTrackingParams(r.url || ""),
420
+ snippet: r.content || "",
421
+ }));
422
+ }
423
+
424
+ /**
425
+ * Performs web search using DuckDuckGo (HTML endpoint first, then duck-duck-scrape).
426
+ */
427
+ async _searchHtml(query, limit, safeSearch) {
428
+ const res = await globalThis.fetch("https://html.duckduckgo.com/html/", {
429
+ method: "POST",
430
+ headers: {
431
+ "User-Agent": this.userAgent,
432
+ "Content-Type": "application/x-www-form-urlencoded",
433
+ Accept: "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
434
+ },
435
+ body: `q=${encodeURIComponent(query)}&kp=${safeSearch === "off" ? "-2" : safeSearch === "strict" ? "1" : "-1"}`,
436
+ signal: AbortSignal.timeout(this.searchTimeoutMs),
437
+ });
438
+
439
+ if (!res.ok) return [];
440
+
441
+ const html = await res.text();
442
+ const dom = new JSDOM(html);
443
+ const document = dom.window.document;
444
+ const entries = document.querySelectorAll(".result");
445
+ const results = [];
446
+
447
+ for (const el of entries) {
448
+ if (results.length >= limit) break;
449
+ const titleEl = el.querySelector(".result__title a");
450
+ const snippetEl = el.querySelector(".result__snippet");
451
+ if (!titleEl) continue;
452
+
453
+ let link = titleEl.href || "";
454
+ try {
455
+ const u = new URL(link, "https://html.duckduckgo.com");
456
+ const uddg = u.searchParams.get("uddg");
457
+ if (uddg) link = decodeURIComponent(uddg);
458
+ } catch {
459
+ /* keep raw link */
460
+ }
461
+
462
+ const title = (titleEl.textContent || "").trim();
463
+ const snippet = snippetEl ? (snippetEl.textContent || "").trim() : "";
464
+ if (title && link) {
465
+ results.push({ title, url: stripTrackingParams(link), snippet });
466
+ }
467
+ }
468
+
469
+ return results;
470
+ }
471
+
472
+ async _searchDuckDuckGo(query, limit, safe_search) {
473
+ await throttleDdgRequest();
474
+
475
+ try {
476
+ const htmlResults = await this._searchHtml(query, limit, safe_search);
477
+ if (htmlResults.length > 0) {
478
+ return htmlResults;
479
+ }
480
+ } catch {
481
+ /* fall through to duck-duck-scrape */
482
+ }
483
+
484
+ let ddgSafeSearch = SafeSearchType.MODERATE;
485
+ if (safe_search === "strict") ddgSafeSearch = SafeSearchType.STRICT;
486
+ else if (safe_search === "off") ddgSafeSearch = SafeSearchType.OFF;
487
+
488
+ const res = await search(query, {
489
+ safeSearch: ddgSafeSearch,
490
+ });
491
+
492
+ if (res.noResults || !Array.isArray(res.results)) {
493
+ return [];
494
+ }
495
+
496
+ return res.results.slice(0, limit).map((r) => ({
497
+ title: r.title || "",
498
+ url: stripTrackingParams(r.url || ""),
499
+ snippet: r.description || r.rawDescription || "",
500
+ }));
501
+ }
502
+
503
+ /**
504
+ * Performs multi-provider web search with automatic failover.
505
+ *
506
+ * @param {object} params
507
+ * @param {string} params.query Search terms or phrase
508
+ * @param {number} [params.max_results=5] Max results to return
509
+ * @param {"strict"|"moderate"|"off"} [params.safe_search="moderate"] SafeSearch level
510
+ * @param {string} [params.provider="auto"] Provider override ('auto', 'brave', 'tavily', 'context7', 'searxng', 'duckduckgo')
511
+ * @returns {Promise<{ query: string, count: number, provider: string, results: Array<{ title: string, url: string, snippet: string }> }>}
512
+ */
513
+ async search({ query, max_results = 5, safe_search = "moderate", provider }) {
514
+ if (!query || typeof query !== "string" || !query.trim()) {
515
+ throw new Error("InvalidQueryError: Search query must be a non-empty string");
516
+ }
517
+
518
+ const trimmedQuery = query.trim();
519
+ const limit = Math.min(Math.max(1, max_results), 25);
520
+ const cfg = getSearchConfig();
521
+ const chosenProvider = provider || cfg.provider || "auto";
522
+
523
+ // Build candidate provider chain
524
+ const chain = [];
525
+ if (chosenProvider !== "auto") {
526
+ chain.push(chosenProvider);
527
+ } else {
528
+ // Doc queries: Context7 -> Tavily -> SearXNG -> Brave -> DuckDuckGo
529
+ // General queries: SearXNG -> Tavily -> Brave -> DuckDuckGo
530
+ if (isDocQuery(trimmedQuery)) {
531
+ if (cfg.context7_api_key) chain.push("context7");
532
+ if (cfg.tavily_api_key) chain.push("tavily");
533
+ if (cfg.searxng_url) chain.push("searxng");
534
+ if (cfg.brave_api_key) chain.push("brave");
535
+ } else {
536
+ if (cfg.searxng_url) chain.push("searxng");
537
+ if (cfg.tavily_api_key) chain.push("tavily");
538
+ if (cfg.brave_api_key) chain.push("brave");
539
+ }
540
+ chain.push("duckduckgo");
541
+ }
542
+
543
+ let lastError = null;
544
+ let anySucceeded = false;
545
+ for (const p of chain) {
546
+ try {
547
+ let results = [];
548
+ if (p === "brave") results = await this._searchBrave(trimmedQuery, limit, cfg.brave_api_key);
549
+ else if (p === "tavily") results = await this._searchTavily(trimmedQuery, limit, cfg.tavily_api_key);
550
+ else if (p === "context7") results = await this._searchContext7(trimmedQuery, limit, cfg.context7_api_key);
551
+ else if (p === "searxng") results = await this._searchSearxng(trimmedQuery, limit, cfg.searxng_url);
552
+ else if (p === "duckduckgo") results = await this._searchDuckDuckGo(trimmedQuery, limit, safe_search);
553
+
554
+ anySucceeded = true;
555
+ if (results && results.length > 0) {
556
+ const seen = new Set();
557
+ const deduped = results.filter((r) => {
558
+ const key = normalizeUrl(r.url);
559
+ if (seen.has(key)) return false;
560
+ seen.add(key);
561
+ return true;
562
+ });
563
+ return {
564
+ query: trimmedQuery,
565
+ count: deduped.length,
566
+ provider: p,
567
+ results: deduped,
568
+ };
569
+ }
570
+ } catch (err) {
571
+ lastError = err;
572
+ // On explicit provider request, fail fast
573
+ if (chosenProvider !== "auto") throw err;
574
+ }
575
+ }
576
+
577
+ // Surface error if all providers in the auto chain failed.
578
+ if (!anySucceeded && lastError) {
579
+ return {
580
+ isError: true,
581
+ text: `SearchError: all ${chain.length} provider(s) in the auto chain failed. Last error: ${lastError.message}`,
582
+ };
583
+ }
584
+
585
+ return {
586
+ query: trimmedQuery,
587
+ count: 0,
588
+ provider: chain[chain.length - 1],
589
+ results: [],
590
+ };
591
+ }
592
+
593
+
594
+ /**
595
+ * Fetches a webpage or API endpoint, extracting clean Markdown.
596
+ *
597
+ * @param {object} params
598
+ * @param {string} params.url HTTP/HTTPS URL
599
+ * @param {boolean} [params.extract_article=true] Whether to use Mozilla Readability
600
+ * @param {number} [params.timeout_ms=20000] Request timeout
601
+ * @param {number} [params.max_chars=60000] Maximum response characters
602
+ * @returns {Promise<object>}
603
+ */
604
+ async fetch({ url, extract_article = true, timeout_ms = this.fetchTimeoutMs, max_chars = MAX_FETCH_CHARS }) {
605
+ if (!url || typeof url !== "string") {
606
+ throw new Error("InvalidUrlError: URL must be a string");
607
+ }
608
+
609
+ let parsedUrl;
610
+ try {
611
+ parsedUrl = new URL(url.trim());
612
+ } catch {
613
+ throw new Error(`InvalidUrlError: Malformed URL '${url}'`);
614
+ }
615
+
616
+ if (parsedUrl.protocol !== "http:" && parsedUrl.protocol !== "https:") {
617
+ throw new Error(`InvalidUrlError: Only HTTP and HTTPS URLs are supported (got '${parsedUrl.protocol}')`);
618
+ }
619
+
620
+ const headers = {
621
+ "User-Agent": this.userAgent,
622
+ Accept: "text/html,application/xhtml+xml,application/xml;q=0.9,application/json;q=0.8,text/plain;q=0.7,*/*;q=0.5",
623
+ "Accept-Language": "en-US,en;q=0.9",
624
+ };
625
+ if (parsedUrl.hostname === "api.github.com" && process.env.GITHUB_TOKEN) {
626
+ headers.Authorization = `Bearer ${process.env.GITHUB_TOKEN}`;
627
+ }
628
+
629
+ const response = await globalThis.fetch(parsedUrl.href, {
630
+ method: "GET",
631
+ headers,
632
+ signal: AbortSignal.timeout(timeout_ms),
633
+ });
634
+
635
+ if (!response.ok) {
636
+ throw new Error(`HttpError: GET '${url}' failed with status ${response.status} ${response.statusText}`);
637
+ }
638
+
639
+ // Read the body once as a Buffer so the content can be classified before
640
+ // any text decoding (prevents binary bodies from being mangled as UTF-8).
641
+ const body = Buffer.from(await response.arrayBuffer());
642
+ const contentType = (response.headers.get("content-type") || "").toLowerCase();
643
+ const category = classifyResponse(contentType, body);
644
+
645
+ // 1. PDF: extract plain text in-process (never decode as UTF-8).
646
+ if (category === "pdf") {
647
+ const text = await extractPdfText(body);
648
+ const truncated = smartTruncate(text, max_chars);
649
+ return {
650
+ url: parsedUrl.href,
651
+ status: response.status,
652
+ content_type: "application/pdf",
653
+ markdown: truncated,
654
+ length: text.length,
655
+ };
656
+ }
657
+
658
+ // 2. Binary media: return a structured stub, never a UTF-8 decode.
659
+ if (category === "binary") {
660
+ return {
661
+ url: parsedUrl.href,
662
+ status: response.status,
663
+ content_type: contentType || "application/octet-stream",
664
+ binary: true,
665
+ byte_length: body.length,
666
+ markdown:
667
+ `[Binary content: ${contentType || "unknown type"}, ${body.length} bytes. ` +
668
+ "Not rendered as text; use a dedicated tool to inspect this file.",
669
+ length: body.length,
670
+ };
671
+ }
672
+
673
+ // 3. JSON handling (with automatic token distillation)
674
+ if (category === "json") {
675
+ const data = JSON.parse(body.toString("utf-8"));
676
+ const distilled = pruneJsonPayload(data);
677
+ const jsonStr = JSON.stringify(distilled, null, 2);
678
+ const truncated = smartTruncate(jsonStr, max_chars);
679
+ return {
680
+ url: parsedUrl.href,
681
+ status: response.status,
682
+ content_type: "application/json",
683
+ markdown: "```json\n" + truncated + "\n```",
684
+ length: jsonStr.length,
685
+ };
686
+ }
687
+
688
+ // 4. Plaintext / Markdown handling
689
+ if (category === "text") {
690
+ const text = body.toString("utf-8");
691
+ const truncated = smartTruncate(text, max_chars);
692
+ return {
693
+ url: parsedUrl.href,
694
+ status: response.status,
695
+ content_type: contentType.includes("markdown") ? "text/markdown" : "text/plain",
696
+ markdown: truncated,
697
+ length: text.length,
698
+ };
699
+ }
700
+
701
+ // 5. HTML handling: Mozilla Readability + Turndown
702
+ const rawHtml = body.toString("utf-8");
703
+ const dom = new JSDOM(rawHtml, { url: parsedUrl.href });
704
+ const document = dom.window.document;
705
+
706
+ // Remove scripts, styles, iframes from DOM before extraction
707
+ const unwanted = document.querySelectorAll("script, style, noscript, iframe");
708
+ unwanted.forEach((el) => el.remove());
709
+
710
+ if (extract_article) {
711
+ try {
712
+ const reader = new Readability(document);
713
+ const article = reader.parse();
714
+ if (article && article.content) {
715
+ let markdown = this.turndown.turndown(article.content);
716
+ const fullLength = markdown.length;
717
+ markdown = smartTruncate(markdown, max_chars);
718
+ return {
719
+ url: parsedUrl.href,
720
+ status: response.status,
721
+ content_type: "article",
722
+ title: article.title || document.title || "",
723
+ byline: article.byline || null,
724
+ excerpt: article.excerpt || null,
725
+ siteName: article.siteName || null,
726
+ markdown: markdown.trim(),
727
+ length: fullLength,
728
+ };
729
+ }
730
+ } catch {
731
+ // If readability parsing threw, fall through to full body conversion
732
+ }
733
+ }
734
+
735
+ // Fall back to entire body HTML -> Markdown
736
+ const bodyHtml = document.body ? document.body.innerHTML : rawHtml;
737
+ let markdown = this.turndown.turndown(bodyHtml);
738
+ const fullLength = markdown.length;
739
+ markdown = smartTruncate(markdown, max_chars);
740
+
741
+ return {
742
+ url: parsedUrl.href,
743
+ status: response.status,
744
+ content_type: "page",
745
+ title: document.title || "",
746
+ markdown: markdown.trim(),
747
+ length: fullLength,
748
+ };
749
+ }
750
+ }
751
+
752
+ /**
753
+ * Castor Plugin to mount WebService into Context.
754
+ */
755
+ export function webPlugin(ctx, options = {}) {
756
+ const web = new WebService(options);
757
+ ctx.provide("web", web);
758
+ registerShutdown();
759
+
760
+ ctx.registerTool("web_search", {
761
+ description:
762
+ "Search the live web with automatic multi-provider routing (Brave, Tavily, Context7, SearXNG, DuckDuckGo). " +
763
+ "Returns top matching results with titles, URLs, and text snippets. " +
764
+ "Use this to find current documentation, technical solutions, and library APIs. " +
765
+ "Set provider to 'context7' to specifically search framework and library documentation.",
766
+ parameters: {
767
+ type: "object",
768
+ properties: {
769
+ query: {
770
+ type: "string",
771
+ description: "Search query keywords or phrase",
772
+ },
773
+ max_results: {
774
+ type: "integer",
775
+ description: "Maximum number of search results to return (default 5, max 25)",
776
+ default: 5,
777
+ },
778
+ safe_search: {
779
+ type: "string",
780
+ enum: ["strict", "moderate", "off"],
781
+ description: "SafeSearch filtering level (default 'moderate')",
782
+ default: "moderate",
783
+ },
784
+ provider: {
785
+ type: "string",
786
+ enum: ["auto", "brave", "tavily", "context7", "searxng", "duckduckgo"],
787
+ description: "Search provider override (default 'auto'). Use 'context7' for framework and library documentation.",
788
+ default: "auto",
789
+ },
790
+ },
791
+ required: ["query"],
792
+ },
793
+ execute: (args) => web.search(args),
794
+ });
795
+
796
+ ctx.registerTool("web_fetch", {
797
+ description:
798
+ "Fetches web page or API content at a URL and converts it into clean Markdown using Mozilla Readability and Turndown. " +
799
+ "Extracts main article text, headers, and code blocks while stripping navigation, scripts, and ads. " +
800
+ "Also supports JSON and plain text endpoints.",
801
+ parameters: {
802
+ type: "object",
803
+ properties: {
804
+ url: {
805
+ type: "string",
806
+ description: "HTTP or HTTPS URL to fetch",
807
+ },
808
+ extract_article: {
809
+ type: "boolean",
810
+ description: "When true (default), use Mozilla Readability to extract main article body. When false, convert entire page body to markdown.",
811
+ default: true,
812
+ },
813
+ timeout_ms: {
814
+ type: "integer",
815
+ description: "Request timeout in milliseconds (default 20000)",
816
+ default: 20000,
817
+ },
818
+ max_chars: {
819
+ type: "integer",
820
+ description: "Maximum characters of markdown to return (default 60000)",
821
+ default: 60000,
822
+ },
823
+ },
824
+ required: ["url"],
825
+ },
826
+ execute: (args) => web.fetch(args),
827
+ });
828
+ }