mcp-castor 2026.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +487 -0
  2. package/bin/castor.js +706 -0
  3. package/index.js +206 -0
  4. package/package.json +97 -0
  5. package/skills/canary-test-staging/SKILL.md +24 -0
  6. package/skills/evo-mutation-rollback/SKILL.md +29 -0
  7. package/skills/hypothesis-generation/SKILL.md +26 -0
  8. package/skills/traceback-condensing/SKILL.md +26 -0
  9. package/src/castor_runner.js +469 -0
  10. package/src/config.js +1204 -0
  11. package/src/env.js +10 -0
  12. package/src/evo_engine.js +214 -0
  13. package/src/harness/core/events.js +75 -0
  14. package/src/harness/core/kernel.js +209 -0
  15. package/src/harness/evo/evaluator.js +156 -0
  16. package/src/harness/evo/evo_operator.js +550 -0
  17. package/src/harness/evo/lineage_dag.js +383 -0
  18. package/src/harness/evo/trace_repair.js +173 -0
  19. package/src/harness/evo/watchdog.js +72 -0
  20. package/src/harness/loop_detector.js +135 -0
  21. package/src/harness/runner.js +1216 -0
  22. package/src/harness/services/ast_service.js +1813 -0
  23. package/src/harness/services/event_logger.js +275 -0
  24. package/src/harness/services/mcp_bridge.js +408 -0
  25. package/src/harness/services/provider_vllm.js +728 -0
  26. package/src/harness/services/sandbox_fs.js +1238 -0
  27. package/src/harness/services/searxng_lifecycle.js +254 -0
  28. package/src/harness/services/shell_executor.js +264 -0
  29. package/src/harness/services/shell_validator.js +506 -0
  30. package/src/harness/services/web_service.js +828 -0
  31. package/src/platform.js +344 -0
  32. package/src/repetition_detector.js +139 -0
  33. package/src/semaphore.js +373 -0
  34. package/src/server_lifecycle.js +781 -0
  35. package/src/skills.js +400 -0
  36. package/src/state_pruner.js +392 -0
  37. package/src/task_registry.js +1357 -0
  38. package/src/telemetry.js +638 -0
  39. package/src/tools.js +997 -0
  40. package/src/wsl_bridge.js +629 -0
  41. package/src/wsl_env.js +171 -0
  42. package/stream_proxy.js +453 -0
@@ -0,0 +1,469 @@
1
+ import crypto from "node:crypto";
2
+ import {
3
+ IS_WINDOWS,
4
+ DEFAULT_TIMEOUT_MS,
5
+ MIN_TIMEOUT_MS,
6
+ EXTENSION_BONUS_TIMEOUT_MS,
7
+ MAX_TURNS,
8
+ BASE_TURN_BUDGET,
9
+ SESSION_TURNS_RECOMMEND,
10
+ MAX_LEN_HUGE,
11
+ getReasoningEffort,
12
+ } from "./config.js";
13
+ import {
14
+ isWslLocation,
15
+ isWindowsDrivePath,
16
+ toPosixWslPath,
17
+ toWindowsPath,
18
+ toMsys2Path,
19
+ canonicalizePath,
20
+ killSessionProcessTreeSync,
21
+ } from "./wsl_bridge.js";
22
+ import { posixShell } from "./platform.js";
23
+ import {
24
+ ensureServerRunning,
25
+ ensureStreamProxyRunning,
26
+ withBootMutex,
27
+ } from "./server_lifecycle.js";
28
+ import { runQueued, clearReclaimableTaskSlots } from "./semaphore.js";
29
+ import {
30
+ tasks,
31
+ saveTaskToDisk,
32
+ notifyWaiters,
33
+ } from "./task_registry.js";
34
+ import { EvoLineageEngine } from "./evo_engine.js";
35
+ import { CastorRunner } from "./harness/runner.js";
36
+ import { injectSkills } from "./skills.js";
37
+ import { recordTaskResult } from "./telemetry.js";
38
+
39
+ const READ_TOOLS = new Set(["read_file", "list_dir", "search_code", "ast_search"]);
40
+ const MUTATION_TOOLS = new Set([
41
+ "write_file",
42
+ "edit_file",
43
+ "apply_patch",
44
+ "ast_replace",
45
+ "ast_replace_batch",
46
+ "evo_propose_candidate",
47
+ "evo_select_candidate",
48
+ "evo_revert_candidate",
49
+ ]);
50
+ const COMMAND_TOOLS = new Set(["bash"]);
51
+ const WEB_TOOLS = new Set(["web_search", "web_fetch"]);
52
+
53
+ /**
54
+ * Pure predicate: does a runner status represent a successful task completion?
55
+ *
56
+ * The Castor runner reports an honest status taxonomy. A task is a SUCCESS when
57
+ * the model produced a complete deliverable:
58
+ * - "completed" — clean stop, never exhausted the continuation budget.
59
+ * - "completed_ceiling" — the model hit the token ceiling the maximum number
60
+ * of times (continuation budget exhausted) but still
61
+ * produced a complete, non-empty deliverable. This is
62
+ * a SUCCESS, not a failure: the work is done, it just
63
+ * ran out of room.
64
+ *
65
+ * Every other status is a FAILURE (isError = true):
66
+ * - "failed" — the runner's outer catch (a real upstream
67
+ * error thrown by the provider, a tool crash,
68
+ * or an aborted fetch).
69
+ * - "engine_empty_response" — the engine kept returning empty generations.
70
+ * - "reasoning_budget_exhausted"— the model burned the whole budget on
71
+ * thinking and emitted no visible content.
72
+ * - "length_limit_reached" — the model hit the ceiling and produced no
73
+ * usable content.
74
+ * - "degenerate_response_truncated" — the stream proxy circuit-broke a
75
+ * runaway repetition loop and the final
76
+ * message was just the guard marker (or a
77
+ * tiny sliver of text + marker) with no tool
78
+ * calls in a short session; the retry budget
79
+ * was exhausted. The original partial+marker
80
+ * is preserved in finalText for honesty.
81
+ * - "turn_limit_reached" — the maxTurns cap was hit.
82
+ * - "context_exhausted" — prompt context exceeded 245K token ceiling.
83
+ * - "aborted" — the client cancelled the run.
84
+ * - unknown / null / undefined — fail closed: treat as an error.
85
+ *
86
+ * This is the single source of truth for the status -> isError mapping used by
87
+ * the task-completion path in startCastorTask. It is a pure function so it can
88
+ * be unit-tested offline without spawning a live engine.
89
+ *
90
+ * @param {string|undefined|null} status
91
+ * @returns {boolean}
92
+ */
93
+ export function isSuccessStatus(status) {
94
+ return (
95
+ status === "completed" ||
96
+ status === "completed_ceiling" ||
97
+ status === "completed_budget_exhausted"
98
+ );
99
+ }
100
+
101
+ /**
102
+ * Assembles the orchestrator-facing result text with advisory metadata.
103
+ *
104
+ * Appends structured advisory banners when thresholds are reached:
105
+ * - Turn budget exhausted: warning banner advising review of partial deliverables.
106
+ * - Session turn limit reached: advisory note recommending context consolidation
107
+ * and the machine-readable SessionTurnLimitRecommendation marker.
108
+ *
109
+ * @param {string} finalText The runner's raw deliverable text.
110
+ * @param {number|undefined|null} sessionTurns Cumulative session turn count.
111
+ * @param {string|undefined|null} [status] Terminal status of the runner execution.
112
+ * @returns {string} Formatted result string for client consumption.
113
+ */
114
+ export function buildResultText(finalText, sessionTurns, status) {
115
+ let text = finalText;
116
+ if (status === "completed_budget_exhausted") {
117
+ text =
118
+ `${text}\n\n` +
119
+ `> [!WARNING] **Turn Limit Reached (Budget Exhausted)**\n` +
120
+ `> The coworker reached the maximum turn budget for this dispatch. All tools were disabled and a mandatory final synthesis was performed.\n` +
121
+ `> **Advisory:** Take this deliverable with a grain of salt. It represents grounded observations and partial progress gathered up to the turn limit, but may be incomplete or lack final verification. Review findings critically and dispatch a focused follow-up slice if needed.`;
122
+ }
123
+ if (
124
+ typeof sessionTurns === "number" &&
125
+ sessionTurns >= SESSION_TURNS_RECOMMEND
126
+ ) {
127
+ if (status !== "completed_budget_exhausted") {
128
+ text =
129
+ `${text}\n\n` +
130
+ `> [!NOTE] **High Turn Count Advisory (Turn ${sessionTurns})**\n` +
131
+ `> The coworker completed this deliverable in a session that has reached ${sessionTurns} cumulative turns (>= ${SESSION_TURNS_RECOMMEND}).\n` +
132
+ `> **Advisory:** As context depth grows, early consolidation can occur and speculative decoding degrades. Validate critical findings and roll to a fresh session_id before the next dispatch.`;
133
+ }
134
+ text =
135
+ `${text}\n` +
136
+ `[SessionTurnLimitRecommendation: session at ${sessionTurns} turns — roll to a fresh session_id before the next dispatch]`;
137
+ }
138
+ return text;
139
+ }
140
+
141
+ /**
142
+ * Permitted character set for session identifiers: [A-Za-z0-9._:-].
143
+ * Validates session IDs fail-fast to prevent shell metacharacter injection
144
+ * during process-tree management and process sweeps.
145
+ * @type {RegExp}
146
+ */
147
+ const SESSION_ID_CHARSET = /^[A-Za-z0-9._:-]+$/;
148
+
149
+ export function resolveSessionId(cwd, requestedSessionId) {
150
+ if (requestedSessionId && requestedSessionId.trim()) {
151
+ const id = requestedSessionId.trim();
152
+ if (!SESSION_ID_CHARSET.test(id)) {
153
+ throw new Error(
154
+ `Invalid session_id "${id}": session ids must match [A-Za-z0-9._:-]+ ` +
155
+ `(the charset the system generates, e.g. qwen_sh_<ts>_<rand>, ` +
156
+ `<milestone>_s1). Refusing to use a session id containing shell ` +
157
+ `metacharacters (quote, ;, backtick, space, newline, ...) to prevent ` +
158
+ `shell injection in the anchored process sweep.`
159
+ );
160
+ }
161
+ return id;
162
+ }
163
+ const hash = crypto.createHash("md5").update(cwd.toLowerCase()).digest("hex").slice(0, 8);
164
+ // Generates a unique default session identifier combining directory hash, process PID, and timestamp.
165
+ return `workspace_${hash}_${process.pid.toString(36)}_${Date.now().toString(36)}`;
166
+ }
167
+
168
+ export function startCastorTask({
169
+ cwd,
170
+ prompt,
171
+ sessionId,
172
+ extensions,
173
+ timeoutMs,
174
+ hypothesis,
175
+ testCommand,
176
+ metricName,
177
+ higherIsBetter,
178
+ skills,
179
+ reasoningEffort,
180
+ }) {
181
+ // Canonicalize task working directory through OS symlink and junction resolution
182
+ // to ensure consistent realpath evaluation across sandboxed services.
183
+ cwd = canonicalizePath(cwd);
184
+ const taskId = `task_${sessionId}_${Date.now()}`;
185
+ const baseTimeoutMs = Math.max(timeoutMs ?? DEFAULT_TIMEOUT_MS, MIN_TIMEOUT_MS);
186
+ const totalTimeoutMs =
187
+ baseTimeoutMs + (extensions && extensions.length ? EXTENSION_BONUS_TIMEOUT_MS : 0);
188
+
189
+ // Resolve task-level reasoning effort: explicit dispatch parameter takes precedence over environment default.
190
+ const effectiveReasoningEffort = reasoningEffort || getReasoningEffort();
191
+
192
+ const taskEntry = {
193
+ id: taskId,
194
+ sessionId,
195
+ cwd,
196
+ prompt,
197
+ reasoningEffort: effectiveReasoningEffort,
198
+ ownerPid: process.pid,
199
+ ownerPlatform: process.platform,
200
+ createdAt: Date.now(),
201
+ startedAt: null,
202
+ finishedAt: null,
203
+ lastActivityAt: null,
204
+ lastHeartbeatAt: Date.now(),
205
+ budgetTurns: BASE_TURN_BUDGET,
206
+ leaseExtensionsCount: 0,
207
+ lastActivityPreview: "",
208
+ toolOpsSummary: { reads: 0, mutations: 0, commands: 0, web: 0 },
209
+ lastTool: null,
210
+ streamBytes: 0,
211
+ streamTail: "",
212
+ status: "queued",
213
+ done: false,
214
+ isError: false,
215
+ child: null,
216
+ lines: [],
217
+ fileOps: [],
218
+ toolCallsCount: 0,
219
+ lastPromptTokens: null,
220
+ contextHeadroom: null,
221
+ result: null,
222
+ waiters: [],
223
+ };
224
+
225
+ tasks.set(taskId, taskEntry);
226
+ saveTaskToDisk(taskEntry);
227
+
228
+ const executionPromise = runQueued(
229
+ async () => {
230
+ if (taskEntry.done || taskEntry.status === "cancelled") {
231
+ return taskEntry.result ?? { isError: true, text: "Task was cancelled before execution." };
232
+ }
233
+ try {
234
+ await withBootMutex(async () => {
235
+ if (taskEntry.done || taskEntry.status === "cancelled") return;
236
+ await ensureServerRunning();
237
+ await ensureStreamProxyRunning();
238
+ });
239
+ if (taskEntry.done || taskEntry.status === "cancelled") {
240
+ return taskEntry.result ?? { isError: true, text: "Task was cancelled before dispatch." };
241
+ }
242
+ } catch (err) {
243
+ taskEntry.done = true;
244
+ taskEntry.isError = true;
245
+ taskEntry.status = "failed";
246
+ taskEntry.result = { isError: true, text: `Failed to boot model server: ${err.message}` };
247
+ try {
248
+ recordTaskResult({ isSuccess: false, effort: effectiveReasoningEffort });
249
+ } catch {}
250
+ notifyWaiters(taskEntry);
251
+ return taskEntry.result;
252
+ }
253
+
254
+ let lineageEngine = null;
255
+ let evoContext = null;
256
+ if (testCommand) {
257
+ try {
258
+ lineageEngine = new EvoLineageEngine(cwd);
259
+ evoContext = await lineageEngine.getLineageContext();
260
+ } catch {}
261
+ }
262
+
263
+ // Reconcile the shell path dialect between the system prompt and the
264
+ // executor. The executor (shell_executor.js) routes by the *actual*
265
+ // execution shell: WSL-native locations (/home/, /root/, ...) go to
266
+ // wsl.exe (WSL dialect /mnt/d/...), while Windows drive locations
267
+ // (D:\, /mnt/d/, /d/) go to Git Bash/MSYS2 (MSYS2 dialect /d/...) or
268
+ // cmd.exe (Windows path). The prompt's working directory must match the
269
+ // dialect the executor will actually use, or the model issues commands
270
+ // against a non-existent path (e.g. `mkdir /mnt/...` in Git Bash).
271
+ //
272
+ // WSL-native = a WSL location that is NOT a Windows drive mount.
273
+ const cwdInWsl = isWslLocation(cwd) && !isWindowsDrivePath(cwd);
274
+ // MCP extension routing (separate concern from the shell dialect):
275
+ // WSL-native, or forced WSL on a Windows host.
276
+ const targetInWsl = cwdInWsl || (IS_WINDOWS && process.env.QWEN_FORCE_WSL !== "0");
277
+ // Prompt/executor dialect: mirror the shell executor's routing decision
278
+ // (isWslLocation of the normalized cwd), NOT QWEN_FORCE_WSL.
279
+ let targetCwd;
280
+ if (cwdInWsl) {
281
+ // wsl.exe target: WSL dialect (/mnt/d/... for drives, /home/... for native).
282
+ targetCwd = toPosixWslPath(cwd);
283
+ } else if (IS_WINDOWS && posixShell()) {
284
+ // Git Bash / MSYS2 target: MSYS2 dialect (/d/... for drives).
285
+ targetCwd = toMsys2Path(cwd);
286
+ } else {
287
+ // cmd.exe target (no POSIX shell): Windows path (D:\...).
288
+ targetCwd = toWindowsPath(cwd);
289
+ }
290
+
291
+ // P9: keyword auto-inject matching skills from the packaged skills/
292
+ // library into the instruction block. Additive and budget-capped; when
293
+ // nothing matches the prompt is returned unchanged. Never throws.
294
+ const effectivePrompt = injectSkills(prompt, targetCwd, undefined, skills);
295
+
296
+ let finalTaskPrompt = `Your working directory is: ${targetCwd}\n\n`;
297
+ if (evoContext) {
298
+ finalTaskPrompt += `=== Evo Lineage Context ===\n${evoContext}\n\n`;
299
+ }
300
+ if (hypothesis) {
301
+ finalTaskPrompt += `=== Current Hypothesis ===\n${hypothesis}\n\n`;
302
+ }
303
+ finalTaskPrompt += effectivePrompt;
304
+
305
+ // Execute task within Castor microkernel and sandboxed service harness.
306
+ taskEntry.startedAt = Date.now();
307
+ taskEntry.status = "running";
308
+ saveTaskToDisk(taskEntry);
309
+
310
+ const runner = new CastorRunner({ cwd: targetCwd });
311
+ const abortController = new AbortController();
312
+ taskEntry.abortController = abortController;
313
+
314
+ const maxTurnsVal = taskEntry.budgetTurns || MAX_TURNS || BASE_TURN_BUDGET;
315
+ const heartbeatTimer = setInterval(() => {
316
+ if (taskEntry.done) {
317
+ clearInterval(heartbeatTimer);
318
+ return;
319
+ }
320
+ taskEntry.lastHeartbeatAt = Date.now();
321
+ saveTaskToDisk(taskEntry);
322
+ }, 5_000);
323
+ heartbeatTimer.unref();
324
+
325
+ try {
326
+ const runResult = await runner.run({
327
+ prompt: finalTaskPrompt,
328
+ cwd: targetCwd,
329
+ sessionId,
330
+ maxTurns: maxTurnsVal,
331
+ getDynamicBudget: () => taskEntry.budgetTurns || maxTurnsVal,
332
+ onActivity: (preview) => {
333
+ taskEntry.lastActivityPreview = preview;
334
+ taskEntry.lastHeartbeatAt = Date.now();
335
+ saveTaskToDisk(taskEntry);
336
+ },
337
+ // Pass dispatch-scoped reasoning effort to the model provider.
338
+ reasoningEffort,
339
+ signal: abortController.signal,
340
+ // Initialize stdio MCP extension servers on the execution harness.
341
+ extensions,
342
+ targetInWsl,
343
+ onToken: (tok) => {
344
+ taskEntry.lastActivityAt = Date.now();
345
+ taskEntry.lastHeartbeatAt = Date.now();
346
+ taskEntry.streamBytes = (taskEntry.streamBytes || 0) + tok.length;
347
+ taskEntry.streamTail = (taskEntry.streamTail + tok).slice(-4096);
348
+ },
349
+ onToolCall: (tc) => {
350
+ taskEntry.toolCallsCount = (taskEntry.toolCallsCount || 0) + 1;
351
+ taskEntry.fileOps.push(`${tc.name}:${tc.args?.path || ""}`);
352
+ if (!taskEntry.toolOpsSummary) {
353
+ taskEntry.toolOpsSummary = { reads: 0, mutations: 0, commands: 0, web: 0 };
354
+ }
355
+ if (READ_TOOLS.has(tc.name)) taskEntry.toolOpsSummary.reads++;
356
+ else if (MUTATION_TOOLS.has(tc.name)) taskEntry.toolOpsSummary.mutations++;
357
+ else if (COMMAND_TOOLS.has(tc.name)) taskEntry.toolOpsSummary.commands++;
358
+ else if (WEB_TOOLS.has(tc.name)) taskEntry.toolOpsSummary.web++;
359
+
360
+ let argSummary = "";
361
+ if (tc.args?.path) argSummary = String(tc.args.path);
362
+ else if (tc.args?.command) argSummary = String(tc.args.command).replace(/[\r\n]+/g, " ").slice(0, 80);
363
+ else if (tc.args?.query) argSummary = String(tc.args.query).slice(0, 60);
364
+ else if (tc.args?.url) argSummary = String(tc.args.url).slice(0, 60);
365
+
366
+ taskEntry.lastTool = {
367
+ name: tc.name,
368
+ summary: argSummary,
369
+ timestamp: Date.now(),
370
+ };
371
+ taskEntry.lastActivityAt = Date.now();
372
+ taskEntry.lastHeartbeatAt = Date.now();
373
+ saveTaskToDisk(taskEntry);
374
+ },
375
+ // E3: live context-headroom tracking. Each turn's promptTokens
376
+ // updates the task state so the status/wait endpoints can surface
377
+ // remaining headroom to the orchestrator in real time.
378
+ onMetrics: (m) => {
379
+ if (typeof m?.promptTokens === "number" && m.promptTokens > 0) {
380
+ taskEntry.lastPromptTokens = m.promptTokens;
381
+ taskEntry.contextHeadroom = Math.max(0, MAX_LEN_HUGE - m.promptTokens);
382
+ }
383
+ },
384
+ });
385
+
386
+ if (lineageEngine && testCommand) {
387
+ try {
388
+ await lineageEngine.recordCandidate({
389
+ hypothesis: hypothesis || "Variation candidate",
390
+ testCommand,
391
+ metricName,
392
+ higherIsBetter,
393
+ stdout: runResult.finalText,
394
+ });
395
+ } catch {}
396
+ }
397
+
398
+ const isSuccess = isSuccessStatus(runResult.status);
399
+ taskEntry.done = true;
400
+ taskEntry.finishedAt = Date.now();
401
+ taskEntry.status = runResult.status;
402
+ taskEntry.isError = !isSuccess;
403
+ // Propagate remaining context headroom telemetry to task state.
404
+ taskEntry.lastPromptTokens = runResult.lastPromptTokens ?? null;
405
+ taskEntry.contextHeadroom = runResult.contextHeadroom ?? null;
406
+
407
+ // Append session-rollover advisory to result text when cumulative turn thresholds are reached.
408
+ const resultText = buildResultText(runResult.finalText, runResult.sessionTurns, runResult.status);
409
+ taskEntry.result = {
410
+ isError: !isSuccess,
411
+ text: resultText,
412
+ toolCalls: taskEntry.toolCallsCount,
413
+ fileOps: (taskEntry.fileOps || []).slice(-5),
414
+ durationMs: runResult.durationMs,
415
+ };
416
+ try {
417
+ recordTaskResult({
418
+ isSuccess,
419
+ effort: effectiveReasoningEffort,
420
+ });
421
+ } catch {}
422
+ saveTaskToDisk(taskEntry);
423
+ notifyWaiters(taskEntry);
424
+ return taskEntry.result;
425
+ } catch (err) {
426
+ taskEntry.done = true;
427
+ taskEntry.finishedAt = Date.now();
428
+ taskEntry.status = "failed";
429
+ taskEntry.isError = true;
430
+ taskEntry.result = {
431
+ isError: true,
432
+ text: `Castor runner error: ${err.message}`,
433
+ toolCalls: taskEntry.toolCallsCount,
434
+ fileOps: taskEntry.fileOps,
435
+ };
436
+ try {
437
+ recordTaskResult({
438
+ isSuccess: false,
439
+ effort: effectiveReasoningEffort,
440
+ });
441
+ } catch {}
442
+ saveTaskToDisk(taskEntry);
443
+ notifyWaiters(taskEntry);
444
+ return taskEntry.result;
445
+ } finally {
446
+ clearInterval(heartbeatTimer);
447
+ // Run idempotent post-task cleanup: terminate session process trees and release stale slot leases.
448
+ try {
449
+ killSessionProcessTreeSync(sessionId);
450
+ } catch (err) {
451
+ const msg = err && err.message ? err.message : String(err);
452
+ console.error(`[castor_runner] session-end sweep failed for ${sessionId}: ${msg}`);
453
+ }
454
+ try {
455
+ clearReclaimableTaskSlots();
456
+ } catch (err) {
457
+ const msg = err && err.message ? err.message : String(err);
458
+ console.error(`[castor_runner] stale slot-lease cleanup failed: ${msg}`);
459
+ }
460
+ }
461
+ },
462
+ taskEntry,
463
+ (entry) => {
464
+ saveTaskToDisk(entry);
465
+ }
466
+ );
467
+
468
+ return { taskId, taskEntry, executionPromise, totalTimeoutMs };
469
+ }