mcp-castor 2026.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +487 -0
  2. package/bin/castor.js +706 -0
  3. package/index.js +206 -0
  4. package/package.json +97 -0
  5. package/skills/canary-test-staging/SKILL.md +24 -0
  6. package/skills/evo-mutation-rollback/SKILL.md +29 -0
  7. package/skills/hypothesis-generation/SKILL.md +26 -0
  8. package/skills/traceback-condensing/SKILL.md +26 -0
  9. package/src/castor_runner.js +469 -0
  10. package/src/config.js +1204 -0
  11. package/src/env.js +10 -0
  12. package/src/evo_engine.js +214 -0
  13. package/src/harness/core/events.js +75 -0
  14. package/src/harness/core/kernel.js +209 -0
  15. package/src/harness/evo/evaluator.js +156 -0
  16. package/src/harness/evo/evo_operator.js +550 -0
  17. package/src/harness/evo/lineage_dag.js +383 -0
  18. package/src/harness/evo/trace_repair.js +173 -0
  19. package/src/harness/evo/watchdog.js +72 -0
  20. package/src/harness/loop_detector.js +135 -0
  21. package/src/harness/runner.js +1216 -0
  22. package/src/harness/services/ast_service.js +1813 -0
  23. package/src/harness/services/event_logger.js +275 -0
  24. package/src/harness/services/mcp_bridge.js +408 -0
  25. package/src/harness/services/provider_vllm.js +728 -0
  26. package/src/harness/services/sandbox_fs.js +1238 -0
  27. package/src/harness/services/searxng_lifecycle.js +254 -0
  28. package/src/harness/services/shell_executor.js +264 -0
  29. package/src/harness/services/shell_validator.js +506 -0
  30. package/src/harness/services/web_service.js +828 -0
  31. package/src/platform.js +344 -0
  32. package/src/repetition_detector.js +139 -0
  33. package/src/semaphore.js +373 -0
  34. package/src/server_lifecycle.js +781 -0
  35. package/src/skills.js +400 -0
  36. package/src/state_pruner.js +392 -0
  37. package/src/task_registry.js +1357 -0
  38. package/src/telemetry.js +638 -0
  39. package/src/tools.js +997 -0
  40. package/src/wsl_bridge.js +629 -0
  41. package/src/wsl_env.js +171 -0
  42. package/stream_proxy.js +453 -0
package/src/tools.js ADDED
@@ -0,0 +1,997 @@
1
+ import { z } from "zod";
2
+ import http from "node:http";
3
+ import path from "node:path";
4
+ import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
5
+ import { normalizeObjectSchema } from "@modelcontextprotocol/sdk/server/zod-compat.js";
6
+ import { toJsonSchemaCompat } from "@modelcontextprotocol/sdk/server/zod-json-schema-compat.js";
7
+ import {
8
+ BASE_URL,
9
+ STATUS_PORT,
10
+ MAX_LEN_HUGE,
11
+ RACE_MS,
12
+ IS_WINDOWS,
13
+ MAX_CONCURRENT_TASKS,
14
+ QWEN_STATE_DIR,
15
+ TASK_DIR,
16
+ AUTO_HEAL,
17
+ WEDGE_STATS_SILENCE_S,
18
+ REASONING_EFFORT_TIERS,
19
+ TASK_RETENTION_MS,
20
+ PROMPT_BUDGET_CHARS,
21
+ ALLOW_ENGINE_INTERRUPT,
22
+ BASE_TURN_BUDGET,
23
+ MAX_ELASTIC_TURNS,
24
+ IS_TEST_ENV,
25
+ MODEL,
26
+ TOOL_PREFIX,
27
+ } from "./config.js";
28
+ import {
29
+ normalizeWorkspacePath,
30
+ canonicalizePath,
31
+ isIdeAppDirectory,
32
+ killProcessTree,
33
+ killSessionProcessTree,
34
+ toPosixWslPath,
35
+ toWindowsPath,
36
+ } from "./wsl_bridge.js";
37
+ import {
38
+ serverInfo,
39
+ readEngineMetrics,
40
+ ensureServerRunning,
41
+ stopServer,
42
+ resetEngineHealthCache,
43
+ engineWedgeState,
44
+ healWedgedEngine,
45
+ readWedgeCounter,
46
+ setHealGatekeeper,
47
+ } from "./server_lifecycle.js";
48
+ import { listTaskSlots, releaseTaskSlot, getSlotStatus } from "./semaphore.js";
49
+ import {
50
+ tasks,
51
+ listTasksFromDisk,
52
+ readTaskFromDisk,
53
+ saveTaskToDisk,
54
+ notifyWaiters,
55
+ cancelAllTasks,
56
+ extendTaskBudget,
57
+ statusServerOwned,
58
+ hasLiveWork,
59
+ } from "./task_registry.js";
60
+ import { startCastorTask, resolveSessionId } from "./castor_runner.js";
61
+ import { formatTelemetrySummary, recordTaskResult } from "./telemetry.js";
62
+
63
+ export function registerTools(server, options = {}) {
64
+ // Effective tool-name prefix: explicit option > MCP_TOOL_PREFIX env /
65
+ // ~/.castor/config.json tool_prefix > "qwen". An empty string yields the
66
+ // canonical unprefixed names (coworker, task, server).
67
+ const prefix = options.prefix ?? TOOL_PREFIX;
68
+ const toolName = (base) => (prefix ? `${prefix}_${base}` : base);
69
+
70
+ // Wire the heal backstop: refuse to stop/reboot the engine while live work
71
+ // is in flight. Uses the injection hook so server_lifecycle.js stays free of
72
+ // a hard dependency on task_registry.js (which has import-time side effects).
73
+ setHealGatekeeper(hasLiveWork);
74
+
75
+ // Tool 1: coworker (Primary Hybrid Agent Interface)
76
+ server.registerTool(
77
+ toolName("coworker"),
78
+ {
79
+ title: "Autonomous Senior Coworker (Castor Microkernel)",
80
+ description:
81
+ `Primary autonomous execution coworker for ${MODEL} via Castor microkernel harness ($0 execution). ` +
82
+ `Has full native access to Filesystem, Shell, and Git across Windows and WSL. Context window nominal ceiling: ${MAX_LEN_HUGE.toLocaleString()} tokens. ` +
83
+ "Executes codebase exploration, refactoring, implementation, diagnostics, live web/docs research, and git operations.\n\n" +
84
+ "ORCHESTRATION RULES:\n" +
85
+ " - Single Logical Concern: Scope each prompt to ONE cohesive subsystem, architectural layer, or target AST slice. Do not bundle disparate subsystems or cross-cutting concerns into a single dispatch.\n" +
86
+ " - Full Objective Fulfillment: Do not instruct the coworker to limit its tool calls or artificially restrict its execution. The coworker operates autonomously with full tool depth once dispatched with a focused objective.\n" +
87
+ " - Session Lifecycle: Maintain a persistent `session_id` across a cohesive milestone to maximize KV prefix caching. Roll to a fresh session_id (e.g. '<milestone>_stage2') upon milestone boundaries, session drift, or ~60–80 cumulative turns.\n" +
88
+ " - Zero-Turn Execution Contract: Tasks completing within ~45s return results synchronously. Long-running tasks yield a `taskId` and a `wait_command`. Execute the `wait_command` immediately in your shell to block at $0 cost and wake on completion. Do not poll manually or execute parallel exploratory tools while waiting.\n" +
89
+ " - Reasoning Effort: optional `reasoning_effort` param (xhigh | medium | low) tunes per-dispatch thinking depth; omit to use the CASTOR_REASONING_EFFORT / QWEN_REASONING_EFFORT env default (medium).\n\n" +
90
+ "BUILT-IN CAPABILITIES:\n" +
91
+ " - Built-in live Web Search & Article Extraction ('web_search', 'web_fetch' with Mozilla Readability & Turndown)\n" +
92
+ " - Transparent AST & Syntax Validation on edits ('edit_file' validates JS, TS, Python, JSON, LaTeX, BibTeX)\n" +
93
+ " - AST Search ('ast_search') and Git Unified Diffs ('apply_patch' with --unidiff-zero)\n" +
94
+ " - Shell Execution ('bash') with process group cleanup and git CLI integration",
95
+ inputSchema: {
96
+ prompt: z
97
+ .string()
98
+ .describe(
99
+ "Task, inquiry, or architectural instruction for Qwen (pure text-only; images must be inspected natively by Lead Architect and summarized into text)"
100
+ ),
101
+ session_id: z
102
+ .string()
103
+ .optional()
104
+ .describe(
105
+ "Named persistent session ID (maintains KV-cache and conversation context across turns)"
106
+ ),
107
+ cwd: z
108
+ .string()
109
+ .optional()
110
+ .describe(
111
+ "Working directory for filesystem and shell tools (defaults to current workspace)"
112
+ ),
113
+ extensions: z
114
+ .array(z.string())
115
+ .optional()
116
+ .describe(
117
+ "Optional stdio extensions (e.g. ['uvx free-search-mcp'], ['npx -y @upstash/context7-mcp'])"
118
+ ),
119
+ hypothesis: z.string().optional().describe("Optional Evo hypothesis being tested"),
120
+ test_command: z
121
+ .string()
122
+ .optional()
123
+ .describe("Optional verification test/benchmark command (e.g. 'pytest tests/test_core.py')"),
124
+ metric_name: z
125
+ .string()
126
+ .optional()
127
+ .describe("Target metric name in benchmark output (e.g. 'throughput', 'accuracy')"),
128
+ higher_is_better: z
129
+ .boolean()
130
+ .optional()
131
+ .describe("Whether higher metric values represent improvement (default true)"),
132
+ timeout_ms: z
133
+ .number()
134
+ .int()
135
+ .positive()
136
+ .optional()
137
+ .describe(
138
+ "Task timeout in ms (default 14,400,000ms (4 hours), minimum 600,000ms (10 min) - budgets are floored because a 27B model on consumer silicon routinely needs tens of minutes)"
139
+ ),
140
+ skills: z
141
+ .array(z.string())
142
+ .optional()
143
+ .describe("Explicit list of skill names to inject (bypasses keyword auto-matching)"),
144
+ reasoning_effort: z
145
+ .enum(REASONING_EFFORT_TIERS)
146
+ .optional()
147
+ .describe(
148
+ "Per-dispatch reasoning-effort tier forwarded to the engine chat template (xhigh = maximal deliberation, medium = balanced, low = brief). Omit to use the QWEN_REASONING_EFFORT env default (medium). Only the engine's supported tiers are accepted; invalid values are rejected."
149
+ ),
150
+ allow_large_prompt: z
151
+ .boolean()
152
+ .optional()
153
+ .describe(
154
+ "Explicit override allowing a prompt up to 2,500 chars when a detailed specification for a single slice is genuinely unavoidable. Prompts > 1,500 chars without this flag are rejected fail-fast to prevent monolithic runaway sessions."
155
+ ),
156
+ },
157
+ annotations: {
158
+ readOnlyHint: true,
159
+ },
160
+ },
161
+ async ({
162
+ prompt,
163
+ session_id,
164
+ cwd,
165
+ extensions,
166
+ hypothesis,
167
+ test_command,
168
+ metric_name,
169
+ higher_is_better,
170
+ timeout_ms,
171
+ skills,
172
+ reasoning_effort,
173
+ allow_large_prompt,
174
+ }) => {
175
+ let raceHandle;
176
+ try {
177
+ // Canonicalize working directory through OS symlink/junction layer to ensure realpath consistency.
178
+ if (!cwd && isIdeAppDirectory(process.cwd())) {
179
+ return {
180
+ content: [
181
+ {
182
+ type: "text",
183
+ text: `${toolName("coworker")}: 'cwd' parameter is required. The MCP server process was started from an IDE application directory (\`${process.cwd()}\`), which cannot be used as a project workspace. Please provide the target repository or directory path in 'cwd'.`,
184
+ },
185
+ ],
186
+ isError: true,
187
+ };
188
+ }
189
+
190
+ const isLargePromptAllowed = Boolean(allow_large_prompt);
191
+ const effectiveBudget = isLargePromptAllowed ? 2500 : PROMPT_BUDGET_CHARS;
192
+
193
+ if (typeof prompt === "string" && prompt.length > effectiveBudget) {
194
+ const errorDetail = isLargePromptAllowed
195
+ ? `Prompt is ${prompt.length} chars, exceeding the absolute maximum cap of 2,500 chars even with 'allow_large_prompt: true'. Point to files on disk and AST coordinates instead of inlining large content. DO NOT spoon-feed or paste verbatim code.`
196
+ : `Prompt is ${prompt.length} chars (budget: ${PROMPT_BUDGET_CHARS}). Your dispatch is oversized. ` +
197
+ `Per AGENTS.md / CLAUDE.md / GEMINI.md protocol rules (§3.1), you are strictly required to decompose tasks into single-concern slices rather than monolithic multi-milestone dumps. ` +
198
+ `DO NOT spoon-feed or paste verbatim code implementations—Qwen authors code locally. Point to files and AST coordinates. ` +
199
+ `(If a detailed specification is genuinely unavoidable for this single slice, pass 'allow_large_prompt: true' up to 2,500 chars).`;
200
+
201
+ return {
202
+ content: [
203
+ {
204
+ type: "text",
205
+ text: `MonolithicDispatchRejected: ${errorDetail}`,
206
+ },
207
+ ],
208
+ isError: true,
209
+ };
210
+ }
211
+
212
+ const workingDir = canonicalizePath(normalizeWorkspacePath(cwd ?? process.cwd()));
213
+
214
+ if (isIdeAppDirectory(workingDir)) {
215
+ return {
216
+ content: [
217
+ {
218
+ type: "text",
219
+ text: `${toolName("coworker")}: Refusing to use IDE application directory (\`${workingDir}\`) as workspace. Please specify a valid project directory in 'cwd'.`,
220
+ },
221
+ ],
222
+ isError: true,
223
+ };
224
+ }
225
+ const resolvedSession = resolveSessionId(workingDir, session_id);
226
+
227
+ if (process.env.TEST_OFFLINE === "1" || (!ALLOW_ENGINE_INTERRUPT && IS_TEST_ENV)) {
228
+ return {
229
+ content: [
230
+ {
231
+ type: "text",
232
+ text: `[offline_protected] task dispatch accepted without engine execution for prompt (${prompt.length} chars)`,
233
+ },
234
+ ],
235
+ isError: false,
236
+ };
237
+ }
238
+
239
+ const { taskId, taskEntry, executionPromise, totalTimeoutMs } = startCastorTask({
240
+ cwd: workingDir,
241
+ prompt,
242
+ sessionId: resolvedSession,
243
+ extensions,
244
+ timeoutMs: timeout_ms,
245
+ hypothesis,
246
+ testCommand: test_command,
247
+ metricName: metric_name,
248
+ higherIsBetter: higher_is_better,
249
+ skills,
250
+ reasoningEffort: reasoning_effort,
251
+ });
252
+
253
+ const raceTimer = new Promise((resolve) => {
254
+ raceHandle = setTimeout(() => resolve({ timedOutOnClientRace: true }), RACE_MS);
255
+ });
256
+ const winner = await Promise.race([executionPromise, raceTimer]);
257
+ clearTimeout(raceHandle);
258
+
259
+ if (!winner.timedOutOnClientRace) {
260
+ return {
261
+ content: [{ type: "text", text: winner.text }],
262
+ isError: winner.isError,
263
+ };
264
+ }
265
+
266
+ const waitCmdWin = `curl.exe -fsS --retry 5 --retry-delay 2 --retry-connrefused http://127.0.0.1:${STATUS_PORT}/task/${taskId}/wait`;
267
+ const waitCmdWsl = `curl -fsS --retry 5 --retry-delay 2 --retry-connrefused http://127.0.0.1:${STATUS_PORT}/task/${taskId}/wait`;
268
+ const elapsedSec = Math.round(RACE_MS / 1000);
269
+ const sessionEventsWin = path.join(QWEN_STATE_DIR, "sessions", resolvedSession, "events.jsonl");
270
+ const sessionEventsWsl = toPosixWslPath(sessionEventsWin);
271
+ const taskFileWin = path.join(TASK_DIR, `${taskId}.json`);
272
+ const taskFileWsl = toPosixWslPath(taskFileWin);
273
+
274
+ const slotStatus = getSlotStatus(process.pid);
275
+ const taskStatus = taskEntry.status;
276
+
277
+ let statusCallout = "";
278
+ if (taskStatus === "queued") {
279
+ if (slotStatus.isAlienActive) {
280
+ const alienPids = [...new Set(slotStatus.alienHolders.map((h) => h.pid))].join(", ");
281
+ const alienTasks = slotStatus.alienHolders.map((h) => `\`${h.taskId}\``).join(", ");
282
+ statusCallout = [
283
+ `> [!IMPORTANT]`,
284
+ `> **Qwen Engine Status: IN USE BY ANOTHER SESSION (QUEUED)**`,
285
+ `> Qwen is currently executing tasks for another active session (tenant PID: ${alienPids}; active task: ${alienTasks}).`,
286
+ `> **DO NOT PANIC, CANCEL, OR RETRY.** The engine enforces single-tenant multi-slot exclusivity (up to 2 concurrent slots for the active session) to protect GPU KV cache and preserve speculative decoding throughput.`,
287
+ `> Your task \`${taskId}\` is safely queued at $0 cost and will execute automatically the instant the other session yields. Run the wait command below to block until complete.`,
288
+ ].join("\n");
289
+ } else if (slotStatus.isSameActive && slotStatus.isFullyOccupied) {
290
+ const sameTasks = slotStatus.sameHolders.map((h) => `\`${h.taskId}\``).join(", ");
291
+ statusCallout = [
292
+ `> [!NOTE]`,
293
+ `> **Qwen Engine Status: SESSION CAPACITY REACHED (QUEUED)**`,
294
+ `> Both concurrent execution slots are actively running tasks from this session (${sameTasks}).`,
295
+ `> Your task \`${taskId}\` is safely queued and will start as soon as an active task finishes. Run the wait command below to block until complete.`,
296
+ ].join("\n");
297
+ } else {
298
+ statusCallout = [
299
+ `> [!NOTE]`,
300
+ `> **Qwen Engine Status: QUEUED FOR EXECUTION**`,
301
+ `> Task \`${taskId}\` is queued and awaiting slot grant. Run the wait command below to block until complete.`,
302
+ ].join("\n");
303
+ }
304
+ } else {
305
+ statusCallout = [
306
+ `> [!TIP]`,
307
+ `> **Qwen Engine Status: ACTIVELY EXECUTING**`,
308
+ `> Task \`${taskId}\` is actively running on the local ${MODEL} engine with full 245K context.`,
309
+ ].join("\n");
310
+ }
311
+
312
+ const responseText = [
313
+ `### Qwen Task Dispatched (Background Execution)`,
314
+ `- **Task ID**: \`${taskId}\` | **Session**: \`${resolvedSession}\` | **Status**: \`${taskStatus}\``,
315
+ `- **Working Directory**: \`${workingDir}\``,
316
+ statusCallout ? `${statusCallout}` : null,
317
+ `- **Wait Command**: \`${waitCmdWin}\` (WSL: \`${waitCmdWsl}\`)`,
318
+ `- **Status Command**: \`${toolName("task")}(action: "status", task_id: "${taskId}")\``,
319
+ ].filter(Boolean);
320
+
321
+ return {
322
+ content: [{ type: "text", text: responseText.join("\n") }],
323
+ isError: false,
324
+ };
325
+ } catch (err) {
326
+ // Return MCP tool-execution failure response with isError: true.
327
+ return {
328
+ content: [{ type: "text", text: `${toolName("coworker")}: ${err && err.message ? err.message : String(err)}` }],
329
+ isError: true,
330
+ };
331
+ }
332
+ }
333
+ );
334
+
335
+ // Tool 2: task (Unified Background Task Management)
336
+ server.registerTool(
337
+ toolName("task"),
338
+ {
339
+ title: "Manage Background Qwen Tasks",
340
+ description: "Check status, retrieve output, cancel, or list background Qwen coworker tasks.",
341
+ inputSchema: {
342
+ action: z
343
+ .enum(["status", "cancel", "cancel_all", "list", "kill", "stats", "extend_lease"])
344
+ .describe("Action to perform on background tasks (status, cancel, list, stats, or extend_lease to grant additional execution turns)"),
345
+ task_id: z
346
+ .string()
347
+ .optional()
348
+ .describe(
349
+ "Task ID (required for 'status' and 'extend_lease', optional for 'cancel'/'cancel_all'/'kill' to cancel all tasks)"
350
+ ),
351
+ turns: z
352
+ .number()
353
+ .int()
354
+ .positive()
355
+ .optional()
356
+ .describe("Additional turns to grant for 'extend_lease' (default 25)"),
357
+ reason: z
358
+ .string()
359
+ .optional()
360
+ .describe("Optional reason for supervisor lease extension"),
361
+ since: z
362
+ .string()
363
+ .optional()
364
+ .describe("Optional ISO-8601 start timestamp for time-sliced stats (e.g. '2026-09-28T00:00:00Z')"),
365
+ until: z
366
+ .string()
367
+ .optional()
368
+ .describe("Optional ISO-8601 end timestamp for time-sliced stats"),
369
+ window: z
370
+ .enum(["1h", "24h", "today", "yesterday", "all"])
371
+ .optional()
372
+ .describe("Optional relative time window for time-sliced stats"),
373
+ date: z
374
+ .string()
375
+ .optional()
376
+ .describe("Optional calendar date for time-sliced stats (YYYY-MM-DD)"),
377
+ hour: z
378
+ .number()
379
+ .int()
380
+ .min(0)
381
+ .max(23)
382
+ .optional()
383
+ .describe("Optional hour of the day (0-23) for time-sliced stats"),
384
+ minute: z
385
+ .number()
386
+ .int()
387
+ .min(0)
388
+ .max(59)
389
+ .optional()
390
+ .describe("Optional minute of the hour (0-59) for time-sliced stats"),
391
+ },
392
+ annotations: {
393
+ readOnlyHint: true,
394
+ },
395
+ },
396
+ async ({ action, task_id, turns, reason, since, until, window, date, hour, minute }) => {
397
+ try {
398
+ if (action === "kill") {
399
+ action = "cancel";
400
+ }
401
+ if (action === "stats") {
402
+ const filter = {};
403
+ if (since !== undefined) filter.since = since;
404
+ if (until !== undefined) filter.until = until;
405
+ if (window !== undefined) filter.window = window;
406
+ if (date !== undefined) filter.date = date;
407
+ if (hour !== undefined) filter.hour = hour;
408
+ if (minute !== undefined) filter.minute = minute;
409
+ const { summary } = formatTelemetrySummary(filter);
410
+ return {
411
+ content: [
412
+ {
413
+ type: "text",
414
+ text: summary,
415
+ },
416
+ ],
417
+ };
418
+ }
419
+ if (action === "list") {
420
+ const merged = new Map();
421
+ for (const dt of listTasksFromDisk()) {
422
+ merged.set(dt.id, {
423
+ id: dt.id,
424
+ sessionId: dt.sessionId,
425
+ status: dt.status,
426
+ createdAt: dt.createdAt,
427
+ elapsed_s: Math.round(((dt.finishedAt || Date.now()) - dt.createdAt) / 1000),
428
+ done: dt.done,
429
+ isError: dt.isError,
430
+ });
431
+ }
432
+ for (const t of tasks.values()) {
433
+ merged.set(t.id, {
434
+ id: t.id,
435
+ sessionId: t.sessionId,
436
+ status: t.status,
437
+ createdAt: t.createdAt,
438
+ elapsed_s: Math.round(((t.finishedAt || Date.now()) - t.createdAt) / 1000),
439
+ done: t.done,
440
+ isError: t.isError,
441
+ });
442
+ }
443
+ const all = Array.from(merged.values());
444
+ const active = all.filter((t) => !t.done);
445
+ const completed = all.filter((t) => t.done);
446
+ completed.sort((a, b) => (b.createdAt || 0) - (a.createdAt || 0));
447
+ // Keep active tasks + 5 most recent completed tasks to prevent massive context bloat
448
+ const recentCompleted = completed.slice(0, 5);
449
+ const payloadTasks = [...active, ...recentCompleted].map(({ createdAt, ...rest }) => rest);
450
+
451
+ return {
452
+ content: [
453
+ {
454
+ type: "text",
455
+ text: JSON.stringify(
456
+ {
457
+ tasks: payloadTasks,
458
+ active_count: active.length,
459
+ total_tasks: all.length,
460
+ archived_completed: Math.max(0, completed.length - recentCompleted.length),
461
+ },
462
+ null,
463
+ 2
464
+ ),
465
+ },
466
+ ],
467
+ };
468
+ }
469
+
470
+ if (action === "extend_lease") {
471
+ if (!task_id) {
472
+ return {
473
+ content: [{ type: "text", text: "Error: `task_id` parameter is required for action: 'extend_lease'." }],
474
+ isError: true,
475
+ };
476
+ }
477
+ const extResult = extendTaskBudget(task_id, turns || 25, reason);
478
+ if (!extResult.success) {
479
+ return {
480
+ content: [{ type: "text", text: `Failed to extend lease for task \`${task_id}\`: ${extResult.error}` }],
481
+ isError: true,
482
+ };
483
+ }
484
+ return {
485
+ content: [
486
+ {
487
+ type: "text",
488
+ text: `Granted ${turns || 25} additional turns to task \`${task_id}\` (budget: ${extResult.previousBudget} -> ${extResult.budgetTurns} turns, extension #${extResult.leaseExtensionsCount}).`,
489
+ },
490
+ ],
491
+ isError: false,
492
+ };
493
+ }
494
+
495
+ if (
496
+ action === "cancel_all" ||
497
+ (action === "cancel" && (!task_id || task_id.toLowerCase() === "all"))
498
+ ) {
499
+ const count = await cancelAllTasks("cancelled by caller");
500
+ return {
501
+ content: [
502
+ {
503
+ type: "text",
504
+ text: `Cancelled ${count} active/queued task(s), stopped execution, and cleared slot leases.`,
505
+ },
506
+ ],
507
+ };
508
+ }
509
+
510
+ if (!task_id) {
511
+ return {
512
+ content: [
513
+ {
514
+ type: "text",
515
+ text: `Error: \`task_id\` parameter is required for action: '${action}'.`,
516
+ },
517
+ ],
518
+ isError: true,
519
+ };
520
+ }
521
+
522
+ let task = tasks.get(task_id) || readTaskFromDisk(task_id);
523
+ if (task && task.corrupted) {
524
+ // Report corrupt task state with error details.
525
+ return {
526
+ content: [
527
+ {
528
+ type: "text",
529
+ text: `Task \`${task_id}\` file is CORRUPT (quarantined to ${task.file}): ${task.error}`,
530
+ },
531
+ ],
532
+ isError: true,
533
+ };
534
+ }
535
+ if (!task) {
536
+ return {
537
+ content: [
538
+ {
539
+ type: "text",
540
+ text: `Task \`${task_id}\` not found in memory or disk (retention is ${Math.round(TASK_RETENTION_MS / 3_600_000)}h).`,
541
+ },
542
+ ],
543
+ isError: true,
544
+ };
545
+ }
546
+
547
+ if (action === "status") {
548
+ // Compute elapsed execution or queued duration.
549
+ const elapsedS = Math.round(
550
+ ((task.finishedAt || Date.now()) - (task.startedAt || task.createdAt)) / 1000
551
+ );
552
+ if (task.done) {
553
+ const header = `[qwen task] id=${task.id} status=${task.status} elapsed_s=${elapsedS} isError=${task.isError}`;
554
+ return {
555
+ content: [{ type: "text", text: `${header}\n${task.result?.text || "Task completed."}` }],
556
+ isError: task.isError,
557
+ };
558
+ }
559
+ const waitCmdWin = `curl.exe -fsS --retry 5 --retry-delay 2 --retry-connrefused http://127.0.0.1:${STATUS_PORT}/task/${task_id}/wait`;
560
+ const waitCmdWsl = `curl -fsS --retry 5 --retry-delay 2 --retry-connrefused http://127.0.0.1:${STATUS_PORT}/task/${task_id}/wait`;
561
+ const hint = `\n\nWait: \`${waitCmdWin}\` (WSL: \`${waitCmdWsl}\`)`;
562
+
563
+ if (task.status === "queued") {
564
+ const slotStatus = getSlotStatus(process.pid);
565
+ let queueDiagnosis = "";
566
+ if (slotStatus.isAlienActive) {
567
+ const alienPids = [...new Set(slotStatus.alienHolders.map((h) => h.pid))].join(", ");
568
+ const alienTasks = slotStatus.alienHolders.map((h) => `\`${h.taskId}\``).join(", ");
569
+ queueDiagnosis = ` Qwen is currently executing tasks for another active session (tenant PID: ${alienPids}; active task: ${alienTasks}). Your task is queued and will execute automatically when the slot yields.`;
570
+ } else if (slotStatus.isSameActive && slotStatus.isFullyOccupied) {
571
+ const sameTasks = slotStatus.sameHolders.map((h) => `\`${h.taskId}\``).join(", ");
572
+ queueDiagnosis = ` Both execution slots are actively running tasks from this session (${sameTasks}). Your task will run as soon as one completes.`;
573
+ } else {
574
+ queueDiagnosis = ` Awaiting execution slot grant (MAX_CONCURRENT_TASKS=${MAX_CONCURRENT_TASKS} machine-wide).`;
575
+ }
576
+
577
+ return {
578
+ content: [
579
+ {
580
+ type: "text",
581
+ text: `Task \`${task_id}\` is QUEUED (${elapsedS}s waiting).${queueDiagnosis}${hint}`,
582
+ },
583
+ ],
584
+ isError: false,
585
+ };
586
+ }
587
+ const lastActiveSec = task.lastHeartbeatAt
588
+ ? Math.max(0, Math.round((Date.now() - task.lastHeartbeatAt) / 1000))
589
+ : null;
590
+ const livenessStr = lastActiveSec !== null ? `${lastActiveSec}s ago` : "active";
591
+
592
+ const ops = task.toolOpsSummary || {
593
+ reads: 0,
594
+ mutations: 0,
595
+ commands: 0,
596
+ web: 0,
597
+ };
598
+ const opsSummary = `reads: ${ops.reads}, mutations: ${ops.mutations}, commands: ${ops.commands}${ops.web > 0 ? `, web: ${ops.web}` : ""}`;
599
+
600
+ let lastActionDetail = "";
601
+ if (task.lastTool?.name) {
602
+ lastActionDetail = `\n- Last action: \`${task.lastTool.name}\`${task.lastTool.summary ? ` (\`${task.lastTool.summary}\`)` : ""} (${livenessStr})`;
603
+ } else {
604
+ lastActionDetail = `\n- Last heartbeat: ${livenessStr}`;
605
+ }
606
+
607
+ const budgetInfo = `budget: ${task.budgetTurns || BASE_TURN_BUDGET} turns (extensions: ${task.leaseExtensionsCount || 0})`;
608
+ let activityBlock = "";
609
+ if (task.lastActivityPreview) {
610
+ const previewClean = String(task.lastActivityPreview)
611
+ .replace(/["\\{}\[\]]|type|message|content|delta|thinking|text|exitCode|timedOut|latencyMs/g, " ")
612
+ .replace(/\s+/g, " ")
613
+ .trim()
614
+ .slice(-150);
615
+ if (previewClean) {
616
+ activityBlock = `\n- Activity: > ${previewClean}`;
617
+ }
618
+ }
619
+
620
+ let steeringAdvisory = "";
621
+ if (ops.commands >= 4 && ops.mutations === 0) {
622
+ steeringAdvisory = `\n\n> [!NOTE]\n> Consecutive shell commands observed without mutations. The coworker may be running exploration, diagnostics, or test suites.`;
623
+ }
624
+
625
+ return {
626
+ content: [
627
+ {
628
+ type: "text",
629
+ text: `Task \`${task_id}\` is actively EXECUTING (${elapsedS}s elapsed, ${task.toolCallsCount || 0} tool calls made [${opsSummary}], ${budgetInfo}).${lastActionDetail}${activityBlock}${steeringAdvisory}${hint}`,
630
+ },
631
+ ],
632
+ isError: false,
633
+ };
634
+ }
635
+
636
+ if (action === "cancel") {
637
+ const memTask = tasks.get(task_id);
638
+ if (memTask && !memTask.done) {
639
+ killProcessTree(memTask.child, memTask.sessionId);
640
+ // Signal abort to stop in-flight execution and clean up bridged tools.
641
+ if (memTask.abortController) {
642
+ try {
643
+ memTask.abortController.abort();
644
+ } catch {}
645
+ }
646
+ // Release task slot lease immediately upon cancellation.
647
+ if (memTask.slot) {
648
+ releaseTaskSlot(memTask.slot);
649
+ memTask.slot = null;
650
+ }
651
+ memTask.status = "cancelled";
652
+ memTask.done = true;
653
+ memTask.isError = true;
654
+ memTask.result = { isError: true, text: `Task ${task_id} was cancelled by caller.` };
655
+ try {
656
+ recordTaskResult({ isSuccess: false, isCancelled: true, effort: memTask.reasoningEffort });
657
+ } catch {}
658
+ saveTaskToDisk(memTask);
659
+ notifyWaiters(memTask);
660
+ return {
661
+ content: [{ type: "text", text: `Task \`${task_id}\` cancelled and process tree killed.` }],
662
+ };
663
+ }
664
+
665
+ // Delegate cancellation to the status coordinator process if not owned locally
666
+ try {
667
+ const httpCancel = await new Promise((resolve) => {
668
+ const postReq = http.request(
669
+ {
670
+ hostname: "127.0.0.1",
671
+ port: STATUS_PORT,
672
+ path: `/task/${encodeURIComponent(task_id)}/cancel`,
673
+ method: "POST",
674
+ timeout: 4000,
675
+ },
676
+ (res) => {
677
+ let data = "";
678
+ res.on("data", (chunk) => (data += chunk));
679
+ res.on("end", () => {
680
+ try {
681
+ resolve(JSON.parse(data));
682
+ } catch {
683
+ resolve(null);
684
+ }
685
+ });
686
+ }
687
+ );
688
+ postReq.on("error", () => resolve(null));
689
+ postReq.on("timeout", () => {
690
+ postReq.destroy();
691
+ resolve(null);
692
+ });
693
+ postReq.end();
694
+ });
695
+ if (httpCancel?.cancelled) {
696
+ return {
697
+ content: [
698
+ { type: "text", text: `Task \`${task_id}\` cancelled via status coordinator.` },
699
+ ],
700
+ };
701
+ }
702
+ } catch {}
703
+
704
+ const diskTask = readTaskFromDisk(task_id);
705
+ if (diskTask && diskTask.corrupted) {
706
+ // Report corrupt task state with error details.
707
+ // "already finished".
708
+ return {
709
+ content: [
710
+ {
711
+ type: "text",
712
+ text: `Task \`${task_id}\` file is CORRUPT (quarantined to ${diskTask.file}): ${diskTask.error}`,
713
+ },
714
+ ],
715
+ isError: true,
716
+ };
717
+ }
718
+ if (diskTask && !diskTask.done) {
719
+ if (diskTask.sessionId) {
720
+ // Terminate child processes matching exact session ID boundaries.
721
+ try {
722
+ await killSessionProcessTree(diskTask.sessionId);
723
+ } catch {}
724
+ }
725
+ diskTask.status = "cancelled";
726
+ diskTask.done = true;
727
+ diskTask.isError = true;
728
+ diskTask.result = { isError: true, text: `Task ${task_id} was cancelled by caller.` };
729
+ saveTaskToDisk(diskTask);
730
+ return {
731
+ content: [{ type: "text", text: `Task \`${task_id}\` marked as cancelled.` }],
732
+ };
733
+ }
734
+ return {
735
+ content: [{ type: "text", text: `Task \`${task_id}\` was already finished.` }],
736
+ };
737
+ }
738
+ } catch (err) {
739
+ // Return MCP tool-execution failure response with isError: true.
740
+ return {
741
+ content: [{ type: "text", text: `${toolName("task")}: ${err && err.message ? err.message : String(err)}` }],
742
+ isError: true,
743
+ };
744
+ }
745
+ }
746
+ );
747
+
748
+ // Tool 3: server (Unified Server Lifecycle)
749
+ server.registerTool(
750
+ toolName("server"),
751
+ {
752
+ title: `Manage Local ${MODEL} vLLM Instance Lifecycle`,
753
+ description:
754
+ "Check status, start, or stop the universal 245K context vLLM server in WSL Ubuntu. Status includes live engine gauges from /metrics (running/waiting requests, KV cache %, prefix-cache hit ratio, spec-decode acceptance) and an end-to-end canary completion - the port answering is NOT proof of health. Note: during active task execution, canary latency will be higher due to GPU batch contention; this is normal under load and is NOT a wedge. Only stop the server if the engine is idle or if the human user explicitly commands it.",
755
+ inputSchema: {
756
+ action: z.enum(["status", "start", "stop"]).describe("Lifecycle action to perform"),
757
+ force: z
758
+ .boolean()
759
+ .optional()
760
+ .describe(
761
+ "Force stop even if a task is actively executing. ONLY permitted if the human USER explicitly requested stopping/rebooting the server or cancelling all tasks. Prohibited for autonomous agent decisions."
762
+ ),
763
+ },
764
+ annotations: {
765
+ readOnlyHint: true,
766
+ },
767
+ },
768
+ async ({ action, force }) => {
769
+ try {
770
+ if (action === "status") {
771
+ if (process.env.TEST_OFFLINE === "1" || (!ALLOW_ENGINE_INTERRUPT && process.env.NODE_ENV === "test")) {
772
+ return {
773
+ content: [
774
+ {
775
+ type: "text",
776
+ text: JSON.stringify(
777
+ {
778
+ status: "stopped",
779
+ endpoint: BASE_URL,
780
+ max_model_len: null,
781
+ context_window_nominal: MAX_LEN_HUGE,
782
+ stack: "vLLM + DFlash2 + KVarN (Universal 245K)",
783
+ engine: null,
784
+ status_endpoint: `http://127.0.0.1:${STATUS_PORT}`,
785
+ status_endpoint_owned_by_this_instance: statusServerOwned,
786
+ wedge_counter: readWedgeCounter(),
787
+ offline_protected: true,
788
+ },
789
+ null,
790
+ 2
791
+ ),
792
+ },
793
+ ],
794
+ };
795
+ }
796
+ const info = await serverInfo();
797
+ const running = !!info;
798
+ const wedge = running ? await engineWedgeState() : { wedged: false, stats: null };
799
+ let autoHeal = null;
800
+ if (running && wedge.wedged && AUTO_HEAL) {
801
+ try {
802
+ autoHeal = await healWedgedEngine(wedge.stats?.ageSec ?? null);
803
+ } catch (err) {
804
+ autoHeal = { healed: false, error: err.message };
805
+ }
806
+ }
807
+ const statusLabel = !running
808
+ ? "stopped"
809
+ : wedge.wedged
810
+ ? autoHeal?.healed
811
+ ? "wedged_restarted"
812
+ : "wedged"
813
+ : "running";
814
+ return {
815
+ content: [
816
+ {
817
+ type: "text",
818
+ text: JSON.stringify(
819
+ {
820
+ status: statusLabel,
821
+ endpoint: BASE_URL,
822
+ max_model_len: running ? info.maxModelLen : null,
823
+ context_window_nominal: MAX_LEN_HUGE,
824
+ stack: "vLLM + DFlash2 + KVarN (Universal 245K)",
825
+ engine: running
826
+ ? {
827
+ running_requests: wedge.gauges?.running_requests ?? null,
828
+ waiting_requests: wedge.gauges?.waiting_requests ?? null,
829
+ kv_cache_pct: wedge.gauges?.kv_cache_pct ?? null,
830
+ prefix_cache_hit_ratio: wedge.gauges?.prefix_cache_hit_ratio ?? null,
831
+ spec_decode_acceptance: wedge.gauges?.spec_decode_acceptance ?? null,
832
+ // BUSY-GATE: when the engine is busy (MAX_SEQS=1), the canary
833
+ // is intentionally NOT fired (it would queue behind the active
834
+ // generation and time out, measuring queue depth not health).
835
+ // Surface that honestly; wedge is then derived from stats
836
+ // silence only.
837
+ engine_busy: wedge.engineBusy ?? null,
838
+ canary_skipped: wedge.canary?.skipped ?? null,
839
+ canary: wedge.canary,
840
+ engine_stats_age_seconds: wedge.stats?.ageSec ?? null,
841
+ wedge_detected: wedge.wedged,
842
+ wedge_threshold_seconds: WEDGE_STATS_SILENCE_S,
843
+ auto_heal: autoHeal,
844
+ }
845
+ : null,
846
+ status_endpoint: `http://127.0.0.1:${STATUS_PORT}`,
847
+ status_endpoint_owned_by_this_instance: statusServerOwned,
848
+ wedge_counter: readWedgeCounter(),
849
+ },
850
+ null,
851
+ 2
852
+ ),
853
+ },
854
+ ],
855
+ };
856
+ }
857
+ if (action === "start") {
858
+ if (process.env.TEST_OFFLINE === "1" || (!ALLOW_ENGINE_INTERRUPT && process.env.NODE_ENV === "test")) {
859
+ return {
860
+ content: [
861
+ {
862
+ type: "text",
863
+ text: JSON.stringify(
864
+ { status: "stopped", switched: false, note: "offline_protected: start refused without ALLOW_ENGINE_INTERRUPT=1" },
865
+ null,
866
+ 2
867
+ ),
868
+ },
869
+ ],
870
+ };
871
+ }
872
+ const res = await ensureServerRunning();
873
+ return {
874
+ content: [
875
+ {
876
+ type: "text",
877
+ text: JSON.stringify(
878
+ { status: "running", result: res.status, endpoint: BASE_URL, context: MAX_LEN_HUGE },
879
+ null,
880
+ 2
881
+ ),
882
+ },
883
+ ],
884
+ };
885
+ }
886
+ if (action === "stop") {
887
+ if (process.env.TEST_OFFLINE === "1" || (IS_TEST_ENV && !ALLOW_ENGINE_INTERRUPT)) {
888
+ return {
889
+ content: [
890
+ {
891
+ type: "text",
892
+ text: JSON.stringify(
893
+ { stopped: false, reason: "stop_refused_offline_protected", note: "stop refused: engine interruption disabled by default (requires ALLOW_ENGINE_INTERRUPT=1 and explicit user approval)" },
894
+ null,
895
+ 2
896
+ ),
897
+ },
898
+ ],
899
+ };
900
+ }
901
+ const activeTasks = listTasksFromDisk().filter((t) => !t.done && t.status === "executing");
902
+ if (activeTasks.length > 0 && !force) {
903
+ return {
904
+ content: [
905
+ {
906
+ type: "text",
907
+ text: JSON.stringify(
908
+ {
909
+ status: "rejected",
910
+ error: `Refusing to stop vLLM server: task '${activeTasks[0].id}' is actively executing.`,
911
+ guidance:
912
+ `To cancel the active task without rebooting vLLM, call ${toolName("task")}(action: 'cancel', task_id: '` +
913
+ activeTasks[0].id +
914
+ "'). Only pass force: true to stop the server if the human USER explicitly commanded stopping the server or cancelling all tasks.",
915
+ },
916
+ null,
917
+ 2
918
+ ),
919
+ },
920
+ ],
921
+ };
922
+ }
923
+ await cancelAllTasks("server stopped by user");
924
+ // Surface actual stop operation status and diagnostics.
925
+ const stopRes = await stopServer();
926
+ resetEngineHealthCache();
927
+ return {
928
+ content: [
929
+ {
930
+ type: "text",
931
+ text: JSON.stringify(
932
+ stopRes.stopped
933
+ ? {
934
+ status: "stopped",
935
+ message: "vLLM server stopped and all active/queued tasks cancelled.",
936
+ forced: !!force,
937
+ }
938
+ : {
939
+ status: "stop_failed",
940
+ message:
941
+ "All active/queued tasks were cancelled, but the vLLM engine is still responding after the stop grace window.",
942
+ reason: stopRes.reason,
943
+ forced: !!force,
944
+ },
945
+ null,
946
+ 2
947
+ ),
948
+ },
949
+ ],
950
+ isError: !stopRes.stopped,
951
+ };
952
+ }
953
+ } catch (err) {
954
+ // Return MCP tool-execution failure response with isError: true.
955
+ return {
956
+ content: [{ type: "text", text: `${toolName("server")}: ${err && err.message ? err.message : String(err)}` }],
957
+ isError: true,
958
+ };
959
+ }
960
+ }
961
+ );
962
+ }
963
+
964
+ /**
965
+ * getToolManifest() — read-only introspection of the registered tool surface.
966
+ *
967
+ * Builds a throwaway McpServer, registers the real tools against it, and
968
+ * converts each registered tool's zod `inputSchema` into a JSON Schema using
969
+ * the EXACT same code path the SDK uses when it serves `tools/list`
970
+ * (normalizeObjectSchema -> toJsonSchemaCompat with strictUnions/pipeStrategy
971
+ * 'input'). This guarantees the manifest is byte-identical to what a client
972
+ * receives over the wire, without touching or mutating any live registration.
973
+ *
974
+ * @param {object} [options]
975
+ * @param {string} [options.prefix] Tool-name prefix override (defaults to
976
+ * TOOL_PREFIX, i.e. MCP_TOOL_PREFIX env / ~/.castor/config.json tool_prefix /
977
+ * "qwen"). An empty string yields the canonical unprefixed names.
978
+ * @returns {Array<{ name: string, description: string, inputSchema: object }>}
979
+ */
980
+ export function getToolManifest(options = {}) {
981
+ const server = new McpServer({ name: "manifest-probe", version: "0.0.0" });
982
+ registerTools(server, options);
983
+
984
+ return Object.entries(server._registeredTools)
985
+ .filter(([, tool]) => tool.enabled)
986
+ .map(([name, tool]) => {
987
+ const obj = normalizeObjectSchema(tool.inputSchema);
988
+ const inputSchema = obj
989
+ ? toJsonSchemaCompat(obj, { strictUnions: true, pipeStrategy: "input" })
990
+ : { type: "object", properties: {} };
991
+ return {
992
+ name,
993
+ description: tool.description,
994
+ inputSchema,
995
+ };
996
+ });
997
+ }