mcp-castor 2026.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +487 -0
  2. package/bin/castor.js +706 -0
  3. package/index.js +206 -0
  4. package/package.json +97 -0
  5. package/skills/canary-test-staging/SKILL.md +24 -0
  6. package/skills/evo-mutation-rollback/SKILL.md +29 -0
  7. package/skills/hypothesis-generation/SKILL.md +26 -0
  8. package/skills/traceback-condensing/SKILL.md +26 -0
  9. package/src/castor_runner.js +469 -0
  10. package/src/config.js +1204 -0
  11. package/src/env.js +10 -0
  12. package/src/evo_engine.js +214 -0
  13. package/src/harness/core/events.js +75 -0
  14. package/src/harness/core/kernel.js +209 -0
  15. package/src/harness/evo/evaluator.js +156 -0
  16. package/src/harness/evo/evo_operator.js +550 -0
  17. package/src/harness/evo/lineage_dag.js +383 -0
  18. package/src/harness/evo/trace_repair.js +173 -0
  19. package/src/harness/evo/watchdog.js +72 -0
  20. package/src/harness/loop_detector.js +135 -0
  21. package/src/harness/runner.js +1216 -0
  22. package/src/harness/services/ast_service.js +1813 -0
  23. package/src/harness/services/event_logger.js +275 -0
  24. package/src/harness/services/mcp_bridge.js +408 -0
  25. package/src/harness/services/provider_vllm.js +728 -0
  26. package/src/harness/services/sandbox_fs.js +1238 -0
  27. package/src/harness/services/searxng_lifecycle.js +254 -0
  28. package/src/harness/services/shell_executor.js +264 -0
  29. package/src/harness/services/shell_validator.js +506 -0
  30. package/src/harness/services/web_service.js +828 -0
  31. package/src/platform.js +344 -0
  32. package/src/repetition_detector.js +139 -0
  33. package/src/semaphore.js +373 -0
  34. package/src/server_lifecycle.js +781 -0
  35. package/src/skills.js +400 -0
  36. package/src/state_pruner.js +392 -0
  37. package/src/task_registry.js +1357 -0
  38. package/src/telemetry.js +638 -0
  39. package/src/tools.js +997 -0
  40. package/src/wsl_bridge.js +629 -0
  41. package/src/wsl_env.js +171 -0
  42. package/stream_proxy.js +453 -0
package/index.js ADDED
@@ -0,0 +1,206 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Castor - Unified Local Model Agent Harness & MCP Server (version: package.json)
4
+ *
5
+ * Architecture:
6
+ * - Lead Architect: Claude 5 Sonnet in Claude Code / Gemini 3.8 Flash in Antigravity
7
+ * - Autonomous Execution Coworker: Qwen3.8-27B via Castor Microkernel Harness ($0 text-only execution)
8
+ * - Serving: Universal 245K context (vLLM + DFlash2 + KVarN @ localhost:18020)
9
+ * - Zero-Turn Async Architecture: Blocking Long-Poll HTTP Coordinator (localhost:18021)
10
+ * - 3 Consolidated SOTA Tools: qwen_coworker, qwen_task, qwen_server
11
+ * - Modularized Clean Architecture (src/)
12
+ */
13
+
14
+ import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
15
+ import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
16
+ import { createRequire } from "node:module";
17
+ import fs from "node:fs";
18
+ import path from "node:path";
19
+ import { fileURLToPath } from "node:url";
20
+ import { MAX_CONCURRENT_TASKS, TASK_DIR } from "./src/config.js";
21
+ import { killProcessTree, killProcessTreeSync } from "./src/wsl_bridge.js";
22
+ import {
23
+ acquireTaskSlot,
24
+ releaseTaskSlot,
25
+ listTaskSlots,
26
+ getSlotStatus,
27
+ slotFilePath,
28
+ readLease,
29
+ } from "./src/semaphore.js";
30
+ import {
31
+ tasks,
32
+ saveTaskToDisk,
33
+ notifyWaiters,
34
+ isTaskOrphaned,
35
+ markTaskOrphanedOnDisk,
36
+ initStatusServer,
37
+ setMcpServerFactory,
38
+ activeSseSessions,
39
+ } from "./src/task_registry.js";
40
+ import { registerTools } from "./src/tools.js";
41
+ import { shutdownSearxng } from "./src/harness/services/searxng_lifecycle.js";
42
+ import { disposeAllBridges } from "./src/harness/services/mcp_bridge.js";
43
+ import { cleanStateDir } from "./src/state_pruner.js";
44
+
45
+ const require = createRequire(import.meta.url);
46
+ const { name: pkgName, version: pkgVersion } = require("./package.json");
47
+ const __filename = fileURLToPath(import.meta.url);
48
+
49
+ function setupProcessLifecycleHandlers() {
50
+ const cleanup = (signal) => {
51
+ try {
52
+ // P8: reap any live MCP extension bridge children (native engine)
53
+ // so none survive process shutdown.
54
+ disposeAllBridges();
55
+ for (const session of activeSseSessions.values()) {
56
+ try {
57
+ session.transport.close();
58
+ session.server.close();
59
+ } catch {}
60
+ }
61
+ activeSseSessions.clear();
62
+ for (const [id, task] of tasks.entries()) {
63
+ if (!task.done) {
64
+ task.done = true;
65
+ task.status = "cancelled";
66
+ task.isError = true;
67
+ task.finishedAt = Date.now();
68
+ task.result = {
69
+ isError: true,
70
+ text: `Task cancelled: MCP server process terminating (${signal || "shutdown"}).`,
71
+ toolCalls: task.toolCallsCount || 0,
72
+ errors: ["SERVER_PROCESS_TERMINATED"],
73
+ fileOps: task.fileOps || [],
74
+ };
75
+ saveTaskToDisk(task);
76
+ notifyWaiters(task);
77
+ if (task.child) {
78
+ killProcessTreeSync(task.child, task.sessionId);
79
+ }
80
+ }
81
+ }
82
+ for (let i = 0; i < MAX_CONCURRENT_TASKS; i++) {
83
+ const file = slotFilePath(i);
84
+ const lease = readLease(file);
85
+ if (lease && lease.pid === process.pid) {
86
+ try {
87
+ fs.rmSync(file, { force: true });
88
+ } catch {}
89
+ }
90
+ }
91
+ try {
92
+ shutdownSearxng().catch(() => {});
93
+ } catch {}
94
+ } catch {}
95
+ };
96
+
97
+ process.once("SIGINT", () => {
98
+ cleanup("SIGINT");
99
+ process.exit(0);
100
+ });
101
+ process.once("SIGTERM", () => {
102
+ cleanup("SIGTERM");
103
+ process.exit(0);
104
+ });
105
+ process.stdin.on("close", () => {
106
+ cleanup("stdin_closed");
107
+ process.exit(0);
108
+ });
109
+ process.stdin.on("end", () => {
110
+ cleanup("stdin_end");
111
+ process.exit(0);
112
+ });
113
+ process.stdout.on("error", (err) => {
114
+ if (err && (err.code === "EPIPE" || err.code === "ERR_STREAM_DESTROYED")) {
115
+ cleanup("stdout_epipe");
116
+ process.exit(0);
117
+ }
118
+ });
119
+ process.stderr.on("error", (err) => {
120
+ if (err && (err.code === "EPIPE" || err.code === "ERR_STREAM_DESTROYED")) {
121
+ process.exit(0);
122
+ }
123
+ });
124
+ process.on("beforeExit", () => {
125
+ cleanup("beforeExit");
126
+ });
127
+ }
128
+
129
+ function createMcpServer() {
130
+ const server = new McpServer({
131
+ name: pkgName ?? "qwen38-local",
132
+ version: pkgVersion,
133
+ });
134
+ registerTools(server);
135
+ return server;
136
+ }
137
+
138
+ setMcpServerFactory(createMcpServer);
139
+
140
+ /**
141
+ * Starts the Castor MCP server on the stdio transport.
142
+ *
143
+ * Exported so that `bin/castor.js` (the packaged CLI entrypoint) can import
144
+ * and invoke it, while direct execution of `index.js` still works as before.
145
+ *
146
+ * Stdio purity contract: this function must never write to stdout except
147
+ * through the MCP transport (JSON-RPC frames). All diagnostics go to stderr.
148
+ */
149
+ export async function startMcpServer() {
150
+ setupProcessLifecycleHandlers();
151
+ initStatusServer();
152
+
153
+ // Opportunistic state-dir hygiene: prune old sessions / orphan .tmp_* files.
154
+ // Throttled to at most once per 24h via the .last_prune timestamp, and
155
+ // best-effort — a failure here must never prevent the MCP server from
156
+ // starting. Stdio purity: diagnostics go to stderr only.
157
+ try {
158
+ const pruneResult = cleanStateDir({ throttled: true });
159
+ if (!pruneResult.skipped) {
160
+ const reclaimedMb =
161
+ (pruneResult.bytesReclaimed + pruneResult.tmpBytesReclaimed) /
162
+ (1024 * 1024);
163
+ process.stderr.write(
164
+ `[state-pruner] pruned ${pruneResult.sessionsPruned} session(s), ` +
165
+ `${pruneResult.tmpFilesCleaned} orphan .tmp_* file(s) ` +
166
+ `(${reclaimedMb.toFixed(2)} MB); ` +
167
+ `${pruneResult.protected} protected\n`
168
+ );
169
+ }
170
+ } catch (err) {
171
+ process.stderr.write(
172
+ `[state-pruner] startup cleanup failed (non-fatal): ${
173
+ err && err.message ? err.message : err
174
+ }\n`
175
+ );
176
+ }
177
+
178
+ const server = createMcpServer();
179
+ const transport = new StdioServerTransport();
180
+ await server.connect(transport);
181
+ }
182
+
183
+ export {
184
+ acquireTaskSlot,
185
+ releaseTaskSlot,
186
+ listTaskSlots,
187
+ getSlotStatus,
188
+ TASK_DIR,
189
+ isTaskOrphaned,
190
+ markTaskOrphanedOnDisk,
191
+ };
192
+
193
+ // Direct-run guard: only auto-start when this file is the actual entrypoint
194
+ // (e.g. `node index.js`), not when imported by bin/castor.js or tests.
195
+ const isMain = Boolean(
196
+ process.argv[1] &&
197
+ (path.resolve(process.argv[1]).toLowerCase() === __filename.toLowerCase() ||
198
+ process.argv[1].toLowerCase().endsWith("index.js"))
199
+ );
200
+
201
+ if (isMain) {
202
+ startMcpServer().catch((err) => {
203
+ console.error("MCP Server Fatal Error:", err);
204
+ process.exit(1);
205
+ });
206
+ }
package/package.json ADDED
@@ -0,0 +1,97 @@
1
+ {
2
+ "name": "mcp-castor",
3
+ "version": "2026.3.0",
4
+ "description": "Universal 245K local Castor agent microkernel & MCP server: collaborative Socratic pair-programming, native multi-provider research (Brave, Tavily, Context7 docs, SearXNG, DuckDuckGo), cooperative landing turn ceiling preservation, in-process structural AST surgery (@ast-grep/napi), in-memory syntax gates (JS/TS, Python, JSON, LaTeX, BibTeX), 137-vector zero-trust sandbox, closed-loop evolutionary optimization (.evo/lineage.json), vLLM + DFlash2 + KVarN @ 245K context, and zero-turn reactive OS wait @18021.",
5
+ "repository": {
6
+ "type": "git",
7
+ "url": "git+https://github.com/ApatheticMioz/Anser.git",
8
+ "directory": "mcp-castor"
9
+ },
10
+ "main": "index.js",
11
+ "bin": {
12
+ "castor": "bin/castor.js"
13
+ },
14
+ "files": [
15
+ "bin/",
16
+ "index.js",
17
+ "stream_proxy.js",
18
+ "src/",
19
+ "skills/",
20
+ "README.md"
21
+ ],
22
+ "scripts": {
23
+ "test": "node tests/canary.test.js && node tests/security.test.js && node tests/syntax_gates.test.js && node tests/edit_file_guard.test.js && node tests/search_code_guard.test.js && node tests/schema_parity.test.js && node tests/platform.test.js && node --test tests/protocol_sync.test.js && node --test tests/prompt_integrity.test.js",
24
+ "test:fast": "npm test",
25
+ "test:all": "node tests/security.test.js && node tests/canary.test.js && node tests/tool_output_spill.test.js && node tests/ast_engine.test.js && node tests/ast_batch.test.js && node tests/edit_file_guard.test.js && node tests/apply_patch.test.js && node tests/evo.test.js && node tests/evo_skills.test.js && node tests/semaphore.test.js && node tests/tenant_concurrency.test.js && node tests/stream_proxy.test.js && node tests/utf8_proxy.test.js && node tests/runner_continuation.test.js && node tests/degenerate_final.test.js && node tests/probe_budget.test.js && node tests/session_rollover.test.js && node tests/context_high_watermark.test.js && node tests/kv_prefix_stability.test.js && node tests/prompt_budget.test.js && node tests/wedge_guard.test.js && node tests/platform.test.js && node tests/posix_routing.test.js && node tests/honesty_drift.test.js && node tests/path_canonicalization.test.js && node tests/syntax_gates.test.js && node tests/syntax_integrity.test.js && node tests/provider_reasoning.test.js && node tests/idle_timeout_tier.test.js && node tests/deep_retry_budget.test.js && node tests/provider_usage.test.js && node tests/reasoning_effort_dispatch.test.js && node tests/lineage_integrity.test.js && node tests/mcp_bridge.test.js && node tests/skills.test.js && node tests/reaping.test.js && node tests/status_lifecycle.test.js && node tests/wait_endpoint.test.js && node tests/task_retention.test.js && node tests/orphan_terminal_event.test.js && node tests/orphan_reaper.test.js && node tests/shell_hardening.test.js && node tests/mcp_client.test.js && node tests/runner_mapping.test.js && node tests/fifo_queue.test.js && node tests/lifecycle_locks.test.js && node tests/signal_hardening.test.js && node tests/benchmark.test.js && node tests/stdio_purity.test.js && node tests/tool_errors.test.js && node tests/schema_parity.test.js && node tests/binary_guard.test.js && node tests/search_code_guard.test.js && node tests/sse_transport.test.js && node tests/edit_file_syntax_gate.test.js && node tests/web_service.test.js && node tests/searxng_lifecycle.test.js && node tests/telemetry.test.js && node tests/cooperative_landing.test.js && node tests/multi_provider_search.test.js && node tests/loop_detector.test.js && node tests/lease_extension.test.js && node tests/salvage_pass.test.js && node tests/read_governor.test.js && node tests/session_end_reap.test.js && node --test tests/protocol_sync.test.js && node --test tests/prompt_integrity.test.js",
26
+ "test:security": "node tests/security.test.js",
27
+ "test:canary": "node tests/canary.test.js",
28
+ "test:evo": "node tests/evo.test.js",
29
+ "test:evo_skills": "node tests/evo_skills.test.js",
30
+ "test:semaphore": "node tests/semaphore.test.js",
31
+ "test:tenant": "node tests/tenant_concurrency.test.js",
32
+ "test:continuation": "node tests/runner_continuation.test.js",
33
+ "test:wedge": "node tests/wedge_guard.test.js",
34
+ "test:reasoning": "node tests/provider_reasoning.test.js",
35
+ "test:usage": "node tests/provider_usage.test.js",
36
+ "test:canonical": "node tests/path_canonicalization.test.js",
37
+ "test:syntax": "node tests/syntax_gates.test.js",
38
+ "test:syntax_integrity": "node tests/syntax_integrity.test.js",
39
+ "test:stdio": "node tests/stdio_purity.test.js",
40
+ "test:batch": "node tests/ast_batch.test.js",
41
+ "test:proxy": "node tests/stream_proxy.test.js && node tests/utf8_proxy.test.js",
42
+ "test:bridge": "node tests/mcp_bridge.test.js",
43
+ "test:skills": "node tests/skills.test.js",
44
+ "test:reaping": "node tests/reaping.test.js",
45
+ "test:retention": "node tests/task_retention.test.js",
46
+ "test:benchmark": "node tests/benchmark.test.js",
47
+ "test:tool_errors": "node tests/tool_errors.test.js",
48
+ "test:schema_parity": "node tests/schema_parity.test.js",
49
+ "test:sse": "node tests/sse_transport.test.js",
50
+ "test:kv_stability": "node tests/kv_prefix_stability.test.js",
51
+ "test:edit_syntax": "node tests/edit_file_syntax_gate.test.js",
52
+ "test:web": "node tests/web_service.test.js",
53
+ "test:telemetry": "node tests/telemetry.test.js",
54
+ "test:cooperative": "node tests/cooperative_landing.test.js",
55
+ "test:search": "node tests/multi_provider_search.test.js",
56
+ "test:prompt_integrity": "node --test tests/prompt_integrity.test.js"
57
+ },
58
+ "keywords": [
59
+ "mcp",
60
+ "qwen3.8-27b",
61
+ "vllm",
62
+ "castor",
63
+ "evo",
64
+ "agent-harness",
65
+ "ast-grep",
66
+ "sandboxing",
67
+ "dflash2",
68
+ "kvarn"
69
+ ],
70
+ "author": "ApatheticMioz",
71
+ "license": "AGPL-3.0-or-later",
72
+ "type": "module",
73
+ "engines": {
74
+ "node": ">=22"
75
+ },
76
+ "dependencies": {
77
+ "@ast-grep/napi": "^0.45.3",
78
+ "@modelcontextprotocol/sdk": "^1.30.0",
79
+ "@mozilla/readability": "^0.6.0",
80
+ "duck-duck-scrape": "^2.2.7",
81
+ "file-type": "^22.1.0",
82
+ "jsdom": "^29.1.1",
83
+ "pdf-parse": "^2.4.5",
84
+ "turndown": "^7.2.4",
85
+ "typescript": "5.9.3",
86
+ "zod": "^4.4.3"
87
+ },
88
+ "devDependencies": {
89
+ "@ast-grep/cli": "^0.45.3"
90
+ },
91
+ "optionalDependencies": {
92
+ "@ast-grep/cli-linux-x64-gnu": "^0.45.3",
93
+ "@ast-grep/cli-win32-x64-msvc": "^0.45.3",
94
+ "@ast-grep/napi-linux-x64-gnu": "^0.45.3",
95
+ "@ast-grep/napi-win32-x64-msvc": "^0.45.3"
96
+ }
97
+ }
@@ -0,0 +1,24 @@
1
+ ---
2
+ name: canary-test-staging
3
+ description: Stage changes behind a targeted test file and run the narrow suite before the full chain.
4
+ keywords: [canary, smoke test, staging suite, test gate, narrow test, targeted test]
5
+ ---
6
+ # Canary Test Staging
7
+
8
+ Before you trust a change, prove it on the smallest possible surface, then
9
+ widen. Do not run the full test chain as your first signal.
10
+
11
+ ## Checklist
12
+ 1. Write or identify ONE targeted test file that exercises exactly the code
13
+ you changed (a canary).
14
+ 2. Run just that file first: `node tests/<canary>.test.js`.
15
+ 3. If it passes, run the next-narrower group (the module's related tests).
16
+ 4. Only after those are green, run the full chain (`npm test`).
17
+ 5. If a canary fails, fix the code — do not weaken the canary to make it pass.
18
+
19
+ ## Rules
20
+ - A canary must be fast (seconds) and deterministic (no network, no flaky
21
+ timing).
22
+ - Never delete or skip a failing canary to "unblock" the build; that hides
23
+ the regression.
24
+ - If you add a new capability, add its canary in the same change.
@@ -0,0 +1,29 @@
1
+ ---
2
+ name: evo-mutation-rollback
3
+ description: Safe propose, evaluate, select, revert sequence for Evo mutations with no irreversible steps.
4
+ keywords: [rollback, revert, mutation, regression, snapshot, candidate]
5
+ ---
6
+ # Evo Mutation Rollback
7
+
8
+ A mutation is only safe if it can always be undone. Never let a budget,
9
+ timeout, or failure land on a step you cannot roll back.
10
+
11
+ ## Sequence
12
+ 1. `evo_propose_candidate` — snapshot the target files FIRST. This is the
13
+ rollback anchor; nothing is mutated until this succeeds.
14
+ 2. Apply the mutation to the working files.
15
+ 3. `evo_evaluate_candidate` — run the verification command and read the
16
+ fitness score plus the compact failure digest.
17
+ 4. Decision:
18
+ - Fitness improved and tests pass -> `evo_select_candidate` (commit to
19
+ the lineage DAG).
20
+ - Fitness regressed, tests failed, or the digest shows a real bug ->
21
+ `evo_revert_candidate` (restore the snapshot exactly).
22
+ 5. Confirm the revert restored the original file contents before continuing.
23
+
24
+ ## Rules
25
+ - The snapshot (step 1) must always precede any file mutation.
26
+ - If a step is interrupted by a timeout or budget, the next action is to
27
+ revert to the last good snapshot, not to resume mid-mutation.
28
+ - Never edit a file that is not in the candidate's snapshot list; those
29
+ changes are not covered by the rollback.
@@ -0,0 +1,26 @@
1
+ ---
2
+ name: hypothesis-generation
3
+ description: Evo discipline — state a measurable hypothesis, snapshot, evaluate, then select or revert.
4
+ keywords: [hypothesis, evo, fitness, benchmark, optimization, candidate, lineage]
5
+ ---
6
+ # Hypothesis Generation (Evo Discipline)
7
+
8
+ Every code change you make to improve a metric is a *candidate*. Treat it as
9
+ a falsifiable hypothesis, not a guess.
10
+
11
+ ## Checklist
12
+ 1. State the hypothesis in one measurable sentence:
13
+ "Changing X will raise <metric> from A to B because <mechanism>."
14
+ 2. Snapshot the files you are about to mutate with `evo_propose_candidate`
15
+ BEFORE editing (pass the exact file list).
16
+ 3. Make the minimal change that tests the mechanism — one variable at a time.
17
+ 4. Run the verification with `evo_evaluate_candidate` (the test command).
18
+ 5. Read the fitness score and the compact failure digest.
19
+ 6. If it improved: `evo_select_candidate`. If it regressed or failed:
20
+ `evo_revert_candidate` to restore the snapshot.
21
+
22
+ ## Rules
23
+ - Never batch unrelated changes into one candidate — you cannot attribute
24
+ the fitness delta.
25
+ - A candidate that cannot be measured is not a hypothesis; do not propose it.
26
+ - Keep the hypothesis text short and specific; it is stored in the lineage.
@@ -0,0 +1,26 @@
1
+ ---
2
+ name: traceback-condensing
3
+ description: Condense a failing test run into a minimal failure digest before retrying.
4
+ keywords: [traceback, stack trace, error digest, trace_repair, assertion failure, pytest, failed test]
5
+ ---
6
+ # Traceback Condensing
7
+
8
+ When a test run or build fails, do NOT paste the raw multi-hundred-line
9
+ traceback into your reasoning. Condense it into a compact failure digest
10
+ first, then act on the digest.
11
+
12
+ ## Checklist
13
+ 1. Identify the FIRST user-code frame (skip site-packages, stdlib, and
14
+ framework internals). That is where the real bug lives.
15
+ 2. Extract the exact file, line, and symbol from that frame.
16
+ 3. Capture the assertion message or exception type + message verbatim.
17
+ 4. Reduce the digest to at most ~100 tokens:
18
+ `[FailureDigest] <type> at <file>:<line> in <symbol>: <message>`
19
+ 5. Use the digest (not the raw traceback) to form your next hypothesis.
20
+
21
+ ## Rules
22
+ - Never re-emit the full traceback into a tool call or the final answer.
23
+ - If the digest points at a test file (not source), the bug is likely in the
24
+ code under test — go read that source file next.
25
+ - If the digest is a network/timeout error, that is an environment issue, not
26
+ a code bug — say so and do not "fix" unrelated code.