mcp-castor 2026.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +487 -0
  2. package/bin/castor.js +706 -0
  3. package/index.js +206 -0
  4. package/package.json +97 -0
  5. package/skills/canary-test-staging/SKILL.md +24 -0
  6. package/skills/evo-mutation-rollback/SKILL.md +29 -0
  7. package/skills/hypothesis-generation/SKILL.md +26 -0
  8. package/skills/traceback-condensing/SKILL.md +26 -0
  9. package/src/castor_runner.js +469 -0
  10. package/src/config.js +1204 -0
  11. package/src/env.js +10 -0
  12. package/src/evo_engine.js +214 -0
  13. package/src/harness/core/events.js +75 -0
  14. package/src/harness/core/kernel.js +209 -0
  15. package/src/harness/evo/evaluator.js +156 -0
  16. package/src/harness/evo/evo_operator.js +550 -0
  17. package/src/harness/evo/lineage_dag.js +383 -0
  18. package/src/harness/evo/trace_repair.js +173 -0
  19. package/src/harness/evo/watchdog.js +72 -0
  20. package/src/harness/loop_detector.js +135 -0
  21. package/src/harness/runner.js +1216 -0
  22. package/src/harness/services/ast_service.js +1813 -0
  23. package/src/harness/services/event_logger.js +275 -0
  24. package/src/harness/services/mcp_bridge.js +408 -0
  25. package/src/harness/services/provider_vllm.js +728 -0
  26. package/src/harness/services/sandbox_fs.js +1238 -0
  27. package/src/harness/services/searxng_lifecycle.js +254 -0
  28. package/src/harness/services/shell_executor.js +264 -0
  29. package/src/harness/services/shell_validator.js +506 -0
  30. package/src/harness/services/web_service.js +828 -0
  31. package/src/platform.js +344 -0
  32. package/src/repetition_detector.js +139 -0
  33. package/src/semaphore.js +373 -0
  34. package/src/server_lifecycle.js +781 -0
  35. package/src/skills.js +400 -0
  36. package/src/state_pruner.js +392 -0
  37. package/src/task_registry.js +1357 -0
  38. package/src/telemetry.js +638 -0
  39. package/src/tools.js +997 -0
  40. package/src/wsl_bridge.js +629 -0
  41. package/src/wsl_env.js +171 -0
  42. package/stream_proxy.js +453 -0
package/src/wsl_env.js ADDED
@@ -0,0 +1,171 @@
1
+ /**
2
+ * src/wsl_env.js — Leaf module: stateless WSL environment resolvers.
3
+ *
4
+ * Extracted from platform.js via dependency inversion so that config.js can
5
+ * import winHomeWsl() WITHOUT importing platform.js. This breaks the
6
+ * config.js <-> platform.js import cycle that produced a TDZ ReferenceError
7
+ * on Linux CI: when platform.js was the entry, its `_wslUser` cache (a
8
+ * module-level `let`) was still in the temporal dead zone while config.js's
9
+ * QWEN_STATE_DIR IIFE called winHomeWsl() at module-eval time.
10
+ *
11
+ * This module is a LEAF: it imports only node builtins and the leaf env.js.
12
+ * It carries the WSL user/home probe cache and the test seams. The
13
+ * Windows<->POSIX path translators remain in wsl_bridge.js (re-exported by
14
+ * platform.js) and are NOT reimplemented here.
15
+ */
16
+ import fs from "node:fs";
17
+ import os from "node:os";
18
+ import { execFileSync } from "node:child_process";
19
+ import { IS_WINDOWS } from "./env.js";
20
+
21
+ // ---------------------------------------------------------------------------
22
+ // WSL distro / user / home
23
+ // ---------------------------------------------------------------------------
24
+
25
+ /** WSL distro name. Env QWEN_WSL_DISTRO; default "Ubuntu". */
26
+ export function wslDistro() {
27
+ return process.env.QWEN_WSL_DISTRO || "Ubuntu";
28
+ }
29
+
30
+ let _wslUser = null;
31
+ let _wslHome = null;
32
+ let _wslUserProbeOk = null; // null = unknown, true = probe succeeded, false = probe failed
33
+
34
+ function _probeWslUser() {
35
+ if (!IS_WINDOWS) {
36
+ // Already inside WSL/Linux: the current user IS the WSL user.
37
+ return process.env.USER || process.env.LOGNAME || "root";
38
+ }
39
+ try {
40
+ const out = execFileSync("wsl.exe", ["-d", wslDistro(), "whoami"], {
41
+ timeout: 5000,
42
+ stdio: ["ignore", "pipe", "ignore"],
43
+ });
44
+ const u = out.toString().trim();
45
+ return u || "root";
46
+ } catch {
47
+ // WSL unavailable / probe failed: signal failure (null) so the caller
48
+ // can fall back to root//root without throwing.
49
+ process.stderr.write(`[wsl_env] whoami probe failed; fabricating root identity.\n`);
50
+ return null;
51
+ }
52
+ }
53
+
54
+ // Test seam: lets offline tests inject a fake whoami probe (mirrors the
55
+ // setWslRunner seam in server_lifecycle.js). Pass null to restore the real
56
+ // probe. The real probe never throws (it returns null on failure).
57
+ let _wslUserProbe = _probeWslUser;
58
+ export function setWslUserProbe(fn) {
59
+ _wslUserProbe = typeof fn === "function" ? fn : _probeWslUser;
60
+ }
61
+
62
+ /** Clear the cached WSL user/home so the next call re-probes. Test-only. */
63
+ export function _resetWslUserCache() {
64
+ _wslUser = null;
65
+ _wslHome = null;
66
+ _wslUserProbeOk = null;
67
+ }
68
+
69
+ /** WSL user name. Env QWEN_WSL_USER; otherwise probed once via whoami.
70
+ * Never throws: any probe failure falls back to "root". */
71
+ export function wslUser() {
72
+ if (process.env.QWEN_WSL_USER) return process.env.QWEN_WSL_USER;
73
+ if (_wslUser) return _wslUser;
74
+ let u;
75
+ let ok = true;
76
+ try {
77
+ u = _wslUserProbe();
78
+ } catch {
79
+ u = null;
80
+ ok = false;
81
+ }
82
+ _wslUserProbeOk = ok;
83
+ _wslUser = u || "root";
84
+ return _wslUser;
85
+ }
86
+
87
+ /** WSL home directory. Env QWEN_WSL_HOME; otherwise /home/<wslUser>.
88
+ * When the whoami probe FAILED (WSL unavailable) it falls back to /root. */
89
+ export function wslHome() {
90
+ if (process.env.QWEN_WSL_HOME) return process.env.QWEN_WSL_HOME;
91
+ if (_wslHome) return _wslHome;
92
+ if (_wslUserProbeOk === false) {
93
+ _wslHome = "/root";
94
+ return _wslHome;
95
+ }
96
+ _wslHome = `/home/${wslUser()}`;
97
+ return _wslHome;
98
+ }
99
+
100
+ /**
101
+ * The Windows user home as seen from WITHIN WSL (e.g. /mnt/c/Users/<user>),
102
+ * used to share the .qwen state dir between the Windows MCP server and the
103
+ * WSL engine. Env QWEN_WIN_HOME_WSL overrides. When unset, derived from
104
+ * QWEN_WIN_HOME if it is a Windows path (C:\Users\X -> /mnt/c/Users/X);
105
+ * when running inside WSL, auto-probes /mnt/c/Users/<user>; otherwise returns
106
+ * "" so callers fall back to the WSL user's own home.
107
+ *
108
+ * NOTE: the fs.existsSync probe below requires the `fs` import (previously
109
+ * missing in platform.js — a latent Linux bug swallowed by the try/catch).
110
+ */
111
+ export function winHomeWsl() {
112
+ if (process.env.QWEN_WIN_HOME_WSL) return process.env.QWEN_WIN_HOME_WSL;
113
+ const win = process.env.QWEN_WIN_HOME || (IS_WINDOWS ? os.homedir() : "");
114
+ const m = win.match(/^([a-zA-Z]):[\\\/](.*)$/);
115
+ if (m) return `/mnt/${m[1].toLowerCase()}/${m[2].replace(/\\/g, "/")}`;
116
+ if (!IS_WINDOWS) {
117
+ const candidates = [
118
+ `/mnt/c/Users/${wslUser()}`,
119
+ `/mnt/c/Users/${process.env.USER || ""}`,
120
+ `/mnt/c/Users/${process.env.LOGNAME || ""}`,
121
+ ];
122
+ for (const c of candidates) {
123
+ try {
124
+ if (c && c !== "/mnt/c/Users/" && fs.existsSync(c)) return c;
125
+ } catch {}
126
+ }
127
+ }
128
+ return "";
129
+ }
130
+
131
+ // ---------------------------------------------------------------------------
132
+ // WSL availability probe
133
+ // ---------------------------------------------------------------------------
134
+
135
+ let _wslAvailable = null;
136
+
137
+ /**
138
+ * Probe whether WSL is available on this host.
139
+ * - Non-Windows (Linux/macOS): always true (no WSL needed; commands run
140
+ * directly via bash).
141
+ * - Windows: runs `wsl.exe --list --quiet` and checks for a non-empty
142
+ * output (at least one installed distro). Result is cached.
143
+ *
144
+ * Never throws. Returns a boolean.
145
+ *
146
+ * @returns {boolean}
147
+ */
148
+ export function wslAvailable() {
149
+ if (!IS_WINDOWS) return true;
150
+ if (_wslAvailable !== null) return _wslAvailable;
151
+ try {
152
+ const out = execFileSync("wsl.exe", ["--list", "--quiet"], {
153
+ timeout: 5000,
154
+ stdio: ["ignore", "pipe", "ignore"],
155
+ });
156
+ const lines = out
157
+ .toString()
158
+ .split(/\r?\n/)
159
+ .map((s) => s.trim())
160
+ .filter(Boolean);
161
+ _wslAvailable = lines.length > 0;
162
+ } catch {
163
+ _wslAvailable = false;
164
+ }
165
+ return _wslAvailable;
166
+ }
167
+
168
+ /** Clear the WSL availability cache. Test-only. */
169
+ export function _resetWslAvailableCache() {
170
+ _wslAvailable = null;
171
+ }
@@ -0,0 +1,453 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Universal Stateful UTF-8, SSE Stream Sanitizer & Inbound Multimodal Guard Proxy
4
+ *
5
+ * Architecture:
6
+ * - Listens on: 127.0.0.1:18022 (STREAM_PROXY_PORT or VLLM_PROXY_PORT)
7
+ * - Upstream: 127.0.0.1:18020 (vLLM Engine)
8
+ *
9
+ * Capabilities:
10
+ * 1. Inbound Multimodal Guard:
11
+ * When clients send image_url/image blocks, intercepts and converts them
12
+ * to descriptive text placeholders before reaching vLLM, preventing
13
+ * "Bad request (400): At most 0 image(s) may be provided in one prompt".
14
+ * 2. Stateful UTF-8 Reconstruction:
15
+ * Maintains per-stream TextDecoder with { stream: true } to assemble split
16
+ * multi-byte UTF-8 sequences (math symbols, superscripts 2³, Greek letters,
17
+ * emojis) across raw TCP/SSE chunk boundaries.
18
+ * 3. Mid-Stream Error Translation:
19
+ * Intercepts mid-stream vLLM error payloads (`data: {"error": ...}`) that
20
+ * lack `choices` and converts them into valid completion delta chunks before
21
+ * client parsers see them, eliminating `Stream decode error`.
22
+ * 4. Transparent Pass-Through:
23
+ * Non-SSE / non-chat requests (GET /v1/models, embeddings, health) are piped directly.
24
+ * 5. Zero External Dependencies:
25
+ * Standard Node.js http/url modules, runs under WSL2 Linux and Windows.
26
+ */
27
+
28
+ import http from "http";
29
+ import { RepetitionDetector, GUARD_MARKER_TEMPLATE } from "./src/repetition_detector.js";
30
+ import { PROXY_MAX_BODY_BYTES } from "./src/config.js";
31
+
32
+ const UPSTREAM_PORT = parseInt(process.env.VLLM_PORT || "18020", 10);
33
+ // Accept either STREAM_PROXY_PORT (preferred) or VLLM_PROXY_PORT (legacy).
34
+ const PROXY_PORT = parseInt(
35
+ process.env.STREAM_PROXY_PORT || process.env.VLLM_PROXY_PORT || "18022",
36
+ 10
37
+ );
38
+ // P15: bind loopback only — a LAN-reachable proxy lets any remote client
39
+ // drive the MAX_SEQS=1 GPU queue and wedge the engine for local clients.
40
+ // WSL2 localhost-forwarding keeps it reachable from the Windows host.
41
+ const PROXY_HOST = process.env.VLLM_PROXY_HOST || "127.0.0.1";
42
+
43
+ function forwardToUpstream(req, res, reqBodyBuffer) {
44
+ // Disable socket-level timeouts on incoming client connection
45
+ if (req.socket) {
46
+ req.socket.setTimeout(0);
47
+ req.socket.setKeepAlive(true, 10000);
48
+ req.socket.setNoDelay(true);
49
+ }
50
+
51
+ const headers = { ...req.headers, host: `127.0.0.1:${UPSTREAM_PORT}` };
52
+ if (reqBodyBuffer) {
53
+ headers["content-length"] = reqBodyBuffer.length;
54
+ }
55
+
56
+ const isStreamRequest = req.url.startsWith("/v1/chat/completions") && (
57
+ (req.headers["accept"] && req.headers["accept"].includes("text/event-stream")) ||
58
+ (reqBodyBuffer && (reqBodyBuffer.includes('"stream":true') || reqBodyBuffer.includes('"stream": true')))
59
+ );
60
+
61
+ let pingInterval = null;
62
+ let hasDone = false;
63
+
64
+ // P7b: per-request lifecycle observability. Before this pass the proxy
65
+ // logged NOTHING per request — when a task died of engine_empty_response,
66
+ // the proxy log was empty and the death was undecidable from here. One
67
+ // stderr line per streaming request now records the whole story:
68
+ // ttft (first upstream byte), bytes forwarded, duration, and the end cause.
69
+ // "upstream-end-no-done" is the killer signature (stream closed by the
70
+ // upstream without any finish_reason); "client-early-close" means the
71
+ // consumer hung up first. The guard makes the first cause win.
72
+ const lifecycle = {
73
+ t0: Date.now(),
74
+ ttft: null,
75
+ bytes: 0,
76
+ logged: false,
77
+ log(cause) {
78
+ if (this.logged) return;
79
+ this.logged = true;
80
+ console.error(
81
+ `[StreamProxy] ${req.method} ${req.url} ttft=${this.ttft ?? "never"} bytes=${this.bytes} dur=${Date.now() - this.t0}ms end=${cause}`
82
+ );
83
+ },
84
+ };
85
+
86
+ const upstreamUrl = `http://127.0.0.1:${UPSTREAM_PORT}${req.url}`;
87
+ const upstreamReq = http.request(
88
+ upstreamUrl,
89
+ {
90
+ method: req.method,
91
+ headers,
92
+ agent: false,
93
+ },
94
+ (upstreamRes) => {
95
+ const contentType = upstreamRes.headers["content-type"] || "";
96
+ const isEventStream = contentType.includes("text/event-stream") || isStreamRequest;
97
+
98
+ // Non-streaming endpoint: pipe directly
99
+ if (!isEventStream) {
100
+ res.writeHead(upstreamRes.statusCode, upstreamRes.headers);
101
+ upstreamRes.pipe(res);
102
+ return;
103
+ }
104
+
105
+ // If upstream returned an HTTP error (e.g. 400 or 500), forward status and error body directly
106
+ if (upstreamRes.statusCode >= 400) {
107
+ let errBody = "";
108
+ const expectedLen = parseInt(upstreamRes.headers["content-length"] || "0", 10);
109
+ function emitError() {
110
+ if (hasDone) return;
111
+ hasDone = true;
112
+ if (pingInterval) clearInterval(pingInterval);
113
+ if (!res.headersSent) {
114
+ res.writeHead(upstreamRes.statusCode, upstreamRes.headers);
115
+ res.end(errBody);
116
+ } else {
117
+ // Mid-stream error: emit explicit SSE error event and end
118
+ res.end(`event: error\ndata: ${errBody}\n\n`);
119
+ }
120
+ upstreamReq.destroy();
121
+ }
122
+ upstreamRes.on("data", (c) => {
123
+ errBody += c.toString("utf8");
124
+ if (expectedLen > 0 && Buffer.byteLength(errBody, "utf8") >= expectedLen) {
125
+ emitError();
126
+ }
127
+ });
128
+ upstreamRes.on("end", emitError);
129
+ return;
130
+ }
131
+
132
+ // If headers were not pre-flushed, flush them now
133
+ if (!res.headersSent) {
134
+ const cleanHeaders = { ...upstreamRes.headers };
135
+ delete cleanHeaders["content-length"];
136
+ delete cleanHeaders["transfer-encoding"];
137
+
138
+ res.writeHead(upstreamRes.statusCode, {
139
+ ...cleanHeaders,
140
+ "content-type": "text/event-stream; charset=utf-8",
141
+ "cache-control": "no-cache, no-transform",
142
+ connection: "keep-alive",
143
+ });
144
+
145
+ pingInterval = setInterval(() => {
146
+ try {
147
+ res.write(": keep-alive\n\n");
148
+ } catch {}
149
+ }, 5000);
150
+ }
151
+
152
+ const decoder = new TextDecoder("utf-8", { fatal: false, ignoreBOM: true });
153
+ const repetitionDetector = new RepetitionDetector();
154
+ let lineBuffer = "";
155
+
156
+ upstreamRes.on("data", (chunk) => {
157
+ const text = decoder.decode(chunk, { stream: true });
158
+ if (!text) return;
159
+ lifecycle.ttft ??= Date.now() - lifecycle.t0;
160
+ lifecycle.bytes += Buffer.byteLength(text);
161
+
162
+ lineBuffer += text;
163
+ if (lineBuffer.includes("data: [DONE]")) {
164
+ hasDone = true;
165
+ if (pingInterval) clearInterval(pingInterval);
166
+ res.write("data: [DONE]\n\n");
167
+ res.end();
168
+ return;
169
+ }
170
+ const lines = lineBuffer.split("\n");
171
+ lineBuffer = lines.pop(); // Retain incomplete trailing line
172
+
173
+ for (const line of lines) {
174
+ const trimmed = line.trim();
175
+ if (trimmed === "data: [DONE]") {
176
+ hasDone = true;
177
+ if (pingInterval) clearInterval(pingInterval);
178
+ res.write("data: [DONE]\n\n");
179
+ res.end();
180
+ return;
181
+ } else if (trimmed.startsWith("data: ")) {
182
+ const jsonPayload = trimmed.slice(6).trim();
183
+ if (jsonPayload.startsWith("{")) {
184
+ try {
185
+ const parsed = JSON.parse(jsonPayload);
186
+ if (parsed.error && !parsed.choices) {
187
+ const errMsg = parsed.error.message || JSON.stringify(parsed.error);
188
+ if (pingInterval) clearInterval(pingInterval);
189
+ hasDone = true;
190
+ res.end(`event: error\ndata: ${JSON.stringify(parsed)}\n\n`);
191
+ upstreamReq.destroy();
192
+ return;
193
+ }
194
+
195
+ // Check for degenerate runaway repetition in token deltas (content or reasoning).
196
+ // Engine emits reasoning as `delta.reasoning` (--reasoning-parser qwen3, P2d live SSE
197
+ // evidence); `reasoning_content` kept as legacy fallback. Both must feed the detector
198
+ // or a reasoning-level repetition loop burns the full max_tokens budget invisibly
199
+ // (P7b root cause: ~18-min single generation, spec-decode acceptance pinned at 8.0).
200
+ const delta = parsed?.choices?.[0]?.delta;
201
+ if (delta) {
202
+ const tokenText = delta.content || delta.reasoning || delta.reasoning_content || "";
203
+ if (tokenText) {
204
+ let rep;
205
+ try {
206
+ rep = repetitionDetector.feed(tokenText);
207
+ } catch (detErr) {
208
+ console.error(`[StreamProxy] Repetition detector error: ${detErr.message}; truncating stream.`);
209
+ if (pingInterval) clearInterval(pingInterval);
210
+ hasDone = true;
211
+ res.write(`data: ${JSON.stringify({ id: "chatcmpl-det-err", object: "chat.completion.chunk", created: Math.floor(Date.now() / 1000), model: "qwen3.8-27b", choices: [{ index: 0, delta: { content: "[Stream truncated: repetition detector error]" }, finish_reason: "stop" }] })}\n\ndata: [DONE]\n\n`);
212
+ res.end();
213
+ upstreamReq.destroy();
214
+ return;
215
+ }
216
+ if (rep) {
217
+ console.error(`[StreamProxy Circuit Breaker] Runaway repetition detected (${rep.type}: ${JSON.stringify(rep.pattern)}, count: ${rep.count}). Safely aborting stream.`);
218
+ lifecycle.log(`repetition-breaker(${rep.type})`);
219
+ if (pingInterval) clearInterval(pingInterval);
220
+ hasDone = true;
221
+ const breakerChunk = {
222
+ id: "chatcmpl-repetition-breaker",
223
+ object: "chat.completion.chunk",
224
+ created: Math.floor(Date.now() / 1000),
225
+ model: "qwen3.8-27b",
226
+ choices: [
227
+ {
228
+ index: 0,
229
+ // M3b: marker text comes from the GUARD_MARKER_TEMPLATE
230
+ // sentinel (single source of truth in repetition_detector.js).
231
+ delta: { content: GUARD_MARKER_TEMPLATE.replaceAll("${type}", rep.type).replaceAll("${pattern}", JSON.stringify(rep.pattern)) },
232
+ finish_reason: "stop",
233
+ },
234
+ ],
235
+ };
236
+ res.write(`data: ${JSON.stringify(breakerChunk)}\n\ndata: [DONE]\n\n`);
237
+ res.end();
238
+ upstreamReq.destroy();
239
+ return;
240
+ }
241
+ }
242
+ }
243
+ } catch {}
244
+ }
245
+ }
246
+ res.write(line + "\n");
247
+ }
248
+ });
249
+
250
+ upstreamRes.on("end", () => {
251
+ lifecycle.log(hasDone ? "done" : "upstream-end-no-done");
252
+ if (pingInterval) clearInterval(pingInterval);
253
+ const tail = decoder.decode();
254
+ if (tail) {
255
+ lineBuffer += tail;
256
+ }
257
+ if (lineBuffer) {
258
+ res.write(lineBuffer + (lineBuffer.endsWith("\n") ? "\n" : "\n\n"));
259
+ if (lineBuffer.includes("[DONE]")) {
260
+ hasDone = true;
261
+ }
262
+ }
263
+ res.end();
264
+ });
265
+
266
+ upstreamRes.on("error", (err) => {
267
+ lifecycle.log(`upstream-error(${err ? (err.message || String(err)) : "unknown"})`);
268
+ if (pingInterval) clearInterval(pingInterval);
269
+ try {
270
+ if (!hasDone) {
271
+ hasDone = true;
272
+ res.end(`event: error\ndata: ${JSON.stringify({ error: { message: errMsg, type: "upstream_error" } })}\n\n`);
273
+ } else {
274
+ res.end();
275
+ }
276
+ } catch {}
277
+ });
278
+
279
+ res.on("close", () => {
280
+ // Normal completions log via the upstream 'end' handler first; a close
281
+ // with hasDone still false means the CONSUMER hung up early.
282
+ if (!hasDone) lifecycle.log("client-early-close");
283
+ if (pingInterval) clearInterval(pingInterval);
284
+ upstreamReq.destroy();
285
+ });
286
+ }
287
+ );
288
+
289
+ upstreamReq.setTimeout(0);
290
+
291
+ upstreamReq.on("error", (err) => {
292
+ if (isStreamRequest) lifecycle.log(`connect-error(${err ? err.message : "unknown"})`);
293
+ if (pingInterval) clearInterval(pingInterval);
294
+ if (!res.headersSent) {
295
+ res.writeHead(502, { "Content-Type": "application/json" });
296
+ res.end(JSON.stringify({ error: { message: `vLLM upstream connection error: ${err.message}`, type: "upstream_connection_error" } }));
297
+ } else {
298
+ try {
299
+ res.end(`event: error\ndata: ${JSON.stringify({ error: { message: err.message, type: "upstream_connection_error" } })}\n\n`);
300
+ } catch {}
301
+ }
302
+ });
303
+
304
+ if (reqBodyBuffer) {
305
+ upstreamReq.end(reqBodyBuffer);
306
+ } else {
307
+ req.pipe(upstreamReq);
308
+ }
309
+ }
310
+
311
+ const server = http.createServer((req, res) => {
312
+ // Local health check
313
+ if (req.url === "/health") {
314
+ res.writeHead(200, { "Content-Type": "application/json" });
315
+ return res.end(
316
+ JSON.stringify({
317
+ status: "ok",
318
+ service: "mcp-castor-stream-proxy",
319
+ upstream_port: UPSTREAM_PORT,
320
+ proxy_port: PROXY_PORT,
321
+ pid: process.pid,
322
+ })
323
+ );
324
+ }
325
+
326
+ // Deep sanitize incoming chat completions requests to guard against multimodal image crashes
327
+ if (req.method === "POST" && req.url.startsWith("/v1/chat/completions")) {
328
+ const chunks = [];
329
+ let byteCount = 0;
330
+ // D13 (FX6): single source for the inbound body limit — imported from
331
+ // config.js (PROXY_MAX_BODY_BYTES). The old inline 50MB literal was a
332
+ // dual-source drift risk against config.js; it is gone.
333
+ let exceeded = false;
334
+
335
+ req.on("data", (chunk) => {
336
+ if (exceeded) return;
337
+ byteCount += chunk.length;
338
+ if (byteCount > PROXY_MAX_BODY_BYTES) {
339
+ exceeded = true;
340
+ res.writeHead(413, { "Content-Type": "application/json" });
341
+ res.end(
342
+ JSON.stringify({
343
+ error: {
344
+ message: `Payload Too Large: request body exceeded ${PROXY_MAX_BODY_BYTES} bytes limit.`,
345
+ type: "payload_too_large",
346
+ code: 413,
347
+ },
348
+ })
349
+ );
350
+ req.destroy();
351
+ return;
352
+ }
353
+ chunks.push(chunk);
354
+ });
355
+ req.on("end", () => {
356
+ if (exceeded) return;
357
+ const rawBody = Buffer.concat(chunks);
358
+ try {
359
+ const bodyStr = rawBody.toString("utf8");
360
+ const visionEnabled =
361
+ process.env.VISION === "1" || process.env.ENABLE_VISION === "1";
362
+ if (
363
+ !visionEnabled &&
364
+ (bodyStr.includes('"image"') ||
365
+ bodyStr.includes('"image_url"') ||
366
+ bodyStr.includes('"input_image"') ||
367
+ bodyStr.includes('"image_file"') ||
368
+ bodyStr.includes("data:image/"))
369
+ ) {
370
+ const body = JSON.parse(bodyStr);
371
+ let modified = false;
372
+
373
+ function sanitizeItem(item) {
374
+ if (!item) return item;
375
+ if (typeof item === "string") return item;
376
+ if (Array.isArray(item)) {
377
+ return item.map(sanitizeItem);
378
+ }
379
+ if (typeof item === "object") {
380
+ const type = item.type;
381
+ if (
382
+ type === "image_url" ||
383
+ type === "image" ||
384
+ type === "input_image" ||
385
+ type === "image_file" ||
386
+ item.image ||
387
+ item.image_url
388
+ ) {
389
+ modified = true;
390
+ return {
391
+ type: "text",
392
+ text: "[Image file omitted: Local Qwen3.8-27B runs in pure text mode for Universal 245K context. Images must be inspected multimodally by the Lead Architect.]",
393
+ };
394
+ }
395
+ for (const k of Object.keys(item)) {
396
+ item[k] = sanitizeItem(item[k]);
397
+ }
398
+ }
399
+ return item;
400
+ }
401
+
402
+ if (Array.isArray(body.messages)) {
403
+ for (const msg of body.messages) {
404
+ msg.content = sanitizeItem(msg.content);
405
+ }
406
+ }
407
+ if (body.prompt) {
408
+ body.prompt = sanitizeItem(body.prompt);
409
+ }
410
+
411
+ if (modified) {
412
+ const sanitizedBuffer = Buffer.from(JSON.stringify(body), "utf8");
413
+ return forwardToUpstream(req, res, sanitizedBuffer);
414
+ }
415
+ }
416
+ } catch (err) {
417
+ // Fall back to piping raw body if parsing fails
418
+ }
419
+ forwardToUpstream(req, res, rawBody);
420
+ });
421
+ return;
422
+ }
423
+
424
+ forwardToUpstream(req, res, null);
425
+ });
426
+
427
+ server.on("error", (err) => {
428
+ if (err.code === "EADDRINUSE") {
429
+ console.error(`[StreamProxy] Port ${PROXY_PORT} is already in use — port conflict is a failure, not a success.`);
430
+ process.exit(1);
431
+ } else {
432
+ console.error("[StreamProxy] Server error:", err);
433
+ process.exit(1);
434
+ }
435
+ });
436
+
437
+ // Disable Node.js server timeouts so long-running reasoning streams on 245K context never get killed
438
+ server.timeout = 0;
439
+ server.requestTimeout = 0;
440
+ server.headersTimeout = 0;
441
+ server.keepAliveTimeout = 0;
442
+
443
+ server.listen(PROXY_PORT, PROXY_HOST, () => {
444
+ console.log(`[StreamProxy] Universal stream proxy listening on ${PROXY_HOST}:${PROXY_PORT} -> 127.0.0.1:${UPSTREAM_PORT}`);
445
+ });
446
+
447
+ process.on("SIGTERM", () => {
448
+ server.close(() => process.exit(0));
449
+ });
450
+
451
+ process.on("SIGINT", () => {
452
+ server.close(() => process.exit(0));
453
+ });