mcp-castor 2026.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +487 -0
  2. package/bin/castor.js +706 -0
  3. package/index.js +206 -0
  4. package/package.json +97 -0
  5. package/skills/canary-test-staging/SKILL.md +24 -0
  6. package/skills/evo-mutation-rollback/SKILL.md +29 -0
  7. package/skills/hypothesis-generation/SKILL.md +26 -0
  8. package/skills/traceback-condensing/SKILL.md +26 -0
  9. package/src/castor_runner.js +469 -0
  10. package/src/config.js +1204 -0
  11. package/src/env.js +10 -0
  12. package/src/evo_engine.js +214 -0
  13. package/src/harness/core/events.js +75 -0
  14. package/src/harness/core/kernel.js +209 -0
  15. package/src/harness/evo/evaluator.js +156 -0
  16. package/src/harness/evo/evo_operator.js +550 -0
  17. package/src/harness/evo/lineage_dag.js +383 -0
  18. package/src/harness/evo/trace_repair.js +173 -0
  19. package/src/harness/evo/watchdog.js +72 -0
  20. package/src/harness/loop_detector.js +135 -0
  21. package/src/harness/runner.js +1216 -0
  22. package/src/harness/services/ast_service.js +1813 -0
  23. package/src/harness/services/event_logger.js +275 -0
  24. package/src/harness/services/mcp_bridge.js +408 -0
  25. package/src/harness/services/provider_vllm.js +728 -0
  26. package/src/harness/services/sandbox_fs.js +1238 -0
  27. package/src/harness/services/searxng_lifecycle.js +254 -0
  28. package/src/harness/services/shell_executor.js +264 -0
  29. package/src/harness/services/shell_validator.js +506 -0
  30. package/src/harness/services/web_service.js +828 -0
  31. package/src/platform.js +344 -0
  32. package/src/repetition_detector.js +139 -0
  33. package/src/semaphore.js +373 -0
  34. package/src/server_lifecycle.js +781 -0
  35. package/src/skills.js +400 -0
  36. package/src/state_pruner.js +392 -0
  37. package/src/task_registry.js +1357 -0
  38. package/src/telemetry.js +638 -0
  39. package/src/tools.js +997 -0
  40. package/src/wsl_bridge.js +629 -0
  41. package/src/wsl_env.js +171 -0
  42. package/stream_proxy.js +453 -0
@@ -0,0 +1,781 @@
1
+ import fs from "node:fs";
2
+ import path from "node:path";
3
+ import { fileURLToPath } from "node:url";
4
+ import {
5
+ BASE_URL,
6
+ BOOT_TIMEOUT_MS,
7
+ BOOT_POLL_MS,
8
+ STREAM_PROXY_PORT,
9
+ STREAM_PROXY_PORT_RESOLVED,
10
+ USE_STREAM_PROXY,
11
+ USE_STREAM_PROXY_RESOLVED,
12
+ IS_WINDOWS,
13
+ TASK_DIR,
14
+ WEDGE_STATS_SILENCE_S,
15
+ AUTO_HEAL,
16
+ HEAL_LOCK_FILE,
17
+ HEAL_LOCK_TTL_MS,
18
+ ENGINE_BOOT_LOCK_FILE,
19
+ ENGINE_BOOT_LOCK_TTL_MS,
20
+ ENGINE_LOG_PATH,
21
+ WEDGE_COUNTER_FILE,
22
+ ALLOW_ENGINE_INTERRUPT,
23
+ IS_TEST_ENV,
24
+ MODEL,
25
+ LAUNCH_COMMAND,
26
+ } from "./config.js";
27
+ import { getApiKeySync, runWslCommand } from "./wsl_bridge.js";
28
+ import { streamProxyPath, launcherScriptPath } from "./platform.js";
29
+ import { wslAvailable } from "./wsl_env.js";
30
+
31
+ // Indirection for the WSL command runner; tests inject a stub to run offline.
32
+ let wslRun = runWslCommand;
33
+ export function setWslRunner(fn) {
34
+ wslRun = typeof fn === "function" ? fn : runWslCommand;
35
+ }
36
+
37
+ // Heal gatekeeper: returns true when live work is in flight and the engine
38
+ // must not be stopped/rebooted. Wired at registration time (index.js /
39
+ // tools.js) to the task registry so this module avoids a hard dependency on
40
+ // task_registry.js (which starts the status HTTP server and a retention
41
+ // interval at import). Default: no gate (heal allowed).
42
+ let healGatekeeper = null;
43
+ export function setHealGatekeeper(fn) {
44
+ healGatekeeper = typeof fn === "function" ? fn : null;
45
+ }
46
+
47
+ const __filename = fileURLToPath(import.meta.url);
48
+ const __dirname = path.dirname(__filename);
49
+
50
+ let bootMutex = Promise.resolve();
51
+
52
+ /**
53
+ * Serializes a function against a module-level mutex so only one invocation
54
+ * runs at a time; each call waits for the previous one to settle (resolve or
55
+ * reject) before starting.
56
+ * @param {() => any} fn - The function to run under the mutex.
57
+ * @returns {Promise<any>} The promise returned by `fn`.
58
+ */
59
+ export function withBootMutex(fn) {
60
+ const result = bootMutex.then(fn, fn);
61
+ bootMutex = result.then(
62
+ () => {},
63
+ () => {}
64
+ );
65
+ return result;
66
+ }
67
+
68
+ /**
69
+ * Atomically acquires an exclusive lockfile using O_EXCL (openSync 'wx').
70
+ * If the lockfile exists:
71
+ * - Reads existing metadata.
72
+ * - If active (age < ttlMs), returns { acquired: false, heldBy: cur }.
73
+ * - If stale (age >= ttlMs), atomically renames to a PID-tagged tombstone,
74
+ * unlinks the tombstone, and retries openSync("wx") once.
75
+ *
76
+ * @param {string} lockPath
77
+ * @param {number} ttlMs
78
+ * @param {object} payload
79
+ * @returns {{ acquired: boolean, heldBy?: object }}
80
+ */
81
+ export function tryAcquireExclusiveLock(lockPath, ttlMs, payload) {
82
+ try {
83
+ fs.mkdirSync(path.dirname(lockPath), { recursive: true });
84
+ } catch {}
85
+
86
+ const writePayload = (fd) => {
87
+ fs.writeSync(fd, JSON.stringify({ ...payload, at: Date.now(), pid: process.pid }));
88
+ fs.closeSync(fd);
89
+ };
90
+
91
+ try {
92
+ const fd = fs.openSync(lockPath, "wx");
93
+ writePayload(fd);
94
+ return { acquired: true };
95
+ } catch (err) {
96
+ if (err.code !== "EEXIST") {
97
+ return { acquired: false, error: err.message };
98
+ }
99
+ }
100
+
101
+ let cur = null;
102
+ try {
103
+ cur = JSON.parse(fs.readFileSync(lockPath, "utf8"));
104
+ } catch {}
105
+
106
+ const age = cur?.at ? Date.now() - cur.at : Infinity;
107
+ if (cur && age < ttlMs) {
108
+ return { acquired: false, heldBy: cur };
109
+ }
110
+
111
+ // Stale recovery: Atomic Rename-to-Tombstone
112
+ const tombstone = `${lockPath}.stale_${Date.now()}_${process.pid}`;
113
+ try {
114
+ fs.renameSync(lockPath, tombstone);
115
+ fs.rmSync(tombstone, { force: true });
116
+ } catch {}
117
+
118
+ try {
119
+ const fd = fs.openSync(lockPath, "wx");
120
+ writePayload(fd);
121
+ return { acquired: true };
122
+ } catch {
123
+ return { acquired: false, heldBy: cur };
124
+ }
125
+ }
126
+
127
+ /**
128
+ * Releases an exclusive lockfile only if owned by this process.
129
+ * @param {string} lockPath
130
+ */
131
+ export function releaseExclusiveLock(lockPath) {
132
+ try {
133
+ const cur = JSON.parse(fs.readFileSync(lockPath, "utf8"));
134
+ if (!cur || cur.pid === process.pid) {
135
+ fs.rmSync(lockPath, { force: true });
136
+ }
137
+ } catch {
138
+ try {
139
+ fs.rmSync(lockPath, { force: true });
140
+ } catch {}
141
+ }
142
+ }
143
+
144
+ /**
145
+ * Queries the vLLM engine's /models endpoint and reports the maximum model
146
+ * length of the first served model.
147
+ * @returns {Promise<{maxModelLen: number}|null>} The max model length, or
148
+ * null when the engine is unreachable or returns a non-OK status.
149
+ */
150
+ export async function serverInfo() {
151
+ try {
152
+ const key = getApiKeySync();
153
+ // Omit the Authorization header when no key file is present.
154
+ const headers = {};
155
+ if (key) headers.Authorization = `Bearer ${key}`;
156
+ const res = await fetch(`${BASE_URL}/models`, {
157
+ headers,
158
+ signal: AbortSignal.timeout(2000),
159
+ });
160
+ if (!res.ok) return null;
161
+ const j = await res.json();
162
+ const len = j?.data?.[0]?.max_model_len;
163
+ return { maxModelLen: len ?? 0 };
164
+ } catch {
165
+ return null;
166
+ }
167
+ }
168
+
169
+ /**
170
+ * Classifies the engine by its max model length.
171
+ * @returns {Promise<"huge"|"fast"|"unknown"|null>} "huge" for >= 200000,
172
+ * "fast" for > 0, "unknown" for 0, or null when the engine is unreachable.
173
+ */
174
+ async function currentMode() {
175
+ const info = await serverInfo();
176
+ if (!info) return null;
177
+ if (info.maxModelLen >= 200_000) return "huge";
178
+ if (info.maxModelLen > 0) return "fast";
179
+ return "unknown";
180
+ }
181
+
182
+ let canaryCache = { at: 0, result: null };
183
+
184
+ /**
185
+ * Probes engine health by issuing a minimal chat completion and inspecting the
186
+ * response for visible content or reasoning. Results are cached for 60s.
187
+ * @param {boolean} [force=false] - Bypass the 60s cache and re-probe.
188
+ * @returns {Promise<object>} A result object with `ok` (boolean),
189
+ * `latency_ms`, and either `skipped` (when the probe was not run) or
190
+ * `reply`/`content_chars`/`has_reasoning`/`finish_reason`/`error`.
191
+ */
192
+ export async function canaryProbe(force = false) {
193
+ if (wslRun === runWslCommand && (process.env.TEST_OFFLINE === "1" || (IS_TEST_ENV && !ALLOW_ENGINE_INTERRUPT))) {
194
+ return { ok: true, skipped: "engine_protected_offline", latency_ms: 0 };
195
+ }
196
+ if (!force && canaryCache.result && Date.now() - canaryCache.at < 60_000) {
197
+ return canaryCache.result;
198
+ }
199
+ const t0 = Date.now();
200
+ let result;
201
+ try {
202
+ const key = getApiKeySync();
203
+ // Omit the Authorization header when no key file is present.
204
+ const headers = { "Content-Type": "application/json" };
205
+ if (key) headers.Authorization = `Bearer ${key}`;
206
+ const res = await fetch(`${BASE_URL}/chat/completions`, {
207
+ method: "POST",
208
+ headers,
209
+ body: JSON.stringify({
210
+ model: MODEL,
211
+ // Cap on total generated tokens (content + reasoning).
212
+ max_tokens: 512,
213
+ messages: [{ role: "user", content: "Reply with: ok" }],
214
+ }),
215
+ signal: AbortSignal.timeout(15_000),
216
+ });
217
+ const dt = Date.now() - t0;
218
+ if (!res.ok) {
219
+ result = { ok: false, latency_ms: dt, error: `HTTP ${res.status}` };
220
+ } else {
221
+ const data = await res.json();
222
+ const content = data?.choices?.[0]?.message?.content ?? "";
223
+ const reasoning =
224
+ data?.choices?.[0]?.message?.reasoning ??
225
+ data?.choices?.[0]?.message?.reasoning_content ??
226
+ "";
227
+ const contentTrimmed = content.trim();
228
+ const reasoningTrimmed = reasoning.toString().trim();
229
+ const hasEvidence = contentTrimmed.length > 0 || reasoningTrimmed.length > 0;
230
+ if (hasEvidence) {
231
+ result = {
232
+ ok: true,
233
+ latency_ms: dt,
234
+ reply: contentTrimmed,
235
+ content_chars: contentTrimmed.length,
236
+ has_reasoning: reasoningTrimmed.length > 0,
237
+ finish_reason: data?.choices?.[0]?.finish_reason ?? null,
238
+ };
239
+ } else {
240
+ // HTTP 200 with neither content nor reasoning is a generation-less
241
+ // response and is not treated as a healthy engine.
242
+ result = {
243
+ ok: false,
244
+ latency_ms: dt,
245
+ reply: "",
246
+ content_chars: 0,
247
+ has_reasoning: false,
248
+ finish_reason: data?.choices?.[0]?.finish_reason ?? null,
249
+ error: "generation-less response (no content, no reasoning)",
250
+ };
251
+ }
252
+ }
253
+ } catch (err) {
254
+ result = {
255
+ ok: false,
256
+ latency_ms: Date.now() - t0,
257
+ error: err.name === "TimeoutError" ? "timeout (15s)" : err.message,
258
+ };
259
+ }
260
+ canaryCache = { at: Date.now(), result };
261
+ return result;
262
+ }
263
+
264
+ async function readLastEngineStatsLine() {
265
+ try {
266
+ const { stdout } = await wslRun(
267
+ `grep -a 'Engine 000:.*Running:' ${ENGINE_LOG_PATH} 2>/dev/null | tail -1`
268
+ );
269
+ return stdout.trim() || null;
270
+ } catch {
271
+ return null;
272
+ }
273
+ }
274
+
275
+ function parseEngineStats(line) {
276
+ if (!line) return null;
277
+ const ts = line.match(/(\d{2})-(\d{2}) (\d{2}):(\d{2}):(\d{2})/);
278
+ if (!ts) return null;
279
+ const now = new Date();
280
+ const stamp = new Date(
281
+ now.getFullYear(),
282
+ Number(ts[1]) - 1,
283
+ Number(ts[2]),
284
+ Number(ts[3]),
285
+ Number(ts[4]),
286
+ Number(ts[5])
287
+ );
288
+ return {
289
+ ageSec: Math.max(0, Math.round((now.getTime() - stamp.getTime()) / 1000)),
290
+ runningReqs: Number(line.match(/Running: (\d+) reqs/)?.[1] ?? -1),
291
+ waitingReqs: Number(line.match(/Waiting: (\d+) reqs/)?.[1] ?? -1),
292
+ };
293
+ }
294
+
295
+ export function readWedgeCounter() {
296
+ try {
297
+ return JSON.parse(fs.readFileSync(WEDGE_COUNTER_FILE, "utf8"));
298
+ } catch {
299
+ return { count: 0, lastAt: null, lastReason: null };
300
+ }
301
+ }
302
+
303
+ export function bumpWedgeCounter(reason) {
304
+ try {
305
+ const cur = readWedgeCounter();
306
+ cur.count = (cur.count ?? 0) + 1;
307
+ cur.lastAt = Date.now();
308
+ cur.lastReason = String(reason ?? "").slice(0, 200);
309
+ fs.mkdirSync(path.dirname(WEDGE_COUNTER_FILE), { recursive: true });
310
+ const tmp = `${WEDGE_COUNTER_FILE}.tmp_${Date.now()}_${process.pid}`;
311
+ fs.writeFileSync(tmp, JSON.stringify(cur), "utf8");
312
+ fs.renameSync(tmp, WEDGE_COUNTER_FILE);
313
+ } catch (err) {
314
+ process.stderr.write(`[server_lifecycle] Failed to bump wedge counter: ${err.message}\n`);
315
+ }
316
+ }
317
+
318
+ /**
319
+ * Determines whether the engine is wedged. The engine runs a single
320
+ * concurrent generation (MAX_SEQS=1), so a canary probe is only meaningful
321
+ * when the engine is idle; while busy, a canary would queue behind the active
322
+ * generation and time out, so only stats silence is used as a wedge signal.
323
+ * @returns {Promise<object>} A state object with `wedged`, `canary`, `stats`,
324
+ * `isSilenceWedged`, `isCanaryWedged`, `engineBusy`, and `gauges`.
325
+ */
326
+ export async function engineWedgeState() {
327
+ // Read the engine gauges first to determine busy state.
328
+ const metrics = await readEngineMetrics();
329
+ // /metrics unavailable is treated as busy (fail-closed).
330
+ const metricsUnavailable = metrics === null;
331
+ const runningReqs = metrics?.["vllm:num_requests_running"] ?? 0;
332
+ const waitingReqs = metrics?.["vllm:num_requests_waiting"] ?? 0;
333
+ const engineBusy = metricsUnavailable || runningReqs > 0 || waitingReqs > 0;
334
+
335
+ const line = await readLastEngineStatsLine();
336
+ const stats = line ? parseEngineStats(line) : null;
337
+
338
+ // Stats silence while requests are running indicates a stalled engine core.
339
+ const isSilenceWedged = Boolean(
340
+ runningReqs > 0 && stats && stats.runningReqs > 0 && stats.ageSec > WEDGE_STATS_SILENCE_S
341
+ );
342
+
343
+ let canary;
344
+ let isCanaryWedged;
345
+ if (engineBusy) {
346
+ // When busy due to a metrics fetch failure, the canary is refused with an
347
+ // explicit error (fail-closed) rather than a neutral sentinel.
348
+ canary = metricsUnavailable
349
+ ? { ok: false, skipped: "metrics_unavailable", error: "/metrics fetch failed; busy-gate fail-closed" }
350
+ : { ok: true, skipped: "engine_busy", latency_ms: 0 };
351
+ isCanaryWedged = false;
352
+ } else {
353
+ canary = await canaryProbe();
354
+ isCanaryWedged = !canary.ok;
355
+ }
356
+
357
+ // When busy, the canary is ignored — only stats silence can declare a wedge.
358
+ // When idle, the canary probe is authoritative.
359
+ const wedged = engineBusy
360
+ ? isSilenceWedged
361
+ : Boolean(isCanaryWedged);
362
+
363
+ return {
364
+ wedged,
365
+ canary,
366
+ stats,
367
+ isSilenceWedged,
368
+ isCanaryWedged,
369
+ engineBusy,
370
+ gauges: metrics
371
+ ? {
372
+ running_requests: metrics["vllm:num_requests_running"] ?? null,
373
+ waiting_requests: metrics["vllm:num_requests_waiting"] ?? null,
374
+ kv_cache_pct: metrics["vllm:gpu_cache_usage_factor"]
375
+ ? Math.round(metrics["vllm:gpu_cache_usage_factor"] * 1000) / 10
376
+ : null,
377
+ prefix_cache_hit_ratio: metrics["vllm:prefix_cache_hit_rate"] ?? null,
378
+ spec_decode_acceptance: metrics["vllm:spec_decode_draft_acceptance_rate"] ?? null,
379
+ }
380
+ : null,
381
+ };
382
+ }
383
+
384
+ /**
385
+ * Stops and restarts a wedged engine. Refuses to run when engine interruption
386
+ * is disabled, when the heal gatekeeper reports live work in flight, or when
387
+ * another process already holds the heal lock.
388
+ * @param {number|null} statsAgeSec - Age of the last engine stats line, in
389
+ * seconds; recorded in the heal lock payload.
390
+ * @returns {Promise<object>} A result object with `healed` (boolean) and a
391
+ * `note` (when refused) or `boot` status (when healed).
392
+ */
393
+ export async function healWedgedEngine(statsAgeSec) {
394
+ if (wslRun === runWslCommand && (process.env.TEST_OFFLINE === "1" || (IS_TEST_ENV && !ALLOW_ENGINE_INTERRUPT))) {
395
+ return { healed: false, note: "heal refused: engine interruption disabled by default (ALLOW_ENGINE_INTERRUPT unset)" };
396
+ }
397
+ // Refuse to stop/reboot the engine while live work is in flight.
398
+ if (healGatekeeper) {
399
+ try {
400
+ if (healGatekeeper()) {
401
+ return { healed: false, note: "tasks in flight; heal refused" };
402
+ }
403
+ } catch {
404
+ return { healed: false, note: "tasks in flight; heal refused" };
405
+ }
406
+ }
407
+
408
+ const lockResult = tryAcquireExclusiveLock(HEAL_LOCK_FILE, HEAL_LOCK_TTL_MS, {
409
+ pid: process.pid,
410
+ statsAgeSec,
411
+ });
412
+
413
+ if (!lockResult.acquired) {
414
+ const lock = lockResult.heldBy;
415
+ const ago = lock?.at ? Math.round((Date.now() - lock.at) / 1000) : "unknown";
416
+ const pid = lock?.pid ?? "unknown";
417
+ return {
418
+ healed: false,
419
+ note: `heal already started ${ago}s ago by pid ${pid}; boot in progress`,
420
+ };
421
+ }
422
+
423
+ bumpWedgeCounter(`stats_age=${statsAgeSec}s`);
424
+ await stopServer();
425
+ const res = await ensureServerRunning();
426
+ resetEngineHealthCache();
427
+ return { healed: true, boot: res.status };
428
+ }
429
+
430
+ async function warmEngine() {
431
+ try {
432
+ const key = getApiKeySync();
433
+ // Omit Authorization header when no key file is present.
434
+ const headers = { "Content-Type": "application/json" };
435
+ if (key) headers.Authorization = `Bearer ${key}`;
436
+ await fetch(`${BASE_URL}/chat/completions`, {
437
+ method: "POST",
438
+ headers,
439
+ body: JSON.stringify({
440
+ model: MODEL,
441
+ messages: [{ role: "user", content: "ping" }],
442
+ max_tokens: 1,
443
+ temperature: 0.0,
444
+ }),
445
+ signal: AbortSignal.timeout(60_000),
446
+ });
447
+ } catch {}
448
+ }
449
+
450
+ // ---------------------------------------------------------------------------
451
+ // Stream-proxy lifecycle
452
+ // ---------------------------------------------------------------------------
453
+ // Pre-spawn cleanup is targeted: it kills the current listener on the port by
454
+ // its specific pid (probed from /health or `ss -ltnp`), never a broad `pkill -f`.
455
+
456
+ // Indirection for the stream-proxy spawner; tests inject a stub to simulate a
457
+ // slow or failed start.
458
+ let spawnStreamProxy = null;
459
+ export function setStreamProxySpawner(fn) {
460
+ spawnStreamProxy = typeof fn === "function" ? fn : null;
461
+ }
462
+
463
+ /**
464
+ * The real stream-proxy spawner. Windows: spawn inside WSL via setsid.
465
+ * Linux: spawn a detached node child directly.
466
+ */
467
+ async function realSpawnStreamProxy() {
468
+ if (IS_WINDOWS) {
469
+ // WSL tears the session down when the `bash -c` leader exits, which can
470
+ // kill a freshly-forked background child before it execs; the 1s linger
471
+ // keeps the session alive long enough for the child to detach.
472
+ await runWslCommand(
473
+ `setsid node ${streamProxyPath()} < /dev/null > /tmp/stream_proxy.log 2>&1 & sleep 1`
474
+ );
475
+ } else {
476
+ const { spawn } = await import("child_process");
477
+ const p = spawn("node", [path.join(__dirname, "..", "stream_proxy.js")], {
478
+ stdio: "ignore",
479
+ detached: true,
480
+ });
481
+ p.unref();
482
+ }
483
+ }
484
+
485
+ /**
486
+ * Finds the pid of the process currently listening on the stream-proxy port,
487
+ * if any. Probes /health first, then falls back to `ss -ltnp` on the port.
488
+ * @returns {Promise<{pid: number, verified: boolean}|null>} The listener pid
489
+ * and whether it was verified as the stream proxy, or null when no listener
490
+ * is found. Never throws.
491
+ */
492
+ async function findStreamProxyListenerPid() {
493
+ const port = STREAM_PROXY_PORT_RESOLVED;
494
+ // 1. Probe /health for a verified pid.
495
+ try {
496
+ const res = await fetch(`http://127.0.0.1:${port}/health`, {
497
+ signal: AbortSignal.timeout(1000),
498
+ });
499
+ if (res.ok) {
500
+ const j = await res.json();
501
+ if (j && (j.service === "mcp-castor-stream-proxy" || j.service === "mcp-qwen-stream-proxy") && Number.isFinite(j.pid) && j.pid > 0) {
502
+ return { pid: j.pid, verified: true };
503
+ }
504
+ }
505
+ } catch {}
506
+ // 2. Fall back to `ss -ltnp` filtered to the port.
507
+ try {
508
+ const { stdout } = await runWslCommand(
509
+ `ss -ltnp 'sport = :${port}' 2>/dev/null | grep -oE 'pid=[0-9]+' | head -1 | cut -d= -f2 || true`
510
+ );
511
+ const pid = parseInt(stdout.trim(), 10);
512
+ if (Number.isFinite(pid) && pid > 0) {
513
+ try {
514
+ const { stdout: cmdline } = await runWslCommand(
515
+ `tr '\\0' ' ' < /proc/${pid}/cmdline 2>/dev/null || true`
516
+ );
517
+ if (cmdline.includes("stream_proxy.js")) {
518
+ return { pid, verified: true };
519
+ }
520
+ } catch {}
521
+ return { pid, verified: false };
522
+ }
523
+ } catch {}
524
+ return null;
525
+ }
526
+
527
+ /**
528
+ * Ensures the universal stream proxy is running and healthy on
529
+ * STREAM_PROXY_PORT.
530
+ * @param {object} [opts]
531
+ * @param {number} [opts.healthPolls=75] Number of 200ms health polls before
532
+ * declaring failure (default 75 = 15s). Tests may pass a smaller value.
533
+ * @returns {Promise<boolean>} true once the proxy is healthy.
534
+ * @throws {Error} If the proxy does not become healthy within the window;
535
+ * the message includes the captured spawn failure when present.
536
+ */
537
+ export async function ensureStreamProxyRunning({ healthPolls = 75 } = {}) {
538
+ // 0a. If the stream proxy is disabled by configuration, skip cleanly.
539
+ // The engine is reached directly (no proxy hop).
540
+ if (!USE_STREAM_PROXY_RESOLVED) {
541
+ process.stderr.write(
542
+ `[stream-proxy] disabled by configuration (USE_STREAM_PROXY=false); using direct engine connection\n`
543
+ );
544
+ return true;
545
+ }
546
+
547
+ // 0b. On Windows without WSL, the proxy (which runs inside WSL) cannot be
548
+ // spawned. Do NOT crash the whole task: warn and allow a direct engine
549
+ // connection so the task can still proceed.
550
+ if (!spawnStreamProxy && IS_WINDOWS && !wslAvailable()) {
551
+ process.stderr.write(
552
+ `[stream-proxy] WSL is not available on this Windows host; skipping proxy and using direct engine connection\n`
553
+ );
554
+ return true;
555
+ }
556
+
557
+ const port = STREAM_PROXY_PORT_RESOLVED;
558
+
559
+ // 1. Is the proxy already healthy? (two quick probes)
560
+ for (let attempt = 0; attempt < 2; attempt++) {
561
+ try {
562
+ const res = await fetch(`http://127.0.0.1:${port}/health`, {
563
+ signal: AbortSignal.timeout(1000),
564
+ });
565
+ if (res.ok) return true;
566
+ } catch {}
567
+ if (attempt === 0) await new Promise((r) => setTimeout(r, 200));
568
+ }
569
+
570
+ // 2. TARGETED pre-spawn cleanup: kill the current listener on the port ONLY if verified.
571
+ const listenerInfo = await findStreamProxyListenerPid();
572
+ if (listenerInfo) {
573
+ if (!listenerInfo.verified) {
574
+ throw new Error(
575
+ `PortConflictError: Port ${port} is occupied by unverified process pid ${listenerInfo.pid}. Refusing to kill non-proxy process.`
576
+ );
577
+ }
578
+ const listenerPid = listenerInfo.pid;
579
+ try {
580
+ if (IS_WINDOWS) {
581
+ await runWslCommand(`kill -9 ${listenerPid} 2>/dev/null || true`);
582
+ } else {
583
+ try {
584
+ process.kill(listenerPid, "SIGKILL");
585
+ } catch {}
586
+ }
587
+ process.stderr.write(
588
+ `[stream-proxy] pre-spawn cleanup: killed verified listener pid ${listenerPid} on port ${port}\n`
589
+ );
590
+ await new Promise((r) => setTimeout(r, 300));
591
+ } catch (err) {
592
+ process.stderr.write(
593
+ `[stream-proxy] pre-spawn cleanup: failed to kill pid ${listenerPid}: ${err.message}\n`
594
+ );
595
+ }
596
+ } else {
597
+ process.stderr.write(
598
+ `[stream-proxy] pre-spawn cleanup: no listener found on port ${port}\n`
599
+ );
600
+ }
601
+
602
+ // 3. SPAWN the proxy (injected spawner for tests, real spawner otherwise).
603
+ // Do NOT swallow spawn errors: capture them for the final error message.
604
+ let spawnFailure = null;
605
+ try {
606
+ if (spawnStreamProxy) {
607
+ await spawnStreamProxy();
608
+ } else {
609
+ await realSpawnStreamProxy();
610
+ }
611
+ } catch (err) {
612
+ spawnFailure = err;
613
+ process.stderr.write(`[stream-proxy] spawn failed: ${err.message}\n`);
614
+ }
615
+
616
+ // 4. Health window: poll until the proxy reports healthy, with early exit.
617
+ for (let i = 0; i < healthPolls; i++) {
618
+ await new Promise((r) => setTimeout(r, 200));
619
+ try {
620
+ const res = await fetch(`http://127.0.0.1:${port}/health`, {
621
+ signal: AbortSignal.timeout(1000),
622
+ });
623
+ if (res.ok) return true;
624
+ } catch {}
625
+ }
626
+
627
+ const seconds = Math.round((healthPolls * 200) / 1000);
628
+ const spawnNote = spawnFailure ? ` (spawn failure: ${spawnFailure.message})` : "";
629
+ throw new Error(
630
+ `Stream proxy failed to become healthy on port ${port} after ${seconds}s${spawnNote}`
631
+ );
632
+ }
633
+
634
+ /**
635
+ * Ensures the vLLM engine is running. If the engine is already up it checks
636
+ * for a wedge (healing one when AUTO_HEAL is enabled); otherwise it boots the
637
+ * engine, coordinating across processes via an atomic boot lock.
638
+ * @returns {Promise<object>} A result object with `switched` (boolean) and a
639
+ * `status` string; when a wedged engine was healed, a `heal` object is
640
+ * included.
641
+ */
642
+ export async function ensureServerRunning() {
643
+ if (wslRun === runWslCommand && (process.env.TEST_OFFLINE === "1" || (IS_TEST_ENV && !ALLOW_ENGINE_INTERRUPT))) {
644
+ return { switched: false, status: "boot_refused_offline_protected" };
645
+ }
646
+ const current = await currentMode();
647
+ if (current) {
648
+ const wedge = await engineWedgeState();
649
+ if (wedge.wedged && AUTO_HEAL) {
650
+ const heal = await healWedgedEngine(wedge.stats?.ageSec ?? null);
651
+ return { switched: true, status: "restarted_wedged_engine", heal };
652
+ }
653
+ return { switched: false, status: "already_running" };
654
+ }
655
+
656
+ // Cross-process atomic boot lock: only ONE instance executes the launcher script.
657
+ const lock = tryAcquireExclusiveLock(ENGINE_BOOT_LOCK_FILE, ENGINE_BOOT_LOCK_TTL_MS, {
658
+ pid: process.pid,
659
+ action: "booting_huge",
660
+ });
661
+
662
+ if (!lock.acquired) {
663
+ // Secondary instance: wait for primary instance to finish booting.
664
+ const deadline = Date.now() + BOOT_TIMEOUT_MS;
665
+ while (Date.now() < deadline) {
666
+ await new Promise((r) => setTimeout(r, BOOT_POLL_MS));
667
+ const now = await currentMode();
668
+ if (now) {
669
+ await ensureStreamProxyRunning();
670
+ resetEngineHealthCache();
671
+ return { switched: true, status: "started_by_peer" };
672
+ }
673
+ }
674
+ throw new Error(`Timed out waiting for peer vLLM server to boot (${BOOT_TIMEOUT_MS}ms)`);
675
+ }
676
+
677
+ try {
678
+ if (LAUNCH_COMMAND) {
679
+ // Use the configured launch command (from QWEN_LAUNCH_COMMAND or
680
+ // ~/.castor/config.json `launch_command`). This is the portable path:
681
+ // the user controls exactly how the engine is started.
682
+ await wslRun(
683
+ `nohup bash -c '${LAUNCH_COMMAND.replace(/'/g, "'\\''")}' > ${ENGINE_LOG_PATH} 2>&1 < /dev/null & disown; sleep 1; true`
684
+ );
685
+ } else {
686
+ // Fall back to the repo's launcher script.
687
+ const launcher = launcherScriptPath();
688
+ await wslRun(
689
+ `cd ~/qwen-serving && if [ -f "${launcher}" ]; then nohup bash "${launcher}"; elif [ -f launchers/start_huge.sh ]; then nohup bash launchers/start_huge.sh; else nohup bash single-user/start_qwen.sh; fi > ${ENGINE_LOG_PATH} 2>&1 < /dev/null & disown; sleep 1; true`
690
+ );
691
+ }
692
+ const deadline = Date.now() + BOOT_TIMEOUT_MS;
693
+ while (Date.now() < deadline) {
694
+ await new Promise((r) => setTimeout(r, BOOT_POLL_MS));
695
+ const now = await currentMode();
696
+ if (now) {
697
+ await ensureStreamProxyRunning();
698
+ await warmEngine();
699
+ resetEngineHealthCache();
700
+ return { switched: true, status: "started" };
701
+ }
702
+ }
703
+ throw new Error(`Timed out waiting for vLLM server to boot (${BOOT_TIMEOUT_MS}ms)`);
704
+ } finally {
705
+ releaseExclusiveLock(ENGINE_BOOT_LOCK_FILE);
706
+ }
707
+ }
708
+
709
+ /**
710
+ * Stops the vLLM engine and waits up to 5s for it to stop responding.
711
+ * @returns {Promise<object>} A result object with `stopped` (boolean) and a
712
+ * `reason` string; when the engine still responds after the grace window,
713
+ * `stopped` is false and `mode` reports the detected engine mode.
714
+ */
715
+ export async function stopServer() {
716
+ if (wslRun === runWslCommand && (process.env.TEST_OFFLINE === "1" || (IS_TEST_ENV && !ALLOW_ENGINE_INTERRUPT))) {
717
+ return {
718
+ stopped: false,
719
+ reason: "stop_refused_offline_protected",
720
+ note: "stopServer refused: engine interruption disabled by default (ALLOW_ENGINE_INTERRUPT unset)",
721
+ };
722
+ }
723
+ await wslRun(`cd ~/qwen-serving && bash launchers/stop_server.sh 2>/dev/null || true`);
724
+ let mode = null;
725
+ for (let i = 0; i < 10; i++) {
726
+ await new Promise((r) => setTimeout(r, 500));
727
+ mode = await currentMode();
728
+ if (!mode) return { stopped: true };
729
+ }
730
+ // The grace window elapsed but the engine still responds: report a failure
731
+ // rather than a success.
732
+ return {
733
+ stopped: false,
734
+ reason: "engine_still_responding",
735
+ mode,
736
+ };
737
+ }
738
+
739
+ let metricsCache = { at: 0, data: null };
740
+
741
+ /**
742
+ * Fetches and parses the vLLM /metrics endpoint, returning a map of metric
743
+ * name to value. Results are cached for `maxAgeMs`.
744
+ * @param {number} [maxAgeMs=5000] Maximum age of a cached result, in
745
+ * milliseconds.
746
+ * @returns {Promise<Record<string, number>|null>} The parsed metrics, or null
747
+ * when the fetch fails.
748
+ */
749
+ export async function readEngineMetrics(maxAgeMs = 5000) {
750
+ if (metricsCache.data && Date.now() - metricsCache.at < maxAgeMs) return metricsCache.data;
751
+ try {
752
+ const res = await fetch(`${BASE_URL.replace(/\/v1$/, "")}/metrics`, {
753
+ signal: AbortSignal.timeout(3000),
754
+ });
755
+ if (!res.ok) throw new Error(`HTTP ${res.status}`);
756
+ const map = {};
757
+ for (const line of (await res.text()).split("\n")) {
758
+ if (!line.startsWith("vllm:")) continue;
759
+ const name = line.match(/^(vllm:[^{ ]+)/)?.[1];
760
+ const val = Number(line.slice(line.lastIndexOf(" ") + 1));
761
+ if (!name || !Number.isFinite(val)) continue;
762
+ map[name] = Math.max(map[name] ?? -Infinity, val);
763
+ }
764
+ metricsCache = { at: Date.now(), data: map };
765
+ return map;
766
+ } catch (err) {
767
+ // A /metrics fetch failure is logged to stderr and reported as null; the
768
+ // caller (engineWedgeState) treats it as busy (fail-closed).
769
+ process.stderr.write(`[server_lifecycle] /metrics fetch failed: ${err.message}\n`);
770
+ metricsCache = { at: Date.now(), data: null };
771
+ return null;
772
+ }
773
+ }
774
+
775
+ /**
776
+ * Clears the canary and metrics caches so the next probe re-fetches.
777
+ */
778
+ export function resetEngineHealthCache() {
779
+ canaryCache = { at: 0, result: null };
780
+ metricsCache = { at: 0, data: null };
781
+ }