mcp-castor 2026.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +487 -0
- package/bin/castor.js +706 -0
- package/index.js +206 -0
- package/package.json +97 -0
- package/skills/canary-test-staging/SKILL.md +24 -0
- package/skills/evo-mutation-rollback/SKILL.md +29 -0
- package/skills/hypothesis-generation/SKILL.md +26 -0
- package/skills/traceback-condensing/SKILL.md +26 -0
- package/src/castor_runner.js +469 -0
- package/src/config.js +1204 -0
- package/src/env.js +10 -0
- package/src/evo_engine.js +214 -0
- package/src/harness/core/events.js +75 -0
- package/src/harness/core/kernel.js +209 -0
- package/src/harness/evo/evaluator.js +156 -0
- package/src/harness/evo/evo_operator.js +550 -0
- package/src/harness/evo/lineage_dag.js +383 -0
- package/src/harness/evo/trace_repair.js +173 -0
- package/src/harness/evo/watchdog.js +72 -0
- package/src/harness/loop_detector.js +135 -0
- package/src/harness/runner.js +1216 -0
- package/src/harness/services/ast_service.js +1813 -0
- package/src/harness/services/event_logger.js +275 -0
- package/src/harness/services/mcp_bridge.js +408 -0
- package/src/harness/services/provider_vllm.js +728 -0
- package/src/harness/services/sandbox_fs.js +1238 -0
- package/src/harness/services/searxng_lifecycle.js +254 -0
- package/src/harness/services/shell_executor.js +264 -0
- package/src/harness/services/shell_validator.js +506 -0
- package/src/harness/services/web_service.js +828 -0
- package/src/platform.js +344 -0
- package/src/repetition_detector.js +139 -0
- package/src/semaphore.js +373 -0
- package/src/server_lifecycle.js +781 -0
- package/src/skills.js +400 -0
- package/src/state_pruner.js +392 -0
- package/src/task_registry.js +1357 -0
- package/src/telemetry.js +638 -0
- package/src/tools.js +997 -0
- package/src/wsl_bridge.js +629 -0
- package/src/wsl_env.js +171 -0
- package/stream_proxy.js +453 -0
|
@@ -0,0 +1,781 @@
|
|
|
1
|
+
import fs from "node:fs";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
import { fileURLToPath } from "node:url";
|
|
4
|
+
import {
|
|
5
|
+
BASE_URL,
|
|
6
|
+
BOOT_TIMEOUT_MS,
|
|
7
|
+
BOOT_POLL_MS,
|
|
8
|
+
STREAM_PROXY_PORT,
|
|
9
|
+
STREAM_PROXY_PORT_RESOLVED,
|
|
10
|
+
USE_STREAM_PROXY,
|
|
11
|
+
USE_STREAM_PROXY_RESOLVED,
|
|
12
|
+
IS_WINDOWS,
|
|
13
|
+
TASK_DIR,
|
|
14
|
+
WEDGE_STATS_SILENCE_S,
|
|
15
|
+
AUTO_HEAL,
|
|
16
|
+
HEAL_LOCK_FILE,
|
|
17
|
+
HEAL_LOCK_TTL_MS,
|
|
18
|
+
ENGINE_BOOT_LOCK_FILE,
|
|
19
|
+
ENGINE_BOOT_LOCK_TTL_MS,
|
|
20
|
+
ENGINE_LOG_PATH,
|
|
21
|
+
WEDGE_COUNTER_FILE,
|
|
22
|
+
ALLOW_ENGINE_INTERRUPT,
|
|
23
|
+
IS_TEST_ENV,
|
|
24
|
+
MODEL,
|
|
25
|
+
LAUNCH_COMMAND,
|
|
26
|
+
} from "./config.js";
|
|
27
|
+
import { getApiKeySync, runWslCommand } from "./wsl_bridge.js";
|
|
28
|
+
import { streamProxyPath, launcherScriptPath } from "./platform.js";
|
|
29
|
+
import { wslAvailable } from "./wsl_env.js";
|
|
30
|
+
|
|
31
|
+
// Indirection for the WSL command runner; tests inject a stub to run offline.
|
|
32
|
+
let wslRun = runWslCommand;
|
|
33
|
+
export function setWslRunner(fn) {
|
|
34
|
+
wslRun = typeof fn === "function" ? fn : runWslCommand;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
// Heal gatekeeper: returns true when live work is in flight and the engine
|
|
38
|
+
// must not be stopped/rebooted. Wired at registration time (index.js /
|
|
39
|
+
// tools.js) to the task registry so this module avoids a hard dependency on
|
|
40
|
+
// task_registry.js (which starts the status HTTP server and a retention
|
|
41
|
+
// interval at import). Default: no gate (heal allowed).
|
|
42
|
+
let healGatekeeper = null;
|
|
43
|
+
export function setHealGatekeeper(fn) {
|
|
44
|
+
healGatekeeper = typeof fn === "function" ? fn : null;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
const __filename = fileURLToPath(import.meta.url);
|
|
48
|
+
const __dirname = path.dirname(__filename);
|
|
49
|
+
|
|
50
|
+
let bootMutex = Promise.resolve();
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Serializes a function against a module-level mutex so only one invocation
|
|
54
|
+
* runs at a time; each call waits for the previous one to settle (resolve or
|
|
55
|
+
* reject) before starting.
|
|
56
|
+
* @param {() => any} fn - The function to run under the mutex.
|
|
57
|
+
* @returns {Promise<any>} The promise returned by `fn`.
|
|
58
|
+
*/
|
|
59
|
+
export function withBootMutex(fn) {
|
|
60
|
+
const result = bootMutex.then(fn, fn);
|
|
61
|
+
bootMutex = result.then(
|
|
62
|
+
() => {},
|
|
63
|
+
() => {}
|
|
64
|
+
);
|
|
65
|
+
return result;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* Atomically acquires an exclusive lockfile using O_EXCL (openSync 'wx').
|
|
70
|
+
* If the lockfile exists:
|
|
71
|
+
* - Reads existing metadata.
|
|
72
|
+
* - If active (age < ttlMs), returns { acquired: false, heldBy: cur }.
|
|
73
|
+
* - If stale (age >= ttlMs), atomically renames to a PID-tagged tombstone,
|
|
74
|
+
* unlinks the tombstone, and retries openSync("wx") once.
|
|
75
|
+
*
|
|
76
|
+
* @param {string} lockPath
|
|
77
|
+
* @param {number} ttlMs
|
|
78
|
+
* @param {object} payload
|
|
79
|
+
* @returns {{ acquired: boolean, heldBy?: object }}
|
|
80
|
+
*/
|
|
81
|
+
export function tryAcquireExclusiveLock(lockPath, ttlMs, payload) {
|
|
82
|
+
try {
|
|
83
|
+
fs.mkdirSync(path.dirname(lockPath), { recursive: true });
|
|
84
|
+
} catch {}
|
|
85
|
+
|
|
86
|
+
const writePayload = (fd) => {
|
|
87
|
+
fs.writeSync(fd, JSON.stringify({ ...payload, at: Date.now(), pid: process.pid }));
|
|
88
|
+
fs.closeSync(fd);
|
|
89
|
+
};
|
|
90
|
+
|
|
91
|
+
try {
|
|
92
|
+
const fd = fs.openSync(lockPath, "wx");
|
|
93
|
+
writePayload(fd);
|
|
94
|
+
return { acquired: true };
|
|
95
|
+
} catch (err) {
|
|
96
|
+
if (err.code !== "EEXIST") {
|
|
97
|
+
return { acquired: false, error: err.message };
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
let cur = null;
|
|
102
|
+
try {
|
|
103
|
+
cur = JSON.parse(fs.readFileSync(lockPath, "utf8"));
|
|
104
|
+
} catch {}
|
|
105
|
+
|
|
106
|
+
const age = cur?.at ? Date.now() - cur.at : Infinity;
|
|
107
|
+
if (cur && age < ttlMs) {
|
|
108
|
+
return { acquired: false, heldBy: cur };
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
// Stale recovery: Atomic Rename-to-Tombstone
|
|
112
|
+
const tombstone = `${lockPath}.stale_${Date.now()}_${process.pid}`;
|
|
113
|
+
try {
|
|
114
|
+
fs.renameSync(lockPath, tombstone);
|
|
115
|
+
fs.rmSync(tombstone, { force: true });
|
|
116
|
+
} catch {}
|
|
117
|
+
|
|
118
|
+
try {
|
|
119
|
+
const fd = fs.openSync(lockPath, "wx");
|
|
120
|
+
writePayload(fd);
|
|
121
|
+
return { acquired: true };
|
|
122
|
+
} catch {
|
|
123
|
+
return { acquired: false, heldBy: cur };
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Releases an exclusive lockfile only if owned by this process.
|
|
129
|
+
* @param {string} lockPath
|
|
130
|
+
*/
|
|
131
|
+
export function releaseExclusiveLock(lockPath) {
|
|
132
|
+
try {
|
|
133
|
+
const cur = JSON.parse(fs.readFileSync(lockPath, "utf8"));
|
|
134
|
+
if (!cur || cur.pid === process.pid) {
|
|
135
|
+
fs.rmSync(lockPath, { force: true });
|
|
136
|
+
}
|
|
137
|
+
} catch {
|
|
138
|
+
try {
|
|
139
|
+
fs.rmSync(lockPath, { force: true });
|
|
140
|
+
} catch {}
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Queries the vLLM engine's /models endpoint and reports the maximum model
|
|
146
|
+
* length of the first served model.
|
|
147
|
+
* @returns {Promise<{maxModelLen: number}|null>} The max model length, or
|
|
148
|
+
* null when the engine is unreachable or returns a non-OK status.
|
|
149
|
+
*/
|
|
150
|
+
export async function serverInfo() {
|
|
151
|
+
try {
|
|
152
|
+
const key = getApiKeySync();
|
|
153
|
+
// Omit the Authorization header when no key file is present.
|
|
154
|
+
const headers = {};
|
|
155
|
+
if (key) headers.Authorization = `Bearer ${key}`;
|
|
156
|
+
const res = await fetch(`${BASE_URL}/models`, {
|
|
157
|
+
headers,
|
|
158
|
+
signal: AbortSignal.timeout(2000),
|
|
159
|
+
});
|
|
160
|
+
if (!res.ok) return null;
|
|
161
|
+
const j = await res.json();
|
|
162
|
+
const len = j?.data?.[0]?.max_model_len;
|
|
163
|
+
return { maxModelLen: len ?? 0 };
|
|
164
|
+
} catch {
|
|
165
|
+
return null;
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* Classifies the engine by its max model length.
|
|
171
|
+
* @returns {Promise<"huge"|"fast"|"unknown"|null>} "huge" for >= 200000,
|
|
172
|
+
* "fast" for > 0, "unknown" for 0, or null when the engine is unreachable.
|
|
173
|
+
*/
|
|
174
|
+
async function currentMode() {
|
|
175
|
+
const info = await serverInfo();
|
|
176
|
+
if (!info) return null;
|
|
177
|
+
if (info.maxModelLen >= 200_000) return "huge";
|
|
178
|
+
if (info.maxModelLen > 0) return "fast";
|
|
179
|
+
return "unknown";
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
let canaryCache = { at: 0, result: null };
|
|
183
|
+
|
|
184
|
+
/**
|
|
185
|
+
* Probes engine health by issuing a minimal chat completion and inspecting the
|
|
186
|
+
* response for visible content or reasoning. Results are cached for 60s.
|
|
187
|
+
* @param {boolean} [force=false] - Bypass the 60s cache and re-probe.
|
|
188
|
+
* @returns {Promise<object>} A result object with `ok` (boolean),
|
|
189
|
+
* `latency_ms`, and either `skipped` (when the probe was not run) or
|
|
190
|
+
* `reply`/`content_chars`/`has_reasoning`/`finish_reason`/`error`.
|
|
191
|
+
*/
|
|
192
|
+
export async function canaryProbe(force = false) {
|
|
193
|
+
if (wslRun === runWslCommand && (process.env.TEST_OFFLINE === "1" || (IS_TEST_ENV && !ALLOW_ENGINE_INTERRUPT))) {
|
|
194
|
+
return { ok: true, skipped: "engine_protected_offline", latency_ms: 0 };
|
|
195
|
+
}
|
|
196
|
+
if (!force && canaryCache.result && Date.now() - canaryCache.at < 60_000) {
|
|
197
|
+
return canaryCache.result;
|
|
198
|
+
}
|
|
199
|
+
const t0 = Date.now();
|
|
200
|
+
let result;
|
|
201
|
+
try {
|
|
202
|
+
const key = getApiKeySync();
|
|
203
|
+
// Omit the Authorization header when no key file is present.
|
|
204
|
+
const headers = { "Content-Type": "application/json" };
|
|
205
|
+
if (key) headers.Authorization = `Bearer ${key}`;
|
|
206
|
+
const res = await fetch(`${BASE_URL}/chat/completions`, {
|
|
207
|
+
method: "POST",
|
|
208
|
+
headers,
|
|
209
|
+
body: JSON.stringify({
|
|
210
|
+
model: MODEL,
|
|
211
|
+
// Cap on total generated tokens (content + reasoning).
|
|
212
|
+
max_tokens: 512,
|
|
213
|
+
messages: [{ role: "user", content: "Reply with: ok" }],
|
|
214
|
+
}),
|
|
215
|
+
signal: AbortSignal.timeout(15_000),
|
|
216
|
+
});
|
|
217
|
+
const dt = Date.now() - t0;
|
|
218
|
+
if (!res.ok) {
|
|
219
|
+
result = { ok: false, latency_ms: dt, error: `HTTP ${res.status}` };
|
|
220
|
+
} else {
|
|
221
|
+
const data = await res.json();
|
|
222
|
+
const content = data?.choices?.[0]?.message?.content ?? "";
|
|
223
|
+
const reasoning =
|
|
224
|
+
data?.choices?.[0]?.message?.reasoning ??
|
|
225
|
+
data?.choices?.[0]?.message?.reasoning_content ??
|
|
226
|
+
"";
|
|
227
|
+
const contentTrimmed = content.trim();
|
|
228
|
+
const reasoningTrimmed = reasoning.toString().trim();
|
|
229
|
+
const hasEvidence = contentTrimmed.length > 0 || reasoningTrimmed.length > 0;
|
|
230
|
+
if (hasEvidence) {
|
|
231
|
+
result = {
|
|
232
|
+
ok: true,
|
|
233
|
+
latency_ms: dt,
|
|
234
|
+
reply: contentTrimmed,
|
|
235
|
+
content_chars: contentTrimmed.length,
|
|
236
|
+
has_reasoning: reasoningTrimmed.length > 0,
|
|
237
|
+
finish_reason: data?.choices?.[0]?.finish_reason ?? null,
|
|
238
|
+
};
|
|
239
|
+
} else {
|
|
240
|
+
// HTTP 200 with neither content nor reasoning is a generation-less
|
|
241
|
+
// response and is not treated as a healthy engine.
|
|
242
|
+
result = {
|
|
243
|
+
ok: false,
|
|
244
|
+
latency_ms: dt,
|
|
245
|
+
reply: "",
|
|
246
|
+
content_chars: 0,
|
|
247
|
+
has_reasoning: false,
|
|
248
|
+
finish_reason: data?.choices?.[0]?.finish_reason ?? null,
|
|
249
|
+
error: "generation-less response (no content, no reasoning)",
|
|
250
|
+
};
|
|
251
|
+
}
|
|
252
|
+
}
|
|
253
|
+
} catch (err) {
|
|
254
|
+
result = {
|
|
255
|
+
ok: false,
|
|
256
|
+
latency_ms: Date.now() - t0,
|
|
257
|
+
error: err.name === "TimeoutError" ? "timeout (15s)" : err.message,
|
|
258
|
+
};
|
|
259
|
+
}
|
|
260
|
+
canaryCache = { at: Date.now(), result };
|
|
261
|
+
return result;
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
async function readLastEngineStatsLine() {
|
|
265
|
+
try {
|
|
266
|
+
const { stdout } = await wslRun(
|
|
267
|
+
`grep -a 'Engine 000:.*Running:' ${ENGINE_LOG_PATH} 2>/dev/null | tail -1`
|
|
268
|
+
);
|
|
269
|
+
return stdout.trim() || null;
|
|
270
|
+
} catch {
|
|
271
|
+
return null;
|
|
272
|
+
}
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
function parseEngineStats(line) {
|
|
276
|
+
if (!line) return null;
|
|
277
|
+
const ts = line.match(/(\d{2})-(\d{2}) (\d{2}):(\d{2}):(\d{2})/);
|
|
278
|
+
if (!ts) return null;
|
|
279
|
+
const now = new Date();
|
|
280
|
+
const stamp = new Date(
|
|
281
|
+
now.getFullYear(),
|
|
282
|
+
Number(ts[1]) - 1,
|
|
283
|
+
Number(ts[2]),
|
|
284
|
+
Number(ts[3]),
|
|
285
|
+
Number(ts[4]),
|
|
286
|
+
Number(ts[5])
|
|
287
|
+
);
|
|
288
|
+
return {
|
|
289
|
+
ageSec: Math.max(0, Math.round((now.getTime() - stamp.getTime()) / 1000)),
|
|
290
|
+
runningReqs: Number(line.match(/Running: (\d+) reqs/)?.[1] ?? -1),
|
|
291
|
+
waitingReqs: Number(line.match(/Waiting: (\d+) reqs/)?.[1] ?? -1),
|
|
292
|
+
};
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
export function readWedgeCounter() {
|
|
296
|
+
try {
|
|
297
|
+
return JSON.parse(fs.readFileSync(WEDGE_COUNTER_FILE, "utf8"));
|
|
298
|
+
} catch {
|
|
299
|
+
return { count: 0, lastAt: null, lastReason: null };
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
export function bumpWedgeCounter(reason) {
|
|
304
|
+
try {
|
|
305
|
+
const cur = readWedgeCounter();
|
|
306
|
+
cur.count = (cur.count ?? 0) + 1;
|
|
307
|
+
cur.lastAt = Date.now();
|
|
308
|
+
cur.lastReason = String(reason ?? "").slice(0, 200);
|
|
309
|
+
fs.mkdirSync(path.dirname(WEDGE_COUNTER_FILE), { recursive: true });
|
|
310
|
+
const tmp = `${WEDGE_COUNTER_FILE}.tmp_${Date.now()}_${process.pid}`;
|
|
311
|
+
fs.writeFileSync(tmp, JSON.stringify(cur), "utf8");
|
|
312
|
+
fs.renameSync(tmp, WEDGE_COUNTER_FILE);
|
|
313
|
+
} catch (err) {
|
|
314
|
+
process.stderr.write(`[server_lifecycle] Failed to bump wedge counter: ${err.message}\n`);
|
|
315
|
+
}
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
/**
|
|
319
|
+
* Determines whether the engine is wedged. The engine runs a single
|
|
320
|
+
* concurrent generation (MAX_SEQS=1), so a canary probe is only meaningful
|
|
321
|
+
* when the engine is idle; while busy, a canary would queue behind the active
|
|
322
|
+
* generation and time out, so only stats silence is used as a wedge signal.
|
|
323
|
+
* @returns {Promise<object>} A state object with `wedged`, `canary`, `stats`,
|
|
324
|
+
* `isSilenceWedged`, `isCanaryWedged`, `engineBusy`, and `gauges`.
|
|
325
|
+
*/
|
|
326
|
+
export async function engineWedgeState() {
|
|
327
|
+
// Read the engine gauges first to determine busy state.
|
|
328
|
+
const metrics = await readEngineMetrics();
|
|
329
|
+
// /metrics unavailable is treated as busy (fail-closed).
|
|
330
|
+
const metricsUnavailable = metrics === null;
|
|
331
|
+
const runningReqs = metrics?.["vllm:num_requests_running"] ?? 0;
|
|
332
|
+
const waitingReqs = metrics?.["vllm:num_requests_waiting"] ?? 0;
|
|
333
|
+
const engineBusy = metricsUnavailable || runningReqs > 0 || waitingReqs > 0;
|
|
334
|
+
|
|
335
|
+
const line = await readLastEngineStatsLine();
|
|
336
|
+
const stats = line ? parseEngineStats(line) : null;
|
|
337
|
+
|
|
338
|
+
// Stats silence while requests are running indicates a stalled engine core.
|
|
339
|
+
const isSilenceWedged = Boolean(
|
|
340
|
+
runningReqs > 0 && stats && stats.runningReqs > 0 && stats.ageSec > WEDGE_STATS_SILENCE_S
|
|
341
|
+
);
|
|
342
|
+
|
|
343
|
+
let canary;
|
|
344
|
+
let isCanaryWedged;
|
|
345
|
+
if (engineBusy) {
|
|
346
|
+
// When busy due to a metrics fetch failure, the canary is refused with an
|
|
347
|
+
// explicit error (fail-closed) rather than a neutral sentinel.
|
|
348
|
+
canary = metricsUnavailable
|
|
349
|
+
? { ok: false, skipped: "metrics_unavailable", error: "/metrics fetch failed; busy-gate fail-closed" }
|
|
350
|
+
: { ok: true, skipped: "engine_busy", latency_ms: 0 };
|
|
351
|
+
isCanaryWedged = false;
|
|
352
|
+
} else {
|
|
353
|
+
canary = await canaryProbe();
|
|
354
|
+
isCanaryWedged = !canary.ok;
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
// When busy, the canary is ignored — only stats silence can declare a wedge.
|
|
358
|
+
// When idle, the canary probe is authoritative.
|
|
359
|
+
const wedged = engineBusy
|
|
360
|
+
? isSilenceWedged
|
|
361
|
+
: Boolean(isCanaryWedged);
|
|
362
|
+
|
|
363
|
+
return {
|
|
364
|
+
wedged,
|
|
365
|
+
canary,
|
|
366
|
+
stats,
|
|
367
|
+
isSilenceWedged,
|
|
368
|
+
isCanaryWedged,
|
|
369
|
+
engineBusy,
|
|
370
|
+
gauges: metrics
|
|
371
|
+
? {
|
|
372
|
+
running_requests: metrics["vllm:num_requests_running"] ?? null,
|
|
373
|
+
waiting_requests: metrics["vllm:num_requests_waiting"] ?? null,
|
|
374
|
+
kv_cache_pct: metrics["vllm:gpu_cache_usage_factor"]
|
|
375
|
+
? Math.round(metrics["vllm:gpu_cache_usage_factor"] * 1000) / 10
|
|
376
|
+
: null,
|
|
377
|
+
prefix_cache_hit_ratio: metrics["vllm:prefix_cache_hit_rate"] ?? null,
|
|
378
|
+
spec_decode_acceptance: metrics["vllm:spec_decode_draft_acceptance_rate"] ?? null,
|
|
379
|
+
}
|
|
380
|
+
: null,
|
|
381
|
+
};
|
|
382
|
+
}
|
|
383
|
+
|
|
384
|
+
/**
|
|
385
|
+
* Stops and restarts a wedged engine. Refuses to run when engine interruption
|
|
386
|
+
* is disabled, when the heal gatekeeper reports live work in flight, or when
|
|
387
|
+
* another process already holds the heal lock.
|
|
388
|
+
* @param {number|null} statsAgeSec - Age of the last engine stats line, in
|
|
389
|
+
* seconds; recorded in the heal lock payload.
|
|
390
|
+
* @returns {Promise<object>} A result object with `healed` (boolean) and a
|
|
391
|
+
* `note` (when refused) or `boot` status (when healed).
|
|
392
|
+
*/
|
|
393
|
+
export async function healWedgedEngine(statsAgeSec) {
|
|
394
|
+
if (wslRun === runWslCommand && (process.env.TEST_OFFLINE === "1" || (IS_TEST_ENV && !ALLOW_ENGINE_INTERRUPT))) {
|
|
395
|
+
return { healed: false, note: "heal refused: engine interruption disabled by default (ALLOW_ENGINE_INTERRUPT unset)" };
|
|
396
|
+
}
|
|
397
|
+
// Refuse to stop/reboot the engine while live work is in flight.
|
|
398
|
+
if (healGatekeeper) {
|
|
399
|
+
try {
|
|
400
|
+
if (healGatekeeper()) {
|
|
401
|
+
return { healed: false, note: "tasks in flight; heal refused" };
|
|
402
|
+
}
|
|
403
|
+
} catch {
|
|
404
|
+
return { healed: false, note: "tasks in flight; heal refused" };
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
const lockResult = tryAcquireExclusiveLock(HEAL_LOCK_FILE, HEAL_LOCK_TTL_MS, {
|
|
409
|
+
pid: process.pid,
|
|
410
|
+
statsAgeSec,
|
|
411
|
+
});
|
|
412
|
+
|
|
413
|
+
if (!lockResult.acquired) {
|
|
414
|
+
const lock = lockResult.heldBy;
|
|
415
|
+
const ago = lock?.at ? Math.round((Date.now() - lock.at) / 1000) : "unknown";
|
|
416
|
+
const pid = lock?.pid ?? "unknown";
|
|
417
|
+
return {
|
|
418
|
+
healed: false,
|
|
419
|
+
note: `heal already started ${ago}s ago by pid ${pid}; boot in progress`,
|
|
420
|
+
};
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
bumpWedgeCounter(`stats_age=${statsAgeSec}s`);
|
|
424
|
+
await stopServer();
|
|
425
|
+
const res = await ensureServerRunning();
|
|
426
|
+
resetEngineHealthCache();
|
|
427
|
+
return { healed: true, boot: res.status };
|
|
428
|
+
}
|
|
429
|
+
|
|
430
|
+
async function warmEngine() {
|
|
431
|
+
try {
|
|
432
|
+
const key = getApiKeySync();
|
|
433
|
+
// Omit Authorization header when no key file is present.
|
|
434
|
+
const headers = { "Content-Type": "application/json" };
|
|
435
|
+
if (key) headers.Authorization = `Bearer ${key}`;
|
|
436
|
+
await fetch(`${BASE_URL}/chat/completions`, {
|
|
437
|
+
method: "POST",
|
|
438
|
+
headers,
|
|
439
|
+
body: JSON.stringify({
|
|
440
|
+
model: MODEL,
|
|
441
|
+
messages: [{ role: "user", content: "ping" }],
|
|
442
|
+
max_tokens: 1,
|
|
443
|
+
temperature: 0.0,
|
|
444
|
+
}),
|
|
445
|
+
signal: AbortSignal.timeout(60_000),
|
|
446
|
+
});
|
|
447
|
+
} catch {}
|
|
448
|
+
}
|
|
449
|
+
|
|
450
|
+
// ---------------------------------------------------------------------------
|
|
451
|
+
// Stream-proxy lifecycle
|
|
452
|
+
// ---------------------------------------------------------------------------
|
|
453
|
+
// Pre-spawn cleanup is targeted: it kills the current listener on the port by
|
|
454
|
+
// its specific pid (probed from /health or `ss -ltnp`), never a broad `pkill -f`.
|
|
455
|
+
|
|
456
|
+
// Indirection for the stream-proxy spawner; tests inject a stub to simulate a
|
|
457
|
+
// slow or failed start.
|
|
458
|
+
let spawnStreamProxy = null;
|
|
459
|
+
export function setStreamProxySpawner(fn) {
|
|
460
|
+
spawnStreamProxy = typeof fn === "function" ? fn : null;
|
|
461
|
+
}
|
|
462
|
+
|
|
463
|
+
/**
|
|
464
|
+
* The real stream-proxy spawner. Windows: spawn inside WSL via setsid.
|
|
465
|
+
* Linux: spawn a detached node child directly.
|
|
466
|
+
*/
|
|
467
|
+
async function realSpawnStreamProxy() {
|
|
468
|
+
if (IS_WINDOWS) {
|
|
469
|
+
// WSL tears the session down when the `bash -c` leader exits, which can
|
|
470
|
+
// kill a freshly-forked background child before it execs; the 1s linger
|
|
471
|
+
// keeps the session alive long enough for the child to detach.
|
|
472
|
+
await runWslCommand(
|
|
473
|
+
`setsid node ${streamProxyPath()} < /dev/null > /tmp/stream_proxy.log 2>&1 & sleep 1`
|
|
474
|
+
);
|
|
475
|
+
} else {
|
|
476
|
+
const { spawn } = await import("child_process");
|
|
477
|
+
const p = spawn("node", [path.join(__dirname, "..", "stream_proxy.js")], {
|
|
478
|
+
stdio: "ignore",
|
|
479
|
+
detached: true,
|
|
480
|
+
});
|
|
481
|
+
p.unref();
|
|
482
|
+
}
|
|
483
|
+
}
|
|
484
|
+
|
|
485
|
+
/**
|
|
486
|
+
* Finds the pid of the process currently listening on the stream-proxy port,
|
|
487
|
+
* if any. Probes /health first, then falls back to `ss -ltnp` on the port.
|
|
488
|
+
* @returns {Promise<{pid: number, verified: boolean}|null>} The listener pid
|
|
489
|
+
* and whether it was verified as the stream proxy, or null when no listener
|
|
490
|
+
* is found. Never throws.
|
|
491
|
+
*/
|
|
492
|
+
async function findStreamProxyListenerPid() {
|
|
493
|
+
const port = STREAM_PROXY_PORT_RESOLVED;
|
|
494
|
+
// 1. Probe /health for a verified pid.
|
|
495
|
+
try {
|
|
496
|
+
const res = await fetch(`http://127.0.0.1:${port}/health`, {
|
|
497
|
+
signal: AbortSignal.timeout(1000),
|
|
498
|
+
});
|
|
499
|
+
if (res.ok) {
|
|
500
|
+
const j = await res.json();
|
|
501
|
+
if (j && (j.service === "mcp-castor-stream-proxy" || j.service === "mcp-qwen-stream-proxy") && Number.isFinite(j.pid) && j.pid > 0) {
|
|
502
|
+
return { pid: j.pid, verified: true };
|
|
503
|
+
}
|
|
504
|
+
}
|
|
505
|
+
} catch {}
|
|
506
|
+
// 2. Fall back to `ss -ltnp` filtered to the port.
|
|
507
|
+
try {
|
|
508
|
+
const { stdout } = await runWslCommand(
|
|
509
|
+
`ss -ltnp 'sport = :${port}' 2>/dev/null | grep -oE 'pid=[0-9]+' | head -1 | cut -d= -f2 || true`
|
|
510
|
+
);
|
|
511
|
+
const pid = parseInt(stdout.trim(), 10);
|
|
512
|
+
if (Number.isFinite(pid) && pid > 0) {
|
|
513
|
+
try {
|
|
514
|
+
const { stdout: cmdline } = await runWslCommand(
|
|
515
|
+
`tr '\\0' ' ' < /proc/${pid}/cmdline 2>/dev/null || true`
|
|
516
|
+
);
|
|
517
|
+
if (cmdline.includes("stream_proxy.js")) {
|
|
518
|
+
return { pid, verified: true };
|
|
519
|
+
}
|
|
520
|
+
} catch {}
|
|
521
|
+
return { pid, verified: false };
|
|
522
|
+
}
|
|
523
|
+
} catch {}
|
|
524
|
+
return null;
|
|
525
|
+
}
|
|
526
|
+
|
|
527
|
+
/**
|
|
528
|
+
* Ensures the universal stream proxy is running and healthy on
|
|
529
|
+
* STREAM_PROXY_PORT.
|
|
530
|
+
* @param {object} [opts]
|
|
531
|
+
* @param {number} [opts.healthPolls=75] Number of 200ms health polls before
|
|
532
|
+
* declaring failure (default 75 = 15s). Tests may pass a smaller value.
|
|
533
|
+
* @returns {Promise<boolean>} true once the proxy is healthy.
|
|
534
|
+
* @throws {Error} If the proxy does not become healthy within the window;
|
|
535
|
+
* the message includes the captured spawn failure when present.
|
|
536
|
+
*/
|
|
537
|
+
export async function ensureStreamProxyRunning({ healthPolls = 75 } = {}) {
|
|
538
|
+
// 0a. If the stream proxy is disabled by configuration, skip cleanly.
|
|
539
|
+
// The engine is reached directly (no proxy hop).
|
|
540
|
+
if (!USE_STREAM_PROXY_RESOLVED) {
|
|
541
|
+
process.stderr.write(
|
|
542
|
+
`[stream-proxy] disabled by configuration (USE_STREAM_PROXY=false); using direct engine connection\n`
|
|
543
|
+
);
|
|
544
|
+
return true;
|
|
545
|
+
}
|
|
546
|
+
|
|
547
|
+
// 0b. On Windows without WSL, the proxy (which runs inside WSL) cannot be
|
|
548
|
+
// spawned. Do NOT crash the whole task: warn and allow a direct engine
|
|
549
|
+
// connection so the task can still proceed.
|
|
550
|
+
if (!spawnStreamProxy && IS_WINDOWS && !wslAvailable()) {
|
|
551
|
+
process.stderr.write(
|
|
552
|
+
`[stream-proxy] WSL is not available on this Windows host; skipping proxy and using direct engine connection\n`
|
|
553
|
+
);
|
|
554
|
+
return true;
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
const port = STREAM_PROXY_PORT_RESOLVED;
|
|
558
|
+
|
|
559
|
+
// 1. Is the proxy already healthy? (two quick probes)
|
|
560
|
+
for (let attempt = 0; attempt < 2; attempt++) {
|
|
561
|
+
try {
|
|
562
|
+
const res = await fetch(`http://127.0.0.1:${port}/health`, {
|
|
563
|
+
signal: AbortSignal.timeout(1000),
|
|
564
|
+
});
|
|
565
|
+
if (res.ok) return true;
|
|
566
|
+
} catch {}
|
|
567
|
+
if (attempt === 0) await new Promise((r) => setTimeout(r, 200));
|
|
568
|
+
}
|
|
569
|
+
|
|
570
|
+
// 2. TARGETED pre-spawn cleanup: kill the current listener on the port ONLY if verified.
|
|
571
|
+
const listenerInfo = await findStreamProxyListenerPid();
|
|
572
|
+
if (listenerInfo) {
|
|
573
|
+
if (!listenerInfo.verified) {
|
|
574
|
+
throw new Error(
|
|
575
|
+
`PortConflictError: Port ${port} is occupied by unverified process pid ${listenerInfo.pid}. Refusing to kill non-proxy process.`
|
|
576
|
+
);
|
|
577
|
+
}
|
|
578
|
+
const listenerPid = listenerInfo.pid;
|
|
579
|
+
try {
|
|
580
|
+
if (IS_WINDOWS) {
|
|
581
|
+
await runWslCommand(`kill -9 ${listenerPid} 2>/dev/null || true`);
|
|
582
|
+
} else {
|
|
583
|
+
try {
|
|
584
|
+
process.kill(listenerPid, "SIGKILL");
|
|
585
|
+
} catch {}
|
|
586
|
+
}
|
|
587
|
+
process.stderr.write(
|
|
588
|
+
`[stream-proxy] pre-spawn cleanup: killed verified listener pid ${listenerPid} on port ${port}\n`
|
|
589
|
+
);
|
|
590
|
+
await new Promise((r) => setTimeout(r, 300));
|
|
591
|
+
} catch (err) {
|
|
592
|
+
process.stderr.write(
|
|
593
|
+
`[stream-proxy] pre-spawn cleanup: failed to kill pid ${listenerPid}: ${err.message}\n`
|
|
594
|
+
);
|
|
595
|
+
}
|
|
596
|
+
} else {
|
|
597
|
+
process.stderr.write(
|
|
598
|
+
`[stream-proxy] pre-spawn cleanup: no listener found on port ${port}\n`
|
|
599
|
+
);
|
|
600
|
+
}
|
|
601
|
+
|
|
602
|
+
// 3. SPAWN the proxy (injected spawner for tests, real spawner otherwise).
|
|
603
|
+
// Do NOT swallow spawn errors: capture them for the final error message.
|
|
604
|
+
let spawnFailure = null;
|
|
605
|
+
try {
|
|
606
|
+
if (spawnStreamProxy) {
|
|
607
|
+
await spawnStreamProxy();
|
|
608
|
+
} else {
|
|
609
|
+
await realSpawnStreamProxy();
|
|
610
|
+
}
|
|
611
|
+
} catch (err) {
|
|
612
|
+
spawnFailure = err;
|
|
613
|
+
process.stderr.write(`[stream-proxy] spawn failed: ${err.message}\n`);
|
|
614
|
+
}
|
|
615
|
+
|
|
616
|
+
// 4. Health window: poll until the proxy reports healthy, with early exit.
|
|
617
|
+
for (let i = 0; i < healthPolls; i++) {
|
|
618
|
+
await new Promise((r) => setTimeout(r, 200));
|
|
619
|
+
try {
|
|
620
|
+
const res = await fetch(`http://127.0.0.1:${port}/health`, {
|
|
621
|
+
signal: AbortSignal.timeout(1000),
|
|
622
|
+
});
|
|
623
|
+
if (res.ok) return true;
|
|
624
|
+
} catch {}
|
|
625
|
+
}
|
|
626
|
+
|
|
627
|
+
const seconds = Math.round((healthPolls * 200) / 1000);
|
|
628
|
+
const spawnNote = spawnFailure ? ` (spawn failure: ${spawnFailure.message})` : "";
|
|
629
|
+
throw new Error(
|
|
630
|
+
`Stream proxy failed to become healthy on port ${port} after ${seconds}s${spawnNote}`
|
|
631
|
+
);
|
|
632
|
+
}
|
|
633
|
+
|
|
634
|
+
/**
|
|
635
|
+
* Ensures the vLLM engine is running. If the engine is already up it checks
|
|
636
|
+
* for a wedge (healing one when AUTO_HEAL is enabled); otherwise it boots the
|
|
637
|
+
* engine, coordinating across processes via an atomic boot lock.
|
|
638
|
+
* @returns {Promise<object>} A result object with `switched` (boolean) and a
|
|
639
|
+
* `status` string; when a wedged engine was healed, a `heal` object is
|
|
640
|
+
* included.
|
|
641
|
+
*/
|
|
642
|
+
export async function ensureServerRunning() {
|
|
643
|
+
if (wslRun === runWslCommand && (process.env.TEST_OFFLINE === "1" || (IS_TEST_ENV && !ALLOW_ENGINE_INTERRUPT))) {
|
|
644
|
+
return { switched: false, status: "boot_refused_offline_protected" };
|
|
645
|
+
}
|
|
646
|
+
const current = await currentMode();
|
|
647
|
+
if (current) {
|
|
648
|
+
const wedge = await engineWedgeState();
|
|
649
|
+
if (wedge.wedged && AUTO_HEAL) {
|
|
650
|
+
const heal = await healWedgedEngine(wedge.stats?.ageSec ?? null);
|
|
651
|
+
return { switched: true, status: "restarted_wedged_engine", heal };
|
|
652
|
+
}
|
|
653
|
+
return { switched: false, status: "already_running" };
|
|
654
|
+
}
|
|
655
|
+
|
|
656
|
+
// Cross-process atomic boot lock: only ONE instance executes the launcher script.
|
|
657
|
+
const lock = tryAcquireExclusiveLock(ENGINE_BOOT_LOCK_FILE, ENGINE_BOOT_LOCK_TTL_MS, {
|
|
658
|
+
pid: process.pid,
|
|
659
|
+
action: "booting_huge",
|
|
660
|
+
});
|
|
661
|
+
|
|
662
|
+
if (!lock.acquired) {
|
|
663
|
+
// Secondary instance: wait for primary instance to finish booting.
|
|
664
|
+
const deadline = Date.now() + BOOT_TIMEOUT_MS;
|
|
665
|
+
while (Date.now() < deadline) {
|
|
666
|
+
await new Promise((r) => setTimeout(r, BOOT_POLL_MS));
|
|
667
|
+
const now = await currentMode();
|
|
668
|
+
if (now) {
|
|
669
|
+
await ensureStreamProxyRunning();
|
|
670
|
+
resetEngineHealthCache();
|
|
671
|
+
return { switched: true, status: "started_by_peer" };
|
|
672
|
+
}
|
|
673
|
+
}
|
|
674
|
+
throw new Error(`Timed out waiting for peer vLLM server to boot (${BOOT_TIMEOUT_MS}ms)`);
|
|
675
|
+
}
|
|
676
|
+
|
|
677
|
+
try {
|
|
678
|
+
if (LAUNCH_COMMAND) {
|
|
679
|
+
// Use the configured launch command (from QWEN_LAUNCH_COMMAND or
|
|
680
|
+
// ~/.castor/config.json `launch_command`). This is the portable path:
|
|
681
|
+
// the user controls exactly how the engine is started.
|
|
682
|
+
await wslRun(
|
|
683
|
+
`nohup bash -c '${LAUNCH_COMMAND.replace(/'/g, "'\\''")}' > ${ENGINE_LOG_PATH} 2>&1 < /dev/null & disown; sleep 1; true`
|
|
684
|
+
);
|
|
685
|
+
} else {
|
|
686
|
+
// Fall back to the repo's launcher script.
|
|
687
|
+
const launcher = launcherScriptPath();
|
|
688
|
+
await wslRun(
|
|
689
|
+
`cd ~/qwen-serving && if [ -f "${launcher}" ]; then nohup bash "${launcher}"; elif [ -f launchers/start_huge.sh ]; then nohup bash launchers/start_huge.sh; else nohup bash single-user/start_qwen.sh; fi > ${ENGINE_LOG_PATH} 2>&1 < /dev/null & disown; sleep 1; true`
|
|
690
|
+
);
|
|
691
|
+
}
|
|
692
|
+
const deadline = Date.now() + BOOT_TIMEOUT_MS;
|
|
693
|
+
while (Date.now() < deadline) {
|
|
694
|
+
await new Promise((r) => setTimeout(r, BOOT_POLL_MS));
|
|
695
|
+
const now = await currentMode();
|
|
696
|
+
if (now) {
|
|
697
|
+
await ensureStreamProxyRunning();
|
|
698
|
+
await warmEngine();
|
|
699
|
+
resetEngineHealthCache();
|
|
700
|
+
return { switched: true, status: "started" };
|
|
701
|
+
}
|
|
702
|
+
}
|
|
703
|
+
throw new Error(`Timed out waiting for vLLM server to boot (${BOOT_TIMEOUT_MS}ms)`);
|
|
704
|
+
} finally {
|
|
705
|
+
releaseExclusiveLock(ENGINE_BOOT_LOCK_FILE);
|
|
706
|
+
}
|
|
707
|
+
}
|
|
708
|
+
|
|
709
|
+
/**
|
|
710
|
+
* Stops the vLLM engine and waits up to 5s for it to stop responding.
|
|
711
|
+
* @returns {Promise<object>} A result object with `stopped` (boolean) and a
|
|
712
|
+
* `reason` string; when the engine still responds after the grace window,
|
|
713
|
+
* `stopped` is false and `mode` reports the detected engine mode.
|
|
714
|
+
*/
|
|
715
|
+
export async function stopServer() {
|
|
716
|
+
if (wslRun === runWslCommand && (process.env.TEST_OFFLINE === "1" || (IS_TEST_ENV && !ALLOW_ENGINE_INTERRUPT))) {
|
|
717
|
+
return {
|
|
718
|
+
stopped: false,
|
|
719
|
+
reason: "stop_refused_offline_protected",
|
|
720
|
+
note: "stopServer refused: engine interruption disabled by default (ALLOW_ENGINE_INTERRUPT unset)",
|
|
721
|
+
};
|
|
722
|
+
}
|
|
723
|
+
await wslRun(`cd ~/qwen-serving && bash launchers/stop_server.sh 2>/dev/null || true`);
|
|
724
|
+
let mode = null;
|
|
725
|
+
for (let i = 0; i < 10; i++) {
|
|
726
|
+
await new Promise((r) => setTimeout(r, 500));
|
|
727
|
+
mode = await currentMode();
|
|
728
|
+
if (!mode) return { stopped: true };
|
|
729
|
+
}
|
|
730
|
+
// The grace window elapsed but the engine still responds: report a failure
|
|
731
|
+
// rather than a success.
|
|
732
|
+
return {
|
|
733
|
+
stopped: false,
|
|
734
|
+
reason: "engine_still_responding",
|
|
735
|
+
mode,
|
|
736
|
+
};
|
|
737
|
+
}
|
|
738
|
+
|
|
739
|
+
let metricsCache = { at: 0, data: null };
|
|
740
|
+
|
|
741
|
+
/**
|
|
742
|
+
* Fetches and parses the vLLM /metrics endpoint, returning a map of metric
|
|
743
|
+
* name to value. Results are cached for `maxAgeMs`.
|
|
744
|
+
* @param {number} [maxAgeMs=5000] Maximum age of a cached result, in
|
|
745
|
+
* milliseconds.
|
|
746
|
+
* @returns {Promise<Record<string, number>|null>} The parsed metrics, or null
|
|
747
|
+
* when the fetch fails.
|
|
748
|
+
*/
|
|
749
|
+
export async function readEngineMetrics(maxAgeMs = 5000) {
|
|
750
|
+
if (metricsCache.data && Date.now() - metricsCache.at < maxAgeMs) return metricsCache.data;
|
|
751
|
+
try {
|
|
752
|
+
const res = await fetch(`${BASE_URL.replace(/\/v1$/, "")}/metrics`, {
|
|
753
|
+
signal: AbortSignal.timeout(3000),
|
|
754
|
+
});
|
|
755
|
+
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
|
756
|
+
const map = {};
|
|
757
|
+
for (const line of (await res.text()).split("\n")) {
|
|
758
|
+
if (!line.startsWith("vllm:")) continue;
|
|
759
|
+
const name = line.match(/^(vllm:[^{ ]+)/)?.[1];
|
|
760
|
+
const val = Number(line.slice(line.lastIndexOf(" ") + 1));
|
|
761
|
+
if (!name || !Number.isFinite(val)) continue;
|
|
762
|
+
map[name] = Math.max(map[name] ?? -Infinity, val);
|
|
763
|
+
}
|
|
764
|
+
metricsCache = { at: Date.now(), data: map };
|
|
765
|
+
return map;
|
|
766
|
+
} catch (err) {
|
|
767
|
+
// A /metrics fetch failure is logged to stderr and reported as null; the
|
|
768
|
+
// caller (engineWedgeState) treats it as busy (fail-closed).
|
|
769
|
+
process.stderr.write(`[server_lifecycle] /metrics fetch failed: ${err.message}\n`);
|
|
770
|
+
metricsCache = { at: Date.now(), data: null };
|
|
771
|
+
return null;
|
|
772
|
+
}
|
|
773
|
+
}
|
|
774
|
+
|
|
775
|
+
/**
|
|
776
|
+
* Clears the canary and metrics caches so the next probe re-fetches.
|
|
777
|
+
*/
|
|
778
|
+
export function resetEngineHealthCache() {
|
|
779
|
+
canaryCache = { at: 0, result: null };
|
|
780
|
+
metricsCache = { at: 0, data: null };
|
|
781
|
+
}
|