mcp-castor 2026.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +487 -0
- package/bin/castor.js +706 -0
- package/index.js +206 -0
- package/package.json +97 -0
- package/skills/canary-test-staging/SKILL.md +24 -0
- package/skills/evo-mutation-rollback/SKILL.md +29 -0
- package/skills/hypothesis-generation/SKILL.md +26 -0
- package/skills/traceback-condensing/SKILL.md +26 -0
- package/src/castor_runner.js +469 -0
- package/src/config.js +1204 -0
- package/src/env.js +10 -0
- package/src/evo_engine.js +214 -0
- package/src/harness/core/events.js +75 -0
- package/src/harness/core/kernel.js +209 -0
- package/src/harness/evo/evaluator.js +156 -0
- package/src/harness/evo/evo_operator.js +550 -0
- package/src/harness/evo/lineage_dag.js +383 -0
- package/src/harness/evo/trace_repair.js +173 -0
- package/src/harness/evo/watchdog.js +72 -0
- package/src/harness/loop_detector.js +135 -0
- package/src/harness/runner.js +1216 -0
- package/src/harness/services/ast_service.js +1813 -0
- package/src/harness/services/event_logger.js +275 -0
- package/src/harness/services/mcp_bridge.js +408 -0
- package/src/harness/services/provider_vllm.js +728 -0
- package/src/harness/services/sandbox_fs.js +1238 -0
- package/src/harness/services/searxng_lifecycle.js +254 -0
- package/src/harness/services/shell_executor.js +264 -0
- package/src/harness/services/shell_validator.js +506 -0
- package/src/harness/services/web_service.js +828 -0
- package/src/platform.js +344 -0
- package/src/repetition_detector.js +139 -0
- package/src/semaphore.js +373 -0
- package/src/server_lifecycle.js +781 -0
- package/src/skills.js +400 -0
- package/src/state_pruner.js +392 -0
- package/src/task_registry.js +1357 -0
- package/src/telemetry.js +638 -0
- package/src/tools.js +997 -0
- package/src/wsl_bridge.js +629 -0
- package/src/wsl_env.js +171 -0
- package/stream_proxy.js +453 -0
|
@@ -0,0 +1,1357 @@
|
|
|
1
|
+
import fs from "node:fs";
|
|
2
|
+
import http from "node:http";
|
|
3
|
+
import path from "node:path";
|
|
4
|
+
import {
|
|
5
|
+
TASK_DIR,
|
|
6
|
+
STATUS_PORT,
|
|
7
|
+
TASK_RETENTION_MS,
|
|
8
|
+
DEFAULT_TIMEOUT_MS,
|
|
9
|
+
INACTIVITY_TIMEOUT_MS,
|
|
10
|
+
ORPHAN_REAP_STALE_MS,
|
|
11
|
+
BASE_TURN_BUDGET,
|
|
12
|
+
MAX_ELASTIC_TURNS,
|
|
13
|
+
SUPERVISOR_PREVIEW_CHARS,
|
|
14
|
+
} from "./config.js";
|
|
15
|
+
import {
|
|
16
|
+
pidAlive,
|
|
17
|
+
listTaskSlots,
|
|
18
|
+
clearReclaimableTaskSlots,
|
|
19
|
+
releaseTaskSlot,
|
|
20
|
+
} from "./semaphore.js";
|
|
21
|
+
import {
|
|
22
|
+
killProcessTree,
|
|
23
|
+
killSessionProcessTreeSync,
|
|
24
|
+
} from "./wsl_bridge.js";
|
|
25
|
+
import { EventLoggerService } from "./harness/services/event_logger.js";
|
|
26
|
+
import { SSEServerTransport } from "@modelcontextprotocol/sdk/server/sse.js";
|
|
27
|
+
|
|
28
|
+
try {
|
|
29
|
+
fs.mkdirSync(TASK_DIR, { recursive: true });
|
|
30
|
+
} catch {}
|
|
31
|
+
|
|
32
|
+
export const tasks = new Map();
|
|
33
|
+
|
|
34
|
+
export function saveTaskToDisk(task) {
|
|
35
|
+
if (!task || !task.id) return;
|
|
36
|
+
const filePath = path.join(TASK_DIR, `${task.id}.json`);
|
|
37
|
+
try {
|
|
38
|
+
const tmpPath = `${filePath}.tmp_${process.pid}_${Date.now()}`;
|
|
39
|
+
const payload = {
|
|
40
|
+
id: task.id,
|
|
41
|
+
sessionId: task.sessionId,
|
|
42
|
+
cwd: task.cwd,
|
|
43
|
+
prompt: task.prompt,
|
|
44
|
+
// Effective reasoning-effort tier for this task (per-dispatch param when
|
|
45
|
+
// provided, else the QWEN_REASONING_EFFORT env default). Persisted for
|
|
46
|
+
// telemetry; undefined when the task predates the field.
|
|
47
|
+
reasoningEffort: task.reasoningEffort ?? null,
|
|
48
|
+
ownerPid: task.ownerPid || process.pid,
|
|
49
|
+
ownerPlatform: task.ownerPlatform || process.platform,
|
|
50
|
+
createdAt: task.createdAt,
|
|
51
|
+
startedAt: task.startedAt,
|
|
52
|
+
finishedAt: task.finishedAt,
|
|
53
|
+
lastHeartbeatAt: task.lastHeartbeatAt || Date.now(),
|
|
54
|
+
status: task.status,
|
|
55
|
+
done: task.done,
|
|
56
|
+
isError: task.isError,
|
|
57
|
+
budgetTurns: task.budgetTurns || BASE_TURN_BUDGET,
|
|
58
|
+
leaseExtensionsCount: task.leaseExtensionsCount || 0,
|
|
59
|
+
lastActivityPreview: task.lastActivityPreview ? String(task.lastActivityPreview).slice(-SUPERVISOR_PREVIEW_CHARS) : "",
|
|
60
|
+
fileOps: task.fileOps || [],
|
|
61
|
+
toolCallsCount: task.toolCallsCount || 0,
|
|
62
|
+
toolOpsSummary: task.toolOpsSummary || { reads: 0, mutations: 0, commands: 0, web: 0 },
|
|
63
|
+
lastTool: task.lastTool || null,
|
|
64
|
+
lastPromptTokens: task.lastPromptTokens ?? null,
|
|
65
|
+
contextHeadroom: task.contextHeadroom ?? null,
|
|
66
|
+
result: task.result || null,
|
|
67
|
+
stderr: task.stderr ? task.stderr.slice(-2000) : "",
|
|
68
|
+
};
|
|
69
|
+
fs.writeFileSync(tmpPath, JSON.stringify(payload, null, 2), "utf8");
|
|
70
|
+
fs.renameSync(tmpPath, filePath);
|
|
71
|
+
} catch (err) {
|
|
72
|
+
// Telemetry persistence must never throw into the runner, but it must not
|
|
73
|
+
// be invisible either: a failed save (disk full, EBUSY, permission) is a
|
|
74
|
+
// real signal — log the path and the error to stderr and continue.
|
|
75
|
+
const msg = err && err.message ? err.message : String(err);
|
|
76
|
+
console.error(`[task_registry] Failed to save task ${task.id} to ${filePath}: ${msg}`);
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
export function isTaskOrphaned(diskTask) {
|
|
81
|
+
if (!diskTask || diskTask.done) return false;
|
|
82
|
+
const now = Date.now();
|
|
83
|
+
const lastActive = diskTask.lastHeartbeatAt || diskTask.startedAt || diskTask.createdAt;
|
|
84
|
+
|
|
85
|
+
// LIVE-OWNER INVARIANT:
|
|
86
|
+
// If the owner process is still alive across platforms, it is not an orphan.
|
|
87
|
+
// Active workers make tool calls, perform deep deliberation, or run long commands.
|
|
88
|
+
if (diskTask.ownerPid && pidAlive(diskTask.ownerPid, diskTask.ownerPlatform)) {
|
|
89
|
+
return false;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
// Owner process is dead or unrecorded: check if heartbeat is stale beyond threshold.
|
|
93
|
+
const staleThreshold = ORPHAN_REAP_STALE_MS;
|
|
94
|
+
if (lastActive && now - lastActive > staleThreshold) {
|
|
95
|
+
return true;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
return false;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
export function markTaskOrphanedOnDisk(diskTask) {
|
|
102
|
+
if (!diskTask || diskTask.done) return diskTask;
|
|
103
|
+
diskTask.done = true;
|
|
104
|
+
diskTask.status = "failed";
|
|
105
|
+
diskTask.isError = true;
|
|
106
|
+
diskTask.finishedAt = Date.now();
|
|
107
|
+
diskTask.result = {
|
|
108
|
+
isError: true,
|
|
109
|
+
text: `Task orphaned: worker process (PID ${diskTask.ownerPid || process.pid || "unknown"}) exited unexpectedly before completion.`,
|
|
110
|
+
toolCalls: diskTask.toolCallsCount || 0,
|
|
111
|
+
errors: ["WORKER_PROCESS_TERMINATED"],
|
|
112
|
+
fileOps: diskTask.fileOps || [],
|
|
113
|
+
};
|
|
114
|
+
saveTaskToDisk(diskTask);
|
|
115
|
+
appendOrphanTerminalEvent(diskTask);
|
|
116
|
+
|
|
117
|
+
// Clean up in-memory task handles and execution slot immediately
|
|
118
|
+
const mem = tasks.get(diskTask.id);
|
|
119
|
+
if (mem) {
|
|
120
|
+
if (mem.abortController) {
|
|
121
|
+
try { mem.abortController.abort(); } catch {}
|
|
122
|
+
mem.abortController = null;
|
|
123
|
+
}
|
|
124
|
+
if (mem.slot) {
|
|
125
|
+
try { releaseTaskSlot(mem.slot); } catch {}
|
|
126
|
+
mem.slot = null;
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
return diskTask;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/**
|
|
133
|
+
* Appends a single-line terminal `session_error` event to the orphaned task's
|
|
134
|
+
* session events.jsonl so a silent infra death (external process-tree kill)
|
|
135
|
+
* leaves a detectable trace.
|
|
136
|
+
*
|
|
137
|
+
* Cross-instance safe: the detecting instance may differ from the owning
|
|
138
|
+
* instance (shared state dir). We only ever APPEND one line (appendFileSync)
|
|
139
|
+
* and never rewrite existing content. The double-terminal guard
|
|
140
|
+
* (hasTerminalEvent) ensures a session that already ended (session_end or
|
|
141
|
+
* session_error) is not given a second terminal event.
|
|
142
|
+
*
|
|
143
|
+
* Never throws: a failure to append the trace must not break the (already
|
|
144
|
+
* working) orphan-marking of the task file.
|
|
145
|
+
*/
|
|
146
|
+
function appendOrphanTerminalEvent(diskTask) {
|
|
147
|
+
if (!diskTask || !diskTask.sessionId) return;
|
|
148
|
+
try {
|
|
149
|
+
const logger = new EventLoggerService({ sessionId: diskTask.sessionId });
|
|
150
|
+
// Double-terminal guard: if the session already has a terminal event, do
|
|
151
|
+
// not append a second one.
|
|
152
|
+
if (logger.hasTerminalEvent()) return;
|
|
153
|
+
logger.append({
|
|
154
|
+
type: "session_error",
|
|
155
|
+
reason: "orphaned",
|
|
156
|
+
detail: describeOrphanCause(diskTask),
|
|
157
|
+
taskId: diskTask.id,
|
|
158
|
+
ownerPid: diskTask.ownerPid || null,
|
|
159
|
+
});
|
|
160
|
+
} catch (err) {
|
|
161
|
+
// Honest, non-fatal: the trace is best-effort. The task file is already
|
|
162
|
+
// marked orphaned (the primary signal). Log and continue.
|
|
163
|
+
const msg = err && err.message ? err.message : String(err);
|
|
164
|
+
console.error(
|
|
165
|
+
`[task_registry] Failed to append orphan terminal event for session ${diskTask.sessionId}: ${msg}`
|
|
166
|
+
);
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
/**
|
|
171
|
+
* Honest, human-readable description of WHY the task was orphaned (the owner
|
|
172
|
+
* pid state), for the terminal event's `detail` field.
|
|
173
|
+
*/
|
|
174
|
+
function describeOrphanCause(diskTask) {
|
|
175
|
+
const pid = diskTask.ownerPid;
|
|
176
|
+
const platform = diskTask.ownerPlatform || process.platform;
|
|
177
|
+
if (pid) {
|
|
178
|
+
if (!pidAlive(pid, platform)) {
|
|
179
|
+
return `owner pid ${pid} (${platform}) is dead (process exited or was killed)`;
|
|
180
|
+
}
|
|
181
|
+
return `owner pid ${pid} (${platform}) is alive but heartbeat is stale`;
|
|
182
|
+
}
|
|
183
|
+
return `no owner pid recorded; heartbeat is stale`;
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
/**
|
|
187
|
+
* Quarantines a corrupt/unreadable task file by renaming it to
|
|
188
|
+
* `<name>.corrupt-<epochms>` (the LineageDag quarantine pattern) and logs
|
|
189
|
+
* loudly to stderr. Returns the underlying error message. Shared by
|
|
190
|
+
* readTaskFromDisk (single read) and listTasksFromDisk (bulk read) so BOTH
|
|
191
|
+
* surface corruption identically instead of silently swallowing it.
|
|
192
|
+
*
|
|
193
|
+
// Report corrupted task disk artifacts with HTTP 500 containing parse details.
|
|
194
|
+
* not-found. It is preserved (renamed, not deleted) so it can be inspected,
|
|
195
|
+
* and the corruption is announced on stderr.
|
|
196
|
+
*/
|
|
197
|
+
function quarantineCorruptTaskFile(filePath, err) {
|
|
198
|
+
const msg = err && err.message ? err.message : String(err);
|
|
199
|
+
const corruptBackup = `${filePath}.corrupt-${Date.now()}`;
|
|
200
|
+
console.error(
|
|
201
|
+
`[task_registry] Corrupt task file ${filePath}: ${msg}. Quarantining to ${corruptBackup}.`
|
|
202
|
+
);
|
|
203
|
+
try {
|
|
204
|
+
fs.renameSync(filePath, corruptBackup);
|
|
205
|
+
} catch (qerr) {
|
|
206
|
+
// The rename itself failed (e.g. the file vanished between the read and
|
|
207
|
+
// the rename, or a transient lock). The corruption is still surfaced via
|
|
208
|
+
// the stderr log and the returned message; we do not fabricate a success.
|
|
209
|
+
console.error(
|
|
210
|
+
`[task_registry] Failed to quarantine ${filePath}: ${
|
|
211
|
+
qerr && qerr.message ? qerr.message : String(qerr)
|
|
212
|
+
}`
|
|
213
|
+
);
|
|
214
|
+
}
|
|
215
|
+
return msg;
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
/**
|
|
219
|
+
* Reads a single task from disk, distinguishing execution signals:
|
|
220
|
+
* - file ABSENT -> null (not found)
|
|
221
|
+
* - transient file lock -> { transientLock: true, id } (retry next tick)
|
|
222
|
+
* - file CORRUPT/unreadable -> { corrupted: true, id, file, error } (quarantined)
|
|
223
|
+
* - healthy file -> parsed task object
|
|
224
|
+
*
|
|
225
|
+
* @param {string} taskId Unique task identifier.
|
|
226
|
+
* @returns {object|null} Parsed task object, lock/corruption descriptor, or null.
|
|
227
|
+
*/
|
|
228
|
+
export function readTaskFromDisk(taskId) {
|
|
229
|
+
const filePath = path.join(TASK_DIR, `${taskId}.json`);
|
|
230
|
+
if (!fs.existsSync(filePath)) {
|
|
231
|
+
return null; // clean not-found
|
|
232
|
+
}
|
|
233
|
+
let raw;
|
|
234
|
+
try {
|
|
235
|
+
raw = fs.readFileSync(filePath, "utf8");
|
|
236
|
+
} catch (err) {
|
|
237
|
+
if (err && (err.code === "EBUSY" || err.code === "EPERM")) {
|
|
238
|
+
// Transient Windows file lock while the worker is mid-write: a real
|
|
239
|
+
// signal, not corruption. The caller retries on the next tick.
|
|
240
|
+
return { transientLock: true, id: taskId };
|
|
241
|
+
}
|
|
242
|
+
// Unreadable for a non-transient reason: corruption.
|
|
243
|
+
const msg = quarantineCorruptTaskFile(filePath, err);
|
|
244
|
+
return { corrupted: true, id: taskId, file: filePath, error: msg };
|
|
245
|
+
}
|
|
246
|
+
let parsed;
|
|
247
|
+
try {
|
|
248
|
+
parsed = JSON.parse(raw);
|
|
249
|
+
} catch (err) {
|
|
250
|
+
// Unparseable: corruption.
|
|
251
|
+
const msg = quarantineCorruptTaskFile(filePath, err);
|
|
252
|
+
return { corrupted: true, id: taskId, file: filePath, error: msg };
|
|
253
|
+
}
|
|
254
|
+
if (parsed && !parsed.done && isTaskOrphaned(parsed)) {
|
|
255
|
+
return markTaskOrphanedOnDisk(parsed);
|
|
256
|
+
}
|
|
257
|
+
return parsed;
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
/**
|
|
261
|
+
* Lists all on-disk tasks, applying retention cleanup.
|
|
262
|
+
*
|
|
263
|
+
// Report corrupted task disk artifacts with HTTP 500 containing parse details.
|
|
264
|
+
* `catch {}` swallowed it, so a corrupt task vanished from the list with no
|
|
265
|
+
* signal). Now each corrupt file is QUARANTINEd and logged loudly to stderr
|
|
266
|
+
* (same helper as readTaskFromDisk); a transient lock (EBUSY/EPERM) is still
|
|
267
|
+
* skipped for this tick (the worker is mid-write) but is a distinct, honest
|
|
268
|
+
* case. The returned list contains only healthy, parseable tasks.
|
|
269
|
+
*
|
|
270
|
+
* Retention cleanup also reaps orphaned `task_*.json.tmp_<pid>_<ts>` files
|
|
271
|
+
* (a saveTaskToDisk whose writeFileSync succeeded but whose renameSync never
|
|
272
|
+
* ran). Those orphans never match the `.json` filter, so without this they
|
|
273
|
+
* accumulate forever; the same mtime age gate reaps them (their lifetime is
|
|
274
|
+
* sub-second, so the gate is sufficient).
|
|
275
|
+
*/
|
|
276
|
+
export function listTasksFromDisk() {
|
|
277
|
+
const result = [];
|
|
278
|
+
let files;
|
|
279
|
+
try {
|
|
280
|
+
files = fs.readdirSync(TASK_DIR);
|
|
281
|
+
} catch {
|
|
282
|
+
return result; // TASK_DIR unreadable this tick: return what we have.
|
|
283
|
+
}
|
|
284
|
+
const now = Date.now();
|
|
285
|
+
for (const f of files) {
|
|
286
|
+
// Match completed task JSON files and orphaned atomic write tmp files.
|
|
287
|
+
const isTmpOrphan = /\.json\.tmp_\d+_\d+$/.test(f);
|
|
288
|
+
if (!f.endsWith(".json") && !isTmpOrphan) continue;
|
|
289
|
+
const filePath = path.join(TASK_DIR, f);
|
|
290
|
+
let stat;
|
|
291
|
+
try {
|
|
292
|
+
stat = fs.statSync(filePath);
|
|
293
|
+
} catch {
|
|
294
|
+
continue; // vanished between readdir and stat
|
|
295
|
+
}
|
|
296
|
+
if (now - stat.mtimeMs > TASK_RETENTION_MS) {
|
|
297
|
+
try {
|
|
298
|
+
fs.unlinkSync(filePath);
|
|
299
|
+
} catch {}
|
|
300
|
+
continue;
|
|
301
|
+
}
|
|
302
|
+
if (isTmpOrphan) continue; // within retention: leave it (sub-second, ages out)
|
|
303
|
+
let raw;
|
|
304
|
+
try {
|
|
305
|
+
raw = fs.readFileSync(filePath, "utf8");
|
|
306
|
+
} catch (err) {
|
|
307
|
+
if (err && (err.code === "EBUSY" || err.code === "EPERM")) continue; // transient lock
|
|
308
|
+
quarantineCorruptTaskFile(filePath, err);
|
|
309
|
+
continue;
|
|
310
|
+
}
|
|
311
|
+
let task;
|
|
312
|
+
try {
|
|
313
|
+
task = JSON.parse(raw);
|
|
314
|
+
} catch (err) {
|
|
315
|
+
quarantineCorruptTaskFile(filePath, err);
|
|
316
|
+
continue;
|
|
317
|
+
}
|
|
318
|
+
result.push(task);
|
|
319
|
+
}
|
|
320
|
+
return result;
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
/**
|
|
324
|
+
* Returns true when live work is in flight and the engine must NOT be
|
|
325
|
+
* stopped/rebooted. Used as the heal gatekeeper:
|
|
326
|
+
* - any in-memory task with status "running"/"queued" (not done), OR
|
|
327
|
+
* - any disk task that is not done, whose owner pid is alive, and whose
|
|
328
|
+
* last heartbeat is within INACTIVITY_TIMEOUT_MS.
|
|
329
|
+
*/
|
|
330
|
+
export function hasLiveWork() {
|
|
331
|
+
for (const t of tasks.values()) {
|
|
332
|
+
if (!t.done && (t.status === "running" || t.status === "queued")) return true;
|
|
333
|
+
}
|
|
334
|
+
const now = Date.now();
|
|
335
|
+
for (const dt of listTasksFromDisk()) {
|
|
336
|
+
if (dt.done) continue;
|
|
337
|
+
if (!dt.ownerPid || !pidAlive(dt.ownerPid, dt.ownerPlatform)) continue;
|
|
338
|
+
const lastActive = dt.lastHeartbeatAt || dt.startedAt || dt.createdAt;
|
|
339
|
+
if (lastActive && now - lastActive <= INACTIVITY_TIMEOUT_MS) return true;
|
|
340
|
+
}
|
|
341
|
+
return false;
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
/**
|
|
345
|
+
// Periodic liveness check: reap orphaned tasks on the regular retention interval.
|
|
346
|
+
*
|
|
347
|
+
* The boot-only orphan sweep (markTaskOrphanedOnDisk, invoked from
|
|
348
|
+
* readTaskFromDisk) only runs when a task file is READ at process start. A
|
|
349
|
+
* task whose owner process dies MID-SESSION — e.g. task_anomaly-probe-s1,
|
|
350
|
+
* frozen at status:"running" with a dead ownerPid and no terminal event — is
|
|
351
|
+
* never re-read by a boot sweep, so it stays "running" forever and misleads
|
|
352
|
+
* every later probe (the N2 defect).
|
|
353
|
+
*
|
|
354
|
+
* This pass closes that gap. For every NOT-DONE task (in-memory AND on-disk)
|
|
355
|
+
* whose status is "queued"/"running" and whose heartbeat is STALE
|
|
356
|
+
* (now - lastHeartbeatAt > ORPHAN_REAP_STALE_MS), it reaps the task ONLY when
|
|
357
|
+
* the owner pid is dead — reusing the SAME liveness helper (pidAlive) the boot
|
|
358
|
+
* sweep uses, and writing the SAME terminal marker by CALLING the existing
|
|
359
|
+
* markTaskOrphanedOnDisk (never duplicating its logic):
|
|
360
|
+
* { type:"session_error", reason:"orphaned" } + status/isError terminal +
|
|
361
|
+
* finishedAt.
|
|
362
|
+
*
|
|
363
|
+
* LIVE-OWNER INVARIANT: a task whose ownerPid is ALIVE is NEVER reaped,
|
|
364
|
+
* regardless of how stale its heartbeat is. A live owner that is simply slow
|
|
365
|
+
* (a long deep-thinking turn, a wedged-but-alive worker) is not an orphan;
|
|
366
|
+
* reaping it would kill a healthy task. The dead-owner check is the gate; the
|
|
367
|
+
* stale-heartbeat check is only a conservative pre-filter so a live owner with
|
|
368
|
+
* a fresh heartbeat is never even considered.
|
|
369
|
+
*
|
|
370
|
+
* Conservative by design: a task with a FRESH heartbeat (within the stale
|
|
371
|
+
* window) is left untouched even if its owner pid is dead — the owner may
|
|
372
|
+
* still be writing its final state, and the next cadence tick re-checks.
|
|
373
|
+
*
|
|
374
|
+
* Idempotent: markTaskOrphanedOnDisk no-ops on an already-done task, and the
|
|
375
|
+
* double-terminal guard in appendOrphanTerminalEvent prevents a second
|
|
376
|
+
* terminal event, so re-running this pass is safe.
|
|
377
|
+
*
|
|
378
|
+
* @returns {number} the count of tasks reaped (marked orphaned) this pass.
|
|
379
|
+
*/
|
|
380
|
+
export function reapOrphans() {
|
|
381
|
+
const now = Date.now();
|
|
382
|
+
let reaped = 0;
|
|
383
|
+
|
|
384
|
+
// 1. In-memory tasks (this instance's live registry).
|
|
385
|
+
for (const task of tasks.values()) {
|
|
386
|
+
if (task.done) continue;
|
|
387
|
+
if (task.status !== "running" && task.status !== "queued") continue;
|
|
388
|
+
const lastActive = task.lastHeartbeatAt || task.startedAt || task.createdAt;
|
|
389
|
+
if (!lastActive || now - lastActive <= ORPHAN_REAP_STALE_MS) continue; // fresh: leave it
|
|
390
|
+
// LIVE-OWNER INVARIANT: never reap an in-memory task whose owner is alive.
|
|
391
|
+
const ownerPid = task.ownerPid || process.pid;
|
|
392
|
+
const ownerPlatform = task.ownerPlatform || process.platform;
|
|
393
|
+
if (pidAlive(ownerPid, ownerPlatform)) continue;
|
|
394
|
+
markTaskOrphanedOnDisk(task);
|
|
395
|
+
reaped++;
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
// 2. On-disk tasks (any instance sharing the state dir).
|
|
399
|
+
for (const diskTask of listTasksFromDisk()) {
|
|
400
|
+
if (diskTask.done) continue;
|
|
401
|
+
if (diskTask.status !== "running" && diskTask.status !== "queued") continue;
|
|
402
|
+
const lastActive = diskTask.lastHeartbeatAt || diskTask.startedAt || diskTask.createdAt;
|
|
403
|
+
if (!lastActive || now - lastActive <= ORPHAN_REAP_STALE_MS) continue; // fresh: leave it
|
|
404
|
+
// LIVE-OWNER INVARIANT: never reap a task whose owner is alive.
|
|
405
|
+
if (diskTask.ownerPid && pidAlive(diskTask.ownerPid, diskTask.ownerPlatform)) continue;
|
|
406
|
+
markTaskOrphanedOnDisk(diskTask);
|
|
407
|
+
reaped++;
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
return reaped;
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
export function cleanOldTasks() {
|
|
414
|
+
const now = Date.now();
|
|
415
|
+
for (const [id, task] of tasks.entries()) {
|
|
416
|
+
if (task.done && now - task.createdAt > TASK_RETENTION_MS) {
|
|
417
|
+
tasks.delete(id);
|
|
418
|
+
}
|
|
419
|
+
}
|
|
420
|
+
listTasksFromDisk(); // Triggers disk retention cleanup
|
|
421
|
+
// Periodic liveness check: reap orphaned tasks on the regular retention interval.
|
|
422
|
+
reapOrphans();
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
export function notifyWaiters(task) {
|
|
426
|
+
saveTaskToDisk(task);
|
|
427
|
+
if (!task.waiters || task.waiters.length === 0) return;
|
|
428
|
+
// Terminal wait responses return HTTP 200 with standard task JSON.
|
|
429
|
+
// Connection:close ensures socket terminates cleanly.
|
|
430
|
+
const now = Date.now();
|
|
431
|
+
const elapsed_s = Math.round(
|
|
432
|
+
((task.finishedAt || now) - task.createdAt) / 1000
|
|
433
|
+
);
|
|
434
|
+
const lastActivitySecAgo = task.lastActivityAt
|
|
435
|
+
? Math.max(0, Math.round((now - task.lastActivityAt) / 1000))
|
|
436
|
+
: null;
|
|
437
|
+
const body = JSON.stringify(
|
|
438
|
+
{
|
|
439
|
+
found: true,
|
|
440
|
+
id: task.id,
|
|
441
|
+
sessionId: task.sessionId,
|
|
442
|
+
cwd: task.cwd,
|
|
443
|
+
status: task.status,
|
|
444
|
+
done: task.done,
|
|
445
|
+
isError: task.isError,
|
|
446
|
+
reasoningEffort: task.reasoningEffort ?? null,
|
|
447
|
+
elapsed_s,
|
|
448
|
+
startedAt: task.startedAt,
|
|
449
|
+
lastActivitySecAgo,
|
|
450
|
+
budgetTurns: task.budgetTurns || BASE_TURN_BUDGET,
|
|
451
|
+
leaseExtensionsCount: task.leaseExtensionsCount || 0,
|
|
452
|
+
lastActivityPreview: task.lastActivityPreview || "",
|
|
453
|
+
streamBytes: task.streamBytes || 0,
|
|
454
|
+
streamTail: (task.streamTail || "")
|
|
455
|
+
.replace(/["\\{}\[\]]|type|message|content|delta|thinking|text/g, " ")
|
|
456
|
+
.replace(/\s+/g, " ")
|
|
457
|
+
.slice(-250),
|
|
458
|
+
fileOps: (task.fileOps || []).slice(-5),
|
|
459
|
+
toolCallsCount: task.toolCallsCount || 0,
|
|
460
|
+
lastPromptTokens: task.lastPromptTokens ?? null,
|
|
461
|
+
contextHeadroom: task.contextHeadroom ?? null,
|
|
462
|
+
result: task.result || null,
|
|
463
|
+
},
|
|
464
|
+
null,
|
|
465
|
+
2
|
|
466
|
+
);
|
|
467
|
+
for (const res of task.waiters) {
|
|
468
|
+
try {
|
|
469
|
+
res.writeHead(200, {
|
|
470
|
+
"Content-Type": "application/json",
|
|
471
|
+
"Connection": "close",
|
|
472
|
+
"X-Task-ID": task.id,
|
|
473
|
+
"X-Task-Status": task.status,
|
|
474
|
+
});
|
|
475
|
+
res.end(body);
|
|
476
|
+
} catch {}
|
|
477
|
+
}
|
|
478
|
+
task.waiters = [];
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
export async function cancelAllTasks(reason = "cancelled by caller") {
|
|
482
|
+
let count = 0;
|
|
483
|
+
// Cancel in-memory tasks and collect session IDs for child process cleanup.
|
|
484
|
+
const memSessionIds = new Set();
|
|
485
|
+
for (const task of tasks.values()) {
|
|
486
|
+
if (!task.done) {
|
|
487
|
+
if (task.child) {
|
|
488
|
+
killProcessTree(task.child, task.sessionId);
|
|
489
|
+
task.child = null;
|
|
490
|
+
}
|
|
491
|
+
if (task.abortController) {
|
|
492
|
+
try {
|
|
493
|
+
task.abortController.abort();
|
|
494
|
+
} catch {}
|
|
495
|
+
task.abortController = null;
|
|
496
|
+
}
|
|
497
|
+
// Release task execution slot immediately on cancel.
|
|
498
|
+
if (task.slot) {
|
|
499
|
+
releaseTaskSlot(task.slot);
|
|
500
|
+
task.slot = null;
|
|
501
|
+
}
|
|
502
|
+
task.status = "cancelled";
|
|
503
|
+
task.done = true;
|
|
504
|
+
task.isError = true;
|
|
505
|
+
task.finishedAt = Date.now();
|
|
506
|
+
task.result = { isError: true, text: `Task ${task.id} was ${reason}.` };
|
|
507
|
+
saveTaskToDisk(task);
|
|
508
|
+
notifyWaiters(task);
|
|
509
|
+
count++;
|
|
510
|
+
if (task.sessionId) memSessionIds.add(task.sessionId);
|
|
511
|
+
}
|
|
512
|
+
}
|
|
513
|
+
|
|
514
|
+
// 2. Cancel disk tasks (only this process's own tasks or dead owners' tasks)
|
|
515
|
+
const diskSessionIds = new Set();
|
|
516
|
+
for (const diskTask of listTasksFromDisk()) {
|
|
517
|
+
if (!diskTask.done) {
|
|
518
|
+
// LIVE-OWNER INVARIANT: Never cancel or kill tasks owned by another living process.
|
|
519
|
+
if (diskTask.ownerPid && diskTask.ownerPid !== process.pid && pidAlive(diskTask.ownerPid, diskTask.ownerPlatform)) {
|
|
520
|
+
continue;
|
|
521
|
+
}
|
|
522
|
+
diskTask.status = "cancelled";
|
|
523
|
+
diskTask.done = true;
|
|
524
|
+
diskTask.isError = true;
|
|
525
|
+
diskTask.finishedAt = Date.now();
|
|
526
|
+
diskTask.result = { isError: true, text: `Task ${diskTask.id} was ${reason}.` };
|
|
527
|
+
saveTaskToDisk(diskTask);
|
|
528
|
+
count++;
|
|
529
|
+
if (diskTask.sessionId) diskSessionIds.add(diskTask.sessionId);
|
|
530
|
+
}
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
// Terminate child processes matching exact session IDs across WSL and Windows.
|
|
534
|
+
// sessions survive. Sync kills are fast; cancel_all still returns promptly.
|
|
535
|
+
const sweepIds = new Set([...memSessionIds, ...diskSessionIds]);
|
|
536
|
+
for (const sessionId of sweepIds) {
|
|
537
|
+
try {
|
|
538
|
+
killSessionProcessTreeSync(sessionId);
|
|
539
|
+
} catch {}
|
|
540
|
+
}
|
|
541
|
+
|
|
542
|
+
// 4. Clean up slot lease locks (only this process's own or dead owners'
|
|
543
|
+
// leases — never another live instance's lease)
|
|
544
|
+
clearReclaimableTaskSlots();
|
|
545
|
+
|
|
546
|
+
return count;
|
|
547
|
+
}
|
|
548
|
+
|
|
549
|
+
/**
|
|
550
|
+
* Extends the execution turn budget for an active background task.
|
|
551
|
+
* Allows supervisor (Claude/Gemini) to grant extra turns dynamically
|
|
552
|
+
* without killing the task or re-dispatching from scratch.
|
|
553
|
+
*
|
|
554
|
+
* @param {string} taskId
|
|
555
|
+
* @param {number} [additionalTurns=25]
|
|
556
|
+
* @param {string} [reason="supervisor request"]
|
|
557
|
+
* @returns {{ success: boolean, taskId?: string, previousBudget?: number, budgetTurns?: number, leaseExtensionsCount?: number, error?: string, reason?: string }}
|
|
558
|
+
*/
|
|
559
|
+
export function extendTaskBudget(taskId, additionalTurns = 25, reason = "supervisor request") {
|
|
560
|
+
if (!taskId) return { success: false, error: "Task ID required" };
|
|
561
|
+
let task = tasks.get(taskId);
|
|
562
|
+
if (!task) {
|
|
563
|
+
task = readTaskFromDisk(taskId);
|
|
564
|
+
}
|
|
565
|
+
if (!task) {
|
|
566
|
+
return { success: false, error: "Task not found" };
|
|
567
|
+
}
|
|
568
|
+
if (task.corrupted) {
|
|
569
|
+
return { success: false, error: `Task file corrupted: ${task.error}` };
|
|
570
|
+
}
|
|
571
|
+
if (task.done) {
|
|
572
|
+
return { success: false, error: "Task already finished" };
|
|
573
|
+
}
|
|
574
|
+
const current = task.budgetTurns || BASE_TURN_BUDGET;
|
|
575
|
+
const newBudget = Math.min(current + additionalTurns, MAX_ELASTIC_TURNS);
|
|
576
|
+
task.budgetTurns = newBudget;
|
|
577
|
+
task.leaseExtensionsCount = (task.leaseExtensionsCount || 0) + 1;
|
|
578
|
+
saveTaskToDisk(task);
|
|
579
|
+
return {
|
|
580
|
+
success: true,
|
|
581
|
+
taskId,
|
|
582
|
+
previousBudget: current,
|
|
583
|
+
budgetTurns: newBudget,
|
|
584
|
+
leaseExtensionsCount: task.leaseExtensionsCount,
|
|
585
|
+
reason,
|
|
586
|
+
};
|
|
587
|
+
}
|
|
588
|
+
|
|
589
|
+
export let statusServerOwned = false;
|
|
590
|
+
|
|
591
|
+
let mcpServerFactory = null;
|
|
592
|
+
export function setMcpServerFactory(fn) {
|
|
593
|
+
mcpServerFactory = typeof fn === "function" ? fn : null;
|
|
594
|
+
}
|
|
595
|
+
function getMcpServerFactory() {
|
|
596
|
+
return mcpServerFactory;
|
|
597
|
+
}
|
|
598
|
+
export const activeSseSessions = new Map();
|
|
599
|
+
|
|
600
|
+
export const statusHttpServer = http.createServer(async (req, res) => {
|
|
601
|
+
const parsedUrl = new URL(req.url, `http://localhost:${STATUS_PORT}`);
|
|
602
|
+
const pathname = parsedUrl.pathname;
|
|
603
|
+
|
|
604
|
+
// Global CORS headers for cross-boundary / local tooling access
|
|
605
|
+
res.setHeader("Access-Control-Allow-Origin", "*");
|
|
606
|
+
res.setHeader("Access-Control-Allow-Methods", "GET, POST, OPTIONS");
|
|
607
|
+
res.setHeader("Access-Control-Allow-Headers", "Content-Type, Authorization");
|
|
608
|
+
|
|
609
|
+
if (req.method === "OPTIONS") {
|
|
610
|
+
res.writeHead(204);
|
|
611
|
+
return res.end();
|
|
612
|
+
}
|
|
613
|
+
|
|
614
|
+
// GET /sse — MCP Server-Sent Events stream endpoint
|
|
615
|
+
if (req.method === "GET" && pathname === "/sse") {
|
|
616
|
+
if (!mcpServerFactory) {
|
|
617
|
+
res.writeHead(503, { "Content-Type": "application/json" });
|
|
618
|
+
return res.end(JSON.stringify({ error: "MCP server factory not initialized" }));
|
|
619
|
+
}
|
|
620
|
+
try {
|
|
621
|
+
const transport = new SSEServerTransport("/message", res);
|
|
622
|
+
const server = mcpServerFactory();
|
|
623
|
+
const sessionId = transport.sessionId;
|
|
624
|
+
activeSseSessions.set(sessionId, { transport, server });
|
|
625
|
+
|
|
626
|
+
const cleanup = () => {
|
|
627
|
+
if (activeSseSessions.has(sessionId)) {
|
|
628
|
+
activeSseSessions.delete(sessionId);
|
|
629
|
+
try {
|
|
630
|
+
server.close();
|
|
631
|
+
} catch {}
|
|
632
|
+
}
|
|
633
|
+
};
|
|
634
|
+
|
|
635
|
+
transport.onclose = cleanup;
|
|
636
|
+
res.on("close", cleanup);
|
|
637
|
+
|
|
638
|
+
await server.connect(transport);
|
|
639
|
+
} catch (err) {
|
|
640
|
+
if (!res.headersSent) {
|
|
641
|
+
res.writeHead(500, { "Content-Type": "application/json" });
|
|
642
|
+
res.end(JSON.stringify({ error: err.message }));
|
|
643
|
+
}
|
|
644
|
+
}
|
|
645
|
+
return;
|
|
646
|
+
}
|
|
647
|
+
|
|
648
|
+
// POST /message — MCP client message endpoint for active SSE session
|
|
649
|
+
if (req.method === "POST" && pathname === "/message") {
|
|
650
|
+
const sessionId = parsedUrl.searchParams.get("sessionId");
|
|
651
|
+
if (!sessionId) {
|
|
652
|
+
res.writeHead(400, { "Content-Type": "application/json" });
|
|
653
|
+
return res.end(JSON.stringify({ error: "Missing sessionId query parameter" }));
|
|
654
|
+
}
|
|
655
|
+
const session = activeSseSessions.get(sessionId);
|
|
656
|
+
if (!session) {
|
|
657
|
+
res.writeHead(404, { "Content-Type": "application/json" });
|
|
658
|
+
return res.end(JSON.stringify({ error: `Session not found: ${sessionId}` }));
|
|
659
|
+
}
|
|
660
|
+
try {
|
|
661
|
+
await session.transport.handlePostMessage(req, res);
|
|
662
|
+
} catch (err) {
|
|
663
|
+
if (!res.headersSent) {
|
|
664
|
+
res.writeHead(500, { "Content-Type": "application/json" });
|
|
665
|
+
res.end(JSON.stringify({ error: err.message }));
|
|
666
|
+
}
|
|
667
|
+
}
|
|
668
|
+
return;
|
|
669
|
+
}
|
|
670
|
+
|
|
671
|
+
// GET /health
|
|
672
|
+
if (req.method === "GET" && pathname === "/health") {
|
|
673
|
+
res.writeHead(200, { "Content-Type": "application/json" });
|
|
674
|
+
res.end(JSON.stringify({
|
|
675
|
+
status: "ok",
|
|
676
|
+
service: STATUS_SERVICE_IDENTITY,
|
|
677
|
+
port: STATUS_PORT,
|
|
678
|
+
sse: true,
|
|
679
|
+
active_sse_sessions: activeSseSessions.size,
|
|
680
|
+
}));
|
|
681
|
+
return;
|
|
682
|
+
}
|
|
683
|
+
|
|
684
|
+
// GET /tasks
|
|
685
|
+
if (req.method === "GET" && pathname === "/tasks") {
|
|
686
|
+
const merged = new Map();
|
|
687
|
+
for (const dt of listTasksFromDisk()) {
|
|
688
|
+
merged.set(dt.id, {
|
|
689
|
+
id: dt.id,
|
|
690
|
+
sessionId: dt.sessionId,
|
|
691
|
+
cwd: dt.cwd,
|
|
692
|
+
status: dt.status,
|
|
693
|
+
createdAt: dt.createdAt,
|
|
694
|
+
elapsed_s: Math.round(((dt.finishedAt || Date.now()) - dt.createdAt) / 1000),
|
|
695
|
+
done: dt.done,
|
|
696
|
+
isError: dt.isError,
|
|
697
|
+
});
|
|
698
|
+
}
|
|
699
|
+
for (const t of tasks.values()) {
|
|
700
|
+
merged.set(t.id, {
|
|
701
|
+
id: t.id,
|
|
702
|
+
sessionId: t.sessionId,
|
|
703
|
+
cwd: t.cwd,
|
|
704
|
+
status: t.status,
|
|
705
|
+
createdAt: t.createdAt,
|
|
706
|
+
elapsed_s: Math.round(((t.finishedAt || Date.now()) - t.createdAt) / 1000),
|
|
707
|
+
done: t.done,
|
|
708
|
+
isError: t.isError,
|
|
709
|
+
streamBytes: t.streamBytes || 0,
|
|
710
|
+
streamTail: (t.streamTail || "")
|
|
711
|
+
.replace(/["\\{}\[\]]|type|message|content|delta|thinking|text/g, " ")
|
|
712
|
+
.replace(/\s+/g, " ")
|
|
713
|
+
.slice(-150),
|
|
714
|
+
});
|
|
715
|
+
}
|
|
716
|
+
res.writeHead(200, { "Content-Type": "application/json" });
|
|
717
|
+
return res.end(JSON.stringify({ active_tasks: Array.from(merged.values()) }, null, 2));
|
|
718
|
+
}
|
|
719
|
+
|
|
720
|
+
// GET /task/:id/wait (Universal blocking long-poll across memory + disk)
|
|
721
|
+
const waitMatch = pathname.match(/^\/task\/([^/]+)\/wait$/);
|
|
722
|
+
if (req.method === "GET" && waitMatch) {
|
|
723
|
+
if (req.socket) {
|
|
724
|
+
try {
|
|
725
|
+
req.socket.setKeepAlive(true, 15_000);
|
|
726
|
+
req.socket.setTimeout(0);
|
|
727
|
+
} catch {}
|
|
728
|
+
}
|
|
729
|
+
const taskId = waitMatch[1];
|
|
730
|
+
let task = tasks.get(taskId);
|
|
731
|
+
if (task && !task.done && isTaskOrphaned(task)) {
|
|
732
|
+
task = markTaskOrphanedOnDisk(task);
|
|
733
|
+
}
|
|
734
|
+
let diskTask = null;
|
|
735
|
+
if (!task) {
|
|
736
|
+
diskTask = readTaskFromDisk(taskId);
|
|
737
|
+
if (diskTask && diskTask.corrupted) {
|
|
738
|
+
// Report corrupted task disk artifacts with HTTP 500 containing parse details.
|
|
739
|
+
res.writeHead(500, {
|
|
740
|
+
"Content-Type": "application/json",
|
|
741
|
+
"Connection": "close",
|
|
742
|
+
});
|
|
743
|
+
return res.end(
|
|
744
|
+
JSON.stringify({
|
|
745
|
+
found: true,
|
|
746
|
+
id: taskId,
|
|
747
|
+
corrupted: true,
|
|
748
|
+
file: diskTask.file,
|
|
749
|
+
error: `Task file corrupted: ${diskTask.error}`,
|
|
750
|
+
})
|
|
751
|
+
);
|
|
752
|
+
}
|
|
753
|
+
if (!diskTask) {
|
|
754
|
+
res.writeHead(404, { "Content-Type": "text/plain" });
|
|
755
|
+
return res.end(`Task not found: ${taskId}`);
|
|
756
|
+
}
|
|
757
|
+
}
|
|
758
|
+
|
|
759
|
+
const isDone = task ? task.done : diskTask.done;
|
|
760
|
+
if (isDone) {
|
|
761
|
+
// Terminal wait responses return HTTP 200 with standard task JSON.
|
|
762
|
+
// Connection:close ensures socket terminates cleanly.
|
|
763
|
+
const t = task || diskTask;
|
|
764
|
+
const now = Date.now();
|
|
765
|
+
const elapsed_s = Math.round(((t.finishedAt || now) - t.createdAt) / 1000);
|
|
766
|
+
const lastActivitySecAgo = t.lastActivityAt
|
|
767
|
+
? Math.max(0, Math.round((now - t.lastActivityAt) / 1000))
|
|
768
|
+
: null;
|
|
769
|
+
res.writeHead(200, {
|
|
770
|
+
"Content-Type": "application/json",
|
|
771
|
+
"Connection": "close",
|
|
772
|
+
});
|
|
773
|
+
return res.end(
|
|
774
|
+
JSON.stringify(
|
|
775
|
+
{
|
|
776
|
+
found: true,
|
|
777
|
+
id: t.id,
|
|
778
|
+
sessionId: t.sessionId,
|
|
779
|
+
cwd: t.cwd,
|
|
780
|
+
status: t.status,
|
|
781
|
+
done: t.done,
|
|
782
|
+
isError: t.isError,
|
|
783
|
+
reasoningEffort: t.reasoningEffort ?? null,
|
|
784
|
+
elapsed_s,
|
|
785
|
+
startedAt: t.startedAt,
|
|
786
|
+
lastActivitySecAgo,
|
|
787
|
+
budgetTurns: t.budgetTurns || BASE_TURN_BUDGET,
|
|
788
|
+
leaseExtensionsCount: t.leaseExtensionsCount || 0,
|
|
789
|
+
lastActivityPreview: t.lastActivityPreview || "",
|
|
790
|
+
streamBytes: t.streamBytes || 0,
|
|
791
|
+
streamTail: (t.streamTail || "")
|
|
792
|
+
.replace(/["\\{}\[\]]|type|message|content|delta|thinking|text/g, " ")
|
|
793
|
+
.replace(/\s+/g, " ")
|
|
794
|
+
.slice(-250),
|
|
795
|
+
fileOps: t.fileOps || [],
|
|
796
|
+
toolCallsCount: t.toolCallsCount || 0,
|
|
797
|
+
lastPromptTokens: t.lastPromptTokens ?? null,
|
|
798
|
+
contextHeadroom: t.contextHeadroom ?? null,
|
|
799
|
+
result: t.result || null,
|
|
800
|
+
},
|
|
801
|
+
null,
|
|
802
|
+
2
|
|
803
|
+
)
|
|
804
|
+
);
|
|
805
|
+
}
|
|
806
|
+
|
|
807
|
+
if (task) {
|
|
808
|
+
task.waiters = task.waiters || [];
|
|
809
|
+
task.waiters.push(res);
|
|
810
|
+
req.on("close", () => {
|
|
811
|
+
task.waiters = task.waiters.filter((w) => w !== res);
|
|
812
|
+
});
|
|
813
|
+
return;
|
|
814
|
+
}
|
|
815
|
+
|
|
816
|
+
// Disk-based task from another instance: poll disk until done
|
|
817
|
+
const waitStartTime = Date.now();
|
|
818
|
+
let absentTicks = 0;
|
|
819
|
+
const diskPoll = setInterval(() => {
|
|
820
|
+
const current = readTaskFromDisk(taskId);
|
|
821
|
+
if (current?.transientLock) {
|
|
822
|
+
// Transient Windows file lock (EBUSY/EPERM) during worker saveTaskToDisk.
|
|
823
|
+
// Worker is actively writing; skip tick and continue polling.
|
|
824
|
+
return;
|
|
825
|
+
}
|
|
826
|
+
if (current?.corrupted) {
|
|
827
|
+
// Report corrupted task disk artifacts with HTTP 500 containing parse details.
|
|
828
|
+
// corruption signal (500) naming the file and the parse error — never
|
|
829
|
+
// debounced into a fabricated "Task failed."
|
|
830
|
+
clearInterval(diskPoll);
|
|
831
|
+
try {
|
|
832
|
+
res.writeHead(500, {
|
|
833
|
+
"Content-Type": "application/json",
|
|
834
|
+
"Connection": "close",
|
|
835
|
+
});
|
|
836
|
+
res.end(
|
|
837
|
+
JSON.stringify({
|
|
838
|
+
id: taskId,
|
|
839
|
+
corrupted: true,
|
|
840
|
+
file: current.file,
|
|
841
|
+
error: `Task file corrupted: ${current.error}`,
|
|
842
|
+
})
|
|
843
|
+
);
|
|
844
|
+
} catch {}
|
|
845
|
+
return;
|
|
846
|
+
}
|
|
847
|
+
|
|
848
|
+
if (!current) {
|
|
849
|
+
absentTicks++;
|
|
850
|
+
// Debounce: require 3 consecutive absent ticks (6 seconds) before treating as missing/failed
|
|
851
|
+
if (absentTicks < 3) {
|
|
852
|
+
return;
|
|
853
|
+
}
|
|
854
|
+
} else {
|
|
855
|
+
absentTicks = 0;
|
|
856
|
+
}
|
|
857
|
+
|
|
858
|
+
if (!current || current.done) {
|
|
859
|
+
clearInterval(diskPoll);
|
|
860
|
+
// Terminal wait responses return HTTP 200 with standard task JSON.
|
|
861
|
+
// Connection:close ensures socket terminates cleanly.
|
|
862
|
+
const t = current || {
|
|
863
|
+
id: taskId,
|
|
864
|
+
done: true,
|
|
865
|
+
isError: true,
|
|
866
|
+
status: "failed",
|
|
867
|
+
result: { isError: true, text: "Task file disappeared during wait." },
|
|
868
|
+
};
|
|
869
|
+
const now = Date.now();
|
|
870
|
+
const elapsed_s = Math.round(((t.finishedAt || now) - t.createdAt) / 1000);
|
|
871
|
+
const lastActivitySecAgo = t.lastActivityAt
|
|
872
|
+
? Math.max(0, Math.round((now - t.lastActivityAt) / 1000))
|
|
873
|
+
: null;
|
|
874
|
+
try {
|
|
875
|
+
res.writeHead(200, {
|
|
876
|
+
"Content-Type": "application/json",
|
|
877
|
+
"Connection": "close",
|
|
878
|
+
});
|
|
879
|
+
res.end(
|
|
880
|
+
JSON.stringify(
|
|
881
|
+
{
|
|
882
|
+
found: true,
|
|
883
|
+
id: t.id,
|
|
884
|
+
sessionId: t.sessionId,
|
|
885
|
+
cwd: t.cwd,
|
|
886
|
+
status: t.status,
|
|
887
|
+
done: t.done,
|
|
888
|
+
isError: t.isError,
|
|
889
|
+
reasoningEffort: t.reasoningEffort ?? null,
|
|
890
|
+
elapsed_s,
|
|
891
|
+
startedAt: t.startedAt,
|
|
892
|
+
lastActivitySecAgo,
|
|
893
|
+
budgetTurns: t.budgetTurns || BASE_TURN_BUDGET,
|
|
894
|
+
leaseExtensionsCount: t.leaseExtensionsCount || 0,
|
|
895
|
+
lastActivityPreview: t.lastActivityPreview || "",
|
|
896
|
+
streamBytes: t.streamBytes || 0,
|
|
897
|
+
streamTail: (t.streamTail || "")
|
|
898
|
+
.replace(/["\\{}\[\]]|type|message|content|delta|thinking|text/g, " ")
|
|
899
|
+
.replace(/\s+/g, " ")
|
|
900
|
+
.slice(-250),
|
|
901
|
+
fileOps: t.fileOps || [],
|
|
902
|
+
toolCallsCount: t.toolCallsCount || 0,
|
|
903
|
+
lastPromptTokens: t.lastPromptTokens ?? null,
|
|
904
|
+
contextHeadroom: t.contextHeadroom ?? null,
|
|
905
|
+
result: t.result || null,
|
|
906
|
+
},
|
|
907
|
+
null,
|
|
908
|
+
2
|
|
909
|
+
)
|
|
910
|
+
);
|
|
911
|
+
} catch {}
|
|
912
|
+
return;
|
|
913
|
+
}
|
|
914
|
+
if (Date.now() - waitStartTime > DEFAULT_TIMEOUT_MS) {
|
|
915
|
+
clearInterval(diskPoll);
|
|
916
|
+
try {
|
|
917
|
+
res.writeHead(504, {
|
|
918
|
+
"Content-Type": "text/markdown; charset=utf-8",
|
|
919
|
+
"Connection": "close",
|
|
920
|
+
});
|
|
921
|
+
res.end("Task wait timed out after maximum duration budget.");
|
|
922
|
+
} catch {}
|
|
923
|
+
}
|
|
924
|
+
}, 2000);
|
|
925
|
+
|
|
926
|
+
req.on("close", () => {
|
|
927
|
+
clearInterval(diskPoll);
|
|
928
|
+
});
|
|
929
|
+
return;
|
|
930
|
+
}
|
|
931
|
+
|
|
932
|
+
// GET /task/:id (Immediate JSON status check & telemetry)
|
|
933
|
+
const getMatch = pathname.match(/^\/task\/([^/]+)$/);
|
|
934
|
+
if (req.method === "GET" && getMatch) {
|
|
935
|
+
const taskId = getMatch[1];
|
|
936
|
+
let task = tasks.get(taskId);
|
|
937
|
+
if (task && !task.done && isTaskOrphaned(task)) {
|
|
938
|
+
task = markTaskOrphanedOnDisk(task);
|
|
939
|
+
}
|
|
940
|
+
if (!task) {
|
|
941
|
+
task = readTaskFromDisk(taskId);
|
|
942
|
+
}
|
|
943
|
+
if (task && task.corrupted) {
|
|
944
|
+
// Return HTTP 500 with error details for corrupted task disk state.
|
|
945
|
+
res.writeHead(500, { "Content-Type": "application/json" });
|
|
946
|
+
return res.end(
|
|
947
|
+
JSON.stringify({
|
|
948
|
+
found: true,
|
|
949
|
+
id: taskId,
|
|
950
|
+
corrupted: true,
|
|
951
|
+
file: task.file,
|
|
952
|
+
error: `Task file corrupted: ${task.error}`,
|
|
953
|
+
})
|
|
954
|
+
);
|
|
955
|
+
}
|
|
956
|
+
if (!task) {
|
|
957
|
+
res.writeHead(404, { "Content-Type": "application/json" });
|
|
958
|
+
return res.end(JSON.stringify({ found: false, id: taskId }));
|
|
959
|
+
}
|
|
960
|
+
const now = Date.now();
|
|
961
|
+
const elapsed_s = Math.round(((task.finishedAt || now) - task.createdAt) / 1000);
|
|
962
|
+
const lastActivitySecAgo = task.lastActivityAt
|
|
963
|
+
? Math.max(0, Math.round((now - task.lastActivityAt) / 1000))
|
|
964
|
+
: null;
|
|
965
|
+
res.writeHead(200, { "Content-Type": "application/json" });
|
|
966
|
+
return res.end(
|
|
967
|
+
JSON.stringify(
|
|
968
|
+
{
|
|
969
|
+
found: true,
|
|
970
|
+
id: task.id,
|
|
971
|
+
sessionId: task.sessionId,
|
|
972
|
+
cwd: task.cwd,
|
|
973
|
+
status: task.status,
|
|
974
|
+
done: task.done,
|
|
975
|
+
isError: task.isError,
|
|
976
|
+
reasoningEffort: task.reasoningEffort ?? null,
|
|
977
|
+
elapsed_s,
|
|
978
|
+
startedAt: task.startedAt,
|
|
979
|
+
lastActivitySecAgo,
|
|
980
|
+
budgetTurns: task.budgetTurns || BASE_TURN_BUDGET,
|
|
981
|
+
leaseExtensionsCount: task.leaseExtensionsCount || 0,
|
|
982
|
+
lastActivityPreview: task.lastActivityPreview || "",
|
|
983
|
+
streamBytes: task.streamBytes || 0,
|
|
984
|
+
streamTail: (task.streamTail || "")
|
|
985
|
+
.replace(/["\\{}\[\]]|type|message|content|delta|thinking|text/g, " ")
|
|
986
|
+
.replace(/\s+/g, " ")
|
|
987
|
+
.slice(-250),
|
|
988
|
+
fileOps: (task.fileOps || []).slice(-5),
|
|
989
|
+
toolCallsCount: task.toolCallsCount || 0,
|
|
990
|
+
lastPromptTokens: task.lastPromptTokens ?? null,
|
|
991
|
+
contextHeadroom: task.contextHeadroom ?? null,
|
|
992
|
+
},
|
|
993
|
+
null,
|
|
994
|
+
2
|
|
995
|
+
)
|
|
996
|
+
);
|
|
997
|
+
}
|
|
998
|
+
|
|
999
|
+
// POST /task/:id/extend_lease (Supervisor Dynamic Lease Extension)
|
|
1000
|
+
const extendMatch = pathname.match(/^\/task\/([^/]+)\/extend_lease$/);
|
|
1001
|
+
if (req.method === "POST" && extendMatch) {
|
|
1002
|
+
const taskId = extendMatch[1];
|
|
1003
|
+
let body = "";
|
|
1004
|
+
req.on("data", (chunk) => {
|
|
1005
|
+
body += chunk;
|
|
1006
|
+
});
|
|
1007
|
+
req.on("end", () => {
|
|
1008
|
+
let parsed;
|
|
1009
|
+
try {
|
|
1010
|
+
parsed = JSON.parse(body || "{}");
|
|
1011
|
+
} catch {
|
|
1012
|
+
res.writeHead(400, { "Content-Type": "application/json", Connection: "close" });
|
|
1013
|
+
res.end(JSON.stringify({ success: false, error: "Malformed JSON body" }));
|
|
1014
|
+
return;
|
|
1015
|
+
}
|
|
1016
|
+
const turns = typeof parsed.turns === "number" ? parsed.turns : 25;
|
|
1017
|
+
const resData = extendTaskBudget(taskId, turns, parsed.reason);
|
|
1018
|
+
res.writeHead(
|
|
1019
|
+
resData.success
|
|
1020
|
+
? 200
|
|
1021
|
+
: resData.error === "Task not found"
|
|
1022
|
+
? 404
|
|
1023
|
+
: 400,
|
|
1024
|
+
{
|
|
1025
|
+
"Content-Type": "application/json",
|
|
1026
|
+
Connection: "close",
|
|
1027
|
+
}
|
|
1028
|
+
);
|
|
1029
|
+
res.end(JSON.stringify(resData, null, 2));
|
|
1030
|
+
});
|
|
1031
|
+
return;
|
|
1032
|
+
}
|
|
1033
|
+
|
|
1034
|
+
// POST /task/:id/cancel
|
|
1035
|
+
const cancelMatch = pathname.match(/^\/task\/([^/]+)\/cancel$/);
|
|
1036
|
+
if ((req.method === "POST" || req.method === "DELETE") && cancelMatch) {
|
|
1037
|
+
const taskId = cancelMatch[1];
|
|
1038
|
+
const task = tasks.get(taskId);
|
|
1039
|
+
if (task) {
|
|
1040
|
+
if (!task.done) {
|
|
1041
|
+
killProcessTree(task.child, task.sessionId);
|
|
1042
|
+
// Abort in-flight task execution via AbortController.
|
|
1043
|
+
if (task.abortController) {
|
|
1044
|
+
try {
|
|
1045
|
+
task.abortController.abort();
|
|
1046
|
+
} catch {}
|
|
1047
|
+
}
|
|
1048
|
+
// Release task execution slot immediately on cancel.
|
|
1049
|
+
// (idempotent — a cancel after natural completion is a no-op).
|
|
1050
|
+
if (task.slot) {
|
|
1051
|
+
releaseTaskSlot(task.slot);
|
|
1052
|
+
task.slot = null;
|
|
1053
|
+
}
|
|
1054
|
+
task.status = "cancelled";
|
|
1055
|
+
task.done = true;
|
|
1056
|
+
task.isError = true;
|
|
1057
|
+
task.result = { isError: true, text: `Task ${taskId} cancelled by request.` };
|
|
1058
|
+
saveTaskToDisk(task);
|
|
1059
|
+
notifyWaiters(task);
|
|
1060
|
+
// E5: single-task cancel must also clear any stale slot-lease locks
|
|
1061
|
+
// (dead-owner or already-terminal leases) so a cancelled task does not
|
|
1062
|
+
// leave a zombie lease that blocks the next dispatch. Idempotent and
|
|
1063
|
+
// LIVE-OWNER safe: clearReclaimableTaskSlots never touches a live
|
|
1064
|
+
// owner's active lease.
|
|
1065
|
+
try {
|
|
1066
|
+
clearReclaimableTaskSlots();
|
|
1067
|
+
} catch (err) {
|
|
1068
|
+
const msg = err && err.message ? err.message : String(err);
|
|
1069
|
+
console.error(`[task_registry] stale slot-lease cleanup on cancel failed: ${msg}`);
|
|
1070
|
+
}
|
|
1071
|
+
}
|
|
1072
|
+
res.writeHead(200, { "Content-Type": "application/json" });
|
|
1073
|
+
return res.end(JSON.stringify({ cancelled: true, id: taskId }));
|
|
1074
|
+
}
|
|
1075
|
+
const diskTask = readTaskFromDisk(taskId);
|
|
1076
|
+
if (diskTask && diskTask.corrupted) {
|
|
1077
|
+
// Report corrupt task file on cancel attempt with HTTP 500.
|
|
1078
|
+
res.writeHead(500, { "Content-Type": "application/json" });
|
|
1079
|
+
return res.end(
|
|
1080
|
+
JSON.stringify({
|
|
1081
|
+
cancelled: false,
|
|
1082
|
+
id: taskId,
|
|
1083
|
+
corrupted: true,
|
|
1084
|
+
file: diskTask.file,
|
|
1085
|
+
error: `Task file corrupted: ${diskTask.error}`,
|
|
1086
|
+
})
|
|
1087
|
+
);
|
|
1088
|
+
}
|
|
1089
|
+
if (diskTask) {
|
|
1090
|
+
if (diskTask.sessionId) {
|
|
1091
|
+
// Perform anchored process sweep targeting exact session ID boundaries.
|
|
1092
|
+
try {
|
|
1093
|
+
killSessionProcessTreeSync(diskTask.sessionId);
|
|
1094
|
+
} catch {}
|
|
1095
|
+
}
|
|
1096
|
+
diskTask.status = "cancelled";
|
|
1097
|
+
diskTask.done = true;
|
|
1098
|
+
diskTask.isError = true;
|
|
1099
|
+
diskTask.result = { isError: true, text: `Task ${taskId} cancelled.` };
|
|
1100
|
+
saveTaskToDisk(diskTask);
|
|
1101
|
+
// E5: clear stale slot-lease locks on disk-task cancel (idempotent,
|
|
1102
|
+
// LIVE-OWNER safe — never touches a live owner's active lease).
|
|
1103
|
+
try {
|
|
1104
|
+
clearReclaimableTaskSlots();
|
|
1105
|
+
} catch (err) {
|
|
1106
|
+
const msg = err && err.message ? err.message : String(err);
|
|
1107
|
+
console.error(`[task_registry] stale slot-lease cleanup on cancel failed: ${msg}`);
|
|
1108
|
+
}
|
|
1109
|
+
res.writeHead(200, { "Content-Type": "application/json" });
|
|
1110
|
+
return res.end(JSON.stringify({ cancelled: true, id: taskId }));
|
|
1111
|
+
}
|
|
1112
|
+
res.writeHead(404, { "Content-Type": "application/json" });
|
|
1113
|
+
return res.end(JSON.stringify({ cancelled: false, error: "Not found" }));
|
|
1114
|
+
}
|
|
1115
|
+
|
|
1116
|
+
// POST /tasks/cancel or /tasks/cancel_all (Universal mass cancellation)
|
|
1117
|
+
if (
|
|
1118
|
+
(req.method === "POST" || req.method === "DELETE") &&
|
|
1119
|
+
(pathname === "/tasks/cancel" || pathname === "/tasks/cancel_all")
|
|
1120
|
+
) {
|
|
1121
|
+
cancelAllTasks("cancelled via HTTP coordinator")
|
|
1122
|
+
.then((count) => {
|
|
1123
|
+
res.writeHead(200, { "Content-Type": "application/json" });
|
|
1124
|
+
res.end(JSON.stringify({ cancelled: true, count }));
|
|
1125
|
+
})
|
|
1126
|
+
.catch((err) => {
|
|
1127
|
+
res.writeHead(500, { "Content-Type": "application/json" });
|
|
1128
|
+
res.end(JSON.stringify({ cancelled: false, error: err.message }));
|
|
1129
|
+
});
|
|
1130
|
+
return;
|
|
1131
|
+
}
|
|
1132
|
+
|
|
1133
|
+
res.writeHead(404, { "Content-Type": "text/plain" });
|
|
1134
|
+
res.end("Not found");
|
|
1135
|
+
});
|
|
1136
|
+
|
|
1137
|
+
statusHttpServer.on("error", (err) => {
|
|
1138
|
+
if (err.code === "EADDRINUSE") {
|
|
1139
|
+
statusServerOwned = false;
|
|
1140
|
+
} else {
|
|
1141
|
+
console.error("Status HTTP Server Error:", err);
|
|
1142
|
+
}
|
|
1143
|
+
});
|
|
1144
|
+
|
|
1145
|
+
/**
|
|
1146
|
+
* Status Server Keeper Re-Election Protocol.
|
|
1147
|
+
*
|
|
1148
|
+
* Coordinates ownership of STATUS_PORT across concurrent client instances:
|
|
1149
|
+
* - Exactly one process binds STATUS_PORT as keeper; other instances act as followers.
|
|
1150
|
+
* - Followers periodically probe /health to verify keeper availability and service identity.
|
|
1151
|
+
* - If the port becomes dark, followers attempt atomic listen() re-election.
|
|
1152
|
+
* - The OS kernel arbitrates port ownership: the winner becomes keeper, losers remain followers.
|
|
1153
|
+
*/
|
|
1154
|
+
|
|
1155
|
+
// The identity our /health endpoint advertises. A responder with this exact
|
|
1156
|
+
// service string is a healthy keeper of OUR service; anything else is foreign.
|
|
1157
|
+
export const STATUS_SERVICE_IDENTITY = "mcp-castor-status";
|
|
1158
|
+
|
|
1159
|
+
// How often a follower re-probes the port. ~60s keeps the test suite fast
|
|
1160
|
+
// (tests inject a shorter interval) while bounding real-world dark-port
|
|
1161
|
+
// recovery to a minute.
|
|
1162
|
+
const STATUS_ELECTION_INTERVAL_MS = 60_000;
|
|
1163
|
+
// Per-probe fetch timeout. A dark port (connection refused) fails fast; this
|
|
1164
|
+
// only bounds a half-open / black-holed socket.
|
|
1165
|
+
const STATUS_ELECTION_PROBE_TIMEOUT_MS = 2_000;
|
|
1166
|
+
|
|
1167
|
+
/**
|
|
1168
|
+
* Pure election decision. Given the raw /health body (or null when the port
|
|
1169
|
+
* is dark / the probe failed) and whether we currently own the port, decide
|
|
1170
|
+
* what to do. Exported as a pure function so it can be unit-tested without
|
|
1171
|
+
* any socket or timer.
|
|
1172
|
+
*
|
|
1173
|
+
* @param {object|null} healthBody parsed /health JSON, or null when dark.
|
|
1174
|
+
* @param {boolean} owned whether this process currently owns the port.
|
|
1175
|
+
* @returns {"stay"|"takeover"|"foreign"}
|
|
1176
|
+
* "stay" = keep the current role (owner, or our live keeper holds it);
|
|
1177
|
+
* "takeover" = the port is dark — attempt re-listen;
|
|
1178
|
+
* "foreign" = a non-Castor process holds the port — never fight it; report
|
|
1179
|
+
* loudly and stay a follower.
|
|
1180
|
+
*/
|
|
1181
|
+
export function decideElection(healthBody, owned) {
|
|
1182
|
+
// Owner path: never re-elect. The one-shot boot listen is the owner's
|
|
1183
|
+
// contract; it is left exactly as-is.
|
|
1184
|
+
if (owned) return "stay";
|
|
1185
|
+
// Follower: a dark port (null) means the keeper is gone — take over.
|
|
1186
|
+
if (healthBody === null) return "takeover";
|
|
1187
|
+
// A healthy responder with OUR identity is a live keeper — stay follower.
|
|
1188
|
+
if (healthBody && (healthBody.service === STATUS_SERVICE_IDENTITY || healthBody.service === "mcp-qwen-status")) return "stay";
|
|
1189
|
+
// A FOREIGN service occupies the port. We never fight a foreign process
|
|
1190
|
+
// for the port (same doctrine as the stream proxy's PortConflictError on
|
|
1191
|
+
// unverified alien listeners): the tick reports it loudly and stays a
|
|
1192
|
+
// follower, re-probing so we can take over the moment the port frees.
|
|
1193
|
+
return "foreign";
|
|
1194
|
+
}
|
|
1195
|
+
|
|
1196
|
+
/**
|
|
1197
|
+
* Probe the status port's /health endpoint. Returns the parsed JSON body, or
|
|
1198
|
+
* null when the port is dark (connection refused / timeout / non-2xx /
|
|
1199
|
+
* unparseable). Never throws.
|
|
1200
|
+
*
|
|
1201
|
+
* @param {number} [port] defaults to STATUS_PORT.
|
|
1202
|
+
* @param {number} [timeoutMs] defaults to STATUS_ELECTION_PROBE_TIMEOUT_MS.
|
|
1203
|
+
* @returns {Promise<object|null>}
|
|
1204
|
+
*/
|
|
1205
|
+
export async function probeStatusHealth(port = STATUS_PORT, timeoutMs = STATUS_ELECTION_PROBE_TIMEOUT_MS) {
|
|
1206
|
+
try {
|
|
1207
|
+
const res = await fetch(`http://127.0.0.1:${port}/health`, {
|
|
1208
|
+
signal: AbortSignal.timeout(timeoutMs),
|
|
1209
|
+
});
|
|
1210
|
+
if (!res.ok) return null;
|
|
1211
|
+
const text = await res.text();
|
|
1212
|
+
const body = JSON.parse(text);
|
|
1213
|
+
return body && typeof body === "object" ? body : null;
|
|
1214
|
+
} catch {
|
|
1215
|
+
// Connection refused / timeout / bad JSON: the port is dark for our
|
|
1216
|
+
// purposes. This is a real signal (keeper gone), not an error to swallow.
|
|
1217
|
+
return null;
|
|
1218
|
+
}
|
|
1219
|
+
}
|
|
1220
|
+
|
|
1221
|
+
/**
|
|
1222
|
+
* Attempt a single atomic re-listen on the status port. Returns true if this
|
|
1223
|
+
* process won the election (became keeper), false if the port is still held
|
|
1224
|
+
* by someone else (EADDRINUSE) or the listen failed for another reason.
|
|
1225
|
+
*
|
|
1226
|
+
* The listen() call is the atomic arbiter: the OS grants the port to exactly
|
|
1227
|
+
* one process, so concurrent callers cannot both win.
|
|
1228
|
+
*
|
|
1229
|
+
* @param {number} [port] defaults to STATUS_PORT.
|
|
1230
|
+
* @returns {Promise<boolean>}
|
|
1231
|
+
*/
|
|
1232
|
+
export function attemptStatusReListen(port = STATUS_PORT) {
|
|
1233
|
+
return new Promise((resolve) => {
|
|
1234
|
+
let settled = false;
|
|
1235
|
+
let onErr = null;
|
|
1236
|
+
const done = (won) => {
|
|
1237
|
+
if (settled) return;
|
|
1238
|
+
settled = true;
|
|
1239
|
+
if (onErr) statusHttpServer.removeListener("error", onErr);
|
|
1240
|
+
resolve(won);
|
|
1241
|
+
};
|
|
1242
|
+
try {
|
|
1243
|
+
onErr = (err) => {
|
|
1244
|
+
if (err && err.code === "EADDRINUSE") {
|
|
1245
|
+
done(false); // someone else owns it — remain follower
|
|
1246
|
+
} else {
|
|
1247
|
+
// A non-EADDRINUSE error: do not claim ownership. Log it (honest
|
|
1248
|
+
// signal) and stay a follower; the next tick will re-probe.
|
|
1249
|
+
console.error("[status-election] re-listen error:", err);
|
|
1250
|
+
done(false);
|
|
1251
|
+
}
|
|
1252
|
+
};
|
|
1253
|
+
statusHttpServer.once("error", onErr);
|
|
1254
|
+
statusHttpServer.listen({ port, host: "127.0.0.1", exclusive: true }, () => {
|
|
1255
|
+
statusServerOwned = true; // we won the election — we are keeper
|
|
1256
|
+
done(true);
|
|
1257
|
+
});
|
|
1258
|
+
} catch (err) {
|
|
1259
|
+
if (err && err.code === "EADDRINUSE") {
|
|
1260
|
+
done(false);
|
|
1261
|
+
} else {
|
|
1262
|
+
console.error("[status-election] re-listen threw:", err);
|
|
1263
|
+
done(false);
|
|
1264
|
+
}
|
|
1265
|
+
}
|
|
1266
|
+
});
|
|
1267
|
+
}
|
|
1268
|
+
|
|
1269
|
+
/**
|
|
1270
|
+
* Start the follower re-election loop. Only meaningful when this process is
|
|
1271
|
+
* a follower (statusServerOwned === false after the boot listen). The timer
|
|
1272
|
+
* is .unref()ed so it never keeps the process alive.
|
|
1273
|
+
*
|
|
1274
|
+
* @param {object} [opts]
|
|
1275
|
+
* @param {number} [opts.intervalMs] probe interval (default 60s).
|
|
1276
|
+
* @param {number} [opts.port] status port (default STATUS_PORT).
|
|
1277
|
+
* @returns {NodeJS.Timeout|null} the unref'd interval, or null if we are
|
|
1278
|
+
* already the owner (nothing to elect).
|
|
1279
|
+
*/
|
|
1280
|
+
export function startStatusServerElection({
|
|
1281
|
+
intervalMs = STATUS_ELECTION_INTERVAL_MS,
|
|
1282
|
+
port = STATUS_PORT,
|
|
1283
|
+
} = {}) {
|
|
1284
|
+
// Owner path: nothing to do. The one-shot boot listen already won.
|
|
1285
|
+
if (statusServerOwned) return null;
|
|
1286
|
+
|
|
1287
|
+
const tick = async () => {
|
|
1288
|
+
// If we became owner some other way (or a prior tick won), stop.
|
|
1289
|
+
if (statusServerOwned) {
|
|
1290
|
+
clearInterval(timer);
|
|
1291
|
+
return;
|
|
1292
|
+
}
|
|
1293
|
+
const body = await probeStatusHealth(port);
|
|
1294
|
+
const verdict = decideElection(body, statusServerOwned);
|
|
1295
|
+
if (verdict === "takeover") {
|
|
1296
|
+
const won = await attemptStatusReListen(port);
|
|
1297
|
+
if (won) {
|
|
1298
|
+
clearInterval(timer); // we are keeper now — stop the election loop
|
|
1299
|
+
}
|
|
1300
|
+
// If we did not win, remain a follower and let the next tick re-probe.
|
|
1301
|
+
} else if (verdict === "foreign") {
|
|
1302
|
+
// Honest, recurring signal (follower-only, unref'd timer): a non-Castor
|
|
1303
|
+
// process holds the status port. We do not fight it; each tick says so
|
|
1304
|
+
// until the situation resolves.
|
|
1305
|
+
console.error(
|
|
1306
|
+
`[status-election] foreign service on 127.0.0.1:${port} (service=${body && body.service}): not ours; staying follower`
|
|
1307
|
+
);
|
|
1308
|
+
}
|
|
1309
|
+
};
|
|
1310
|
+
|
|
1311
|
+
const timer = setInterval(() => {
|
|
1312
|
+
// Fire-and-forget each tick; a slow probe must not block the next.
|
|
1313
|
+
tick().catch((err) => {
|
|
1314
|
+
// A probe/listen failure is a real signal, not a crash. Log it and
|
|
1315
|
+
// keep the election alive on the next tick.
|
|
1316
|
+
console.error("[status-election] tick error:", err);
|
|
1317
|
+
});
|
|
1318
|
+
}, intervalMs);
|
|
1319
|
+
timer.unref(); // never hold the process open
|
|
1320
|
+
return timer;
|
|
1321
|
+
}
|
|
1322
|
+
|
|
1323
|
+
export function initStatusServer() {
|
|
1324
|
+
// Disable Node.js server request and socket timeouts for long-poll blocking waits
|
|
1325
|
+
statusHttpServer.requestTimeout = 0;
|
|
1326
|
+
statusHttpServer.headersTimeout = 0;
|
|
1327
|
+
statusHttpServer.keepAliveTimeout = 0;
|
|
1328
|
+
statusHttpServer.timeout = 0;
|
|
1329
|
+
const onBootError = (err) => {
|
|
1330
|
+
if (err && err.code === "EADDRINUSE") {
|
|
1331
|
+
statusServerOwned = false;
|
|
1332
|
+
} else {
|
|
1333
|
+
console.error("[status-server] boot listen error:", err);
|
|
1334
|
+
statusServerOwned = false;
|
|
1335
|
+
}
|
|
1336
|
+
startStatusServerElection();
|
|
1337
|
+
};
|
|
1338
|
+
|
|
1339
|
+
statusHttpServer.once("error", onBootError);
|
|
1340
|
+
|
|
1341
|
+
try {
|
|
1342
|
+
statusHttpServer.listen({ port: STATUS_PORT, host: "127.0.0.1", exclusive: true }, () => {
|
|
1343
|
+
statusHttpServer.removeListener("error", onBootError);
|
|
1344
|
+
statusServerOwned = true;
|
|
1345
|
+
});
|
|
1346
|
+
} catch (err) {
|
|
1347
|
+
statusHttpServer.removeListener("error", onBootError);
|
|
1348
|
+
if (err.code === "EADDRINUSE") {
|
|
1349
|
+
statusServerOwned = false;
|
|
1350
|
+
} else {
|
|
1351
|
+
console.error("[status-server] boot listen threw:", err);
|
|
1352
|
+
statusServerOwned = false;
|
|
1353
|
+
}
|
|
1354
|
+
startStatusServerElection();
|
|
1355
|
+
}
|
|
1356
|
+
setInterval(cleanOldTasks, 300_000).unref();
|
|
1357
|
+
}
|