@mmerterden/multi-agent-pipeline 17.4.0 → 17.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +136 -0
- package/README.md +23 -5
- package/README.tr.md +23 -5
- package/docs/adr/0013-lsp-code-intelligence.md +102 -0
- package/docs/adr/README.md +1 -0
- package/docs/token-budget-history.md +1 -1
- package/install/templates/copilot-instructions.md +9 -3
- package/package.json +1 -1
- package/pipeline/commands/multi-agent/analysis/SKILL.md +3 -3
- package/pipeline/commands/multi-agent/autopilot/SKILL.md +3 -3
- package/pipeline/commands/multi-agent/autopilot-off/SKILL.md +5 -3
- package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/local/SKILL.md +17 -6
- package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +3 -3
- package/pipeline/lib/multi-repo-pipeline.sh +26 -0
- package/pipeline/multi-agent-refs/features/base-branch-evidence.md +222 -0
- package/pipeline/multi-agent-refs/features/code-intelligence.md +80 -0
- package/pipeline/multi-agent-refs/phases/modes.md +23 -3
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +96 -71
- package/pipeline/multi-agent-refs/phases/phase-7-report.md +1 -1
- package/pipeline/multi-agent-refs/phases.md +7 -2
- package/pipeline/multi-agent-refs/picker-contract.md +37 -5
- package/pipeline/multi-agent-refs/tracker-contract.md +25 -14
- package/pipeline/schemas/agent-state.schema.json +88 -4
- package/pipeline/schemas/prefs.schema.json +22 -0
- package/pipeline/schemas/token-budget.json +2 -2
- package/pipeline/scripts/autopilot-runner.mjs +292 -45
- package/pipeline/scripts/base-branch-candidates.mjs +599 -0
- package/pipeline/scripts/gc-abandoned.sh +5 -3
- package/pipeline/scripts/gen-mode-dispatch.mjs +39 -16
- package/pipeline/scripts/phase-tracker.sh +39 -2
- package/pipeline/scripts/phase0-exit-gate.mjs +128 -0
- package/pipeline/scripts/verify-citations.mjs +84 -2
- package/pipeline/skills/.skill-manifest.json +2 -2
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +1 -1
|
@@ -969,6 +969,28 @@
|
|
|
969
969
|
}
|
|
970
970
|
}
|
|
971
971
|
},
|
|
972
|
+
"baseBranchEvidence": {
|
|
973
|
+
"type": "object",
|
|
974
|
+
"additionalProperties": false,
|
|
975
|
+
"description": "v17.5.0+ - Phase 0 Step 3 base-branch evidence collection (refs/features/base-branch-evidence.md). Candidates are collected with the evidence behind them (issue version field, linked release issue, the branch convention learned from the refs that exist, recent branches, repo default), ranked, and shown. Interactive runs always ask; the evidence only reorders the rows.",
|
|
976
|
+
"properties": {
|
|
977
|
+
"enabled": {
|
|
978
|
+
"type": "boolean",
|
|
979
|
+
"default": true,
|
|
980
|
+
"description": "Collect issue-derived candidates at all. Off falls back to the develop/release/main sort order, which is still surfaced through the picker."
|
|
981
|
+
},
|
|
982
|
+
"preferLinkedRelease": {
|
|
983
|
+
"type": "boolean",
|
|
984
|
+
"default": false,
|
|
985
|
+
"description": "On boards where opening a development sub-task requires selecting the related release issue, that link is the authoritative base-branch answer and the version field is corroboration. Raises the linked-release weight above the version-field one rather than adding a second rule."
|
|
986
|
+
},
|
|
987
|
+
"autopilotAsksOnIssue": {
|
|
988
|
+
"type": "boolean",
|
|
989
|
+
"default": false,
|
|
990
|
+
"description": "OFF by default, and it is an outward-facing write. When on, an autopilot run whose base-branch derivation is ambiguous posts ONE comment on the Jira or GitHub issue asking which branch to develop from, then halts on circuit-breaker trigger 6 and waits for resume. A question, never a state change: no transition, no resolution, no assignee, no close. The body uses Ref:, never Closes:/Fixes:/Resolves:, and its human-facing copy follows outputLanguage. Autopilot never posts the question and then answers it itself."
|
|
991
|
+
}
|
|
992
|
+
}
|
|
993
|
+
},
|
|
972
994
|
"autopilotCircuitBreaker": {
|
|
973
995
|
"type": "object",
|
|
974
996
|
"additionalProperties": false,
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
"description": "Per-phase token ceilings for the lazy-loaded pipeline docs, enforced by smoke-token-budget.sh. Only the ACTIVE phase is loaded at run time and nothing truncates a phase document, so these numbers govern what may be written, not what a run receives. Two rules, and the split is the point. `max_tokens` and `total_max_tokens` are committed constants a human owns: computing them would end the gate, because a ceiling that is always current x k can never fail. The warn tier is NOT stored - the gate derives it per phase from that phase's own git history (median + 3*MAD of its historical per-commit deltas, which is robust to the one 980-token release that makes sigma meaningless on phase-4), because a soft line maintained by hand rots, and this one rotted twice: it was reset at v13.6.0 when five lines had gone permanently amber, and six of eight were amber again by v17.1.0. A ceiling more than 25% above the measurement is stale and fails, which is the 'only ratchets down' rule the SKILL.md grace list always had and this budget never did. Change history: docs/token-budget-history.md.",
|
|
5
5
|
"phases": {
|
|
6
6
|
"phase-0-init": {
|
|
7
|
-
"max_tokens":
|
|
7
|
+
"max_tokens": 13400
|
|
8
8
|
},
|
|
9
9
|
"phase-1-analysis": {
|
|
10
10
|
"max_tokens": 4600
|
|
@@ -28,5 +28,5 @@
|
|
|
28
28
|
"max_tokens": 6350
|
|
29
29
|
}
|
|
30
30
|
},
|
|
31
|
-
"total_max_tokens":
|
|
31
|
+
"total_max_tokens": 63000
|
|
32
32
|
}
|
|
@@ -12,7 +12,8 @@
|
|
|
12
12
|
* 2. INTAKE autopilot-intake.mjs writes the ordered queue. Read-only.
|
|
13
13
|
* 3. ARM autopilot-arming.mjs answers spend and credentials.
|
|
14
14
|
* 4. TAKE the head of the queue, and only if this repo is free.
|
|
15
|
-
* 5. RUN one `claude --bg` child,
|
|
15
|
+
* 5. RUN one `claude --bg` child, then SUPERVISED to the end. `--bg`
|
|
16
|
+
* returns immediately, so the spawn proves only that a run began.
|
|
16
17
|
* 6. RECORD attempted.jsonl, clear the pid file, refresh status.json.
|
|
17
18
|
*
|
|
18
19
|
* LIVENESS WITHOUT A LEASE. Lease arithmetic exists to arbitrate between rival
|
|
@@ -26,9 +27,9 @@
|
|
|
26
27
|
* runner was live and do nothing, forever.
|
|
27
28
|
*
|
|
28
29
|
* WHAT IT WILL NOT DO. It does not merge. It does not remove a worktree holding
|
|
29
|
-
* uncommitted work - that is stashed
|
|
30
|
-
* losing a day of edits is worse than 750 MB. It does not remove
|
|
31
|
-
* cannot prove it created: the measured tree here had 19 worktrees with no state
|
|
30
|
+
* uncommitted work - that is stashed under a named message and the worktree
|
|
31
|
+
* kept, because losing a day of edits is worse than 750 MB. It does not remove
|
|
32
|
+
* a worktree it cannot prove it created: the measured tree here had 19 worktrees with no state
|
|
32
33
|
* file, several of them hand-made and one in active use.
|
|
33
34
|
*
|
|
34
35
|
* Usage:
|
|
@@ -48,6 +49,8 @@ import {
|
|
|
48
49
|
mkdirSync,
|
|
49
50
|
rmSync,
|
|
50
51
|
chmodSync,
|
|
52
|
+
readdirSync,
|
|
53
|
+
statSync,
|
|
51
54
|
} from "node:fs";
|
|
52
55
|
import { join } from "node:path";
|
|
53
56
|
import { homedir } from "node:os";
|
|
@@ -62,6 +65,17 @@ const ROOT = process.env.MA_AUTOPILOT_ROOT || join(homedir(), ".claude", "autopi
|
|
|
62
65
|
const SCRIPTS = process.env.MA_AP_SCRIPTS || import.meta.dirname;
|
|
63
66
|
const CLAUDE_BIN = process.env.MA_AP_CLAUDE_BIN || "claude";
|
|
64
67
|
const DRY = process.argv.includes("--dry-run");
|
|
68
|
+
const POLL_MS = Number(process.env.MA_AP_POLL_MS || 15000);
|
|
69
|
+
// A run that has not finished in this long is not going to, and holding the
|
|
70
|
+
// queue behind it is worse than recording it as stuck. launchd restarts the
|
|
71
|
+
// tick, so nothing is lost by returning.
|
|
72
|
+
const MAX_RUN_MS = Number(process.env.MA_AP_MAX_RUN_MS || 6 * 60 * 60 * 1000);
|
|
73
|
+
// `claude agents --json` lists a session a moment AFTER `--bg` returns, so an
|
|
74
|
+
// absence in the first few seconds is not evidence of anything.
|
|
75
|
+
const REGISTER_GRACE_MS = Number(process.env.MA_AP_GRACE_MS || 60000);
|
|
76
|
+
// A session in one of these is still working. Anything else - and a row with
|
|
77
|
+
// no status at all is not "anything else" - ends supervision.
|
|
78
|
+
const LIVE_STATUSES = new Set(["busy", "running", "waiting"]);
|
|
65
79
|
|
|
66
80
|
const log = (m) => process.stdout.write(`autopilot-runner: ${m}\n`);
|
|
67
81
|
|
|
@@ -122,17 +136,32 @@ function pidAlive(pid) {
|
|
|
122
136
|
}
|
|
123
137
|
}
|
|
124
138
|
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
139
|
+
/** Block this thread. The tick IS the supervisor, so it has nothing else to do. */
|
|
140
|
+
function sleepSync(ms) {
|
|
141
|
+
Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* The session's row in `claude agents --json`.
|
|
146
|
+
*
|
|
147
|
+
* Three answers, not two, and the third is the one that matters: `undefined`
|
|
148
|
+
* means the list could not be read, which is NOT the same as the session being
|
|
149
|
+
* gone. Collapsing them is how a live run gets retired because one command's
|
|
150
|
+
* output changed shape.
|
|
151
|
+
*
|
|
152
|
+
* @returns {object|null|undefined} the row, null when listed-and-absent,
|
|
153
|
+
* undefined when the list itself could not be read
|
|
154
|
+
*/
|
|
155
|
+
export function agentRow(sessionId, agentsJson) {
|
|
156
|
+
if (!sessionId) return null;
|
|
157
|
+
const out = agentsJson ?? process.env.MA_AP_AGENTS_JSON ?? run(CLAUDE_BIN, ["agents", "--json"]);
|
|
158
|
+
if (!out || !out.trim()) return undefined;
|
|
128
159
|
try {
|
|
129
|
-
const rows = JSON.parse(out
|
|
160
|
+
const rows = JSON.parse(out);
|
|
130
161
|
const list = Array.isArray(rows) ? rows : rows.agents || [];
|
|
131
|
-
return list.
|
|
162
|
+
return list.find((a) => a.sessionId === sessionId) || null;
|
|
132
163
|
} catch {
|
|
133
|
-
|
|
134
|
-
// that stops forever because one command's output changed shape.
|
|
135
|
-
return false;
|
|
164
|
+
return undefined;
|
|
136
165
|
}
|
|
137
166
|
}
|
|
138
167
|
|
|
@@ -143,20 +172,55 @@ function sessionLive(sessionId) {
|
|
|
143
172
|
* restart: a pid recorded before the current boot is stale by definition and
|
|
144
173
|
* needs no probe at all.
|
|
145
174
|
*/
|
|
146
|
-
export function
|
|
147
|
-
if (!holder || !holder.pid) return
|
|
148
|
-
if (holder.bootTime && now && holder.bootTime !== now) return
|
|
149
|
-
if (!pidAlive(holder.pid)) return
|
|
150
|
-
|
|
175
|
+
export function holderLiveness(holder, { now = bootTime(), agentsJson } = {}) {
|
|
176
|
+
if (!holder || !holder.pid) return "dead";
|
|
177
|
+
if (holder.bootTime && now && holder.bootTime !== now) return "dead";
|
|
178
|
+
if (!pidAlive(holder.pid)) return "dead";
|
|
179
|
+
const row = agentRow(holder.sessionId, agentsJson);
|
|
180
|
+
if (row === undefined) return "unknown";
|
|
181
|
+
return row ? "live" : "dead";
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
export function holderIsLive(holder, opts = {}) {
|
|
185
|
+
return holderLiveness(holder, opts) === "live";
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* The run's own state file, which is what actually knows the worktree.
|
|
190
|
+
*
|
|
191
|
+
* The runner does not create the worktree - Phase 0 of the run does - and the
|
|
192
|
+
* runner's `taskId` (`ap-<epoch>`) is not the run's task id. So the link is made
|
|
193
|
+
* from disk: a state file whose `worktreePath` sits under this repo and which
|
|
194
|
+
* was written after this claim started. `gc-abandoned.sh` builds the same index
|
|
195
|
+
* for the same reason.
|
|
196
|
+
*
|
|
197
|
+
* Returns null rather than guessing. Every field it fills is one `retire()` can
|
|
198
|
+
* act on; without them retire still records the attempt, it just cannot clean
|
|
199
|
+
* anything up.
|
|
200
|
+
*/
|
|
201
|
+
export function findRunState(repoPath, startedAtSec, logsRoot) {
|
|
202
|
+
const root = logsRoot || join(homedir(), ".claude", "logs", "multi-agent");
|
|
203
|
+
if (!repoPath || !existsSync(root)) return null;
|
|
204
|
+
let best = null;
|
|
205
|
+
for (const dir of readdirSync(root, { withFileTypes: true })) {
|
|
206
|
+
if (!dir.isDirectory()) continue;
|
|
207
|
+
const p = join(root, dir.name, "agent-state.json");
|
|
208
|
+
if (!existsSync(p)) continue;
|
|
209
|
+
let st;
|
|
151
210
|
try {
|
|
152
|
-
|
|
153
|
-
const list = Array.isArray(rows) ? rows : rows.agents || [];
|
|
154
|
-
return list.some((a) => a.sessionId === holder.sessionId);
|
|
211
|
+
st = statSync(p);
|
|
155
212
|
} catch {
|
|
156
|
-
|
|
213
|
+
continue;
|
|
157
214
|
}
|
|
215
|
+
// A second of slack: the claim is stamped before the child is spawned.
|
|
216
|
+
if (st.mtimeMs < (startedAtSec - 1) * 1000) continue;
|
|
217
|
+
const doc = readJson(p, null);
|
|
218
|
+
const wt = doc && doc.worktreePath;
|
|
219
|
+
if (!wt || !(wt === repoPath || wt.startsWith(repoPath + "/"))) continue;
|
|
220
|
+
if (!best || st.mtimeMs > best.mtimeMs)
|
|
221
|
+
best = { statePath: p, worktree: wt, mtimeMs: st.mtimeMs };
|
|
158
222
|
}
|
|
159
|
-
return
|
|
223
|
+
return best;
|
|
160
224
|
}
|
|
161
225
|
|
|
162
226
|
/**
|
|
@@ -172,17 +236,31 @@ function retire(holder) {
|
|
|
172
236
|
if (wt && existsSync(wt)) {
|
|
173
237
|
const dirty = run("git", ["-C", wt, "status", "--porcelain"]).trim();
|
|
174
238
|
if (dirty) {
|
|
175
|
-
// A day of edits is worth more than 750 MB. The
|
|
176
|
-
//
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
239
|
+
// A day of edits is worth more than 750 MB. The stash MESSAGE carries the
|
|
240
|
+
// task, so `git stash list` finds it without remembering this ran.
|
|
241
|
+
//
|
|
242
|
+
// No branch is created. A branch made after `stash push` points at HEAD
|
|
243
|
+
// and therefore contains none of the stashed work - verified: the branch's
|
|
244
|
+
// tree held the ORIGINAL file and no untracked file at all. Branching at
|
|
245
|
+
// `refs/stash` is no better, because with `-u` the untracked files live in
|
|
246
|
+
// the stash commit's third parent, not in its tree. One honest reference
|
|
247
|
+
// beats two, and this is the contract gc-abandoned.sh already keeps.
|
|
248
|
+
const label = `autopilot/abandoned/${holder.taskId || item.id || "unknown"}`;
|
|
249
|
+
run("git", ["-C", wt, "stash", "push", "-u", "-m", label]);
|
|
250
|
+
stashed = label;
|
|
251
|
+
log(`kept ${wt}: uncommitted work stashed as "${label}" (git stash list)`);
|
|
182
252
|
} else if (holder.createdWorktree) {
|
|
183
|
-
|
|
184
|
-
run
|
|
185
|
-
|
|
253
|
+
// A clean worktree is only safe to remove once nothing is writing to it.
|
|
254
|
+
// Supervision can end while the run does not - a timeout, or an agent
|
|
255
|
+
// list that could not be read - and `worktree remove --force` under a
|
|
256
|
+
// live session is the one mistake here with nothing to undo it.
|
|
257
|
+
if (agentRow(holder.sessionId)) {
|
|
258
|
+
log(`kept ${wt}: session ${holder.sessionId} is still listed`);
|
|
259
|
+
} else {
|
|
260
|
+
run("git", ["-C", wt, "worktree", "remove", "--force", wt]);
|
|
261
|
+
run("git", ["-C", holder.repoPath || wt, "worktree", "prune"]);
|
|
262
|
+
log(`removed ${wt}`);
|
|
263
|
+
}
|
|
186
264
|
}
|
|
187
265
|
}
|
|
188
266
|
|
|
@@ -209,20 +287,156 @@ function refreshStatus() {
|
|
|
209
287
|
if (existsSync(s)) run("bash", [s, "--write"]);
|
|
210
288
|
}
|
|
211
289
|
|
|
290
|
+
/**
|
|
291
|
+
* Put the slot back and drop the claim. One place, so every exit agrees.
|
|
292
|
+
*
|
|
293
|
+
* It re-reads queue.json rather than trusting a copy, because intake runs
|
|
294
|
+
* between the read and here. The recovery path calls this too: `retire()` used
|
|
295
|
+
* to record the outcome and delete the claim while leaving the item in
|
|
296
|
+
* `running`, and nothing else on this machine ever removes one - intake copies
|
|
297
|
+
* the list forward verbatim. One crashed runner was enough to wedge the queue
|
|
298
|
+
* at "1/1 slots in use" on every tick from then on.
|
|
299
|
+
*/
|
|
300
|
+
function releaseSlot(id, pidPath, queue = { queued: [], running: [] }) {
|
|
301
|
+
const after = readJson(join(ROOT, "queue.json"), queue);
|
|
302
|
+
writeState("queue.json", {
|
|
303
|
+
...after,
|
|
304
|
+
running: (after.running || []).filter((r) => r.id !== id),
|
|
305
|
+
});
|
|
306
|
+
rmSync(pidPath, { force: true });
|
|
307
|
+
refreshStatus();
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
/** The PR a run opened, wherever its state file recorded it. */
|
|
311
|
+
export function prFromState(state) {
|
|
312
|
+
const pr = state && state.pr;
|
|
313
|
+
if (!pr) return null;
|
|
314
|
+
if (typeof pr === "string") return pr.startsWith("http") ? pr : null;
|
|
315
|
+
if (typeof pr === "object") return pr.url || null;
|
|
316
|
+
return null;
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
/**
|
|
320
|
+
* What to call this attempt.
|
|
321
|
+
*
|
|
322
|
+
* The run's own state is the authority; the supervisor only says why waiting
|
|
323
|
+
* stopped. "finished with no PR" is deliberately not "failed": a run can stop
|
|
324
|
+
* legitimately without opening one, and calling that a failure would put a
|
|
325
|
+
* healthy item into the retry path.
|
|
326
|
+
*/
|
|
327
|
+
export function outcomeFor(supervised, state, prUrl) {
|
|
328
|
+
if (supervised.reason === "timeout") return "timed-out";
|
|
329
|
+
if (supervised.reason === "unknown") return "unknown";
|
|
330
|
+
if (supervised.reason === "needs-input") return "needs-input";
|
|
331
|
+
if (prUrl) return "pr-opened";
|
|
332
|
+
const status = state && state.status;
|
|
333
|
+
if (status === "complete" || status === "completed") return "completed-no-pr";
|
|
334
|
+
if (status === "failed") return "failed";
|
|
335
|
+
return state ? `stopped-${status || "unknown"}` : "no-state";
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
/**
|
|
339
|
+
* Wait for the background run, and learn what it made while waiting.
|
|
340
|
+
*
|
|
341
|
+
* Polling `claude agents --json` rather than waiting on the child, because the
|
|
342
|
+
* child is gone: `--bg` returns immediately. Three things end the wait - the
|
|
343
|
+
* session leaves the list, it reports a status that is not live, or the ceiling
|
|
344
|
+
* is reached. An unreadable list ends it too, but says "unknown" rather than
|
|
345
|
+
* inventing an outcome.
|
|
346
|
+
*
|
|
347
|
+
* The claim is enriched on the way: the first poll that finds the run's state
|
|
348
|
+
* file writes `worktree` and `statePath` into runner.pid, which is what lets a
|
|
349
|
+
* LATER tick's retire() actually clean up. Those fields were read by retire()
|
|
350
|
+
* from the day it was written and never once set by anything.
|
|
351
|
+
*/
|
|
352
|
+
function supervise(sessionId, claim, item) {
|
|
353
|
+
const startedMs = Date.now();
|
|
354
|
+
const deadline = startedMs + MAX_RUN_MS;
|
|
355
|
+
let sawLive = false;
|
|
356
|
+
let sawBusy = false;
|
|
357
|
+
let reason;
|
|
358
|
+
|
|
359
|
+
for (;;) {
|
|
360
|
+
const row = agentRow(sessionId);
|
|
361
|
+
|
|
362
|
+
if (row === undefined) {
|
|
363
|
+
reason = "unknown";
|
|
364
|
+
break;
|
|
365
|
+
}
|
|
366
|
+
if (row) {
|
|
367
|
+
sawLive = true;
|
|
368
|
+
if (!claim.statePath) {
|
|
369
|
+
const found = findRunState(claim.repoPath, claim.startedAt);
|
|
370
|
+
if (found) {
|
|
371
|
+
claim.statePath = found.statePath;
|
|
372
|
+
claim.worktree = found.worktree;
|
|
373
|
+
// Phase 0 opened it, not this runner - but this runner is the only
|
|
374
|
+
// thing that will be around to remove it.
|
|
375
|
+
claim.createdWorktree = true;
|
|
376
|
+
writeState("runner.pid", claim);
|
|
377
|
+
log(`tracking ${found.worktree}`);
|
|
378
|
+
}
|
|
379
|
+
}
|
|
380
|
+
if (row.status === "busy" || row.status === "running") sawBusy = true;
|
|
381
|
+
if (row.status === "waiting") {
|
|
382
|
+
// Only terminal once the run has actually been working. A session can
|
|
383
|
+
// read as "waiting" in the second after launch, and ending supervision
|
|
384
|
+
// there would abandon every run at birth.
|
|
385
|
+
if (sawBusy) {
|
|
386
|
+
reason = "needs-input";
|
|
387
|
+
break;
|
|
388
|
+
}
|
|
389
|
+
} else if (row.status && !LIVE_STATUSES.has(row.status)) {
|
|
390
|
+
reason = "finished";
|
|
391
|
+
break;
|
|
392
|
+
}
|
|
393
|
+
} else if (sawLive || Date.now() - startedMs > REGISTER_GRACE_MS) {
|
|
394
|
+
reason = "gone";
|
|
395
|
+
break;
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
if (Date.now() >= deadline) {
|
|
399
|
+
reason = "timeout";
|
|
400
|
+
break;
|
|
401
|
+
}
|
|
402
|
+
sleepSync(POLL_MS);
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
if (!claim.statePath) {
|
|
406
|
+
const found = findRunState(claim.repoPath, claim.startedAt);
|
|
407
|
+
if (found) {
|
|
408
|
+
claim.statePath = found.statePath;
|
|
409
|
+
claim.worktree = found.worktree;
|
|
410
|
+
claim.createdWorktree = true;
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
const waitedSec = Math.round((Date.now() - startedMs) / 1000);
|
|
414
|
+
log(`${item.id}: supervision ended (${reason}) after ${waitedSec}s`);
|
|
415
|
+
return { reason, waitedSec };
|
|
416
|
+
}
|
|
417
|
+
|
|
212
418
|
function main() {
|
|
213
419
|
const pidPath = join(ROOT, "runner.pid");
|
|
214
420
|
|
|
215
421
|
// ---- 1. RECOVER --------------------------------------------------------
|
|
216
422
|
const holder = readJson(pidPath, null);
|
|
217
423
|
if (holder) {
|
|
218
|
-
|
|
424
|
+
const liveness = holderLiveness(holder);
|
|
425
|
+
if (liveness === "live") {
|
|
219
426
|
log(`another runner is live (pid ${holder.pid}) - nothing to do`);
|
|
220
427
|
return 0;
|
|
221
428
|
}
|
|
429
|
+
if (liveness === "unknown") {
|
|
430
|
+
// The pid answers but the agent list could not be read. "I cannot tell"
|
|
431
|
+
// is not "it died": retiring here deletes a running run's claim and
|
|
432
|
+
// records an outcome that never happened.
|
|
433
|
+
log(`cannot read the agent list - leaving pid ${holder.pid} alone this tick`);
|
|
434
|
+
return 0;
|
|
435
|
+
}
|
|
222
436
|
log(`previous runner is gone (pid ${holder.pid}) - retiring its item`);
|
|
223
437
|
if (!DRY) {
|
|
224
438
|
retire(holder);
|
|
225
|
-
|
|
439
|
+
releaseSlot(holder.item && holder.item.id, pidPath);
|
|
226
440
|
}
|
|
227
441
|
}
|
|
228
442
|
|
|
@@ -332,20 +546,53 @@ function main() {
|
|
|
332
546
|
],
|
|
333
547
|
{ cwd: next.localPath || process.cwd(), encoding: "utf-8", stdio: ["ignore", "pipe", "pipe"] },
|
|
334
548
|
);
|
|
549
|
+
if (child.status !== 0) {
|
|
550
|
+
const why = (child.stderr || child.error?.message || "").trim().slice(0, 300);
|
|
551
|
+
record({ source: next.source, id: next.id, taskId, outcome: "launch-failed", note: why });
|
|
552
|
+
releaseSlot(next.id, pidPath, queue);
|
|
553
|
+
log(`${next.id}: launch-failed ${why}`);
|
|
554
|
+
return 0;
|
|
555
|
+
}
|
|
335
556
|
|
|
336
|
-
// ----
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
557
|
+
// ---- 5b. SUPERVISE -----------------------------------------------------
|
|
558
|
+
// `claude --bg` returns as soon as the session is started - `claude --help`
|
|
559
|
+
// says so in as many words. The spawn above therefore proves only that a run
|
|
560
|
+
// BEGAN. Everything below used to run immediately after it: the outcome was
|
|
561
|
+
// recorded as "pr-opened" before any work happened, the slot was freed within
|
|
562
|
+
// a second, and the per-repo guard three steps up became decorative because
|
|
563
|
+
// `running` was already empty when the next tick read it.
|
|
564
|
+
const supervised = supervise(sessionId, claim, next);
|
|
340
565
|
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
566
|
+
// ---- 6. RECORD ---------------------------------------------------------
|
|
567
|
+
// From the run's own state file, not from the child's exit code and not from
|
|
568
|
+
// its stdout - with `--bg` the stdout is a session id, so the PR regex that
|
|
569
|
+
// used to read it could never match.
|
|
570
|
+
const finalState = claim.statePath ? readJson(claim.statePath, null) : null;
|
|
571
|
+
const prUrl = prFromState(finalState);
|
|
572
|
+
const outcome = outcomeFor(supervised, finalState, prUrl);
|
|
573
|
+
record({
|
|
574
|
+
source: next.source,
|
|
575
|
+
id: next.id,
|
|
576
|
+
taskId,
|
|
577
|
+
outcome,
|
|
578
|
+
prUrl,
|
|
579
|
+
waitedSec: supervised.waitedSec,
|
|
580
|
+
usd: 0,
|
|
345
581
|
});
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
582
|
+
|
|
583
|
+
// "gone" and "finished" are the two reasons that mean the run ENDED. The
|
|
584
|
+
// other three - a ceiling, an unreadable list, a session parked on a question
|
|
585
|
+
// - say only that waiting stopped, and the run may still hold its worktree.
|
|
586
|
+
// Deleting the claim there would strand it: the worktree and state path this
|
|
587
|
+
// tick just learned are the only record, and the next tick would take another
|
|
588
|
+
// item in the same repo. Keeping it means the next tick's RECOVER decides,
|
|
589
|
+
// with retire() free to clean up and releaseSlot to free the slot.
|
|
590
|
+
if (supervised.reason === "gone" || supervised.reason === "finished") {
|
|
591
|
+
releaseSlot(next.id, pidPath, queue);
|
|
592
|
+
} else {
|
|
593
|
+
log(`${next.id}: claim kept for the next tick (${supervised.reason})`);
|
|
594
|
+
}
|
|
595
|
+
log(`${next.id}: ${outcome}${prUrl ? ` ${prUrl}` : ""} (${supervised.waitedSec}s)`);
|
|
349
596
|
return 0;
|
|
350
597
|
}
|
|
351
598
|
|