dorfl 0.11.0 → 0.11.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/arbiter-refs.d.ts +139 -0
- package/dist/arbiter-refs.d.ts.map +1 -0
- package/dist/arbiter-refs.js +114 -0
- package/dist/arbiter-refs.js.map +1 -0
- package/dist/cli.d.ts.map +1 -1
- package/dist/cli.js +32 -18
- package/dist/cli.js.map +1 -1
- package/dist/config.d.ts +73 -12
- package/dist/config.d.ts.map +1 -1
- package/dist/config.js +28 -0
- package/dist/config.js.map +1 -1
- package/dist/do-config.d.ts +21 -1
- package/dist/do-config.d.ts.map +1 -1
- package/dist/do-config.js +19 -0
- package/dist/do-config.js.map +1 -1
- package/dist/do.d.ts +14 -2
- package/dist/do.d.ts.map +1 -1
- package/dist/do.js +108 -14
- package/dist/do.js.map +1 -1
- package/dist/env-config.d.ts.map +1 -1
- package/dist/env-config.js +3 -0
- package/dist/env-config.js.map +1 -1
- package/dist/harness.d.ts +29 -0
- package/dist/harness.d.ts.map +1 -1
- package/dist/harness.js.map +1 -1
- package/dist/ledger-write.d.ts.map +1 -1
- package/dist/ledger-write.js +38 -3
- package/dist/ledger-write.js.map +1 -1
- package/dist/needs-attention.d.ts +43 -0
- package/dist/needs-attention.d.ts.map +1 -1
- package/dist/needs-attention.js +211 -50
- package/dist/needs-attention.js.map +1 -1
- package/dist/pi-harness.d.ts.map +1 -1
- package/dist/pi-harness.js +109 -13
- package/dist/pi-harness.js.map +1 -1
- package/dist/protocol/WORK-CONTRACT.md +11 -2
- package/dist/protocol/spec-template.md +1 -1
- package/dist/protocol/task-template.md +1 -1
- package/dist/reap-agent-tree.d.ts +108 -0
- package/dist/reap-agent-tree.d.ts.map +1 -0
- package/dist/reap-agent-tree.js +173 -0
- package/dist/reap-agent-tree.js.map +1 -0
- package/dist/repo-config.d.ts +1 -1
- package/dist/repo-config.d.ts.map +1 -1
- package/dist/repo-config.js +19 -2
- package/dist/repo-config.js.map +1 -1
- package/dist/run.d.ts.map +1 -1
- package/dist/run.js +4 -3
- package/dist/run.js.map +1 -1
- package/dist/skills/drive-tasks/SKILL.md +20 -2
- package/dist/skills/setup/protocol/WORK-CONTRACT.md +11 -2
- package/dist/skills/setup/protocol/spec-template.md +1 -1
- package/dist/skills/setup/protocol/task-template.md +1 -1
- package/dist/worktree-writer-lock.d.ts +99 -0
- package/dist/worktree-writer-lock.d.ts.map +1 -0
- package/dist/worktree-writer-lock.js +158 -0
- package/dist/worktree-writer-lock.js.map +1 -0
- package/package.json +1 -1
- package/src/arbiter-refs.ts +222 -0
- package/src/cli.ts +57 -17
- package/src/config.ts +90 -12
- package/src/do-config.ts +38 -0
- package/src/do.ts +158 -19
- package/src/env-config.ts +3 -0
- package/src/harness.ts +30 -0
- package/src/ledger-write.ts +41 -5
- package/src/needs-attention.ts +282 -59
- package/src/pi-harness.ts +109 -15
- package/src/reap-agent-tree.ts +221 -0
- package/src/repo-config.ts +19 -1
- package/src/run.ts +4 -3
- package/src/worktree-writer-lock.ts +217 -0
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* **Reap a stopped agent's whole PROCESS TREE, and VERIFY it is gone** (spec
|
|
3
|
+
* `graceful-pre-timeout-wip-checkpoint`, observation
|
|
4
|
+
* `checkpoint-releases-lock-while-predecessor-agent-still-writes`).
|
|
5
|
+
*
|
|
6
|
+
* ## The gap this closes
|
|
7
|
+
*
|
|
8
|
+
* The deadline checkpoint used to `child.kill('SIGTERM')` the agent process and
|
|
9
|
+
* treat the signal as if it were the outcome: the moment pi's own `exit` fired,
|
|
10
|
+
* the runner saved the WIP, RELEASED the item lock, and let the next tick
|
|
11
|
+
* dispatch a CONTINUATION agent into the SAME worktree. But `child.kill` signals
|
|
12
|
+
* exactly ONE pid, and a modern agent is a TREE (subagent processes, MCP servers,
|
|
13
|
+
* model proxies, tool subshells). Those descendants survive the parent's SIGTERM,
|
|
14
|
+
* and once pi exits they are re-parented to init — so they can no longer even be
|
|
15
|
+
* FOUND by walking `ppid`, while they keep writing into the worktree.
|
|
16
|
+
*
|
|
17
|
+
* Observed on a real run: the checkpoint saved WIP at 02:13, a continuation agent
|
|
18
|
+
* onboarded into the same worktree at ~02:15, and the predecessor's session log
|
|
19
|
+
* kept being written until 02:19:29 — four minutes INTO the successor's run,
|
|
20
|
+
* whose opening `git status` had already read the tree as clean. Nothing was lost
|
|
21
|
+
* only because the two happened to touch different files. The shape is the
|
|
22
|
+
* defect: one lock, one working tree, two live writers. A write landing after the
|
|
23
|
+
* successor's `git status` is invisible to it; a write landing during its edits
|
|
24
|
+
* can be clobbered either way; and the successor can commit the predecessor's
|
|
25
|
+
* half-finished edits as its own, under a message describing something else.
|
|
26
|
+
*
|
|
27
|
+
* ## Why the PROCESS GROUP is the handle
|
|
28
|
+
*
|
|
29
|
+
* A pid-tree walk cannot work here: the descendants we must reap are precisely
|
|
30
|
+
* the ones that OUTLIVE the parent, and an orphan's `ppid` is gone. A process
|
|
31
|
+
* GROUP id, by contrast, is inherited by every descendant and is NOT changed by
|
|
32
|
+
* re-parenting. So if the agent is spawned as a group LEADER (`detached: true`,
|
|
33
|
+
* making its pgid equal its pid), `kill(-pgid, …)` reaches the entire tree,
|
|
34
|
+
* orphans included — which is why {@link reapProcessGroup} takes a pgid and why
|
|
35
|
+
* `pi-harness.ts` spawns the deadline-capable async launch detached.
|
|
36
|
+
*
|
|
37
|
+
* ## Signal, then VERIFY — never assume
|
|
38
|
+
*
|
|
39
|
+
* The point of the whole module is that sending a signal is not evidence that
|
|
40
|
+
* anything died. So: SIGTERM the group, POLL until it is actually gone, escalate
|
|
41
|
+
* to SIGKILL after a grace, keep polling, and if it STILL will not die, say so
|
|
42
|
+
* LOUDLY and let the caller refuse to release the lock. A checkpoint that cannot
|
|
43
|
+
* prove the predecessor is dead must not hand the worktree to a successor.
|
|
44
|
+
*/
|
|
45
|
+
|
|
46
|
+
/** Poll interval while waiting for a signalled group to actually exit. */
|
|
47
|
+
const REAP_POLL_MS = 50;
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* How long to wait after SIGTERM before escalating to SIGKILL. Matches
|
|
51
|
+
* `pi-harness.ts`'s `DEADLINE_SIGKILL_GRACE_MS` intent: enough for an agent to
|
|
52
|
+
* flush its session log and exit cleanly, short enough not to stall a CI leg.
|
|
53
|
+
*/
|
|
54
|
+
export const REAP_SIGTERM_GRACE_MS = 10_000;
|
|
55
|
+
|
|
56
|
+
/** How long to keep waiting after SIGKILL before declaring the reap FAILED. */
|
|
57
|
+
export const REAP_SIGKILL_TIMEOUT_MS = 5_000;
|
|
58
|
+
|
|
59
|
+
/** The outcome of a {@link reapProcessGroup} attempt. */
|
|
60
|
+
export interface ReapResult {
|
|
61
|
+
/**
|
|
62
|
+
* True iff the group is VERIFIED gone (observed non-existent, not merely
|
|
63
|
+
* signalled). Only a `true` here licenses releasing the item lock and
|
|
64
|
+
* dispatching a successor into the same worktree.
|
|
65
|
+
*/
|
|
66
|
+
reaped: boolean;
|
|
67
|
+
/** True iff SIGKILL was needed (the tree ignored SIGTERM) — worth reporting. */
|
|
68
|
+
escalatedToSigkill: boolean;
|
|
69
|
+
/** Total wall-clock ms spent waiting for the tree to die. */
|
|
70
|
+
waitedMs: number;
|
|
71
|
+
/** A human-readable account, always populated (the LOUD failure text). */
|
|
72
|
+
detail: string;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** Sleep helper (injectable clock is not needed: callers inject `wait` in tests). */
|
|
76
|
+
function sleep(ms: number): Promise<void> {
|
|
77
|
+
return new Promise((resolve) => {
|
|
78
|
+
const timer = setTimeout(resolve, ms);
|
|
79
|
+
timer.unref?.();
|
|
80
|
+
});
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/**
|
|
84
|
+
* Is any process still alive in process group `pgid`?
|
|
85
|
+
*
|
|
86
|
+
* `process.kill(-pgid, 0)` is the liveness probe: signal 0 performs the
|
|
87
|
+
* permission/existence check WITHOUT delivering a signal (the same technique
|
|
88
|
+
* `harness.ts`'s {@link pidAlive} uses for a single pid, widened to the group).
|
|
89
|
+
*
|
|
90
|
+
* - It THROWS `ESRCH` when no process in the group exists ⇒ the group is gone.
|
|
91
|
+
* - It THROWS `EPERM` when the group exists but we may not signal it. That is
|
|
92
|
+
* still "alive", and reporting it as dead would be the very assumption this
|
|
93
|
+
* module exists to remove — so `EPERM` reads as ALIVE.
|
|
94
|
+
* - It succeeds ⇒ alive.
|
|
95
|
+
*/
|
|
96
|
+
export function processGroupAlive(pgid: number): boolean {
|
|
97
|
+
if (!Number.isInteger(pgid) || pgid <= 1) {
|
|
98
|
+
// pgid 0/1 (or a bogus value) would mean "our own group" / init — signalling
|
|
99
|
+
// those would be catastrophic, so never claim they are ours to reap.
|
|
100
|
+
return false;
|
|
101
|
+
}
|
|
102
|
+
try {
|
|
103
|
+
process.kill(-pgid, 0);
|
|
104
|
+
return true;
|
|
105
|
+
} catch (err) {
|
|
106
|
+
const code = (err as NodeJS.ErrnoException).code;
|
|
107
|
+
if (code === 'EPERM') {
|
|
108
|
+
return true; // exists, just not signallable by us.
|
|
109
|
+
}
|
|
110
|
+
return false; // ESRCH (or anything else): treat as gone.
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/** Signal a whole process group, tolerating an already-dead group. */
|
|
115
|
+
function signalGroup(pgid: number, signal: NodeJS.Signals): void {
|
|
116
|
+
try {
|
|
117
|
+
process.kill(-pgid, signal);
|
|
118
|
+
} catch {
|
|
119
|
+
// ESRCH: already gone — the wait loop below observes that and succeeds.
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* SIGTERM process group `pgid`, wait for it to ACTUALLY exit, escalate to
|
|
125
|
+
* SIGKILL after {@link REAP_SIGTERM_GRACE_MS}, and report whether the tree is
|
|
126
|
+
* VERIFIED gone.
|
|
127
|
+
*
|
|
128
|
+
* Bounded by construction: at worst `sigtermGraceMs + sigkillTimeoutMs` before it
|
|
129
|
+
* returns `reaped: false` with a loud `detail`. It never waits indefinitely, so
|
|
130
|
+
* it cannot reintroduce the runner-hang the async launch's resolve-on-`exit`
|
|
131
|
+
* discipline exists to avoid.
|
|
132
|
+
*/
|
|
133
|
+
export async function reapProcessGroup(params: {
|
|
134
|
+
/** The process GROUP id to reap (the group leader's pid). */
|
|
135
|
+
pgid: number;
|
|
136
|
+
sigtermGraceMs?: number;
|
|
137
|
+
sigkillTimeoutMs?: number;
|
|
138
|
+
/** Injectable sleep so tests need not burn real seconds. */
|
|
139
|
+
wait?: (ms: number) => Promise<void>;
|
|
140
|
+
/**
|
|
141
|
+
* Injectable liveness probe (default {@link processGroupAlive}). Exists so the
|
|
142
|
+
* REFUSAL path — a tree that survives SIGTERM *and* SIGKILL — can be tested
|
|
143
|
+
* deterministically. There is no portable way to create a genuinely unkillable
|
|
144
|
+
* process, and the alternative (a group we truly cannot signal) would risk the
|
|
145
|
+
* test process itself, so the probe is the seam.
|
|
146
|
+
*/
|
|
147
|
+
alive?: (pgid: number) => boolean;
|
|
148
|
+
}): Promise<ReapResult> {
|
|
149
|
+
const {
|
|
150
|
+
pgid,
|
|
151
|
+
sigtermGraceMs = REAP_SIGTERM_GRACE_MS,
|
|
152
|
+
sigkillTimeoutMs = REAP_SIGKILL_TIMEOUT_MS,
|
|
153
|
+
wait = sleep,
|
|
154
|
+
alive = processGroupAlive,
|
|
155
|
+
} = params;
|
|
156
|
+
const started = Date.now();
|
|
157
|
+
|
|
158
|
+
if (!alive(pgid)) {
|
|
159
|
+
return {
|
|
160
|
+
reaped: true,
|
|
161
|
+
escalatedToSigkill: false,
|
|
162
|
+
waitedMs: 0,
|
|
163
|
+
detail: `agent process group ${pgid} was already gone (nothing to reap).`,
|
|
164
|
+
};
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
// 1. SOFT: ask the whole tree to stop, then WAIT for it to be observably gone.
|
|
168
|
+
signalGroup(pgid, 'SIGTERM');
|
|
169
|
+
while (Date.now() - started < sigtermGraceMs) {
|
|
170
|
+
if (!alive(pgid)) {
|
|
171
|
+
const waitedMs = Date.now() - started;
|
|
172
|
+
return {
|
|
173
|
+
reaped: true,
|
|
174
|
+
escalatedToSigkill: false,
|
|
175
|
+
waitedMs,
|
|
176
|
+
detail:
|
|
177
|
+
`agent process group ${pgid} exited on SIGTERM after ${waitedMs}ms ` +
|
|
178
|
+
'(verified gone).',
|
|
179
|
+
};
|
|
180
|
+
}
|
|
181
|
+
await wait(REAP_POLL_MS);
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
// 2. HARD: the tree ignored SIGTERM through the grace. SIGKILL it and keep
|
|
185
|
+
// verifying — a wedged agent still holding the worktree must not survive
|
|
186
|
+
// into the successor's run.
|
|
187
|
+
signalGroup(pgid, 'SIGKILL');
|
|
188
|
+
const killDeadline = Date.now() + sigkillTimeoutMs;
|
|
189
|
+
while (Date.now() < killDeadline) {
|
|
190
|
+
if (!alive(pgid)) {
|
|
191
|
+
const waitedMs = Date.now() - started;
|
|
192
|
+
return {
|
|
193
|
+
reaped: true,
|
|
194
|
+
escalatedToSigkill: true,
|
|
195
|
+
waitedMs,
|
|
196
|
+
detail:
|
|
197
|
+
`agent process group ${pgid} ignored SIGTERM and was SIGKILLed; ` +
|
|
198
|
+
`exited after ${waitedMs}ms (verified gone).`,
|
|
199
|
+
};
|
|
200
|
+
}
|
|
201
|
+
await wait(REAP_POLL_MS);
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
// 3. LOUD FAILURE: we cannot prove the predecessor is dead. Say exactly that,
|
|
205
|
+
// and exactly what the caller must not do as a result.
|
|
206
|
+
const waitedMs = Date.now() - started;
|
|
207
|
+
return {
|
|
208
|
+
reaped: false,
|
|
209
|
+
escalatedToSigkill: true,
|
|
210
|
+
waitedMs,
|
|
211
|
+
detail:
|
|
212
|
+
`REFUSING TO PROCEED: agent process group ${pgid} is STILL ALIVE ${waitedMs}ms ` +
|
|
213
|
+
'after SIGTERM and SIGKILL (an unkillable/uninterruptible descendant — e.g. a ' +
|
|
214
|
+
'process wedged in a kernel call, or one we lack permission to signal). It may ' +
|
|
215
|
+
'still be WRITING to the worktree, so the item lock must NOT be released and no ' +
|
|
216
|
+
'successor agent may onboard here: two live writers in one working tree can ' +
|
|
217
|
+
'silently clobber each other and let a successor commit the predecessor’s ' +
|
|
218
|
+
`half-finished edits. Inspect and kill it by hand (\`ps -g ${pgid}\`, ` +
|
|
219
|
+
`\`kill -9 -${pgid}\`) before re-running this item.`,
|
|
220
|
+
};
|
|
221
|
+
}
|
package/src/repo-config.ts
CHANGED
|
@@ -2,6 +2,7 @@ import {readFileSync, existsSync} from 'node:fs';
|
|
|
2
2
|
import {join} from 'node:path';
|
|
3
3
|
import {
|
|
4
4
|
mergeConfig,
|
|
5
|
+
applyModelFallbacks,
|
|
5
6
|
validateDeadlineConfig,
|
|
6
7
|
validateDorflCmdConfig,
|
|
7
8
|
warnDeprecatedConfigKeys,
|
|
@@ -209,8 +210,12 @@ export const REPO_ALLOWED_KEYS = [
|
|
|
209
210
|
// `model` (which model this repo's work runs on) and `harness` (which adapter)
|
|
210
211
|
// are legitimate repo properties (ADR §13) — model is routing intent, not auth,
|
|
211
212
|
// and a repo may prefer a given harness. `piBin`/`agentCmd` stay host-only
|
|
212
|
-
// (machine paths/commands), so they are rejected below.
|
|
213
|
+
// (machine paths/commands), so they are rejected below. `buildModel` is the
|
|
214
|
+
// builder-specific override of the general `model` (falls back to `model` when
|
|
215
|
+
// unset at all levels — see `applyModelFallbacks`); like `model` it is routing
|
|
216
|
+
// intent (not auth), so it is repo-appropriate.
|
|
213
217
|
'model',
|
|
218
|
+
'buildModel',
|
|
214
219
|
'harness',
|
|
215
220
|
// Gate 2 (PR/code review) policy is a genuine repo property (GATES spec
|
|
216
221
|
// `work/specs/tasked/review.md`), resolved per-repo like `integration`/`autoBuild`:
|
|
@@ -230,6 +235,14 @@ export const REPO_ALLOWED_KEYS = [
|
|
|
230
235
|
'taskerLoop',
|
|
231
236
|
'taskerLoopMax',
|
|
232
237
|
'taskerLoopModel',
|
|
238
|
+
// `triageModel` (the model the advance lifecycle gates — surface, apply,
|
|
239
|
+
// triage — run on) is a genuine repo property like `reviewModel`/
|
|
240
|
+
// `taskerLoopModel`: routing intent (not auth), resolved per-repo through the
|
|
241
|
+
// same chain. `intakeModel` (the front-door intake decision agent's model) is
|
|
242
|
+
// likewise repo-appropriate. Each falls back to `model` when unset at all
|
|
243
|
+
// levels (see `applyModelFallbacks`).
|
|
244
|
+
'triageModel',
|
|
245
|
+
'intakeModel',
|
|
233
246
|
// `freshWorktreeGate` (run the acceptance gate against the REBASED tip in a
|
|
234
247
|
// clean throwaway worktree, ON by default) is a genuine repo property exactly
|
|
235
248
|
// like `verify`/`prepare`/`review`: whether this repo's gate tests the merged
|
|
@@ -646,6 +659,11 @@ export function resolveRepoConfigFromLoaded(
|
|
|
646
659
|
// (flag / env / per-repo / global) surfaces the same clear error (ADR
|
|
647
660
|
// `dorfl-cmd-repo-settable-exception-to-host-only`).
|
|
648
661
|
validateDorflCmdConfig(config);
|
|
662
|
+
// Apply the per-role model fallbacks: `buildModel`, `reviewModel`, and
|
|
663
|
+
// `taskerLoopModel` each inherit the general `model` when unset at EVERY level
|
|
664
|
+
// (flag / env / per-repo / global). Applied HERE — the resolution FINAL point —
|
|
665
|
+
// so the fallback uses the FULLY resolved `model`, never an intermediate layer.
|
|
666
|
+
applyModelFallbacks(config);
|
|
649
667
|
return {
|
|
650
668
|
config,
|
|
651
669
|
rejected: repo.rejected,
|
package/src/run.ts
CHANGED
|
@@ -834,8 +834,9 @@ async function runOneItem(
|
|
|
834
834
|
|
|
835
835
|
// 4. Run the agent — via the injected runner (tests) or the harness seam
|
|
836
836
|
// (null adapter by default), shelling out to the configured agentCmd. The
|
|
837
|
-
// resolved per-repo `
|
|
838
|
-
//
|
|
837
|
+
// resolved per-repo `buildModel` (ADR §13, which inherits the general
|
|
838
|
+
// `model` when unset) flows through the seam to the adapter; a
|
|
839
|
+
// `{model}`-in-agentCmd misconfiguration surfaces as agent-failed.
|
|
839
840
|
let agent: {ok: boolean; detail?: string; output?: string};
|
|
840
841
|
try {
|
|
841
842
|
agent = runAgent(
|
|
@@ -844,7 +845,7 @@ async function runOneItem(
|
|
|
844
845
|
prompt,
|
|
845
846
|
slug,
|
|
846
847
|
config.agentCmd,
|
|
847
|
-
config.
|
|
848
|
+
config.buildModel,
|
|
848
849
|
config.sessionsDir,
|
|
849
850
|
);
|
|
850
851
|
} catch (err) {
|
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
import {
|
|
2
|
+
existsSync,
|
|
3
|
+
mkdirSync,
|
|
4
|
+
readFileSync,
|
|
5
|
+
rmSync,
|
|
6
|
+
writeFileSync,
|
|
7
|
+
} from 'node:fs';
|
|
8
|
+
import {dirname, join} from 'node:path';
|
|
9
|
+
import {run} from './git.js';
|
|
10
|
+
import {pidAlive} from './harness.js';
|
|
11
|
+
import {processGroupAlive} from './reap-agent-tree.js';
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* **The per-WORKING-TREE writer sentinel** (observation
|
|
15
|
+
* `checkpoint-releases-lock-while-predecessor-agent-still-writes`).
|
|
16
|
+
*
|
|
17
|
+
* The per-item lock (`item-lock.ts`) guards the ITEM: it answers "who owns this
|
|
18
|
+
* task?" and it is what claim / requeue / bounce move around. Nothing guarded the
|
|
19
|
+
* WORKING TREE. Those are different resources, and the deadline checkpoint is
|
|
20
|
+
* exactly where they come apart: the checkpoint releases the item lock so the
|
|
21
|
+
* next tick can continue the task, but the tree the previous agent was editing is
|
|
22
|
+
* reused by the successor. If the predecessor is still alive, two agents write to
|
|
23
|
+
* one tree.
|
|
24
|
+
*
|
|
25
|
+
* The primary fix is to reap the predecessor and VERIFY it is gone before the
|
|
26
|
+
* lock moves (`reap-agent-tree.ts`). This sentinel is the INDEPENDENT backstop:
|
|
27
|
+
* even if a live writer survives by some route the reap did not cover (a
|
|
28
|
+
* deliberately `setsid`-ed grandchild, a stale run from a crashed runner, an
|
|
29
|
+
* operator manually re-driving an item), a second agent physically cannot onboard
|
|
30
|
+
* into a tree that already has a LIVE holder. It is deliberately keyed on the
|
|
31
|
+
* TREE, not on the item: two different items sharing one worktree is just as
|
|
32
|
+
* unsafe as two attempts at the same item.
|
|
33
|
+
*
|
|
34
|
+
* ## Where the sentinel lives, and why not in the tree
|
|
35
|
+
*
|
|
36
|
+
* It is written to the worktree's PRIVATE git directory (`git rev-parse
|
|
37
|
+
* --absolute-git-dir`, which for a linked worktree is
|
|
38
|
+
* `.../.git/worktrees/<name>/`), NOT to a file inside the working tree. That
|
|
39
|
+
* placement is load-bearing:
|
|
40
|
+
*
|
|
41
|
+
* - it is per-worktree (linked worktrees each get their own git dir), which is
|
|
42
|
+
* precisely the granularity we are guarding;
|
|
43
|
+
* - it can never appear in `git status`, so it cannot be mistaken for agent work,
|
|
44
|
+
* cannot be swept into a commit by a `git add -A`, and needs no new exclusion
|
|
45
|
+
* in the empty-diff backstop / `gc`'s cleanliness predicate (unlike
|
|
46
|
+
* `.dorfl-job.json`, which each of those has to filter out by name);
|
|
47
|
+
* - it is removed with the worktree, so it cannot outlive what it guards.
|
|
48
|
+
*
|
|
49
|
+
* ## Liveness, not presence
|
|
50
|
+
*
|
|
51
|
+
* A pid file that only records presence becomes a permanent blocker the first
|
|
52
|
+
* time a runner is `kill -9`ed. So the holder is checked for LIVENESS (its
|
|
53
|
+
* process group first, falling back to its pid) and a dead holder's sentinel is
|
|
54
|
+
* treated as stale and taken over. Only a genuinely live foreign writer refuses.
|
|
55
|
+
*/
|
|
56
|
+
|
|
57
|
+
/** The sentinel filename inside the worktree's private git directory. */
|
|
58
|
+
export const WRITER_SENTINEL_FILENAME = 'dorfl-writer.json';
|
|
59
|
+
|
|
60
|
+
/** The recorded holder of a worktree's writer sentinel. */
|
|
61
|
+
export interface WorktreeWriter {
|
|
62
|
+
/** The runner process that owns the agent writing in this tree. */
|
|
63
|
+
pid: number;
|
|
64
|
+
/** The agent's process GROUP, when the harness spawned a killable one. */
|
|
65
|
+
pgid?: number;
|
|
66
|
+
/** The item being built in this tree (diagnostics: names the other writer). */
|
|
67
|
+
slug: string;
|
|
68
|
+
/** ISO timestamp of acquisition (diagnostics: how long it has been held). */
|
|
69
|
+
startedAt: string;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** The outcome of trying to become a worktree's sole writer. */
|
|
73
|
+
export type WorktreeWriterLock =
|
|
74
|
+
| {
|
|
75
|
+
acquired: true;
|
|
76
|
+
/** Release the sentinel. Idempotent, and safe if it was already stolen. */
|
|
77
|
+
release(): void;
|
|
78
|
+
}
|
|
79
|
+
| {
|
|
80
|
+
acquired: false;
|
|
81
|
+
/** The LIVE holder that refused us (when it could be parsed). */
|
|
82
|
+
holder?: WorktreeWriter;
|
|
83
|
+
/** Human-readable refusal, naming the other writer. */
|
|
84
|
+
reason: string;
|
|
85
|
+
};
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* The worktree's PRIVATE git directory, or `undefined` when `dir` is not a git
|
|
89
|
+
* worktree (in which case there is no sentinel location and the caller proceeds
|
|
90
|
+
* unguarded rather than failing — this is a backstop, not a gate).
|
|
91
|
+
*/
|
|
92
|
+
function worktreeGitDir(
|
|
93
|
+
dir: string,
|
|
94
|
+
env: NodeJS.ProcessEnv | undefined,
|
|
95
|
+
): string | undefined {
|
|
96
|
+
const result = run('git', ['rev-parse', '--absolute-git-dir'], dir, {env});
|
|
97
|
+
if (result.status !== 0) {
|
|
98
|
+
return undefined;
|
|
99
|
+
}
|
|
100
|
+
const path = result.stdout.trim();
|
|
101
|
+
return path === '' ? undefined : path;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/** The sentinel path for `dir`, or `undefined` when `dir` is not a worktree. */
|
|
105
|
+
export function writerSentinelPath(
|
|
106
|
+
dir: string,
|
|
107
|
+
env?: NodeJS.ProcessEnv,
|
|
108
|
+
): string | undefined {
|
|
109
|
+
const gitDir = worktreeGitDir(dir, env);
|
|
110
|
+
return gitDir === undefined
|
|
111
|
+
? undefined
|
|
112
|
+
: join(gitDir, WRITER_SENTINEL_FILENAME);
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/** Read + parse the sentinel, or `undefined` when absent/corrupt. */
|
|
116
|
+
export function readWorktreeWriter(
|
|
117
|
+
dir: string,
|
|
118
|
+
env?: NodeJS.ProcessEnv,
|
|
119
|
+
): WorktreeWriter | undefined {
|
|
120
|
+
const path = writerSentinelPath(dir, env);
|
|
121
|
+
if (path === undefined || !existsSync(path)) {
|
|
122
|
+
return undefined;
|
|
123
|
+
}
|
|
124
|
+
try {
|
|
125
|
+
const parsed = JSON.parse(readFileSync(path, 'utf8')) as WorktreeWriter;
|
|
126
|
+
return typeof parsed?.pid === 'number' ? parsed : undefined;
|
|
127
|
+
} catch {
|
|
128
|
+
// A corrupt sentinel records nothing we can trust; treat it as absent so it
|
|
129
|
+
// self-heals on the next acquire rather than wedging the worktree forever.
|
|
130
|
+
return undefined;
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* Is the recorded holder still running? Prefers the agent's process GROUP (which
|
|
136
|
+
* survives the group leader's death and so catches exactly the orphaned-writer
|
|
137
|
+
* case this exists for), and falls back to the runner pid.
|
|
138
|
+
*/
|
|
139
|
+
export function writerAlive(holder: WorktreeWriter): boolean {
|
|
140
|
+
if (holder.pgid !== undefined && processGroupAlive(holder.pgid)) {
|
|
141
|
+
return true;
|
|
142
|
+
}
|
|
143
|
+
return pidAlive(holder.pid);
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Claim `dir` as the SOLE agent-writable working tree for `slug`.
|
|
148
|
+
*
|
|
149
|
+
* Refuses when a DIFFERENT, still-LIVE writer holds it — the second-agent case
|
|
150
|
+
* the observation describes. A dead holder's sentinel is stale and is taken over
|
|
151
|
+
* silently (a `kill -9`ed runner must not poison the worktree forever), and our
|
|
152
|
+
* OWN pid re-acquiring is a no-op re-entry rather than a refusal.
|
|
153
|
+
*
|
|
154
|
+
* When `dir` is not a git worktree there is nowhere private to record the
|
|
155
|
+
* sentinel; that is reported as acquired with a no-op release, because this is a
|
|
156
|
+
* defence-in-depth backstop and must never become a new way for a legitimate run
|
|
157
|
+
* to fail.
|
|
158
|
+
*/
|
|
159
|
+
export function acquireWorktreeWriterLock(params: {
|
|
160
|
+
dir: string;
|
|
161
|
+
slug: string;
|
|
162
|
+
/** The agent's process group, when known (the strongest liveness anchor). */
|
|
163
|
+
pgid?: number;
|
|
164
|
+
env?: NodeJS.ProcessEnv;
|
|
165
|
+
}): WorktreeWriterLock {
|
|
166
|
+
const {dir, slug, pgid, env} = params;
|
|
167
|
+
const path = writerSentinelPath(dir, env);
|
|
168
|
+
if (path === undefined) {
|
|
169
|
+
return {acquired: true, release: () => {}};
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
const existing = readWorktreeWriter(dir, env);
|
|
173
|
+
if (
|
|
174
|
+
existing !== undefined &&
|
|
175
|
+
existing.pid !== process.pid &&
|
|
176
|
+
writerAlive(existing)
|
|
177
|
+
) {
|
|
178
|
+
return {
|
|
179
|
+
acquired: false,
|
|
180
|
+
holder: existing,
|
|
181
|
+
reason:
|
|
182
|
+
`worktree ${dir} already has a LIVE agent writer: pid ${existing.pid}` +
|
|
183
|
+
(existing.pgid !== undefined ? ` (group ${existing.pgid})` : '') +
|
|
184
|
+
` building '${existing.slug}' since ${existing.startedAt}. Refusing to ` +
|
|
185
|
+
`onboard '${slug}' into the same working tree: two live agents in one ` +
|
|
186
|
+
'tree can clobber each other’s edits, and the second can commit the ' +
|
|
187
|
+
'first’s half-finished work under a message describing something else. ' +
|
|
188
|
+
'Wait for it to exit, or kill it, then retry.',
|
|
189
|
+
};
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
const record: WorktreeWriter = {
|
|
193
|
+
pid: process.pid,
|
|
194
|
+
...(pgid !== undefined ? {pgid} : {}),
|
|
195
|
+
slug,
|
|
196
|
+
startedAt: new Date().toISOString(),
|
|
197
|
+
};
|
|
198
|
+
mkdirSync(dirname(path), {recursive: true});
|
|
199
|
+
writeFileSync(path, `${JSON.stringify(record, null, 2)}\n`, 'utf8');
|
|
200
|
+
|
|
201
|
+
let released = false;
|
|
202
|
+
return {
|
|
203
|
+
acquired: true,
|
|
204
|
+
release: (): void => {
|
|
205
|
+
if (released) {
|
|
206
|
+
return;
|
|
207
|
+
}
|
|
208
|
+
released = true;
|
|
209
|
+
// Only remove a sentinel that is still OURS: if it was stolen as stale by
|
|
210
|
+
// another runner, deleting it would silently un-guard that runner's tree.
|
|
211
|
+
const current = readWorktreeWriter(dir, env);
|
|
212
|
+
if (current === undefined || current.pid === process.pid) {
|
|
213
|
+
rmSync(path, {force: true});
|
|
214
|
+
}
|
|
215
|
+
},
|
|
216
|
+
};
|
|
217
|
+
}
|