pi-claude-supervisor 0.5.0 → 0.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  # Pi Claude Supervisor 完整方案
2
2
 
3
- > 文档状态:方案设计稿 / MVP 实施基线
3
+ > 文档状态:`v0.5.1` 已发布;Phase A–D 加固和真实 repair/reacceptance 已在当前工作树实现;待 exact-head 独立只读 Review 门禁
4
4
  > 目标项目目录:`pi-claude-supervisor`
5
5
  > 适用对象:W、项目负责人、实现人员、评审人员
6
6
 
@@ -36,7 +36,7 @@ Claude Code Worker
36
36
  4. Supervisor 因误判导致无限循环、危险操作或不可审计的修改。
37
37
  5. 人工无法随时接管或恢复任务。
38
38
 
39
- **总体判断:架构方向可以 GO。先完成兼容性 Spike、生命周期和故障恢复验证;低权限用户、OS sandbox 与网络隔离不作为当前主线或硬性阻塞,按调用者明确授权和宿主机策略运行,后续再做安全加固。**
39
+ **当前状态:`v0.5.1` 已正式发布,已完成固定 Claude Code `2.1.270` 稳定性统计、单 Worker recovery、真实只读 Review drill 和隔离临时 worktree 的允许编辑 repair/reacceptance drill。当前工作树已落实 repairable/persistent 能力拆分、verifying stop、paused watchdog、完整 repository evidence、可取消验收/Reviewer、启动 preflight、权限门禁修复和阶段进度通知;下一项也是发布前最后硬门禁的是当前 exact head 的独立只读 Review。协同多 Worker 仍延期;低权限用户、OS sandbox 与网络隔离仍是后续安全加固。**
40
40
 
41
41
  ---
42
42
 
@@ -74,7 +74,9 @@ Claude Code 适合作为实际开发 Worker,但在长时间任务中可能出
74
74
  - 不自动 merge、deploy 或 release。
75
75
 
76
76
  当前扩展已支持多个独立任务会话并行推进,但不允许活动会话共享同一
77
- 工作目录。事件日志由跨进程锁协调,状态和 watchdog 按会话隔离。
77
+ 工作目录。事件日志由跨进程锁协调,状态和 watchdog 按会话隔离。这里要区分两种
78
+ “多 Worker”:**独立会话并行**已经属于当前能力;**有依赖、交接和汇总验收的协同多
79
+ Worker**属于后续开发任务,不能通过简单地放宽 cwd 限制来实现。
78
80
 
79
81
  暂不支持:
80
82
 
@@ -979,3 +981,74 @@ Reviewer 必须使用独立 Pi session,只允许 `read`、`grep`、`find`、`l
979
981
  - OS sandbox、低权限执行和网络隔离;
980
982
  - 自动 merge、deploy、release、publish;
981
983
  - 多 Worker 在同一工作树协作。
984
+
985
+ ## 21. 后续开发路线图
986
+
987
+ `v0.5.0` 的发布不代表所有自动化目标都已完成。后续任务按“稳定性 → 恢复 → 协同
988
+ 调度 → 安全加固”推进;多 Worker 可以纳入开发任务,但应作为独立阶段,不能与当前
989
+ 单 Worker 稳定性门禁混在一起。
990
+
991
+ ### 21.1 短期:稳定性收尾
992
+
993
+ - 完成真实 Claude Code `2.1.270` 重复 Spike:普通任务连续 10 次,权限和问题回退各至少 5 次;
994
+ - 补齐 replay:多轮修复、验收失败修复、repair budget 耗尽、takeover、recover 和 Pi shutdown;
995
+ - 补齐边界测试:Reviewer 流式输出上限、`DecisionSessionStore.list()` 任务 ID 校验、跨进程恢复和超时/输出截断;
996
+ - 继续观察 npm `0.5.0`、GitHub Release 资产、provenance 和回滚路径;
997
+ - 验收标准:无重复动作、错误 complete、未清理 Worker 或未审计的自动放行。
998
+
999
+ ### 21.2 中期:恢复能力
1000
+
1001
+ - 设计安全的 Claude session resume;明确 `--resume` 与实时 PTY attach 的边界;
1002
+ - 完善 takeover、recover、Pi shutdown、Worker 崩溃和部分完成的状态语义;
1003
+ - 增加跨进程恢复端到端测试,包括 cwd lease、Decision Worker session、Worker 身份和事件日志一致性;
1004
+ - 恢复失败必须进入 `HUMAN_REQUIRED`,不能静默重放原始任务或重复发送输入。
1005
+
1006
+ ### 21.3 后续:多 Worker 协作与调度
1007
+
1008
+ 第一阶段只做**独立 worktree 的多 Worker 编排**,不允许共享工作树写入。建议拆成以下
1009
+ 开发任务:
1010
+
1011
+ 1. **任务图与角色模型**:增加 `parentTaskId`、Worker role、`dependsOn`、worktree、handoff
1012
+ artifact 和子任务状态;明确 root task 与 child task 的审计关联。
1013
+ 2. **受限调度器**:实现并发上限、依赖就绪、全局时间/修复预算、取消传播和失败隔离;
1014
+ 不让任意 Worker 自行启动、停止或批准另一个 Worker。
1015
+ 3. **结构化交接**:Worker 之间只通过受限 artifact、事件引用和验收报告交接,不直接共享
1016
+ 控制通道;交接内容必须经过 schema 校验和大小限制。
1017
+ 4. **汇总验收**:每个 child 先独立验收,root task 再汇总目标、diff、测试和 Reviewer 结果;
1018
+ 冲突、缺失证据或任一 P0/P1 自动升级人工。
1019
+ 5. **恢复与关闭**:支持单个 child、整棵任务图和 Pi shutdown 的一致性恢复;父任务不能在
1020
+ 子任务状态未知时报告 `completed`。
1021
+ 6. **冲突检测和人工整合**:只允许在独立 integration worktree 中进行显式整合;不自动
1022
+ merge/publish,冲突和整合动作必须保留人工控制权。
1023
+
1024
+ 多 Worker 阶段的最小验收矩阵:两个独立 Worker 并行、依赖顺序、一个 Worker 失败、取消
1025
+ 传播、重复交接、工作树冲突、单 child 恢复、整棵任务图恢复和 shutdown 中断。通过这些
1026
+ 门禁后,才评估是否需要更复杂的 lead-worker 或动态任务分解。
1027
+
1028
+ ### 21.4 后置:安全加固
1029
+
1030
+ - CLI 多版本兼容矩阵;
1031
+ - OS sandbox、低权限执行、网络隔离/allowlist;
1032
+ - 更深的供应链、SBOM、密钥隔离和生产监控。
1033
+
1034
+ ## 22. v0.5.1 真实演练后的自动化加固计划
1035
+
1036
+ `v0.5.1` 发布后的真实 Claude Code `2.1.270` 演练完成了
1037
+ Worker → 验收 → 独立 Reviewer → fail-closed 人工介入链路。验收六项全部通过,
1038
+ 但发现两个 P1 和两个 P2 生命周期/证据问题。正式记录、复现结果、实施阶段和门禁
1039
+ 见 [`docs/automation-hardening-plan.md`](automation-hardening-plan.md)。
1040
+
1041
+ 本轮实现顺序固定为:
1042
+
1043
+ 1. **生命周期与能力模型**:拆分 `persistentSession` 与 `repairableSession`,修复非持久
1044
+ JSONL repair 的非法终态转换,并支持 `verifying` 状态的 stop/shutdown;
1045
+ 2. **watchdog 与证据完整性**:暂停 no-output 时钟,补齐 HEAD-relative staged/unstaged
1046
+ diff 和安全的 untracked evidence;
1047
+ 3. **自动化协议**:按 assistant message 边界解析 Reviewer/Decision Worker 输出,增加
1048
+ 启动 preflight、可观测 heartbeat、permission gate 一致性和 signal 生命周期;
1049
+ 4. **验证门禁**:已补齐真实 capability 矩阵并完成隔离 worktree 的真实
1050
+ repair/reacceptance 演练;当前 exact-head 独立 Reviewer 通过前不进行任何合并或发布。
1051
+
1052
+ 本轮不放宽以下边界:Reviewer 仍只读,验收仍使用 argv/`execFile`,不伪造 Claude
1053
+ `--resume`,不自动 merge/deploy/release,P0/P1、重复 finding、超时、API 错误和不完整
1054
+ 证据继续 fail-closed。多 Worker 协作继续后置。
@@ -0,0 +1,24 @@
1
+ # Recovery follow-up review
2
+
3
+ The independent read-only review of the recovery changes initially returned
4
+ `BLOCK` with three P1 findings, one P2 finding, and a coverage note. The findings
5
+ were addressed in this follow-up.
6
+
7
+ - Stale `starting`/`registered`/`recovered_idle` claims now persist recovery
8
+ owner PID/start time, are reconciled only after the old owner and Worker
9
+ boundary are independently gone, and return to an explicit `interrupted`
10
+ state.
11
+ - `--takeover` requires a dead owner, dead Worker/process group, and a real,
12
+ readable empty cgroup boundary; missing or unverifiable Worker evidence is
13
+ rejected.
14
+ - Worker registration and `recovered_idle` transitions require an active record,
15
+ persist atomically, and are read back and checked before the recovered session
16
+ is exposed.
17
+ - Adopted tmux handoff reports the replaced task and closes its old Decision
18
+ session mapping only after worker identity registration succeeds.
19
+ - Session closure now carries cleanup evidence and intent; uncertain cleanup
20
+ retains the active record and cwd lease.
21
+
22
+ Validation: `npm run check`, `npm run build`, the pinned Claude Code 2.1.270
23
+ question matrix (5/5 verified), and the bounded stability evidence in
24
+ [`stability-matrix-2.1.270.md`](stability-matrix-2.1.270.md).
@@ -0,0 +1,32 @@
1
+ # Claude Code 2.1.270 stability matrix
2
+
3
+ Run date: 2026-09-14. The executable was resolved as
4
+ `/home/yancao/.local/share/mise/installs/claude/2.1.270/claude` and reported
5
+ `2.1.270 (Claude Code)`. Runs used the authenticated local provider and
6
+ `scripts/spike-claude-automation.mjs`; no Claude `--resume` was used.
7
+
8
+ ## Required matrix (120 s bounded run)
9
+
10
+ | Scenario | Runs | Completed + verified | Fail-closed outcomes |
11
+ | --- | ---: | ---: | --- |
12
+ | ordinary task | 10 | 9 | 1 stopped at the bounded deadline without verification |
13
+ | permission handling | 5 | 5 | 0 |
14
+ | question handling | 5 | 5 | 0 |
15
+
16
+ The ordinary outlier was not treated as success: the spike exited non-zero,
17
+ kept verification false, and stopped rather than retrying or replaying the task.
18
+ This is the intended timeout fail-closed behavior. An earlier question spike also
19
+ escalated after invalid Decision Worker output; it was likewise not counted as a
20
+ success. The five-run question matrix above was then rerun with a 180 s bound and
21
+ all five completed with `verified=true` and zero human interventions.
22
+
23
+ ## Bounded follow-up
24
+
25
+ Three additional ordinary and three additional question runs with the same
26
+ pinned executable and a 180 s bound also completed with `verified=true` and zero
27
+ human interventions. These runs support a provider-latency explanation for the
28
+ 120 s outliers; they do not turn an outlier into a success.
29
+
30
+ The deterministic replay, acceptance, recovery-state, lease, cleanup and
31
+ package checks remain the CI evidence. Authenticated Claude runs are manual
32
+ release evidence only.
package/docs/testing.md CHANGED
@@ -65,10 +65,15 @@ cause a later run to fail closed as human-required after the bounded Decision
65
65
  Worker or Reviewer timeout; this is evidence for the manual spike only, not a CI
66
66
  guarantee.
67
67
  The extension persists each automatic Decision Worker session as Pi JSONL plus a
68
- 0600 task mapping. Recovery is explicit and safe: after an unclean Pi restart,
68
+ 0600 task mapping. Automatic startup preflights the state/lease directories, cwd,
69
+ worker executable, transport dependency and required cgroup boundary before model
70
+ execution. Progress callbacks expose the current phase and periodic Worker
71
+ heartbeat. Recovery is explicit and safe: after an unclean Pi restart,
69
72
  `/supervise sessions` shows the task as `recoverable`, and `/supervise recover
70
- <task-id>` restores the Decision Worker history before starting a new Claude
71
- Worker.
73
+ [--takeover] <task-id>` restores the Decision Worker history before starting a new
74
+ Claude Worker. `--takeover` is accepted only when the old Pi owner is dead, the
75
+ Worker process group is gone, and its cgroup is a real readable empty boundary;
76
+ persistent tmux sessions use `adopt-tmux`.
72
77
  Run the permission and signal probes explicitly when validating a CLI release:
73
78
 
74
79
  ```bash
@@ -94,7 +99,9 @@ response, and an exact result marker; it also records metadata only.
94
99
 
95
100
  For each release, pin and record the validated Claude Code version, resolved
96
101
  executable path, and model. The spikes reject an unpinned/mismatched executable
97
- version. For this release the validated version is `2.1.270` with model `opus`.
102
+ version. For this release the validated version is `2.1.270` with model `opus`;
103
+ the bounded matrix and its fail-closed outliers are recorded in
104
+ [`docs/stability-matrix-2.1.270.md`](stability-matrix-2.1.270.md).
98
105
  Record:
99
106
 
100
107
  1. exact version and resolved executable path;
@@ -116,7 +123,9 @@ escalation is outbound-only through `PI_CLAUDE_SUPERVISOR_HUMAN_WEBHOOK_URL`;
116
123
  approval callbacks are deliberately not accepted without a separately
117
124
  authenticated endpoint.
118
125
 
119
- The tmux transport is selected with `PI_CLAUDE_SUPERVISOR_TRANSPORT=tmux`. Before
126
+ The tmux transport is selected with `PI_CLAUDE_SUPERVISOR_TRANSPORT=tmux`.
127
+ Automatic mode rejects an explicit `process-pipe` transport; use JSONL or tmux for
128
+ bounded decisions and repair. Before
120
129
  release, verify: private-socket attach, multi-line paste, prompt stability while
121
130
  Claude is busy, trust/permission dialog takeover, duplicate send prevention, pane
122
131
  replacement refusal, pause/resume, owned-session stop, adopted-session
@@ -133,28 +142,38 @@ response.
133
142
 
134
143
  The automated adapter matrix covers external `SIGTERM`, `SIGINT`, `SIGKILL`,
135
144
  `SIGSTOP`/`SIGCONT`, SIGTERM refusal/escalation, leader-early-exit descendant
136
- cleanup, required cgroup cleanup of a `setsid()` descendant, repeated stop,
137
- spawn failure, output truncation, blocked stdin write timeouts, and immediate
138
- JSONL results. The Supervisor matrix also covers
139
- retrying failed lifecycle events, preserving startup event order, stopping under
140
- persistent timeout-event failure, and restoring output after event-log failure. The Supervisor matrix covers startup rejection, externally terminated
141
- workers, lifecycle serialization and stop races.
145
+ cleanup, required cgroup bootstrap containment of a pre-attachment detached
146
+ and `setsid()` descendant, repeated stop, spawn failure, output truncation,
147
+ blocked stdin write timeouts, and immediate JSONL results. The Supervisor
148
+ matrix also covers retrying failed lifecycle events, preserving startup event
149
+ order, stopping under persistent timeout-event failure, and restoring output
150
+ after event-log failure. The Supervisor matrix covers startup rejection,
151
+ externally terminated workers, lifecycle serialization and stop races.
142
152
 
143
153
  Before release, manually test at least: immediate crash, hung process, malformed
144
154
  output, duplicate send, send/exit race, Pi `SIGTERM`/`SIGINT` shutdown,
145
155
  verification failure, blocked stdin writes/stop preemption, corrupt event-log
146
156
  tails, and descendants that call `setsid()` when cgroup mode is unavailable (expected
147
- fallback limitation). The cgroup test proves cleanup after attachment but does
148
- not eliminate the post-spawn attachment window. `SIGSTOP` and `SIGKILL` of the Pi host cannot be handled;
149
- verify and document the resulting orphan behavior.
157
+ fallback limitation). In cgroup mode, a small bootstrap joins the cgroup before
158
+ launching Worker code, and cleanup also validates and reaps the detached process
159
+ group; this closes the post-spawn attachment window. `SIGSTOP` and `SIGKILL` of
160
+ the Pi host cannot be handled; verify and document the resulting orphan behavior.
150
161
  Default behavior must be fail-closed and leave no orphaned worker process within
151
162
  the managed process group.
152
163
 
153
164
  ## Near-term automation acceptance gate
154
165
 
155
- The next implementation milestone focuses on real stability for the pinned
156
- Claude Code `2.1.270` CLI. It does not add a multi-version matrix or wait for
157
- OS sandbox, low-privilege or network-isolation work.
166
+ The `v0.5.0` implementation of the acceptance—independent Review—repair—reacceptance
167
+ loop is shipped. The `v0.5.1` real read-only drill reached acceptance and independent
168
+ Review, then correctly stopped at human intervention after two P1 and two P2 findings.
169
+ A separate real edit-capable Claude Code `2.1.270` drill then exercised one bounded
170
+ acceptance failure, repair turn, reacceptance and independent Reviewer `pass` in an
171
+ isolated temporary worktree. The current hardening plan and evidence paths are recorded
172
+ in [`docs/automation-hardening-plan.md`](automation-hardening-plan.md). Deterministic
173
+ coverage now includes repairable-vs-persistent capability assertions, cancellation
174
+ of acceptance commands, stop-from-verifying precedence, paused watchdog baselining,
175
+ staged/untracked evidence and untracked symlink rejection. The remaining release gate
176
+ is the exact-head independent read-only review.
158
177
 
159
178
  ### Acceptance and Reviewer fixtures
160
179
 
@@ -165,6 +184,9 @@ Deterministic tests must cover:
165
184
  - independent read-only Reviewer pass/revise/human results;
166
185
  - invalid Reviewer JSON and Reviewer API failure escalating to human;
167
186
  - repair rounds, repeated finding detection, P0/P1 escalation and repair-budget exhaustion;
187
+ - non-persistent JSONL verification failure without duplicate terminal transitions;
188
+ - repairable-but-not-persistent JSONL multi-turn repair;
189
+ - stop and Pi shutdown from `verifying`, including Decision Worker closure and cwd lease release;
168
190
  - completion being impossible without passing all required checks and review.
169
191
 
170
192
  Reviewer sessions use only `read`, `grep`, `find` and `ls`; they must not modify
@@ -182,10 +204,57 @@ The adapter/replay matrix must cover:
182
204
  - duplicate Supervisor idempotency keys without duplicate input;
183
205
  - stop, SIGTERM, SIGINT and Pi shutdown while a JSONL request is active;
184
206
  - ordinary completion, low-risk permission allow, AskUserQuestion deny-to-text,
185
- multi-turn, verifier failure/repair, takeover and explicit recovery.
207
+ multi-turn, verifier failure/repair, takeover and explicit recovery;
208
+ - staged and untracked repository evidence, symlink rejection and truncation fail-closed;
209
+ - paused watchdog behavior and resume-time no-output rebasing;
210
+ - assistant-message-bounded Reviewer and Decision Worker output parsing.
186
211
 
187
212
  Real Claude tests remain authenticated manual Spikes and are pinned to
188
213
  `2.1.270`; they are not part of normal CI. Normal CI runs deterministic fake
189
- Worker and replay fixtures. Stability evidence should include ten consecutive
190
- ordinary automatic runs and at least five runs each for permission and question
191
- handling, with no duplicate action, false completion or unreaped Worker.
214
+ Worker and replay fixtures. The pinned stability matrix is the compatibility evidence
215
+ for this release line; any future CLI change must rerun ten consecutive ordinary
216
+ automatic runs and at least five runs each for permission and question handling, with
217
+ no duplicate action, false completion or unreaped Worker.
218
+
219
+ ## Future multi-worker test plan
220
+
221
+ Multiple independent task sessions already run concurrently when their canonical
222
+ cwd/worktrees do not overlap. This is not yet coordinated multi-worker
223
+ collaboration. The future multi-worker milestone must be tested as a task graph,
224
+ not as unrestricted shared-worker access.
225
+
226
+ Required deterministic and integration coverage:
227
+
228
+ - two independent Workers with separate worktrees and aggregated parent status;
229
+ - dependency ordering and a blocked child that must not start early;
230
+ - child failure, timeout, cancellation propagation and bounded global budgets;
231
+ - schema-validated handoff artifacts with duplicate/oversized/stale handoffs;
232
+ - conflicting diffs detected before integration, with no same-worktree writes;
233
+ - independent child acceptance followed by root-task aggregate acceptance and Review;
234
+ - single-child recovery, whole-graph recovery and Pi shutdown during scheduling;
235
+ - no child can grant permissions, send control input to another child or bypass Policy Gate;
236
+ - explicit human-controlled integration in a separate worktree; no automatic merge or publish.
237
+
238
+ The multi-worker gate should be added only after the pinned single-worker stability
239
+ and recovery gates pass. CI should use fake Workers and replay fixtures; authenticated
240
+ Claude multi-worker Spikes remain manual and version-pinned.
241
+
242
+ ## Live drill and hardening gate
243
+
244
+ A live review must record the exact Claude executable/version, transport, cgroup mode,
245
+ permission flags, task id, acceptance result, Reviewer result, cleanup status and cwd lease
246
+ status. A human-required result is a valid safety outcome and must not be converted into a
247
+ pass by retrying the same task automatically.
248
+
249
+ For the hardening release, the completed gate record is:
250
+
251
+ 1. `npm run check`, `npm run test:pi`, `npm run test:install`, `npm run build`;
252
+ 2. deterministic lifecycle/evidence/capability tests;
253
+ 3. a disposable temporary-worktree real Claude repair/reacceptance spike with an edit-capable
254
+ Worker;
255
+ 4. cleanup verification: no Worker, no Decision Worker, no unreconciled lease and clean Git
256
+ worktree.
257
+
258
+ The remaining gate is a read-only review of the exact resulting commit. The drill evidence is
259
+ recorded in `docs/automation-hardening-plan.md`; it did not run against this release worktree,
260
+ did not use Claude `--resume`, and did not merge, publish or release automatically.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-claude-supervisor",
3
- "version": "0.5.0",
3
+ "version": "0.5.2",
4
4
  "description": "A policy-gated Pi supervisor for observing and verifying Claude Code workers.",
5
5
  "license": "MIT",
6
6
  "publishConfig": {
package/src/cwd-lease.ts CHANGED
@@ -36,8 +36,25 @@ export interface CwdLeaseHandoff {
36
36
  tmuxSocket?: string;
37
37
  }
38
38
 
39
+ /**
40
+ * Explicit recovery takeover. This is intentionally separate from ordinary
41
+ * acquisition: an old Pi owner may have died while its Worker survived.
42
+ */
43
+ export interface CwdLeaseTakeover {
44
+ taskId: string;
45
+ /** Run before the old lease is removed; failure leaves the old lease intact. */
46
+ beforeReplace?: (lease: CwdLeaseRecord) => Promise<void>;
47
+ }
48
+
49
+ export interface CwdLeaseAcquireOptions {
50
+ handoff?: CwdLeaseHandoff;
51
+ takeover?: CwdLeaseTakeover;
52
+ }
53
+
39
54
  export interface CwdLeaseHandle {
40
55
  readonly record: CwdLeaseRecord;
56
+ /** Task id whose lease was explicitly handed off/taken over, if any. */
57
+ readonly replacedTaskId?: string;
41
58
  updateWorker(worker: CwdLeaseWorker): Promise<void>;
42
59
  release(): Promise<void>;
43
60
  }
@@ -63,7 +80,9 @@ export class CwdLeaseStore {
63
80
  return this.#directory;
64
81
  }
65
82
 
66
- async acquire(cwd: string, taskId: string, transport: CwdLeaseTransport, options: { handoff?: CwdLeaseHandoff } = {}): Promise<CwdLeaseHandle> {
83
+ async acquire(cwd: string, taskId: string, transport: CwdLeaseTransport, options: CwdLeaseAcquireOptions = {}): Promise<CwdLeaseHandle> {
84
+ assertTaskId(taskId);
85
+ if (options.takeover) assertTaskId(options.takeover.taskId);
67
86
  const canonicalCwd = await realpath(resolve(cwd));
68
87
  const now = new Date().toISOString();
69
88
  const lease: CwdLeaseRecord = {
@@ -77,6 +96,7 @@ export class CwdLeaseStore {
77
96
  updatedAt: now,
78
97
  worker: { transport },
79
98
  };
99
+ let replacedTaskId: string | undefined;
80
100
  await this.#withLock(async () => {
81
101
  const leases = await this.#readAll();
82
102
  let handoffLease: CwdLeaseRecord | undefined;
@@ -88,23 +108,35 @@ export class CwdLeaseStore {
88
108
  continue;
89
109
  }
90
110
  if (!pathsOverlap(existing.cwd, canonicalCwd)) continue;
111
+ if (options.takeover?.taskId === existing.taskId
112
+ && existing.cwd === canonicalCwd
113
+ && await canTakeoverLease(existing)) {
114
+ await options.takeover.beforeReplace?.(existing);
115
+ replacedTaskId = existing.taskId;
116
+ await rm(this.#path(existing.leaseId), { force: true });
117
+ continue;
118
+ }
91
119
  throw new Error(`working-directory lease is held by task ${existing.taskId}: ${redactText(existing.cwd)}`);
92
120
  }
93
121
  await this.#write(lease);
94
- if (handoffLease) await rm(this.#path(handoffLease.leaseId), { force: true });
122
+ if (handoffLease) {
123
+ replacedTaskId = handoffLease.taskId;
124
+ await rm(this.#path(handoffLease.leaseId), { force: true });
125
+ }
95
126
  });
96
- return this.#handle(lease);
127
+ return this.#handle(lease, replacedTaskId);
97
128
  }
98
129
 
99
130
  async list(): Promise<CwdLeaseRecord[]> {
100
131
  return this.#withLock(() => this.#readAll());
101
132
  }
102
133
 
103
- #handle(initial: CwdLeaseRecord): CwdLeaseHandle {
134
+ #handle(initial: CwdLeaseRecord, replacedTaskId?: string): CwdLeaseHandle {
104
135
  let current = { ...initial, worker: initial.worker ? { ...initial.worker } : undefined };
105
136
  let released = false;
106
137
  return {
107
138
  get record() { return current; },
139
+ get replacedTaskId() { return replacedTaskId; },
108
140
  updateWorker: async (worker) => {
109
141
  if (released) throw new Error("cwd lease has already been released");
110
142
  assertWorker(worker);
@@ -157,8 +189,8 @@ export class CwdLeaseStore {
157
189
  async #write(lease: CwdLeaseRecord): Promise<void> {
158
190
  await this.#ensureDirectory();
159
191
  const target = this.#path(lease.leaseId);
160
- const temporary = `${target}.${process.pid}.${Date.now()}.tmp`;
161
- await writeFile(temporary, `${JSON.stringify(lease, null, 2)}\n`, { encoding: "utf8", mode: 0o600 });
192
+ const temporary = `${target}.${process.pid}.${Date.now()}.${randomUUID()}.tmp`;
193
+ await writeFile(temporary, `${JSON.stringify(lease, null, 2)}\n`, { encoding: "utf8", mode: 0o600, flag: "wx" });
162
194
  await rename(temporary, target);
163
195
  await chmod(target, 0o600);
164
196
  }
@@ -252,6 +284,10 @@ async function processIdentityLive(pid: number, expectedStartTime?: string): Pro
252
284
  const currentStartTime = await processStartTime(pid);
253
285
  if (currentStartTime && expectedStartTime) return currentStartTime === expectedStartTime;
254
286
  if (currentStartTime) return true;
287
+ return processExists(pid);
288
+ }
289
+
290
+ async function processExists(pid: number): Promise<boolean> {
255
291
  try {
256
292
  process.kill(pid, 0);
257
293
  return true;
@@ -260,6 +296,55 @@ async function processIdentityLive(pid: number, expectedStartTime?: string): Pro
260
296
  }
261
297
  }
262
298
 
299
+ async function processGroupExists(pid: number): Promise<boolean> {
300
+ try {
301
+ process.kill(-pid, 0);
302
+ return true;
303
+ } catch (error) {
304
+ return error instanceof Error && /EPERM/u.test(error.message);
305
+ }
306
+ }
307
+
308
+ async function cgroupHasProcesses(path: string): Promise<boolean> {
309
+ // Lease files are untrusted state. Never read an arbitrary path during
310
+ // takeover; only a real, canonical cgroup below the kernel cgroup root is
311
+ // eligible for this check. Missing/unreadable evidence is not proof of an
312
+ // empty cgroup: a descendant may have escaped before the cgroup disappeared.
313
+ const cgroupRoot = resolve("/sys/fs/cgroup");
314
+ const cgroupPath = resolve(path);
315
+ if (cgroupPath === cgroupRoot || !cgroupPath.startsWith(`${cgroupRoot}/`)) return true;
316
+ try {
317
+ const cgroupInfo = await lstat(cgroupPath);
318
+ if (!cgroupInfo.isDirectory() || cgroupInfo.isSymbolicLink()) return true;
319
+ const canonicalPath = await realpath(cgroupPath);
320
+ if (canonicalPath !== cgroupPath || !canonicalPath.startsWith(`${cgroupRoot}/`)) return true;
321
+ const procsPath = join(canonicalPath, "cgroup.procs");
322
+ const procsInfo = await lstat(procsPath);
323
+ if (!procsInfo.isFile() || procsInfo.isSymbolicLink()) return true;
324
+ const contents = await readFile(procsPath, "utf8");
325
+ return contents.split(/\s+/u).some((pid) => /^\d+$/u.test(pid));
326
+ } catch {
327
+ return true;
328
+ }
329
+ }
330
+
331
+ async function canTakeoverLease(lease: CwdLeaseRecord): Promise<boolean> {
332
+ // The old supervisor owner must be gone. A dead owner is not enough when
333
+ // the detached Worker itself is still alive.
334
+ if (await processExists(lease.ownerPid)) return false;
335
+ const worker = lease.worker;
336
+ if (!worker) return false;
337
+ if (worker.transport === "tmux") return false;
338
+ if (!worker.pid) return false;
339
+ if (await processExists(worker.pid)) return false;
340
+ if (await processGroupExists(worker.pid)) return false;
341
+ // A process-group check cannot see a setsid descendant. Explicit takeover
342
+ // therefore requires the verified cgroup boundary used by the adapter.
343
+ if (!worker.cgroupPath || !resolve(worker.cgroupPath).startsWith(`${resolve("/sys/fs/cgroup")}/`)) return false;
344
+ if (await cgroupHasProcesses(worker.cgroupPath)) return false;
345
+ return true;
346
+ }
347
+
263
348
  async function processStartTime(pid: number): Promise<string | undefined> {
264
349
  try {
265
350
  const statText = await readFile(`/proc/${pid}/stat`, "utf8");
@@ -288,6 +373,10 @@ async function matchesHandoff(lease: CwdLeaseRecord, handoff: CwdLeaseHandoff):
288
373
  return !(await processIdentityLive(lease.ownerPid, lease.ownerStartTime));
289
374
  }
290
375
 
376
+ function assertTaskId(taskId: string): void {
377
+ if (!/^[0-9a-f-]{36}$/iu.test(taskId)) throw new Error("invalid cwd lease task id");
378
+ }
379
+
291
380
  function assertWorker(worker: CwdLeaseWorker): void {
292
381
  if (!["process-pipe", "jsonl", "pty", "tmux"].includes(worker.transport)
293
382
  || (worker.pid !== undefined && (!Number.isSafeInteger(worker.pid) || worker.pid < 1))