pi-claude-supervisor 0.5.2 → 0.5.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -1
- package/README.cn.md +45 -36
- package/README.md +70 -41
- package/docs/architecture.md +65 -38
- package/docs/automation-hardening-plan.md +23 -22
- package/docs/autonomy-target.md +127 -0
- package/docs/engineering-plan.md +136 -138
- package/docs/implementation-review.md +29 -14
- package/docs/independent-review.md +23 -18
- package/docs/releasing.md +7 -2
- package/docs/testing.md +36 -22
- package/package.json +1 -1
- package/src/acceptance.ts +16 -0
- package/src/config.ts +31 -0
- package/src/decision-session-store.ts +19 -1
- package/src/decision-worker.ts +46 -18
- package/src/index.ts +37 -48
- package/src/notifications.ts +29 -11
- package/src/policy.ts +196 -21
- package/src/reviewer.ts +26 -1
- package/src/state.ts +7 -6
- package/src/supervisor.ts +329 -119
- package/src/types.ts +23 -0
- package/src/verifier.ts +88 -6
- package/src/worker/environment.ts +169 -0
- package/src/worker/process-adapter.ts +15 -0
|
@@ -1,15 +1,30 @@
|
|
|
1
1
|
# Implementation Review
|
|
2
2
|
|
|
3
3
|
An independent read-only reviewer examined the implementation before the final
|
|
4
|
-
hardening pass. The review found
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
4
|
+
hardening pass. The review found release blockers in automatic startup validation,
|
|
5
|
+
shell-policy lexical handling, malformed custom Reviewer results, credential
|
|
6
|
+
filtering, and stale security documentation. Those findings are addressed by the
|
|
7
|
+
current implementation and regression tests.
|
|
8
|
+
|
|
9
|
+
Automatic mode is fail-closed at its supported Worker boundary: it requires a
|
|
10
|
+
validated non-bare Git worktree, an existing full baseline commit, a non-protected
|
|
11
|
+
branch, the Claude JSONL transport, and the bare `claude`/`claude.exe` command name.
|
|
12
|
+
Startup resolves and pins an operator-owned, non-writable executable path (or an
|
|
13
|
+
explicit `PI_CLAUDE_SUPERVISOR_TRUSTED_CLAUDE` path), and the final pre-spawn check
|
|
14
|
+
compares the current repository HEAD with the exact startup HEAD. The built-in Claude
|
|
15
|
+
path requests Claude Code's fail-closed Bash sandbox with no outbound domains.
|
|
16
|
+
Arbitrary custom executables and explicit paths are not admitted to automatic mode;
|
|
17
|
+
manual/custom integrations remain responsible for their own host sandbox and network
|
|
18
|
+
boundary.
|
|
8
19
|
|
|
9
20
|
## Findings addressed in this pass
|
|
10
21
|
|
|
11
|
-
- Policy now evaluates the executable and argv
|
|
12
|
-
|
|
22
|
+
- Policy now evaluates every shell argument as well as the executable and argv
|
|
23
|
+
together; dynamic arguments are denied because their capability cannot be checked,
|
|
24
|
+
including Claude permission-bypass flags.
|
|
25
|
+
- Automatic startup pins a secure resolved Claude executable identity and rejects
|
|
26
|
+
explicit paths, persists that identity for recovery, and rechecks the exact startup
|
|
27
|
+
HEAD through the built-in adapter's final `preSpawnCheck` immediately before spawn.
|
|
13
28
|
- Worker and verifier processes use a minimal environment; explicit worker
|
|
14
29
|
variables can be selected with `PI_CLAUDE_SUPERVISOR_WORKER_ENV` or an
|
|
15
30
|
embedding caller's `WorkerStartInput.env`.
|
|
@@ -24,12 +39,12 @@ not blockers for the lifecycle and recovery track.
|
|
|
24
39
|
|
|
25
40
|
## Residual risks and follow-up hardening
|
|
26
41
|
|
|
27
|
-
These are verified limitations and follow-up work
|
|
28
|
-
|
|
42
|
+
These are verified limitations and follow-up work after the automatic boundary
|
|
43
|
+
hardening:
|
|
29
44
|
|
|
30
|
-
-
|
|
31
|
-
|
|
32
|
-
|
|
45
|
+
- The Claude Code sandbox is a requested runtime boundary and is fail-closed when
|
|
46
|
+
unavailable; it is not a substitute for a host-level sandbox, lower-privilege
|
|
47
|
+
account, or container policy for manual integrations.
|
|
33
48
|
- Event contents can contain worker output or user messages; common credential
|
|
34
49
|
patterns are now redacted and sequence recovery is persisted, but broader
|
|
35
50
|
structured-secret coverage remains follow-up work.
|
|
@@ -40,9 +55,9 @@ lifecycle track:
|
|
|
40
55
|
- Fault injection coverage now includes lifecycle-log failure, SIGTERM refusal,
|
|
41
56
|
blocked stdin and managed orphan descendants; shutdown cleanup under injected
|
|
42
57
|
adapter failure remains a follow-up failure-injection case.
|
|
43
|
-
- The default transport is process-pipe. Claude JSONL
|
|
44
|
-
|
|
45
|
-
|
|
58
|
+
- The default manual transport is process-pipe. Claude JSONL is mandatory for
|
|
59
|
+
automatic mode; the remaining transport work concerns signal, shutdown and
|
|
60
|
+
descendant-cleanup evidence rather than weakening the automatic boundary.
|
|
46
61
|
|
|
47
62
|
## Evidence
|
|
48
63
|
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# 独立子 Agent 方案评审报告
|
|
2
2
|
|
|
3
|
+
> 本评审按已确认目标更新:本地开发完全无人值守;远程 push 和 main/integration merge 必须经过独立边界。文中旧的“人工升级”表述表示候选不可发布/可挂起状态,不是要求人工在线。
|
|
4
|
+
|
|
3
5
|
> 评审对象:`requirement.txt`、`docs/engineering-plan.md`
|
|
4
6
|
> 评审重点:复用现有扩展、减少重复开发、形成可整合架构并验证到生产
|
|
5
7
|
|
|
@@ -18,7 +20,7 @@
|
|
|
18
20
|
- `pi-interactive-shell` 如果通过版本、API、许可证和故障测试,应优先复用其 PTY 和人工接管实现。
|
|
19
21
|
- `pi-claude-code`、`pi-harness-delegate` 在完成供应链、API 和故障语义审计前,不作为核心依赖。
|
|
20
22
|
|
|
21
|
-
**结论:架构方向 GO;先完成固定版本 Spike
|
|
23
|
+
**结论:架构方向 GO;先完成固定版本 Spike、生命周期和故障恢复主线。当前自动模式已把直接 Claude 的 fail-closed sandbox、无出站域名、JSONL transport、Git baseline 和分支校验作为启动边界,并拒绝任意自定义可执行文件;手动集成的 host-level 低权限和网络隔离仍是后续加固。另经产品确认,本地开发必须完全无人值守;远程 push 和 main/integration merge 是 Worker 不具备权限的独立边界。详见 [autonomy-target.md](autonomy-target.md)。**
|
|
22
24
|
|
|
23
25
|
## 2. 外部参考源审查结果
|
|
24
26
|
|
|
@@ -163,7 +165,7 @@ interface WorkerAdapter {
|
|
|
163
165
|
- Policy Gate;
|
|
164
166
|
- 预算和最大轮数;
|
|
165
167
|
- 重复停顿检测;
|
|
166
|
-
-
|
|
168
|
+
- 自动决策、重试和候选挂起;
|
|
167
169
|
- 事件日志;
|
|
168
170
|
- 证据收集;
|
|
169
171
|
- 独立验收;
|
|
@@ -194,11 +196,11 @@ interface WorkerAdapter {
|
|
|
194
196
|
创建任务
|
|
195
197
|
→ 启动 Claude
|
|
196
198
|
→ 读取输出
|
|
197
|
-
→
|
|
198
|
-
→
|
|
199
|
-
→ 恢复执行
|
|
200
|
-
→ 测试
|
|
199
|
+
→ 自动处理普通提问和决策
|
|
200
|
+
→ 测试 / 修复 / 重试
|
|
201
201
|
→ 独立验收
|
|
202
|
+
→ 本地提交候选
|
|
203
|
+
→ 独立远程/main 边界
|
|
202
204
|
→ 输出报告
|
|
203
205
|
```
|
|
204
206
|
|
|
@@ -219,8 +221,8 @@ interface WorkerAdapter {
|
|
|
219
221
|
MVP 必须满足:
|
|
220
222
|
|
|
221
223
|
- 单仓库、单 worktree、单 Worker;
|
|
222
|
-
-
|
|
223
|
-
-
|
|
224
|
+
- 本地开发动作按任务授权自动继续、修复或挂起;
|
|
225
|
+
- 不授予 Worker 远程 push 或 main/integration merge 权限;
|
|
224
226
|
- 事件可以完整回放;
|
|
225
227
|
- 人工 takeover 后零自动发送;
|
|
226
228
|
- 验收失败绝不进入 `COMPLETE`;
|
|
@@ -235,17 +237,18 @@ MVP 必须满足:
|
|
|
235
237
|
|
|
236
238
|
- 依赖 lockfile 和 SBOM;
|
|
237
239
|
- 包来源校验;
|
|
238
|
-
-
|
|
239
|
-
-
|
|
240
|
-
-
|
|
240
|
+
- 手动/自定义集成的最小权限和 host-level sandbox;
|
|
241
|
+
- 手动/自定义集成的网络白名单;自动 Claude 路径已请求无出站域名并在不可用时失败;
|
|
242
|
+
- 更广泛的密钥隔离;
|
|
241
243
|
- 日志脱敏;
|
|
242
244
|
- 成本和时间告警;
|
|
243
245
|
- 灰度 feature flag;
|
|
244
246
|
- 可随时关闭自动化;
|
|
245
|
-
-
|
|
247
|
+
- 故障回滚、候选挂起和可选通知。
|
|
246
248
|
|
|
247
|
-
|
|
248
|
-
|
|
249
|
+
其中手动/自定义集成的 host-level 低权限、sandbox 和网络白名单不阻塞当前生命周期验证;
|
|
250
|
+
自动模式不接受没有 Claude sandbox 边界的自定义 Worker。本地运行权限由任务和调用方授权策略
|
|
251
|
+
控制,远程 push/main merge 仍由独立边界控制。
|
|
249
252
|
|
|
250
253
|
### 5.4 建议 Go / No-Go 门槛
|
|
251
254
|
|
|
@@ -262,11 +265,11 @@ MVP 必须满足:
|
|
|
262
265
|
- 通过一组固定任务的成功率、恢复率和费用门槛。
|
|
263
266
|
|
|
264
267
|
未满足生命周期、恢复、审计和独立验收门槛时只能称为 PoC 或生产候选;
|
|
265
|
-
|
|
268
|
+
候选不得进入远程或 main/integration 分支。安全加固项的完成度应单独标注,不能以本地无人值守目标替代安全证据。
|
|
266
269
|
|
|
267
270
|
## 6. 回滚方案
|
|
268
271
|
|
|
269
|
-
任何以下情况发生时,立即停止自动发送并进入 `
|
|
272
|
+
任何以下情况发生时,立即停止自动发送并进入 `blocked`/`candidate_failed` 或恢复检查状态:
|
|
270
273
|
|
|
271
274
|
- 状态不一致;
|
|
272
275
|
- 重复发送指令;
|
|
@@ -277,15 +280,17 @@ MVP 必须满足:
|
|
|
277
280
|
- 发现权限越界或密钥风险;
|
|
278
281
|
- watchdog 与 Worker 状态矛盾。
|
|
279
282
|
|
|
283
|
+
上述情况应保留现场并可选发送通知,但不得把人工在线作为恢复前提,也不得因此放行远程 push 或 main/integration merge。
|
|
284
|
+
|
|
280
285
|
回滚动作:
|
|
281
286
|
|
|
282
287
|
1. 停止自动控制;
|
|
283
288
|
2. 保留 worktree、事件日志和原始输出;
|
|
284
289
|
3. 终止或交还 Worker 控制权;
|
|
285
|
-
4.
|
|
290
|
+
4. 将任务挂起并保留候选证据;
|
|
286
291
|
5. 使用锁定的旧 Adapter 或直接使用 Claude CLI;
|
|
287
292
|
6. 生成故障报告;
|
|
288
|
-
7.
|
|
293
|
+
7. 未完成根因分析前,不重新打开自动化开关;需要人工接管时作为可选后续操作。
|
|
289
294
|
|
|
290
295
|
第三方包升级必须通过 lockfile、独立 worktree、回放测试和 feature flag 灰度,不能直接替换生产版本。
|
|
291
296
|
|
package/docs/releasing.md
CHANGED
|
@@ -1,9 +1,14 @@
|
|
|
1
1
|
# Releasing
|
|
2
2
|
|
|
3
|
+
Local automation may develop, test, repair and commit a candidate, but it does not
|
|
4
|
+
push to a remote or merge into `main`. Remote repository entry and main-branch
|
|
5
|
+
integration are independent protected boundaries; this release workflow is the
|
|
6
|
+
publication path after those boundaries pass. See [the confirmed autonomy target](autonomy-target.md).
|
|
7
|
+
|
|
3
8
|
## Pull request gate
|
|
4
9
|
|
|
5
|
-
All changes must enter `main` through a pull request
|
|
6
|
-
PR title, for example:
|
|
10
|
+
All changes must enter `main` through a pull request and its independent CI/review
|
|
11
|
+
boundary. Use a Conventional Commit PR title, for example:
|
|
7
12
|
|
|
8
13
|
```text
|
|
9
14
|
feat: persist Decision Worker sessions
|
package/docs/testing.md
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# Testing
|
|
2
2
|
|
|
3
|
+
> Autonomy target: local editing, testing, repair and local commits run without a human being online. Invalid output, unavailable evidence, duplicate findings, P0/P1 findings and exhausted budgets become parked/non-publishable candidates rather than synchronous human gates. Remote push and main/integration merge remain independent-boundary tests. See [autonomy-target.md](autonomy-target.md).
|
|
4
|
+
|
|
3
5
|
## Local checks
|
|
4
6
|
|
|
5
7
|
```bash
|
|
@@ -17,7 +19,7 @@ the published TypeScript source directly and there is no second runtime bundle.
|
|
|
17
19
|
|
|
18
20
|
## Test layers
|
|
19
21
|
|
|
20
|
-
- `policy.test.ts`: deterministic allow
|
|
22
|
+
- `policy.test.ts`: deterministic local allow and hard-boundary deny behavior; legacy approval cannot override denial.
|
|
21
23
|
- `state.test.ts`: legal and illegal lifecycle transitions.
|
|
22
24
|
- `events.test.ts`: ordered JSONL persistence, sequence recovery and credential-shaped redaction.
|
|
23
25
|
- `decision-session-store.test.ts`: atomic Decision Worker task mapping, permissions, restart discovery and corrupt-record isolation.
|
|
@@ -57,13 +59,12 @@ owned tmux turns, pause/resume, and restart re-adoption.
|
|
|
57
59
|
The adapter regression suite also verifies event subscription, parsed
|
|
58
60
|
`permission_request` events, and the exact nested `control_response` envelope.
|
|
59
61
|
The automation spike additionally exercises a real Pi SDK Decision Worker with
|
|
60
|
-
Claude: ordinary completion, harmless Bash permission
|
|
62
|
+
Claude: ordinary completion, harmless Bash permission handling, and an
|
|
61
63
|
`AskUserQuestion` denial-to-text fallback followed by automatic verification.
|
|
62
64
|
A local pinned-CLI run completed all three scenarios with `state=completed`,
|
|
63
65
|
`verified=true`, and zero human interventions. Provider/model latency can still
|
|
64
|
-
cause a later run to fail closed as
|
|
65
|
-
Worker or Reviewer timeout; this is evidence for the manual spike only, not a CI
|
|
66
|
-
guarantee.
|
|
66
|
+
cause a later run to fail closed as a parked/non-publishable candidate after the bounded Decision
|
|
67
|
+
Worker or Reviewer timeout; this is evidence for the manual spike only, not a CI guarantee.
|
|
67
68
|
The extension persists each automatic Decision Worker session as Pi JSONL plus a
|
|
68
69
|
0600 task mapping. Automatic startup preflights the state/lease directories, cwd,
|
|
69
70
|
worker executable, transport dependency and required cgroup boundary before model
|
|
@@ -92,7 +93,7 @@ PI_CLAUDE_SUPERVISOR_REAL_CLAUDE=1 npm run spike:tmux-interactive
|
|
|
92
93
|
The tmux spike is gated, authenticated, and excluded from normal CI. It uses
|
|
93
94
|
plan mode with a fixed `opus` model, records only protocol metadata, and
|
|
94
95
|
verifies three real Claude turns, exact screen-result markers, pause/resume,
|
|
95
|
-
|
|
96
|
+
manual takeover, direct PTY input, owned detach,
|
|
96
97
|
and identity-bound restart re-adoption. The interactive spike uses a fresh temporary
|
|
97
98
|
cwd to verify Claude's trust prompt, a real Bash permission prompt, an allow-once
|
|
98
99
|
response, and an exact result marker; it also records metadata only.
|
|
@@ -117,21 +118,33 @@ relevant checks from a clean host perspective.
|
|
|
117
118
|
|
|
118
119
|
Automatic mode is enabled with `PI_CLAUDE_SUPERVISOR_MODE=auto`; it defaults to
|
|
119
120
|
JSONL and routes `result`, permission, and process-exit events to the persistent
|
|
120
|
-
Pi Decision Worker.
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
121
|
+
Pi Decision Worker. Task autonomy defaults to unattended local work, a required
|
|
122
|
+
local commit on a non-protected task branch and two bounded Decision Worker retries. Configure
|
|
123
|
+
`PI_CLAUDE_SUPERVISOR_REQUIRE_LOCAL_COMMIT=0` or task `autonomy.requireLocalCommit`
|
|
124
|
+
only to disable the local-commit deliverability check; automatic mode still requires a Git
|
|
125
|
+
baseline and non-protected worktree. An explicit `PI_CLAUDE_SUPERVISOR_TRANSPORT=tmux` selection
|
|
126
|
+
remains screen-based and does not use the JSONL permission protocol. `process-pipe` remains the manual compatibility mode. Candidate/failure notification is optional and outbound-only through
|
|
127
|
+
`PI_CLAUDE_SUPERVISOR_HUMAN_WEBHOOK_URL`; it is not a synchronous approval
|
|
128
|
+
callback. Approval callbacks are deliberately not accepted without a separately
|
|
124
129
|
authenticated endpoint.
|
|
125
130
|
|
|
126
131
|
The tmux transport is selected with `PI_CLAUDE_SUPERVISOR_TRANSPORT=tmux`.
|
|
127
|
-
Automatic mode rejects
|
|
128
|
-
bounded decisions and repair.
|
|
129
|
-
|
|
130
|
-
|
|
132
|
+
Automatic mode rejects explicit `process-pipe` and `tmux` transports; use JSONL for
|
|
133
|
+
bounded decisions and repair. Built-in Claude workers also receive a fail-closed
|
|
134
|
+
sandbox setting (`failIfUnavailable`, `allowUnsandboxedCommands=false`, no outbound
|
|
135
|
+
network domains); verify that startup fails if the sandbox cannot be initialized.
|
|
136
|
+
Automatic startup also requires a full existing Git baseline, non-bare non-protected
|
|
137
|
+
worktree and the bare `claude`/`claude.exe` command name. It resolves and pins an
|
|
138
|
+
operator-owned, non-writable executable path (or the path configured by
|
|
139
|
+
`PI_CLAUDE_SUPERVISOR_TRUSTED_CLAUDE`), rejects explicit/custom executable paths, and
|
|
140
|
+
compares the exact startup HEAD again immediately before spawn. Before release, verify:
|
|
141
|
+
private-socket attach,
|
|
142
|
+
multi-line paste, prompt stability while
|
|
143
|
+
Claude is busy, trust/permission policy handling, duplicate send prevention, pane
|
|
131
144
|
replacement refusal, pause/resume, owned-session stop, adopted-session
|
|
132
145
|
detach/re-adoption, bounded shutdown, and Pi shutdown without closing an
|
|
133
146
|
attached window. The two gated real-Claude spikes above cover the trust prompt,
|
|
134
|
-
permission prompt, exact output,
|
|
147
|
+
permission prompt, exact output, optional takeover, and adopted detach paths. Use
|
|
135
148
|
`--permission-mode plan` and read-only tools for ordinary live Claude checks;
|
|
136
149
|
the interactive spike is restricted to one harmless `rm -f` in a disposable
|
|
137
150
|
fresh directory. Do not run JSONL and tmux control against the same Claude
|
|
@@ -165,7 +178,7 @@ the managed process group.
|
|
|
165
178
|
|
|
166
179
|
The `v0.5.0` implementation of the acceptance—independent Review—repair—reacceptance
|
|
167
180
|
loop is shipped. The `v0.5.1` real read-only drill reached acceptance and independent
|
|
168
|
-
Review, then correctly
|
|
181
|
+
Review, then correctly produced a non-publishable candidate after two P1 and two P2 findings under the then-current human-gated compatibility path.
|
|
169
182
|
A separate real edit-capable Claude Code `2.1.270` drill then exercised one bounded
|
|
170
183
|
acceptance failure, repair turn, reacceptance and independent Reviewer `pass` in an
|
|
171
184
|
isolated temporary worktree. The current hardening plan and evidence paths are recorded
|
|
@@ -182,8 +195,8 @@ Deterministic tests must cover:
|
|
|
182
195
|
- legacy text tasks normalized to a Goal with the default `git diff --check`;
|
|
183
196
|
- multiple required/optional checks with bounded output, timeout and exit-code evidence;
|
|
184
197
|
- independent read-only Reviewer pass/revise/human results;
|
|
185
|
-
- invalid Reviewer JSON and Reviewer API failure
|
|
186
|
-
- repair rounds, repeated finding detection, P0/P1
|
|
198
|
+
- invalid Reviewer JSON and Reviewer API failure becoming a parked/non-publishable candidate without requiring a live callback;
|
|
199
|
+
- repair rounds, repeated finding detection, P0/P1 parking and repair-budget exhaustion;
|
|
187
200
|
- non-persistent JSONL verification failure without duplicate terminal transitions;
|
|
188
201
|
- repairable-but-not-persistent JSONL multi-turn repair;
|
|
189
202
|
- stop and Pi shutdown from `verifying`, including Decision Worker closure and cwd lease release;
|
|
@@ -191,7 +204,8 @@ Deterministic tests must cover:
|
|
|
191
204
|
|
|
192
205
|
Reviewer sessions use only `read`, `grep`, `find` and `ls`; they must not modify
|
|
193
206
|
the worktree or send Worker input. Decision Worker and Reviewer model calls are
|
|
194
|
-
bounded; timeout or API failure
|
|
207
|
+
bounded; timeout or API failure parks the candidate instead of auto-completing or
|
|
208
|
+
requiring a human to be online. Review reports
|
|
195
209
|
are persisted as bounded event payloads and are not treated as permission grants.
|
|
196
210
|
|
|
197
211
|
### JSONL protocol and replay fixtures
|
|
@@ -233,7 +247,7 @@ Required deterministic and integration coverage:
|
|
|
233
247
|
- independent child acceptance followed by root-task aggregate acceptance and Review;
|
|
234
248
|
- single-child recovery, whole-graph recovery and Pi shutdown during scheduling;
|
|
235
249
|
- no child can grant permissions, send control input to another child or bypass Policy Gate;
|
|
236
|
-
-
|
|
250
|
+
- independent integration in a separate worktree; the Worker has no remote push or main/integration merge authority, and no candidate bypasses that boundary.
|
|
237
251
|
|
|
238
252
|
The multi-worker gate should be added only after the pinned single-worker stability
|
|
239
253
|
and recovery gates pass. CI should use fake Workers and replay fixtures; authenticated
|
|
@@ -243,8 +257,8 @@ Claude multi-worker Spikes remain manual and version-pinned.
|
|
|
243
257
|
|
|
244
258
|
A live review must record the exact Claude executable/version, transport, cgroup mode,
|
|
245
259
|
permission flags, task id, acceptance result, Reviewer result, cleanup status and cwd lease
|
|
246
|
-
status. A
|
|
247
|
-
pass by retrying the same task automatically.
|
|
260
|
+
status. A parked/non-publishable result is a valid safety outcome and must not be converted into a
|
|
261
|
+
pass by retrying the same task automatically or by treating the absence of a human callback as approval.
|
|
248
262
|
|
|
249
263
|
For the hardening release, the completed gate record is:
|
|
250
264
|
|
package/package.json
CHANGED
package/src/acceptance.ts
CHANGED
|
@@ -18,6 +18,7 @@ export function normalizeTaskSpec(value: unknown, fallbackGoal: string): TaskSpe
|
|
|
18
18
|
forbidden: stringList(source.forbidden, "forbidden"),
|
|
19
19
|
acceptance: normalizeChecks(source.acceptance),
|
|
20
20
|
maxRepairRounds: normalizeRepairRounds(source.maxRepairRounds),
|
|
21
|
+
autonomy: normalizeAutonomy(source.autonomy),
|
|
21
22
|
};
|
|
22
23
|
}
|
|
23
24
|
|
|
@@ -82,3 +83,18 @@ function normalizeRepairRounds(value: unknown): number {
|
|
|
82
83
|
if (typeof value !== "number" || !Number.isSafeInteger(value) || value < 0 || value > 10) throw new Error("maxRepairRounds must be between 0 and 10");
|
|
83
84
|
return value;
|
|
84
85
|
}
|
|
86
|
+
|
|
87
|
+
function normalizeAutonomy(value: unknown): TaskSpec["autonomy"] {
|
|
88
|
+
if (value === undefined) return { unattended: true, requireLocalCommit: true, maxDecisionRetries: 2 };
|
|
89
|
+
if (!value || typeof value !== "object" || Array.isArray(value)) throw new Error("task spec autonomy must be an object");
|
|
90
|
+
const source = value as Record<string, unknown>;
|
|
91
|
+
if (source.unattended !== undefined && typeof source.unattended !== "boolean") throw new Error("task spec autonomy.unattended must be boolean");
|
|
92
|
+
if (source.requireLocalCommit !== undefined && typeof source.requireLocalCommit !== "boolean") throw new Error("task spec autonomy.requireLocalCommit must be boolean");
|
|
93
|
+
const retries = source.maxDecisionRetries ?? 2;
|
|
94
|
+
if (typeof retries !== "number" || !Number.isSafeInteger(retries) || retries < 0 || retries > 10) throw new Error("task spec autonomy.maxDecisionRetries must be between 0 and 10");
|
|
95
|
+
return {
|
|
96
|
+
unattended: source.unattended !== false,
|
|
97
|
+
requireLocalCommit: source.requireLocalCommit !== false,
|
|
98
|
+
maxDecisionRetries: retries,
|
|
99
|
+
};
|
|
100
|
+
}
|
package/src/config.ts
CHANGED
|
@@ -10,14 +10,32 @@ const allowed = new Set([
|
|
|
10
10
|
"PI_CLAUDE_SUPERVISOR_CGROUP_MODE",
|
|
11
11
|
"PI_CLAUDE_SUPERVISOR_TMUX_SOCKET",
|
|
12
12
|
"PI_CLAUDE_SUPERVISOR_WORKER",
|
|
13
|
+
"PI_CLAUDE_SUPERVISOR_TRUSTED_CLAUDE",
|
|
13
14
|
"PI_CLAUDE_SUPERVISOR_STATE_DIR",
|
|
14
15
|
"PI_CLAUDE_SUPERVISOR_CWD_LEASE_DIR",
|
|
15
16
|
"PI_CLAUDE_SUPERVISOR_WORKER_ENV",
|
|
16
17
|
"PI_CLAUDE_SUPERVISOR_HUMAN_WEBHOOK_URL",
|
|
17
18
|
"PI_CLAUDE_SUPERVISOR_HUMAN_WEBHOOK_FORMAT",
|
|
18
19
|
"PI_CLAUDE_SUPERVISOR_HUMAN_WEBHOOK_SECRET",
|
|
20
|
+
"PI_CLAUDE_SUPERVISOR_UNATTENDED",
|
|
21
|
+
"PI_CLAUDE_SUPERVISOR_REQUIRE_LOCAL_COMMIT",
|
|
22
|
+
"PI_CLAUDE_SUPERVISOR_MAX_DECISION_RETRIES",
|
|
19
23
|
]);
|
|
20
24
|
|
|
25
|
+
export interface AutonomyDefaults {
|
|
26
|
+
unattended: boolean;
|
|
27
|
+
requireLocalCommit: boolean;
|
|
28
|
+
maxDecisionRetries: number;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export function autonomyDefaults(env: NodeJS.ProcessEnv = process.env): AutonomyDefaults {
|
|
32
|
+
return {
|
|
33
|
+
unattended: readBoolean(env.PI_CLAUDE_SUPERVISOR_UNATTENDED, true),
|
|
34
|
+
requireLocalCommit: readBoolean(env.PI_CLAUDE_SUPERVISOR_REQUIRE_LOCAL_COMMIT, true),
|
|
35
|
+
maxDecisionRetries: readBoundedInteger(env.PI_CLAUDE_SUPERVISOR_MAX_DECISION_RETRIES, 2, 0, 10),
|
|
36
|
+
};
|
|
37
|
+
}
|
|
38
|
+
|
|
21
39
|
export function loadSupervisorEnvironment(): string | undefined {
|
|
22
40
|
const path = process.env.PI_CLAUDE_SUPERVISOR_ENV_FILE ?? join(homedir(), ".config", "pi-claude-supervisor", "env");
|
|
23
41
|
if (!existsSync(path)) return undefined;
|
|
@@ -37,3 +55,16 @@ export function loadSupervisorEnvironment(): string | undefined {
|
|
|
37
55
|
return undefined;
|
|
38
56
|
}
|
|
39
57
|
}
|
|
58
|
+
|
|
59
|
+
function readBoolean(value: string | undefined, fallback: boolean): boolean {
|
|
60
|
+
if (value === undefined) return fallback;
|
|
61
|
+
if (/^(?:1|true|yes|on)$/iu.test(value.trim())) return true;
|
|
62
|
+
if (/^(?:0|false|no|off)$/iu.test(value.trim())) return false;
|
|
63
|
+
return fallback;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function readBoundedInteger(value: string | undefined, fallback: number, minimum: number, maximum: number): number {
|
|
67
|
+
if (value === undefined) return fallback;
|
|
68
|
+
const parsed = Number(value);
|
|
69
|
+
return Number.isSafeInteger(parsed) && parsed >= minimum && parsed <= maximum ? parsed : fallback;
|
|
70
|
+
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { randomUUID } from "node:crypto";
|
|
2
2
|
import { chmod, lstat, mkdir, readdir, readFile, rename, rm, writeFile } from "node:fs/promises";
|
|
3
|
-
import { basename, dirname, join, resolve } from "node:path";
|
|
3
|
+
import { basename, dirname, isAbsolute, join, resolve } from "node:path";
|
|
4
4
|
import { redactSensitive } from "./redaction.ts";
|
|
5
5
|
import { normalizeTaskSpec } from "./acceptance.ts";
|
|
6
6
|
import type { TaskSpec } from "./types.ts";
|
|
@@ -27,6 +27,10 @@ export interface DecisionSessionRecord {
|
|
|
27
27
|
deadlineMs: number;
|
|
28
28
|
noOutputTimeoutMs: number;
|
|
29
29
|
startedAt: string;
|
|
30
|
+
baseCommit?: string;
|
|
31
|
+
baseBranch?: string;
|
|
32
|
+
/** Real executable identity pinned by automatic startup and recovery. */
|
|
33
|
+
resolvedExecutable?: string;
|
|
30
34
|
turn: number;
|
|
31
35
|
repairRound?: number;
|
|
32
36
|
lastFindingSignature?: string;
|
|
@@ -75,6 +79,7 @@ export class DecisionSessionStore {
|
|
|
75
79
|
assertSessionPath(decisionSessionFile, this.#directory, record.taskId);
|
|
76
80
|
assertNoCredentialPath(decisionSessionFile);
|
|
77
81
|
assertNoCredentialPath(record.cwd);
|
|
82
|
+
if (record.resolvedExecutable !== undefined) assertResolvedExecutable(record.resolvedExecutable);
|
|
78
83
|
await this.#withLock(() => this.#saveUnlocked({ ...record, decisionSessionFile }));
|
|
79
84
|
}
|
|
80
85
|
|
|
@@ -354,6 +359,9 @@ function normalizeRecord(value: Partial<DecisionSessionRecord>, directory: strin
|
|
|
354
359
|
|| (value.recoveryOwnerStartTime !== undefined && (typeof value.recoveryOwnerStartTime !== "string" || !/^\d+$/u.test(value.recoveryOwnerStartTime)))
|
|
355
360
|
|| (value.recoveryState !== undefined && !isRecoveryState(value.recoveryState))
|
|
356
361
|
|| (value.lastFindingSignature !== undefined && (typeof value.lastFindingSignature !== "string" || value.lastFindingSignature.length > 128))
|
|
362
|
+
|| (value.baseCommit !== undefined && (typeof value.baseCommit !== "string" || !/^[0-9a-f]{40,64}$/iu.test(value.baseCommit)))
|
|
363
|
+
|| (value.baseBranch !== undefined && (typeof value.baseBranch !== "string" || !/^[A-Za-z0-9._/-]+$/u.test(value.baseBranch)))
|
|
364
|
+
|| (value.resolvedExecutable !== undefined && (typeof value.resolvedExecutable !== "string" || !isAbsolute(value.resolvedExecutable) || value.resolvedExecutable.length > 4_096))
|
|
357
365
|
|| (value.startedAt !== undefined && (typeof value.startedAt !== "string" || !Number.isFinite(Date.parse(value.startedAt))))
|
|
358
366
|
|| (value.recoveryWorker !== undefined && !isRecoveryWorker(value.recoveryWorker))) {
|
|
359
367
|
throw new Error("invalid Decision Worker session record");
|
|
@@ -362,6 +370,7 @@ function normalizeRecord(value: Partial<DecisionSessionRecord>, directory: strin
|
|
|
362
370
|
assertSessionPath(decisionSessionFile, directory, value.taskId);
|
|
363
371
|
assertNoCredentialPath(decisionSessionFile);
|
|
364
372
|
assertNoCredentialPath(value.cwd);
|
|
373
|
+
if (value.resolvedExecutable !== undefined) assertResolvedExecutable(value.resolvedExecutable);
|
|
365
374
|
const recoveryWorker = value.recoveryWorker && {
|
|
366
375
|
id: value.recoveryWorker.id,
|
|
367
376
|
...(value.recoveryWorker.pid !== undefined ? { pid: value.recoveryWorker.pid } : {}),
|
|
@@ -381,6 +390,9 @@ function normalizeRecord(value: Partial<DecisionSessionRecord>, directory: strin
|
|
|
381
390
|
deadlineMs: value.deadlineMs ?? 4 * 60 * 60_000,
|
|
382
391
|
noOutputTimeoutMs: value.noOutputTimeoutMs ?? 20 * 60_000,
|
|
383
392
|
startedAt: value.startedAt ?? value.updatedAt,
|
|
393
|
+
...(typeof value.baseCommit === "string" ? { baseCommit: value.baseCommit } : {}),
|
|
394
|
+
...(typeof value.baseBranch === "string" ? { baseBranch: value.baseBranch } : {}),
|
|
395
|
+
...(typeof value.resolvedExecutable === "string" ? { resolvedExecutable: value.resolvedExecutable } : {}),
|
|
384
396
|
turn: value.turn ?? 0,
|
|
385
397
|
repairRound: value.repairRound ?? 0,
|
|
386
398
|
...(typeof value.lastFindingSignature === "string" ? { lastFindingSignature: value.lastFindingSignature } : {}),
|
|
@@ -415,6 +427,12 @@ function assertNoCredentialPath(path: string): void {
|
|
|
415
427
|
if (!path || String(redactSensitive(path)) !== path) throw new Error("Decision Worker session path contains credential-shaped text");
|
|
416
428
|
}
|
|
417
429
|
|
|
430
|
+
function assertResolvedExecutable(path: string): void {
|
|
431
|
+
if (!isAbsolute(path) || path.length > 4_096 || String(redactSensitive(path)) !== path) {
|
|
432
|
+
throw new Error("Decision Worker resolved executable identity is invalid");
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
|
|
418
436
|
function assertTaskId(taskId: string): void {
|
|
419
437
|
if (!/^[0-9a-f-]{36}$/iu.test(taskId) || basename(taskId) !== taskId) throw new Error("invalid task id");
|
|
420
438
|
}
|
package/src/decision-worker.ts
CHANGED
|
@@ -12,7 +12,8 @@ const MAX_DECISION_FIELD_BYTES = 8 * 1024;
|
|
|
12
12
|
export type DecisionAction =
|
|
13
13
|
| { action: "continue" | "redirect" | "answer"; message: string; reason: string; confidence?: number }
|
|
14
14
|
| { action: "allow_permission" | "deny_permission"; requestId: string; toolUseId: string; reason: string; confidence?: number }
|
|
15
|
-
| { action: "verify" | "
|
|
15
|
+
| { action: "verify" | "stop" | "park" | "ask_human" | "noop"; reason: string; question?: string; confidence?: number }
|
|
16
|
+
| { action: "retry"; reason: string; message?: string; confidence?: number };
|
|
16
17
|
|
|
17
18
|
export interface DecisionContext {
|
|
18
19
|
taskId: string;
|
|
@@ -130,12 +131,7 @@ export class PiDecisionWorker implements DecisionWorkerLike {
|
|
|
130
131
|
const first = this.#seenEvents.values().next().value;
|
|
131
132
|
if (first) this.#seenEvents.delete(first);
|
|
132
133
|
}
|
|
133
|
-
this.#tail = this.#tail.then(async () => {
|
|
134
|
-
if (!this.#session || this.#closed) return;
|
|
135
|
-
const text = await askDecision(this.#session, event, this.#context, this.#timeoutMs);
|
|
136
|
-
const action = parseDecision(text, event);
|
|
137
|
-
await this.#options.onAction(action, event);
|
|
138
|
-
}).catch(async (error) => {
|
|
134
|
+
this.#tail = this.#tail.then(() => this.#processEvent(event)).catch(async (error) => {
|
|
139
135
|
try {
|
|
140
136
|
if (this.#options.onFailure) await this.#options.onFailure(event, error);
|
|
141
137
|
} catch {
|
|
@@ -144,6 +140,29 @@ export class PiDecisionWorker implements DecisionWorkerLike {
|
|
|
144
140
|
});
|
|
145
141
|
}
|
|
146
142
|
|
|
143
|
+
async #processEvent(event: WorkerEvent): Promise<void> {
|
|
144
|
+
if (!this.#session || this.#closed) return;
|
|
145
|
+
const maxRetries = this.#context.spec?.autonomy.maxDecisionRetries ?? 2;
|
|
146
|
+
let attempt = 0;
|
|
147
|
+
while (true) {
|
|
148
|
+
let text: string;
|
|
149
|
+
try {
|
|
150
|
+
text = await askDecision(this.#session, event, this.#context, this.#timeoutMs);
|
|
151
|
+
} catch (error) {
|
|
152
|
+
if (attempt >= maxRetries || this.#closed) throw error;
|
|
153
|
+
attempt += 1;
|
|
154
|
+
await new Promise<void>((resolve) => setTimeout(resolve, Math.min(1_000, 100 * attempt)));
|
|
155
|
+
if (!this.#session || this.#closed) return;
|
|
156
|
+
continue;
|
|
157
|
+
}
|
|
158
|
+
const action = parseDecision(text, event);
|
|
159
|
+
// An action handler may stop or close the session. Do not replay an
|
|
160
|
+
// already-decoded action if the handler itself fails.
|
|
161
|
+
await this.#options.onAction(action, event);
|
|
162
|
+
return;
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
|
|
147
166
|
get sessionFile(): string | undefined {
|
|
148
167
|
return this.#sessionFile;
|
|
149
168
|
}
|
|
@@ -181,14 +200,20 @@ Current repair round: ${context.repairRound ?? 0}
|
|
|
181
200
|
Task specification: ${boundedJson(context.spec ?? { goal: context.task })}
|
|
182
201
|
|
|
183
202
|
Return exactly one JSON object and no markdown:
|
|
184
|
-
{"action":"continue|redirect|answer|allow_permission|deny_permission|verify|retry|stop|
|
|
203
|
+
{"action":"continue|redirect|answer|allow_permission|deny_permission|verify|retry|stop|park|noop",...}
|
|
185
204
|
For continue/redirect/answer include message and reason. For permission actions include
|
|
186
|
-
requestId and toolUseId.
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
205
|
+
requestId and toolUseId. Retry may include a corrective message. Never choose allow_permission
|
|
206
|
+
for a command that crosses the remote push or main/integration merge boundary; the deterministic
|
|
207
|
+
policy will deny it.
|
|
208
|
+
For AskUserQuestion, choose deny_permission when the question can be converted into ordinary
|
|
209
|
+
Claude text, then use answer on the resulting turn. For product ambiguity or an architecture
|
|
210
|
+
choice, inspect the repository and task evidence, select the best task-compatible option, state
|
|
211
|
+
the assumption in reason, and instruct Claude Code with answer or redirect. Do not ask a human
|
|
212
|
+
for ordinary uncertainty. Use verify when a turn result indicates the task is complete, even if
|
|
213
|
+
Claude says it will stop; choose stop only for an explicit stop or technical containment reason.
|
|
214
|
+
Use park only when the task cannot safely produce a candidate because required evidence,
|
|
215
|
+
authority, or runtime capability is unavailable. A parked candidate is asynchronous and must not
|
|
216
|
+
wait for a human to be online.`;
|
|
192
217
|
}
|
|
193
218
|
|
|
194
219
|
async function askDecision(session: AgentSession, event: WorkerEvent, context: DecisionContext, timeoutMs: number): Promise<string> {
|
|
@@ -249,12 +274,12 @@ function textFromMessage(content: unknown): string {
|
|
|
249
274
|
|
|
250
275
|
function parseDecision(text: string, event: WorkerEvent): DecisionAction {
|
|
251
276
|
const candidate = text.trim();
|
|
252
|
-
if (!candidate) return { action: "
|
|
277
|
+
if (!candidate) return { action: "park", reason: "Decision Worker returned no JSON action" };
|
|
253
278
|
try {
|
|
254
279
|
const value = JSON.parse(candidate) as Record<string, unknown>;
|
|
255
280
|
const action = value.action;
|
|
256
281
|
if (typeof action !== "string") throw new Error("missing action");
|
|
257
|
-
const allowed = new Set(["continue", "redirect", "answer", "allow_permission", "deny_permission", "verify", "retry", "stop", "ask_human", "noop"]);
|
|
282
|
+
const allowed = new Set(["continue", "redirect", "answer", "allow_permission", "deny_permission", "verify", "retry", "stop", "park", "ask_human", "noop"]);
|
|
258
283
|
if (!allowed.has(action)) throw new Error(`unsupported action: ${action}`);
|
|
259
284
|
const reason = typeof value.reason === "string" && value.reason.trim() ? boundedDecisionText(value.reason, "reason") : "no reason provided";
|
|
260
285
|
const confidence = value.confidence === undefined ? undefined : typeof value.confidence === "number" && Number.isFinite(value.confidence) && value.confidence >= 0 && value.confidence <= 1
|
|
@@ -271,9 +296,12 @@ function parseDecision(text: string, event: WorkerEvent): DecisionAction {
|
|
|
271
296
|
if (!requestId || !toolUseId) throw new Error("permission requestId/toolUseId required");
|
|
272
297
|
return { action: action as "allow_permission" | "deny_permission", requestId, toolUseId, reason, confidence };
|
|
273
298
|
}
|
|
274
|
-
|
|
299
|
+
if (action === "retry") {
|
|
300
|
+
return { action, reason, message: typeof value.message === "string" ? boundedDecisionText(value.message, "message") : undefined, confidence };
|
|
301
|
+
}
|
|
302
|
+
return { action: action as "verify" | "stop" | "park" | "ask_human" | "noop", reason, question: typeof value.question === "string" ? boundedDecisionText(value.question, "question") : undefined, confidence };
|
|
275
303
|
} catch (error) {
|
|
276
|
-
return { action: "
|
|
304
|
+
return { action: "park", reason: `invalid Decision Worker action: ${error instanceof Error ? error.message : String(error)}` };
|
|
277
305
|
}
|
|
278
306
|
}
|
|
279
307
|
|