pi-claude-supervisor 0.4.1 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +16 -0
- package/README.cn.md +22 -0
- package/README.md +25 -0
- package/docs/architecture.md +27 -1
- package/docs/engineering-plan.md +63 -0
- package/docs/testing.md +48 -3
- package/package.json +1 -1
- package/src/acceptance.ts +84 -0
- package/src/decision-session-store.ts +17 -5
- package/src/decision-worker.ts +44 -6
- package/src/index.ts +37 -7
- package/src/reviewer.ts +205 -0
- package/src/supervisor.ts +199 -35
- package/src/types.ts +60 -0
- package/src/verifier.ts +99 -10
- package/src/worker/process-adapter.ts +103 -34
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,22 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to this project will be documented here.
|
|
4
4
|
|
|
5
|
+
## [0.5.0](https://github.com/btnalit/pi-claude-supervisor/compare/v0.4.1...v0.5.0) (2026-09-14)
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
### Features
|
|
9
|
+
|
|
10
|
+
* add acceptance review repair loop ([fe29c78](https://github.com/btnalit/pi-claude-supervisor/commit/fe29c78a228494768aa52b03ee1bf1545b079119))
|
|
11
|
+
|
|
12
|
+
## [Unreleased]
|
|
13
|
+
|
|
14
|
+
### Added
|
|
15
|
+
|
|
16
|
+
- Structured task specifications with Goal, scope, constraints, forbidden actions and multiple argv-based acceptance checks.
|
|
17
|
+
- Independent read-only Reviewer results with bounded findings and automatic repair rounds.
|
|
18
|
+
- JSONL malformed-record handling and duplicate result/permission suppression fixtures.
|
|
19
|
+
- Active JSONL request shutdown coverage and deterministic acceptance/review tests.
|
|
20
|
+
|
|
5
21
|
## [0.4.1](https://github.com/btnalit/pi-claude-supervisor/compare/v0.4.0...v0.4.1) (2026-09-14)
|
|
6
22
|
|
|
7
23
|
|
package/README.cn.md
CHANGED
|
@@ -51,6 +51,7 @@ Claude CLI `2.1.270` 运行,跨版本兼容性不在本轮范围内。
|
|
|
51
51
|
```text
|
|
52
52
|
/supervise capabilities
|
|
53
53
|
/supervise start inspect the current repository
|
|
54
|
+
/supervise start --spec ./task.json
|
|
54
55
|
/supervise poll
|
|
55
56
|
/supervise sessions
|
|
56
57
|
/supervise recover <task-id>
|
|
@@ -61,6 +62,21 @@ Claude CLI `2.1.270` 运行,跨版本兼容性不在本轮范围内。
|
|
|
61
62
|
/supervise resume-auto <task-id>
|
|
62
63
|
```
|
|
63
64
|
|
|
65
|
+
`--spec` 接受 JSON 文件;验收命令始终使用 argv 执行,不经过 shell。例如:
|
|
66
|
+
|
|
67
|
+
```json
|
|
68
|
+
{
|
|
69
|
+
"goal": "实现请求的修改",
|
|
70
|
+
"scope": ["src/"],
|
|
71
|
+
"constraints": ["保持公共 API 兼容"],
|
|
72
|
+
"forbidden": ["不要发布构建产物"],
|
|
73
|
+
"acceptance": [
|
|
74
|
+
{ "id": "tests", "name": "tests", "command": "npm", "args": ["test"], "required": true }
|
|
75
|
+
],
|
|
76
|
+
"maxRepairRounds": 3
|
|
77
|
+
}
|
|
78
|
+
```
|
|
79
|
+
|
|
64
80
|
人工升级通知的 generic JSON 格式为:
|
|
65
81
|
|
|
66
82
|
```json
|
|
@@ -84,6 +100,12 @@ Claude CLI `2.1.270` 运行,跨版本兼容性不在本轮范围内。
|
|
|
84
100
|
再根据任务和仓库证据自动回答;无法确定时才升级人工。如需微信内闭环,需要另建带签名验证、
|
|
85
101
|
一次性 action token 和重放保护的入站 callback 服务。
|
|
86
102
|
|
|
103
|
+
近期自动化目标是先稳定完成“多命令验收—独立只读 Reviewer—结构化修复轮次—再次验收”闭环。
|
|
104
|
+
任务可通过 API 或 JSON spec 提供 `goal`、`scope`、`constraints`、`forbidden` 和多个
|
|
105
|
+
`acceptance` 命令;旧的纯文本任务继续使用默认 `git diff --check`。Reviewer 只能使用
|
|
106
|
+
`read`、`grep`、`find`、`ls`,不会修改工作树或批准权限。当前稳定性验证固定针对 Claude Code
|
|
107
|
+
`2.1.270`,暂不把多版本兼容、sandbox、低权限和网络隔离作为本阶段门禁。
|
|
108
|
+
|
|
87
109
|
### tmux/PTY 交互模式
|
|
88
110
|
|
|
89
111
|
如果希望在可见的 Claude Code 终端中工作,可显式启用 tmux transport:
|
package/README.md
CHANGED
|
@@ -57,6 +57,7 @@ export PI_CLAUDE_SUPERVISOR_WORKER=claude
|
|
|
57
57
|
```text
|
|
58
58
|
/supervise capabilities
|
|
59
59
|
/supervise start inspect the current repository and report what should be changed
|
|
60
|
+
/supervise start --spec ./task.json
|
|
60
61
|
/supervise sessions
|
|
61
62
|
/supervise recover <task-id>
|
|
62
63
|
/supervise poll
|
|
@@ -68,6 +69,22 @@ export PI_CLAUDE_SUPERVISOR_WORKER=claude
|
|
|
68
69
|
/supervise verify
|
|
69
70
|
```
|
|
70
71
|
|
|
72
|
+
`--spec` accepts a JSON file; checks are always executed with argv (never through
|
|
73
|
+
a shell), for example:
|
|
74
|
+
|
|
75
|
+
```json
|
|
76
|
+
{
|
|
77
|
+
"goal": "Implement the requested change",
|
|
78
|
+
"scope": ["src/"],
|
|
79
|
+
"constraints": ["Keep the public API compatible"],
|
|
80
|
+
"forbidden": ["Do not publish artifacts"],
|
|
81
|
+
"acceptance": [
|
|
82
|
+
{ "id": "tests", "name": "tests", "command": "npm", "args": ["test"], "required": true }
|
|
83
|
+
],
|
|
84
|
+
"maxRepairRounds": 3
|
|
85
|
+
}
|
|
86
|
+
```
|
|
87
|
+
|
|
71
88
|
The default MVP writes the task to the worker's stdin as plain process-pipe
|
|
72
89
|
text. After running the transport spike for the target CLI, JSONL framing can
|
|
73
90
|
be selected explicitly:
|
|
@@ -88,6 +105,14 @@ remaining lifecycle, signal and recovery checks. Host permissions and network
|
|
|
88
105
|
access follow explicit caller authorization and host policy; there is no
|
|
89
106
|
automatic merge, deploy, release or publish.
|
|
90
107
|
|
|
108
|
+
The near-term automation milestone adds a structured acceptance pipeline:
|
|
109
|
+
multiple argv-based checks, an independent read-only Reviewer, bounded structured
|
|
110
|
+
findings and repair rounds. Legacy text tasks keep the default `git diff --check`.
|
|
111
|
+
The Reviewer only has `read`, `grep`, `find` and `ls`; it cannot edit files or grant
|
|
112
|
+
permissions. Stability evidence is pinned to Claude Code `2.1.270`; CLI
|
|
113
|
+
multi-version compatibility, sandboxing, low-privilege execution and network
|
|
114
|
+
isolation are not part of this milestone.
|
|
115
|
+
|
|
91
116
|
Automatic mode persists the Pi Decision Worker session under the configured state
|
|
92
117
|
directory. After an unclean Pi restart, `/supervise sessions` lists recoverable
|
|
93
118
|
tasks; `/supervise recover <task-id>` explicitly restores the Decision Worker
|
package/docs/architecture.md
CHANGED
|
@@ -181,6 +181,29 @@ human operator; it does not attempt a second LLM fallback. Alert delivery is
|
|
|
181
181
|
kept independent from event-log persistence so an audit write failure cannot
|
|
182
182
|
suppress the alert.
|
|
183
183
|
|
|
184
|
+
## Acceptance, review and repair loop
|
|
185
|
+
|
|
186
|
+
A task may provide a structured `TaskSpec` with `goal`, `scope`, `constraints`,
|
|
187
|
+
`forbidden` and an ordered list of required or optional acceptance checks. A
|
|
188
|
+
legacy plain-text task is normalized to a goal with the default `git diff
|
|
189
|
+
--check` acceptance check. The verifier runs every configured check with argv,
|
|
190
|
+
bounded output and the same deterministic command policy; a Worker completion
|
|
191
|
+
claim never substitutes for these results.
|
|
192
|
+
|
|
193
|
+
When automatic supervision is enabled, a successful check set is passed to a
|
|
194
|
+
fresh read-only Reviewer session. The Reviewer receives the task specification, repository status/diff evidence,
|
|
195
|
+
check results and bounded Worker completion evidence, but not the Decision Worker
|
|
196
|
+
conversation or control channel. It can inspect only `read`, `grep`, `find` and `ls`, and must return
|
|
197
|
+
`pass`, `revise` or `human` with bounded structured findings. Invalid Reviewer
|
|
198
|
+
output or a Reviewer API failure is a human-required condition.
|
|
199
|
+
|
|
200
|
+
A `revise` result produces an audited repair round and sends a bounded corrective
|
|
201
|
+
instruction to a still-live JSONL/tmux Worker. Checks and review then run again.
|
|
202
|
+
The repair budget defaults to three rounds; repeated findings and P0/P1 findings
|
|
203
|
+
stop automation and escalate. A non-persistent Worker that has already exited
|
|
204
|
+
cannot be silently recreated for repair; it remains failed/recoverable rather
|
|
205
|
+
than replaying the original task.
|
|
206
|
+
|
|
184
207
|
## Deliberate non-goals
|
|
185
208
|
|
|
186
209
|
- automatic merge/deploy/release;
|
|
@@ -192,4 +215,7 @@ suppress the alert.
|
|
|
192
215
|
- shell command interpolation;
|
|
193
216
|
- automatic network denial or a fake domain allowlist. Network access follows
|
|
194
217
|
Claude's own permission model and the command policy; suspicious download-to-
|
|
195
|
-
shell patterns require human review rather than blanket network rejection
|
|
218
|
+
shell patterns require human review rather than blanket network rejection;
|
|
219
|
+
- Claude CLI multi-version compatibility in the current stability milestone;
|
|
220
|
+
- OS sandbox, low-privilege execution and network isolation in the current
|
|
221
|
+
lifecycle milestone.
|
package/docs/engineering-plan.md
CHANGED
|
@@ -916,3 +916,66 @@ PTY 和 headless JSONL 只能选择一个作为 MVP 的主 transport,禁止两
|
|
|
916
916
|
## 附录 B:一句话版本
|
|
917
917
|
|
|
918
918
|
> 先用最小、可审计、可接管的 Supervisor 闭环证明可靠性,再逐步开放 LLM 判断和自动化权限;不要从“自动化最多”开始,而要从“边界最清楚、证据最可靠”开始。
|
|
919
|
+
|
|
920
|
+
## 20. 近期落地与剩余门禁:稳定的自动验收闭环
|
|
921
|
+
|
|
922
|
+
本轮已落地 TaskSpec、多命令验收、独立只读 Reviewer、结构化 repair round、重复 finding/P0/P1 人工升级、JSONL 去重和确定性 replay fixture。剩余门禁是固定 CLI 的重复运行统计,而不是继续扩大安全边界。近期目标从“扩大安全边界”调整为先证明单一已验证 Claude CLI 版本上的真实功能稳定性。当前只验证固定的 Claude Code `2.1.270`,不把多版本兼容作为本阶段任务。OS sandbox、低权限用户、网络 allowlist、SBOM 和更深的供应链加固后置,不作为本阶段门禁;现有 no-shell、Policy Gate、人工接管和独立验收边界继续保留。
|
|
923
|
+
|
|
924
|
+
### 20.1 Goal / Evidence / Sign-off 模型
|
|
925
|
+
|
|
926
|
+
任务规格统一为:
|
|
927
|
+
|
|
928
|
+
```yaml
|
|
929
|
+
goal: 实现用户邀请接口
|
|
930
|
+
scope:
|
|
931
|
+
- 新增接口
|
|
932
|
+
- 添加权限校验
|
|
933
|
+
constraints:
|
|
934
|
+
- 不修改核心数据库模型
|
|
935
|
+
forbidden:
|
|
936
|
+
- 不执行生产部署
|
|
937
|
+
acceptance:
|
|
938
|
+
- id: tests
|
|
939
|
+
command: npm
|
|
940
|
+
args: [test]
|
|
941
|
+
- id: typecheck
|
|
942
|
+
command: npm
|
|
943
|
+
args: [run, typecheck]
|
|
944
|
+
- id: diff-check
|
|
945
|
+
command: git
|
|
946
|
+
args: [diff, --check]
|
|
947
|
+
maxRepairRounds: 3
|
|
948
|
+
```
|
|
949
|
+
|
|
950
|
+
兼容旧任务时,普通任务文本作为 `goal`,默认验收仍为 `git diff --check`。所有验收命令使用 argv 和确定性 Policy Gate,不经过 shell。
|
|
951
|
+
|
|
952
|
+
验收流程固定为:
|
|
953
|
+
|
|
954
|
+
```text
|
|
955
|
+
Worker result
|
|
956
|
+
→ 多命令 acceptance checks
|
|
957
|
+
→ 独立只读 Reviewer
|
|
958
|
+
├── pass → completed
|
|
959
|
+
├── revise → 结构化修复指令 → Worker → 重新验收
|
|
960
|
+
└── human → 人工接管
|
|
961
|
+
```
|
|
962
|
+
|
|
963
|
+
Reviewer 必须使用独立 Pi session,只允许 `read`、`grep`、`find`、`ls`,输出结构化 verdict 和 findings;不能修改工作树或直接批准权限。默认最多三轮修复;相同 finding 重复出现或出现 P0/P1 问题时升级人工。
|
|
964
|
+
|
|
965
|
+
### 20.2 JSONL 稳定性证据
|
|
966
|
+
|
|
967
|
+
只对 Claude Code `2.1.270` 建立证据,覆盖:
|
|
968
|
+
|
|
969
|
+
- JSONL 跨 chunk 拆分、单 chunk 多记录和 malformed 行;malformed 行不能触发完成事件;
|
|
970
|
+
- 重复 result、重复 permission request、重复 Supervisor idempotency key 不产生重复动作;
|
|
971
|
+
- active request 期间的 SIGTERM、SIGINT、stop 和 Pi shutdown;
|
|
972
|
+
- 普通完成、低风险 Bash allow、`AskUserQuestion` deny-to-text、多轮、验收失败修复和恢复回放。
|
|
973
|
+
|
|
974
|
+
真实 Claude Spike 不进入普通 CI;确定性 fake Worker/replay fixture 进入 CI。自动模式继续保持显式 opt-in,直到以下门禁通过:普通任务连续十次成功,权限和问题转文本场景各至少五次成功,无重复动作、无错误 complete、无未清理 Worker,且关键事件可以完整回放。
|
|
975
|
+
|
|
976
|
+
### 20.3 明确不属于本阶段
|
|
977
|
+
|
|
978
|
+
- Claude CLI 多版本兼容;
|
|
979
|
+
- OS sandbox、低权限执行和网络隔离;
|
|
980
|
+
- 自动 merge、deploy、release、publish;
|
|
981
|
+
- 多 Worker 在同一工作树协作。
|
package/docs/testing.md
CHANGED
|
@@ -49,9 +49,9 @@ non-sensitive prompt, and prints protocol metadata rather than raw model output.
|
|
|
49
49
|
It must not be added to the normal CI gate because authentication is an owner
|
|
50
50
|
controlled prerequisite.
|
|
51
51
|
|
|
52
|
-
The
|
|
53
|
-
permission allow/deny and SIGTERM/SIGINT behavior
|
|
54
|
-
|
|
52
|
+
The transport fixtures validate one prompt, multiple turns, session resume,
|
|
53
|
+
permission allow/deny and SIGTERM/SIGINT behavior. The current release validation
|
|
54
|
+
uses Claude Code 2.1.270 at
|
|
55
55
|
`/home/yancao/.local/share/mise/installs/claude/2.1.270/claude`, including real
|
|
56
56
|
owned tmux turns, pause/resume, and restart re-adoption.
|
|
57
57
|
The adapter regression suite also verifies event subscription, parsed
|
|
@@ -59,6 +59,11 @@ The adapter regression suite also verifies event subscription, parsed
|
|
|
59
59
|
The automation spike additionally exercises a real Pi SDK Decision Worker with
|
|
60
60
|
Claude: ordinary completion, harmless Bash permission approval, and an
|
|
61
61
|
`AskUserQuestion` denial-to-text fallback followed by automatic verification.
|
|
62
|
+
A local pinned-CLI run completed all three scenarios with `state=completed`,
|
|
63
|
+
`verified=true`, and zero human interventions. Provider/model latency can still
|
|
64
|
+
cause a later run to fail closed as human-required after the bounded Decision
|
|
65
|
+
Worker or Reviewer timeout; this is evidence for the manual spike only, not a CI
|
|
66
|
+
guarantee.
|
|
62
67
|
The extension persists each automatic Decision Worker session as Pi JSONL plus a
|
|
63
68
|
0600 task mapping. Recovery is explicit and safe: after an unclean Pi restart,
|
|
64
69
|
`/supervise sessions` shows the task as `recoverable`, and `/supervise recover
|
|
@@ -144,3 +149,43 @@ not eliminate the post-spawn attachment window. `SIGSTOP` and `SIGKILL` of the P
|
|
|
144
149
|
verify and document the resulting orphan behavior.
|
|
145
150
|
Default behavior must be fail-closed and leave no orphaned worker process within
|
|
146
151
|
the managed process group.
|
|
152
|
+
|
|
153
|
+
## Near-term automation acceptance gate
|
|
154
|
+
|
|
155
|
+
The next implementation milestone focuses on real stability for the pinned
|
|
156
|
+
Claude Code `2.1.270` CLI. It does not add a multi-version matrix or wait for
|
|
157
|
+
OS sandbox, low-privilege or network-isolation work.
|
|
158
|
+
|
|
159
|
+
### Acceptance and Reviewer fixtures
|
|
160
|
+
|
|
161
|
+
Deterministic tests must cover:
|
|
162
|
+
|
|
163
|
+
- legacy text tasks normalized to a Goal with the default `git diff --check`;
|
|
164
|
+
- multiple required/optional checks with bounded output, timeout and exit-code evidence;
|
|
165
|
+
- independent read-only Reviewer pass/revise/human results;
|
|
166
|
+
- invalid Reviewer JSON and Reviewer API failure escalating to human;
|
|
167
|
+
- repair rounds, repeated finding detection, P0/P1 escalation and repair-budget exhaustion;
|
|
168
|
+
- completion being impossible without passing all required checks and review.
|
|
169
|
+
|
|
170
|
+
Reviewer sessions use only `read`, `grep`, `find` and `ls`; they must not modify
|
|
171
|
+
the worktree or send Worker input. Decision Worker and Reviewer model calls are
|
|
172
|
+
bounded; timeout or API failure escalates instead of auto-completing. Review reports
|
|
173
|
+
are persisted as bounded event payloads and are not treated as permission grants.
|
|
174
|
+
|
|
175
|
+
### JSONL protocol and replay fixtures
|
|
176
|
+
|
|
177
|
+
The adapter/replay matrix must cover:
|
|
178
|
+
|
|
179
|
+
- JSON split across stdout chunks and multiple records in one chunk;
|
|
180
|
+
- malformed JSON between valid records without a false completion event;
|
|
181
|
+
- duplicate result and permission records without duplicate actions;
|
|
182
|
+
- duplicate Supervisor idempotency keys without duplicate input;
|
|
183
|
+
- stop, SIGTERM, SIGINT and Pi shutdown while a JSONL request is active;
|
|
184
|
+
- ordinary completion, low-risk permission allow, AskUserQuestion deny-to-text,
|
|
185
|
+
multi-turn, verifier failure/repair, takeover and explicit recovery.
|
|
186
|
+
|
|
187
|
+
Real Claude tests remain authenticated manual Spikes and are pinned to
|
|
188
|
+
`2.1.270`; they are not part of normal CI. Normal CI runs deterministic fake
|
|
189
|
+
Worker and replay fixtures. Stability evidence should include ten consecutive
|
|
190
|
+
ordinary automatic runs and at least five runs each for permission and question
|
|
191
|
+
handling, with no duplicate action, false completion or unreaped Worker.
|
package/package.json
CHANGED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
import type { AcceptanceCheck, TaskSpec } from "./types.ts";
|
|
2
|
+
|
|
3
|
+
const DEFAULT_TIMEOUT_MS = 120_000;
|
|
4
|
+
const DEFAULT_MAX_REPAIR_ROUNDS = 3;
|
|
5
|
+
|
|
6
|
+
/** Normalize legacy plain-text tasks into the structured acceptance model. */
|
|
7
|
+
export function normalizeTaskSpec(value: unknown, fallbackGoal: string): TaskSpec {
|
|
8
|
+
if (value !== undefined && (!value || typeof value !== "object" || Array.isArray(value))) {
|
|
9
|
+
throw new Error("task spec must be a JSON object");
|
|
10
|
+
}
|
|
11
|
+
const source = (value ?? {}) as Record<string, unknown>;
|
|
12
|
+
const goal = typeof source.goal === "string" && source.goal.trim() ? source.goal.trim() : fallbackGoal.trim();
|
|
13
|
+
if (!goal) throw new Error("task goal must not be empty");
|
|
14
|
+
return {
|
|
15
|
+
goal,
|
|
16
|
+
scope: stringList(source.scope, "scope"),
|
|
17
|
+
constraints: stringList(source.constraints, "constraints"),
|
|
18
|
+
forbidden: stringList(source.forbidden, "forbidden"),
|
|
19
|
+
acceptance: normalizeChecks(source.acceptance),
|
|
20
|
+
maxRepairRounds: normalizeRepairRounds(source.maxRepairRounds),
|
|
21
|
+
};
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export function defaultAcceptanceChecks(): AcceptanceCheck[] {
|
|
25
|
+
return [{
|
|
26
|
+
id: "diff-check",
|
|
27
|
+
name: "git diff check",
|
|
28
|
+
command: "git",
|
|
29
|
+
args: ["diff", "--check"],
|
|
30
|
+
required: true,
|
|
31
|
+
timeoutMs: DEFAULT_TIMEOUT_MS,
|
|
32
|
+
}];
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
function normalizeChecks(value: unknown): AcceptanceCheck[] {
|
|
36
|
+
if (value === undefined) return defaultAcceptanceChecks();
|
|
37
|
+
if (!Array.isArray(value)) throw new Error("task spec acceptance must be an array");
|
|
38
|
+
if (value.length === 0) return defaultAcceptanceChecks();
|
|
39
|
+
const checks = value.map((item, index) => normalizeCheck(item, index));
|
|
40
|
+
const ids = new Set<string>();
|
|
41
|
+
for (const check of checks) {
|
|
42
|
+
if (ids.has(check.id)) throw new Error(`duplicate acceptance check id: ${check.id}`);
|
|
43
|
+
ids.add(check.id);
|
|
44
|
+
}
|
|
45
|
+
return checks;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
function normalizeCheck(value: unknown, index: number): AcceptanceCheck {
|
|
49
|
+
if (!value || typeof value !== "object" || Array.isArray(value)) throw new Error(`acceptance[${index}] must be an object`);
|
|
50
|
+
const source = value as Record<string, unknown>;
|
|
51
|
+
const id = typeof source.id === "string" && /^[A-Za-z0-9._-]+$/u.test(source.id.trim())
|
|
52
|
+
? source.id.trim()
|
|
53
|
+
: `check-${index + 1}`;
|
|
54
|
+
const name = typeof source.name === "string" && source.name.trim() ? source.name.trim() : id;
|
|
55
|
+
const command = typeof source.command === "string" ? source.command.trim() : "";
|
|
56
|
+
if (!command) throw new Error(`acceptance[${index}] command must not be empty`);
|
|
57
|
+
const args = source.args === undefined
|
|
58
|
+
? []
|
|
59
|
+
: Array.isArray(source.args) && source.args.every((arg) => typeof arg === "string")
|
|
60
|
+
? source.args.map((arg) => arg)
|
|
61
|
+
: undefined;
|
|
62
|
+
if (!args) throw new Error(`acceptance[${index}] args must be an array of strings`);
|
|
63
|
+
const timeoutValue = source.timeoutMs;
|
|
64
|
+
const timeoutMs = timeoutValue === undefined ? DEFAULT_TIMEOUT_MS : timeoutValue;
|
|
65
|
+
if (typeof timeoutMs !== "number" || !Number.isSafeInteger(timeoutMs) || timeoutMs < 1 || timeoutMs > 60 * 60_000) {
|
|
66
|
+
throw new Error(`acceptance[${index}] timeoutMs must be between 1 and 3600000`);
|
|
67
|
+
}
|
|
68
|
+
if (typeof source.required !== "undefined" && typeof source.required !== "boolean") {
|
|
69
|
+
throw new Error(`acceptance[${index}] required must be boolean`);
|
|
70
|
+
}
|
|
71
|
+
return { id, name, command, args, required: source.required !== false, timeoutMs };
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
function stringList(value: unknown, field: string): string[] {
|
|
75
|
+
if (value === undefined) return [];
|
|
76
|
+
if (!Array.isArray(value) || value.some((item) => typeof item !== "string")) throw new Error(`task spec ${field} must be an array of strings`);
|
|
77
|
+
return value.map((item) => item.trim()).filter(Boolean);
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function normalizeRepairRounds(value: unknown): number {
|
|
81
|
+
if (value === undefined) return DEFAULT_MAX_REPAIR_ROUNDS;
|
|
82
|
+
if (typeof value !== "number" || !Number.isSafeInteger(value) || value < 0 || value > 10) throw new Error("maxRepairRounds must be between 0 and 10");
|
|
83
|
+
return value;
|
|
84
|
+
}
|
|
@@ -1,11 +1,14 @@
|
|
|
1
1
|
import { chmod, lstat, mkdir, readdir, readFile, rename, writeFile } from "node:fs/promises";
|
|
2
2
|
import { basename, dirname, join, resolve } from "node:path";
|
|
3
3
|
import { redactSensitive } from "./redaction.ts";
|
|
4
|
+
import { normalizeTaskSpec } from "./acceptance.ts";
|
|
5
|
+
import type { TaskSpec } from "./types.ts";
|
|
4
6
|
|
|
5
7
|
export interface DecisionSessionRecord {
|
|
6
8
|
version: 1;
|
|
7
9
|
taskId: string;
|
|
8
10
|
task: string;
|
|
11
|
+
spec?: TaskSpec;
|
|
9
12
|
cwd: string;
|
|
10
13
|
command: string;
|
|
11
14
|
args: string[];
|
|
@@ -16,6 +19,8 @@ export interface DecisionSessionRecord {
|
|
|
16
19
|
noOutputTimeoutMs: number;
|
|
17
20
|
startedAt: string;
|
|
18
21
|
turn: number;
|
|
22
|
+
repairRound?: number;
|
|
23
|
+
lastFindingSignature?: string;
|
|
19
24
|
state: "active" | "closed";
|
|
20
25
|
updatedAt: string;
|
|
21
26
|
}
|
|
@@ -71,7 +76,7 @@ export class DecisionSessionStore {
|
|
|
71
76
|
await this.save({ ...record, state: "closed", updatedAt: new Date().toISOString() });
|
|
72
77
|
}
|
|
73
78
|
|
|
74
|
-
async update(taskId: string, patch: Partial<Pick<DecisionSessionRecord, "turn" | "updatedAt">>): Promise<void> {
|
|
79
|
+
async update(taskId: string, patch: Partial<Pick<DecisionSessionRecord, "turn" | "repairRound" | "lastFindingSignature" | "updatedAt">>): Promise<void> {
|
|
75
80
|
const record = await this.load(taskId);
|
|
76
81
|
if (!record || record.state !== "active") return;
|
|
77
82
|
await this.save({ ...record, ...patch, updatedAt: patch.updatedAt ?? new Date().toISOString() });
|
|
@@ -94,7 +99,7 @@ export class DecisionSessionStore {
|
|
|
94
99
|
try {
|
|
95
100
|
const value = JSON.parse(await readFile(this.#recordPath(taskId), "utf8")) as Partial<DecisionSessionRecord>;
|
|
96
101
|
assertNoCredentialPath(typeof value.decisionSessionFile === "string" ? resolve(value.decisionSessionFile) : "");
|
|
97
|
-
return normalizeRecord(redactRecord(value), this.#directory);
|
|
102
|
+
return normalizeRecord(redactRecord(value), this.#directory, taskId);
|
|
98
103
|
} catch (error) {
|
|
99
104
|
if (error instanceof Error && /ENOENT/u.test(error.message)) return undefined;
|
|
100
105
|
throw error;
|
|
@@ -109,7 +114,8 @@ export class DecisionSessionStore {
|
|
|
109
114
|
try {
|
|
110
115
|
const value = JSON.parse(await readFile(join(this.#directory, name), "utf8")) as Partial<DecisionSessionRecord>;
|
|
111
116
|
assertNoCredentialPath(typeof value.decisionSessionFile === "string" ? resolve(value.decisionSessionFile) : "");
|
|
112
|
-
const
|
|
117
|
+
const expectedTaskId = name.slice(0, -".json".length);
|
|
118
|
+
const record = normalizeRecord(redactRecord(value), this.#directory, expectedTaskId);
|
|
113
119
|
if (!options.activeOnly || record.state === "active") records.push(record);
|
|
114
120
|
} catch {
|
|
115
121
|
// A torn or manually edited registry record is not recoverable.
|
|
@@ -132,14 +138,17 @@ async function writeJson(path: string, value: unknown): Promise<void> {
|
|
|
132
138
|
await writeFile(path, `${JSON.stringify(value, null, 2)}\n`, { encoding: "utf8", mode: 0o600 });
|
|
133
139
|
}
|
|
134
140
|
|
|
135
|
-
function normalizeRecord(value: Partial<DecisionSessionRecord>, directory: string): DecisionSessionRecord {
|
|
141
|
+
function normalizeRecord(value: Partial<DecisionSessionRecord>, directory: string, expectedTaskId?: string): DecisionSessionRecord {
|
|
142
|
+
if (expectedTaskId !== undefined && value.taskId !== expectedTaskId) throw new Error("Decision Worker session task id mismatch");
|
|
136
143
|
if (value.version !== 1 || typeof value.taskId !== "string" || !/^[0-9a-f-]{36}$/iu.test(value.taskId)
|
|
137
144
|
|| typeof value.task !== "string" || typeof value.cwd !== "string" || typeof value.command !== "string"
|
|
138
145
|
|| !Array.isArray(value.args) || value.args.some((arg) => typeof arg !== "string")
|
|
139
146
|
|| typeof value.decisionSessionFile !== "string" || (value.state !== "active" && value.state !== "closed")
|
|
140
147
|
|| typeof value.updatedAt !== "string" || !Number.isFinite(Date.parse(value.updatedAt))
|
|
141
148
|
|| !validLimit(value.maxTurns, 0) || !validLimit(value.deadlineMs, 0) || !validLimit(value.noOutputTimeoutMs, 0)
|
|
142
|
-
|| !validLimit(value.turn, 0) || (value.
|
|
149
|
+
|| !validLimit(value.turn, 0) || !validLimit(value.repairRound, 0)
|
|
150
|
+
|| (value.lastFindingSignature !== undefined && (typeof value.lastFindingSignature !== "string" || value.lastFindingSignature.length > 128))
|
|
151
|
+
|| (value.startedAt !== undefined && (typeof value.startedAt !== "string" || !Number.isFinite(Date.parse(value.startedAt))))) {
|
|
143
152
|
throw new Error("invalid Decision Worker session record");
|
|
144
153
|
}
|
|
145
154
|
const decisionSessionFile = resolve(value.decisionSessionFile);
|
|
@@ -150,6 +159,7 @@ function normalizeRecord(value: Partial<DecisionSessionRecord>, directory: strin
|
|
|
150
159
|
version: 1,
|
|
151
160
|
taskId: value.taskId,
|
|
152
161
|
task: value.task,
|
|
162
|
+
spec: normalizeTaskSpec(value.spec, value.task),
|
|
153
163
|
cwd: value.cwd,
|
|
154
164
|
command: value.command,
|
|
155
165
|
args: [...value.args],
|
|
@@ -160,6 +170,8 @@ function normalizeRecord(value: Partial<DecisionSessionRecord>, directory: strin
|
|
|
160
170
|
noOutputTimeoutMs: value.noOutputTimeoutMs ?? 20 * 60_000,
|
|
161
171
|
startedAt: value.startedAt ?? value.updatedAt,
|
|
162
172
|
turn: value.turn ?? 0,
|
|
173
|
+
repairRound: value.repairRound ?? 0,
|
|
174
|
+
...(typeof value.lastFindingSignature === "string" ? { lastFindingSignature: value.lastFindingSignature } : {}),
|
|
163
175
|
state: value.state,
|
|
164
176
|
updatedAt: value.updatedAt,
|
|
165
177
|
};
|
package/src/decision-worker.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { access, readFile, writeFile } from "node:fs/promises";
|
|
2
2
|
import { createAgentSession, DefaultResourceLoader, getAgentDir, SessionManager, type AgentSession } from "@earendil-works/pi-coding-agent";
|
|
3
|
-
import type { WorkerEvent } from "./types.ts";
|
|
3
|
+
import type { TaskSpec, WorkerEvent } from "./types.ts";
|
|
4
4
|
import { redactSensitive } from "./redaction.ts";
|
|
5
5
|
|
|
6
6
|
export type DecisionAction =
|
|
@@ -15,13 +15,26 @@ export interface DecisionContext {
|
|
|
15
15
|
state: string;
|
|
16
16
|
turn: number;
|
|
17
17
|
maxTurns: number;
|
|
18
|
+
repairRound?: number;
|
|
19
|
+
spec?: TaskSpec;
|
|
18
20
|
}
|
|
19
21
|
|
|
22
|
+
export interface DecisionWorkerLike {
|
|
23
|
+
start(): Promise<void>;
|
|
24
|
+
updateContext(patch: Partial<DecisionContext>): void;
|
|
25
|
+
notify(event: WorkerEvent): void;
|
|
26
|
+
close(): Promise<void>;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export type DecisionWorkerFactory = (options: DecisionWorkerOptions) => DecisionWorkerLike;
|
|
30
|
+
|
|
20
31
|
export interface DecisionWorkerOptions {
|
|
21
32
|
context: DecisionContext;
|
|
22
33
|
onAction: (action: DecisionAction, event: WorkerEvent) => Promise<void> | void;
|
|
23
34
|
onFailure?: (event: WorkerEvent, error: unknown) => Promise<void> | void;
|
|
24
35
|
onStartupFailure?: (error: unknown) => Promise<void> | void;
|
|
36
|
+
/** Bound each Decision Worker model request so failure handling cannot wait forever. */
|
|
37
|
+
timeoutMs?: number;
|
|
25
38
|
/** Existing Pi session JSONL to restore after a Supervisor/Pi restart. */
|
|
26
39
|
sessionFile?: string;
|
|
27
40
|
/** Directory for newly created Pi session JSONL files. */
|
|
@@ -34,7 +47,7 @@ export interface DecisionWorkerOptions {
|
|
|
34
47
|
* It has read-only repository tools and can request typed actions, but it
|
|
35
48
|
* cannot directly spawn processes, modify files, or answer Claude's stdin.
|
|
36
49
|
*/
|
|
37
|
-
export class PiDecisionWorker {
|
|
50
|
+
export class PiDecisionWorker implements DecisionWorkerLike {
|
|
38
51
|
readonly #options: DecisionWorkerOptions;
|
|
39
52
|
readonly #seenEvents = new Set<string>();
|
|
40
53
|
#session?: AgentSession;
|
|
@@ -43,10 +56,12 @@ export class PiDecisionWorker {
|
|
|
43
56
|
#initialized = false;
|
|
44
57
|
#sessionFile?: string;
|
|
45
58
|
#context: DecisionContext;
|
|
59
|
+
readonly #timeoutMs: number;
|
|
46
60
|
|
|
47
61
|
constructor(options: DecisionWorkerOptions) {
|
|
48
62
|
this.#options = options;
|
|
49
63
|
this.#context = { ...options.context };
|
|
64
|
+
this.#timeoutMs = options.timeoutMs ?? 120_000;
|
|
50
65
|
}
|
|
51
66
|
|
|
52
67
|
async start(): Promise<void> {
|
|
@@ -76,6 +91,7 @@ export class PiDecisionWorker {
|
|
|
76
91
|
cwd: this.#options.context.cwd,
|
|
77
92
|
resourceLoader,
|
|
78
93
|
sessionManager,
|
|
94
|
+
thinkingLevel: "low",
|
|
79
95
|
tools: ["read", "grep", "find", "ls"],
|
|
80
96
|
});
|
|
81
97
|
this.#session = session;
|
|
@@ -86,8 +102,9 @@ export class PiDecisionWorker {
|
|
|
86
102
|
}
|
|
87
103
|
if (!restored) {
|
|
88
104
|
try {
|
|
89
|
-
await session.prompt(decisionInstructions(this.#context));
|
|
105
|
+
await withTimeout(session.prompt(decisionInstructions(this.#context)), this.#timeoutMs, "Decision Worker startup");
|
|
90
106
|
} catch (error) {
|
|
107
|
+
await session.abort().catch(() => {});
|
|
91
108
|
try { await this.#options.onStartupFailure?.(error); } catch { /* preserve the original startup failure */ }
|
|
92
109
|
throw error;
|
|
93
110
|
}
|
|
@@ -109,7 +126,7 @@ export class PiDecisionWorker {
|
|
|
109
126
|
}
|
|
110
127
|
this.#tail = this.#tail.then(async () => {
|
|
111
128
|
if (!this.#session || this.#closed) return;
|
|
112
|
-
const text = await askDecision(this.#session, event, this.#context);
|
|
129
|
+
const text = await askDecision(this.#session, event, this.#context, this.#timeoutMs);
|
|
113
130
|
const action = parseDecision(text, event);
|
|
114
131
|
await this.#options.onAction(action, event);
|
|
115
132
|
}).catch(async (error) => {
|
|
@@ -139,6 +156,7 @@ export class PiDecisionWorker {
|
|
|
139
156
|
this.#closed = true;
|
|
140
157
|
const session = this.#session;
|
|
141
158
|
this.#session = undefined;
|
|
159
|
+
if (session) await session.abort().catch(() => {});
|
|
142
160
|
session?.dispose();
|
|
143
161
|
}
|
|
144
162
|
}
|
|
@@ -153,6 +171,8 @@ Task: ${redactText(context.task)}
|
|
|
153
171
|
Task id: ${redactText(context.taskId)}
|
|
154
172
|
Working directory: ${redactText(context.cwd)}
|
|
155
173
|
Maximum automatic turns: ${context.maxTurns}
|
|
174
|
+
Current repair round: ${context.repairRound ?? 0}
|
|
175
|
+
Task specification: ${boundedJson(context.spec ?? { goal: context.task })}
|
|
156
176
|
|
|
157
177
|
Return exactly one JSON object and no markdown:
|
|
158
178
|
{"action":"continue|redirect|answer|allow_permission|deny_permission|verify|retry|stop|ask_human|noop",...}
|
|
@@ -165,14 +185,19 @@ Use ask_human for product ambiguity, architecture tradeoffs with material risk,
|
|
|
165
185
|
secrets, deployment, or any uncertainty. Never invent missing information.`;
|
|
166
186
|
}
|
|
167
187
|
|
|
168
|
-
async function askDecision(session: AgentSession, event: WorkerEvent, context: DecisionContext): Promise<string> {
|
|
188
|
+
async function askDecision(session: AgentSession, event: WorkerEvent, context: DecisionContext, timeoutMs: number): Promise<string> {
|
|
169
189
|
let text = "";
|
|
170
190
|
const unsubscribe = session.subscribe((value) => {
|
|
171
191
|
const record = value as unknown as { type?: string; assistantMessageEvent?: { type?: string; delta?: string } };
|
|
172
192
|
if (record.type === "message_update" && record.assistantMessageEvent?.type === "text_delta") text += record.assistantMessageEvent.delta ?? "";
|
|
173
193
|
});
|
|
174
194
|
try {
|
|
175
|
-
|
|
195
|
+
try {
|
|
196
|
+
await withTimeout(session.prompt(`UNTRUSTED SUPERVISOR EVENT:\n${boundedJson(event)}\n\nCURRENT CONTEXT:\n${boundedJson(context)}\n\nChoose one action now.`), timeoutMs, "Decision Worker request");
|
|
197
|
+
} catch (error) {
|
|
198
|
+
await session.abort().catch(() => {});
|
|
199
|
+
throw error;
|
|
200
|
+
}
|
|
176
201
|
} finally {
|
|
177
202
|
unsubscribe();
|
|
178
203
|
}
|
|
@@ -215,6 +240,19 @@ function eventKey(event: WorkerEvent): string {
|
|
|
215
240
|
return `${event.handle.id}:output:${event.chunk.at}:${event.chunk.text.slice(0, 80)}`;
|
|
216
241
|
}
|
|
217
242
|
|
|
243
|
+
async function withTimeout<T>(promise: Promise<T>, timeoutMs: number, label: string): Promise<T> {
|
|
244
|
+
let timer: NodeJS.Timeout | undefined;
|
|
245
|
+
const timeout = new Promise<never>((_, reject) => {
|
|
246
|
+
timer = setTimeout(() => reject(new Error(`${label} timed out after ${timeoutMs}ms`)), timeoutMs);
|
|
247
|
+
timer.unref();
|
|
248
|
+
});
|
|
249
|
+
try {
|
|
250
|
+
return await Promise.race([promise, timeout]);
|
|
251
|
+
} finally {
|
|
252
|
+
if (timer) clearTimeout(timer);
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
|
|
218
256
|
function boundedJson(value: unknown): string {
|
|
219
257
|
const text = JSON.stringify(redactDecisionValue(value), null, 2) ?? "null";
|
|
220
258
|
return text.length <= 32_000 ? text : `${text.slice(0, 32_000)}\n[TRUNCATED]`;
|