pi-plans 0.6.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +58 -0
- package/CONTRIBUTING.md +8 -15
- package/README.md +39 -37
- package/agents/execution-reviewer.md +40 -0
- package/agents/reviewer.md +12 -3
- package/index.ts +55 -58
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +50 -60
- package/references/plan-artifact-template.md +81 -60
- package/references/state-and-config.md +60 -44
- package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
- package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
- package/scripts/run-tests.ts +12 -1
- package/scripts/validate.ts +39 -11
- package/skills/debug-and-plan/SKILL.md +3 -3
- package/skills/plan-big/SKILL.md +3 -3
- package/skills/plan-normal/SKILL.md +3 -3
- package/skills/plan-small/SKILL.md +4 -4
- package/skills/plan-with-refs/SKILL.md +6 -6
- package/skills/planning/SKILL.md +1 -1
- package/src/ask-form.ts +4 -4
- package/src/auditor.ts +227 -0
- package/src/auto-approve.ts +1 -1
- package/src/autocomplete.ts +19 -17
- package/src/code-graph/commands.ts +8 -3
- package/src/code-graph/community.ts +1 -1
- package/src/code-graph/paths.ts +1 -1
- package/src/code-graph/watch.ts +2 -2
- package/src/compaction.ts +3 -3
- package/src/config-command.ts +146 -73
- package/src/dashboard.ts +303 -0
- package/src/exec.ts +1185 -924
- package/src/global-state.ts +304 -0
- package/src/guard.ts +18 -19
- package/src/messaging.ts +44 -0
- package/src/plan.ts +421 -112
- package/src/query-hook.ts +4 -4
- package/src/refine-prompts.ts +12 -70
- package/src/refine-ui-helpers.ts +24 -5
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +19 -3
- package/src/resume-command.ts +45 -129
- package/src/resume.ts +5 -1
- package/src/role-panels.ts +542 -0
- package/src/run-context.ts +3 -10
- package/src/staleness.ts +53 -0
- package/src/state.ts +273 -72
- package/src/subagent.ts +19 -29
- package/src/task-tool.ts +100 -0
- package/src/tasks.ts +223 -0
- package/src/thinking-levels.ts +67 -0
- package/src/ui-language.ts +7 -54
- package/src/workflow-state.ts +76 -58
- package/tests/analyze-refs.test.ts +35 -18
- package/tests/ask-choice-schema.test.ts +0 -12
- package/tests/ask-choice.test.ts +2 -49
- package/tests/ask-form-tool.test.ts +4 -5
- package/tests/ask-form.test.ts +2 -2
- package/tests/auditor.test.ts +210 -0
- package/tests/auto-approve.test.ts +7 -10
- package/tests/autocomplete.test.ts +8 -11
- package/tests/code-graph-apply-action.test.ts +2 -2
- package/tests/code-graph-commands.test.ts +2 -2
- package/tests/code-graph-index.test.ts +2 -2
- package/tests/code-graph-loop.e2e.test.ts +1 -1
- package/tests/code-graph-mutations.test.ts +1 -1
- package/tests/code-graph-rollback.test.ts +1 -1
- package/tests/code-graph-v05.test.ts +2 -2
- package/tests/compaction.test.ts +1 -1
- package/tests/config-command.test.ts +103 -100
- package/tests/dashboard.test.ts +402 -0
- package/tests/exec-lifecycle.test.ts +181 -115
- package/tests/exec-panel-lifecycle.test.ts +106 -251
- package/tests/exec-review-loop.test.ts +331 -0
- package/tests/exec.test.ts +771 -1706
- package/tests/execute-plan.test.ts +44 -19
- package/tests/extension-load.test.ts +48 -0
- package/tests/global-state.test.ts +371 -0
- package/tests/graph-aware-file-tools.test.ts +5 -5
- package/tests/guard.test.ts +1 -1
- package/tests/multi-run.test.ts +3 -103
- package/tests/plan.test.ts +139 -62
- package/tests/plans.test.ts +7 -79
- package/tests/refine-prompts.test.ts +20 -71
- package/tests/refine-resume.test.ts +27 -22
- package/tests/refine-ui.test.ts +6 -15
- package/tests/resume-lifecycle.test.ts +41 -22
- package/tests/resume.test.ts +39 -81
- package/tests/role-panels.test.ts +391 -0
- package/tests/run-context.test.ts +1 -1
- package/tests/run-ownership.test.ts +1 -1
- package/tests/stale-ctx.test.ts +218 -0
- package/tests/staleness.test.ts +76 -0
- package/tests/state.test.ts +155 -32
- package/tests/subagent-thinking.test.ts +65 -0
- package/tests/subagent-usage.test.ts +1 -1
- package/tests/task-tool.test.ts +61 -0
- package/tests/tasks.test.ts +142 -0
- package/tests/thinking-levels.test.ts +77 -0
- package/tests/ui-language.test.ts +2 -17
- package/tests/workflow-state.test.ts +73 -90
- package/tools/analyze-refs.ts +67 -32
- package/tools/ask-choice.ts +7 -53
- package/tools/code-graph.ts +2 -2
- package/tools/execute-plan.ts +55 -99
- package/tools/graph-aware-file-tools.ts +4 -10
- package/tools/plans.ts +41 -67
- package/tools/refine.ts +101 -164
- package/agents/criticizer.md +0 -18
- package/agents/executor.md +0 -26
- package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
- package/src/panel.ts +0 -473
- package/src/termination-prompt.ts +0 -73
- package/tests/goal-wait.test.ts +0 -269
- package/tests/panel-i-zero.test.ts +0 -420
- package/tests/panel.test.ts +0 -355
package/tests/exec.test.ts
CHANGED
|
@@ -1,4 +1,9 @@
|
|
|
1
|
-
/**
|
|
1
|
+
/**
|
|
2
|
+
* Task-tree execution core tests (v0.6.1): startExecution, task-tool-driven
|
|
3
|
+
* progress persistence, the per-turn injection, the stall watchdog, the
|
|
4
|
+
* completion-audit flow (pass, rollback, round cap under auto-approve), the
|
|
5
|
+
* checkpoint schema round-trip, and restoreFromSession.
|
|
6
|
+
*/
|
|
2
7
|
|
|
3
8
|
import * as assert from "node:assert/strict";
|
|
4
9
|
import * as fs from "node:fs";
|
|
@@ -6,1765 +11,825 @@ import * as os from "node:os";
|
|
|
6
11
|
import * as path from "node:path";
|
|
7
12
|
import { after, before, describe, it } from "node:test";
|
|
8
13
|
import {
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
applyImplMarkers,
|
|
12
|
-
applyCurrentIMarker,
|
|
13
|
-
buildExecutionCompactionResult,
|
|
14
|
-
buildPlanningCompactionResult,
|
|
15
|
-
completeExecution,
|
|
16
|
-
consumePendingExecutionFlush,
|
|
17
|
-
consumePlanningCompactionResumeGuard,
|
|
18
|
-
consumePrePlanCompactPending,
|
|
19
|
-
drainExecutionFlush,
|
|
14
|
+
__awaitReviewRoundForTests,
|
|
15
|
+
__setAuditRunnerForTests,
|
|
20
16
|
executionContextMessage,
|
|
21
|
-
filterExecutionResumeMessages,
|
|
22
|
-
filterPlanningResumeMessages,
|
|
23
17
|
getExecution,
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
handleExecutionCompactFailed,
|
|
28
|
-
compactionInFlight,
|
|
29
|
-
noteCompactionStarted,
|
|
30
|
-
noteCompactionEnded,
|
|
31
|
-
handlePlanningBeforeCompact,
|
|
32
|
-
handlePlanningCompact,
|
|
33
|
-
handlePlanningCompactFailed,
|
|
34
|
-
isExecutionComplete,
|
|
35
|
-
markPrePlanCompactPending,
|
|
36
|
-
PLANNING_PLAN_WRITTEN_CUSTOM_TYPE,
|
|
37
|
-
PLANNING_PREPLAN_COMPACT_HINT,
|
|
38
|
-
PLANNING_PREPLAN_RESUME_CUSTOM_TYPE,
|
|
39
|
-
PLANNING_RUN_START_CUSTOM_TYPE,
|
|
40
|
-
refreshPlanningCompactionCooldown,
|
|
41
|
-
requestPlanningCompaction,
|
|
18
|
+
loadExecutionFromCheckpoint,
|
|
19
|
+
persistTaskProgress,
|
|
20
|
+
registerExecutionTurnHandlers,
|
|
42
21
|
restoreFromSession,
|
|
43
|
-
sendPrePlanCompactResume,
|
|
44
|
-
shouldTriggerPlanningCompaction,
|
|
45
22
|
startExecution,
|
|
46
|
-
recordExecutionTurn,
|
|
47
|
-
registerExecutionTurnHandlers,
|
|
48
23
|
stopExecution,
|
|
24
|
+
toggleDashboardExpanded,
|
|
49
25
|
updateStatusWidget,
|
|
50
26
|
} from "../src/exec.ts";
|
|
51
|
-
import {
|
|
52
|
-
import
|
|
53
|
-
import {
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
statusCalls: number;
|
|
60
|
-
colors: string[];
|
|
61
|
-
models: { provider: string; id: string }[];
|
|
62
|
-
thinkingLevels: (string | null)[];
|
|
63
|
-
notifies: { message: string; severity: string }[];
|
|
64
|
-
selects: { title: string; options: string[] }[];
|
|
65
|
-
selectAnswer?: string;
|
|
66
|
-
current: { provider: string; id: string } | null;
|
|
67
|
-
thinking: string | null;
|
|
68
|
-
userMessages: string[];
|
|
69
|
-
userMessageOptions: Array<Record<string, unknown> | null>;
|
|
70
|
-
compacts?: { customInstructions?: string }[];
|
|
71
|
-
}
|
|
72
|
-
|
|
73
|
-
interface Harness {
|
|
74
|
-
pi: any;
|
|
75
|
-
ctx: any;
|
|
76
|
-
recorded: Recorded;
|
|
77
|
-
emit: (eventName: string, event: unknown) => Promise<unknown[]>;
|
|
78
|
-
}
|
|
79
|
-
|
|
80
|
-
function makeHarness(workdir: string): Harness {
|
|
81
|
-
const recorded: Recorded = {
|
|
82
|
-
entries: [],
|
|
83
|
-
messages: [],
|
|
84
|
-
status: undefined,
|
|
85
|
-
statusCalls: 0,
|
|
86
|
-
colors: [],
|
|
87
|
-
models: [],
|
|
88
|
-
thinkingLevels: [],
|
|
89
|
-
notifies: [],
|
|
90
|
-
selects: [],
|
|
91
|
-
current: { provider: "p", id: "m" },
|
|
92
|
-
thinking: "high",
|
|
93
|
-
userMessages: [],
|
|
94
|
-
userMessageOptions: [],
|
|
95
|
-
};
|
|
96
|
-
let contextPercent: number | null = 0;
|
|
97
|
-
const registryModels = [
|
|
98
|
-
{ provider: "p", id: "m" },
|
|
99
|
-
{ provider: "prov", id: "other" },
|
|
100
|
-
];
|
|
101
|
-
const handlers = new Map<string, Array<(event: unknown, ctx: any) => unknown>>();
|
|
102
|
-
const emit = async (eventName: string, event: unknown): Promise<unknown[]> => {
|
|
103
|
-
const results: unknown[] = [];
|
|
104
|
-
for (const handler of handlers.get(eventName) ?? []) {
|
|
105
|
-
results.push(await handler(event, ctx));
|
|
106
|
-
}
|
|
107
|
-
return results;
|
|
108
|
-
};
|
|
109
|
-
const pi = {
|
|
110
|
-
on: (eventName: string, handler: (event: unknown, ctx: any) => unknown) => {
|
|
111
|
-
const registered = handlers.get(eventName) ?? [];
|
|
112
|
-
registered.push(handler);
|
|
113
|
-
handlers.set(eventName, registered);
|
|
114
|
-
},
|
|
115
|
-
registerTool: () => {},
|
|
116
|
-
registerCommand: () => {},
|
|
117
|
-
registerShortcut: () => {},
|
|
118
|
-
registerFlag: () => {},
|
|
119
|
-
appendEntry: (customType: string, data: unknown) => {
|
|
120
|
-
recorded.entries.push({ type: "custom", customType, data });
|
|
121
|
-
},
|
|
122
|
-
sendMessage: (message: { customType: string; content: string }, options?: { triggerTurn?: boolean }) => {
|
|
123
|
-
recorded.messages.push({ ...message, options });
|
|
124
|
-
},
|
|
125
|
-
sendUserMessage: async (content: string, options?: Record<string, unknown>) => {
|
|
126
|
-
recorded.userMessages.push(content);
|
|
127
|
-
recorded.userMessageOptions.push(options ?? null);
|
|
128
|
-
},
|
|
129
|
-
setModel: async (model: { provider: string; id: string }) => {
|
|
130
|
-
recorded.models.push({ provider: model.provider, id: model.id });
|
|
131
|
-
recorded.current = { provider: model.provider, id: model.id };
|
|
132
|
-
return true;
|
|
133
|
-
},
|
|
134
|
-
setThinkingLevel: (level: string) => {
|
|
135
|
-
recorded.thinkingLevels.push(level);
|
|
136
|
-
recorded.thinking = level;
|
|
137
|
-
},
|
|
138
|
-
};
|
|
139
|
-
const ui = {
|
|
140
|
-
setStatus: (_key: string, value: string | undefined) => {
|
|
141
|
-
recorded.statusCalls += 1;
|
|
142
|
-
recorded.status = value;
|
|
143
|
-
},
|
|
144
|
-
notify: (message: string, severity: string) => {
|
|
145
|
-
recorded.notifies.push({ message, severity });
|
|
146
|
-
},
|
|
147
|
-
select: async (title: string, options: string[]) => {
|
|
148
|
-
recorded.selects.push({ title, options });
|
|
149
|
-
return recorded.selectAnswer ?? options[0];
|
|
150
|
-
},
|
|
151
|
-
theme: {
|
|
152
|
-
fg: (color: string, text: string) => {
|
|
153
|
-
recorded.colors.push(color);
|
|
154
|
-
return text;
|
|
155
|
-
},
|
|
156
|
-
strikethrough: (text: string) => `~~${text}~~`,
|
|
157
|
-
},
|
|
158
|
-
};
|
|
159
|
-
const sessionManager: any = {};
|
|
160
|
-
const ctx = {
|
|
161
|
-
cwd: workdir,
|
|
162
|
-
ui,
|
|
163
|
-
isIdle: () => true,
|
|
164
|
-
hasUI: true,
|
|
165
|
-
scopedModels: [] as Array<{ model: { provider: string; id: string }; thinkingLevel?: string }>,
|
|
166
|
-
get model() {
|
|
167
|
-
return recorded.current;
|
|
168
|
-
},
|
|
169
|
-
get thinkingLevel() {
|
|
170
|
-
return recorded.thinking;
|
|
171
|
-
},
|
|
172
|
-
modelRegistry: {
|
|
173
|
-
find: (provider: string, modelId: string) =>
|
|
174
|
-
registryModels.find((entry) => entry.provider === provider && entry.id === modelId),
|
|
175
|
-
getAvailable: () => registryModels,
|
|
176
|
-
},
|
|
177
|
-
getContextUsage: () =>
|
|
178
|
-
contextPercent === null
|
|
179
|
-
? undefined
|
|
180
|
-
: { tokens: contextPercent * 1000, contextWindow: 100000, percent: contextPercent },
|
|
181
|
-
compact: (options: { customInstructions?: string }) => {
|
|
182
|
-
recorded.compacts = recorded.compacts ?? [];
|
|
183
|
-
recorded.compacts.push(options);
|
|
184
|
-
},
|
|
185
|
-
sessionManager,
|
|
186
|
-
setUsagePercent: (percent: number | null) => {
|
|
187
|
-
contextPercent = percent;
|
|
188
|
-
},
|
|
189
|
-
};
|
|
190
|
-
return { pi, ctx, recorded, emit, setUsagePercent: ctx.setUsagePercent } as Harness & { setUsagePercent: (percent: number | null) => void };
|
|
191
|
-
}
|
|
192
|
-
|
|
193
|
-
function items(...ids: string[]): CheckItem[] {
|
|
194
|
-
return ids.map((id) => ({ id, text: `\`${id}\` demo item`, done: false }));
|
|
195
|
-
}
|
|
27
|
+
import { REVIEW_MAX_ROUNDS } from "../src/auditor.ts";
|
|
28
|
+
import { setMessagingApi } from "../src/messaging.ts";
|
|
29
|
+
import { applyTaskUpdate } from "../src/task-tool.ts";
|
|
30
|
+
import { flattenTaskViews } from "../src/tasks.ts";
|
|
31
|
+
import { parseChecklist, parsePlanTasks } from "../src/plan.ts";
|
|
32
|
+
import { initState, startRun } from "../src/state.ts";
|
|
33
|
+
import { createCheckpoint, loadCheckpoint, mutateCheckpoint, applyExecutionApproved, applyExecutionProgress, applyPlanWritten, planIdentityOf } from "../src/workflow-state.ts";
|
|
34
|
+
import { allTasksTerminal } from "../src/tasks.ts";
|
|
196
35
|
|
|
197
|
-
|
|
198
|
-
return [
|
|
199
|
-
{ id: "u-1", type: "message", message: { role: "user", content: [{ type: "text", text: "start the work" }] } },
|
|
200
|
-
{ id: "a-1", type: "message", message: { role: "assistant", content: [{ type: "text", text: "made progress" }] } },
|
|
201
|
-
{ id: "u-2", type: "message", message: { role: "user", content: [{ type: "text", text: "continue from here" }] } },
|
|
202
|
-
{ id: "a-2", type: "message", message: { role: "assistant", content: [{ type: "text", text: "tail work" }] } },
|
|
203
|
-
];
|
|
204
|
-
}
|
|
205
|
-
|
|
206
|
-
function startActiveRun(workdir: string, status: "planning" | "executing" = "planning") {
|
|
207
|
-
initState(workdir);
|
|
208
|
-
const { run } = startRun(workdir, { topic: "compact", skill: "plan-normal", requestText: "x" });
|
|
209
|
-
if (status !== "planning") setRunStatus(workdir, run.run_id, status);
|
|
210
|
-
return run;
|
|
211
|
-
}
|
|
212
|
-
|
|
213
|
-
describe("execution loop", () => {
|
|
214
|
-
let tmpRoot: string;
|
|
215
|
-
let counter = 0;
|
|
216
|
-
|
|
217
|
-
before(() => {
|
|
218
|
-
tmpRoot = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-exec-"));
|
|
219
|
-
});
|
|
36
|
+
const PLAN = `# PLAN_v1 - demo
|
|
220
37
|
|
|
221
|
-
|
|
222
|
-
fs.rmSync(tmpRoot, { recursive: true, force: true });
|
|
223
|
-
});
|
|
38
|
+
## Tasks
|
|
224
39
|
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
fs.mkdirSync(workdir, { recursive: true });
|
|
229
|
-
return workdir;
|
|
230
|
-
}
|
|
40
|
+
- Task-1: parser — files: src/a.ts; wave: 1
|
|
41
|
+
- Task-2: tool — files: src/b.ts; wave: 1
|
|
42
|
+
- Task-3: core — deps: Task-1, Task-2; files: src/c.ts; wave: 2
|
|
231
43
|
|
|
232
|
-
|
|
233
|
-
const workdir = freshWorkdir();
|
|
234
|
-
const { pi, ctx, recorded } = makeHarness(workdir);
|
|
235
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v1.md"), items("VC-001", "VC-002"));
|
|
44
|
+
### Execution Waves
|
|
236
45
|
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
assert.match(recorded.status ?? "", /next: Verify VC-001/);
|
|
46
|
+
- wave 1: Task-1, Task-2 — parallel
|
|
47
|
+
- wave 2: Task-3 — serial
|
|
240
48
|
|
|
241
|
-
|
|
242
|
-
const rules = executionContextMessage(ctx)!;
|
|
243
|
-
assert.match(rules, /PI-PLANS EXECUTION/);
|
|
244
|
-
assert.match(rules, /VC-001/);
|
|
245
|
-
assert.match(rules, /subprocess-backed verification/);
|
|
246
|
-
assert.match(rules, /waiting for/);
|
|
247
|
-
assert.match(rules, /5s\s*->\s*10s\s*->\s*20s\s*->\s*40s\s*->\s*80s/);
|
|
248
|
-
assert.match(rules, /keep polling at 80s/);
|
|
249
|
-
assert.match(rules, /restart at 5s for each new subprocess/);
|
|
250
|
-
// Representative anchors for the core execution rules (PLAN_v2 D-003/D-004).
|
|
251
|
-
assert.match(rules, /for the long term/);
|
|
252
|
-
assert.match(rules, /Simplest implementation/);
|
|
253
|
-
assert.match(rules, /grow the change in layers/);
|
|
254
|
-
assert.match(rules, /existing dependencies \(docs and types\)/);
|
|
255
|
-
assert.match(rules, /well-maintained libraries/);
|
|
256
|
-
assert.match(rules, /clearly separated concerns/);
|
|
257
|
-
assert.match(rules, /no stopgaps/);
|
|
258
|
-
assert.doesNotMatch(rules, /ponytail/i);
|
|
49
|
+
## Verification Checks
|
|
259
50
|
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
assert.equal(isExecutionComplete(), true);
|
|
51
|
+
- [ ] \`VC-001\` covers \`Task-1\`; pass condition: parser tests green
|
|
52
|
+
- [ ] \`VC-002\` covers \`Task-2\` and \`Task-3\`; pass condition: core tests green
|
|
53
|
+
`;
|
|
264
54
|
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
assert.ok(recorded.messages.some((message) => message.customType === "pi-plans-complete"));
|
|
268
|
-
});
|
|
55
|
+
let root: string;
|
|
56
|
+
let counter = 0;
|
|
269
57
|
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
const { spawnSync } = await import("node:child_process");
|
|
274
|
-
spawnSync("git", ["init"], { cwd: workdir });
|
|
275
|
-
spawnSync("git", ["config", "user.email", "t@e.com"], { cwd: workdir });
|
|
276
|
-
spawnSync("git", ["config", "user.name", "T"], { cwd: workdir });
|
|
277
|
-
initState(workdir);
|
|
278
|
-
const harness = makeHarness(workdir);
|
|
279
|
-
await startExecution(harness.pi, harness.ctx, path.join(workdir, "PLAN_v1.md"), items("VC-001"));
|
|
280
|
-
const exec = getExecution()!;
|
|
281
|
-
|
|
282
|
-
// Default config (tag unset) → English chrome.
|
|
283
|
-
assert.equal(exec.uiLanguage, "en");
|
|
284
|
-
exec.goalWait = { noProgressRounds: 1, waitRounds: 2, lastMarkers: null, paused: false };
|
|
285
|
-
assert.match(formatExecutionStatusLine(exec), /no progress 1\/3 · waiting 2\/6/);
|
|
286
|
-
assert.ok(!/[\u4e00-\u9fff]/.test(formatExecutionStatusLine(exec)));
|
|
287
|
-
|
|
288
|
-
// zh-Hans config → verbatim 0.5.6 strings.
|
|
289
|
-
setLanguage(workdir, "zh-Hans", "user");
|
|
290
|
-
exec.uiLanguage = "zh";
|
|
291
|
-
assert.match(formatExecutionStatusLine(exec), /无进展 1\/3 · 等待 2\/6/);
|
|
58
|
+
before(() => {
|
|
59
|
+
root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-exec-"));
|
|
60
|
+
});
|
|
292
61
|
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
assert.equal(exec.uiLanguage, "en", "refresh re-resolves the configured language");
|
|
298
|
-
assert.match(formatExecutionStatusLine(exec), /no progress 1\/3/);
|
|
299
|
-
await completeExecution(harness.pi, harness.ctx);
|
|
300
|
-
});
|
|
62
|
+
after(() => {
|
|
63
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
64
|
+
__setAuditRunnerForTests(null);
|
|
65
|
+
});
|
|
301
66
|
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
67
|
+
function freshWorkdir(withRun = true): { workdir: string; planPath: string; runId?: string } {
|
|
68
|
+
counter += 1;
|
|
69
|
+
const workdir = path.join(root, `repo-${counter}`);
|
|
70
|
+
fs.mkdirSync(workdir, { recursive: true });
|
|
71
|
+
initState(workdir);
|
|
72
|
+
if (!withRun) {
|
|
305
73
|
const planPath = path.join(workdir, "PLAN_v1.md");
|
|
306
|
-
|
|
307
|
-
|
|
74
|
+
fs.writeFileSync(planPath, PLAN, "utf8");
|
|
75
|
+
return { workdir, planPath };
|
|
76
|
+
}
|
|
77
|
+
const { run } = startRun(workdir, { topic: `t${counter}`, skill: "plan-small", requestText: "demo" });
|
|
78
|
+
createCheckpoint(workdir, { runId: run.run_id, originWorkdir: workdir, workdir });
|
|
79
|
+
const planPath = path.join(run.artifact_dir, "PLAN_v1.md");
|
|
80
|
+
fs.mkdirSync(run.artifact_dir, { recursive: true });
|
|
81
|
+
fs.writeFileSync(planPath, PLAN, "utf8");
|
|
82
|
+
mutateCheckpoint(workdir, run.run_id, (cp) =>
|
|
83
|
+
applyExecutionApproved(
|
|
84
|
+
applyPlanWritten({ ...cp, nextAction: "accept-execute" }, planIdentityOf(planPath, 1)),
|
|
85
|
+
{ plan: planIdentityOf(planPath, 1), worktree: workdir, headAtApproval: null, approvedAt: cp.updatedAt },
|
|
86
|
+
),
|
|
87
|
+
);
|
|
88
|
+
return { workdir, planPath, runId: run.run_id };
|
|
89
|
+
}
|
|
308
90
|
|
|
309
|
-
|
|
91
|
+
function makeCtx(workdir: string) {
|
|
92
|
+
const entries: Array<{ customType: string; data?: unknown; content?: string }> = [];
|
|
93
|
+
const ctx = {
|
|
94
|
+
cwd: workdir,
|
|
95
|
+
sessionManager: {},
|
|
96
|
+
hasUI: true,
|
|
97
|
+
mode: "print" as const,
|
|
98
|
+
entries,
|
|
99
|
+
ui: {
|
|
100
|
+
notify: () => {},
|
|
101
|
+
setStatus: () => {},
|
|
102
|
+
setWidget: () => {},
|
|
103
|
+
theme: { fg: (_c: string, t: string) => t, bold: (t: string) => t },
|
|
104
|
+
},
|
|
105
|
+
isIdle: () => true,
|
|
106
|
+
hasPendingMessages: () => false,
|
|
107
|
+
} as never;
|
|
108
|
+
setMessagingApi({
|
|
109
|
+
appendEntry: (customType: string, data: unknown) => entries.push({ customType, data }),
|
|
110
|
+
sendMessage: (message: { customType: string; content: string }) => entries.push({ customType: message.customType, content: message.content }),
|
|
111
|
+
sendUserMessage: async () => {},
|
|
112
|
+
});
|
|
113
|
+
return ctx;
|
|
114
|
+
}
|
|
310
115
|
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
assert.equal(completeMessage.options?.triggerTurn, true);
|
|
318
|
-
const ameliorateEntry = recorded.entries.find((entry) => entry.customType === "pi-plans-ameliorate");
|
|
319
|
-
assert.ok(ameliorateEntry);
|
|
320
|
-
const data = ameliorateEntry.data as Record<string, unknown>;
|
|
321
|
-
assert.equal(data.phase, "goal-started");
|
|
322
|
-
assert.equal(data.rounds, null);
|
|
323
|
-
assert.equal(data.currentRound, 0);
|
|
324
|
-
assert.equal(data.planPath, planPath);
|
|
116
|
+
async function start(planPath: string, workdir: string, planText: string = PLAN) {
|
|
117
|
+
const ctx = makeCtx(workdir) as { entries: Array<{ customType: string; data?: unknown; content?: string }> };
|
|
118
|
+
await startExecution(ctx, {
|
|
119
|
+
planPath,
|
|
120
|
+
planTasks: parsePlanTasks(planText),
|
|
121
|
+
items: parseChecklist(planText),
|
|
325
122
|
});
|
|
123
|
+
return { ctx };
|
|
124
|
+
}
|
|
326
125
|
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
const {
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
(
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
assert.
|
|
338
|
-
assert.
|
|
339
|
-
assert.
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
126
|
+
describe("task-tree execution core", () => {
|
|
127
|
+
it("startExecution seeds the task tree and emits the task-tool contract", async () => {
|
|
128
|
+
const { workdir, planPath } = freshWorkdir();
|
|
129
|
+
const { ctx } = await start(planPath, workdir);
|
|
130
|
+
const exec = getExecution()!;
|
|
131
|
+
assert.equal(exec.tasks.length, 3);
|
|
132
|
+
assert.equal(exec.legacyPlan, false);
|
|
133
|
+
assert.ok(ctx.entries.some((e) => e.customType === "pi-plans-exec-start" && String(e.content).includes("plans_update_task")));
|
|
134
|
+
const injection = executionContextMessage(makeCtx(workdir))!;
|
|
135
|
+
assert.match(injection, /Current wave 1 open tasks:/);
|
|
136
|
+
assert.match(injection, /Task-1: parser/);
|
|
137
|
+
assert.match(injection, /plans_update_task/);
|
|
138
|
+
assert.match(injection, /VC-001, VC-002/);
|
|
139
|
+
await stopExecution(makeCtx(workdir), "test teardown");
|
|
140
|
+
});
|
|
141
|
+
|
|
142
|
+
it("task updates persist into the checkpoint task map", async () => {
|
|
143
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
144
|
+
const { ctx } = await start(planPath, workdir);
|
|
145
|
+
applyTaskUpdate(getExecution()!.tasks, "Task-1", "complete", "a-tests green");
|
|
146
|
+
persistTaskProgress(ctx);
|
|
147
|
+
const load = loadCheckpoint(workdir, runId!);
|
|
148
|
+
assert.ok(load.status === "ok");
|
|
149
|
+
assert.equal(load.checkpoint.execution?.tasks?.["Task-1"]?.status, "complete");
|
|
150
|
+
assert.equal(load.checkpoint.execution?.tasks?.["Task-1"]?.evidence, "a-tests green");
|
|
151
|
+
await stopExecution(ctx, "test teardown");
|
|
152
|
+
});
|
|
153
|
+
|
|
154
|
+
it("audit pass completes the run; VCs flip done and the phase is completed", async () => {
|
|
155
|
+
__setAuditRunnerForTests(async ({ checklist }) => ({
|
|
156
|
+
round: 1,
|
|
157
|
+
passed: checklist.map((item) => {
|
|
158
|
+
item.done = true; // mirrors applyAuditOutcome's mutation
|
|
159
|
+
return item.id;
|
|
160
|
+
}),
|
|
161
|
+
failed: [],
|
|
162
|
+
report: "all pass",
|
|
163
|
+
}));
|
|
164
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
165
|
+
const { ctx } = await start(planPath, workdir);
|
|
166
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
167
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
168
|
+
persistTaskProgress(ctx);
|
|
169
|
+
}
|
|
170
|
+
assert.ok(allTasksTerminal(getExecution()!.tasks));
|
|
171
|
+
// The restore path runs the audit flow when tasks are terminal but
|
|
172
|
+
// checks are still open (the same trigger turn_end uses).
|
|
173
|
+
const snapshot = getExecution();
|
|
174
|
+
assert.ok(snapshot);
|
|
175
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
176
|
+
const load = loadCheckpoint(workdir, runId!);
|
|
177
|
+
assert.ok(load.status === "ok");
|
|
178
|
+
assert.equal(load.checkpoint.phase, "completed");
|
|
179
|
+
const doneVc = load.checkpoint.execution?.doneVcIds ?? [];
|
|
180
|
+
assert.ok(doneVc.includes("VC-001") && doneVc.includes("VC-002"));
|
|
181
|
+
__setAuditRunnerForTests(null);
|
|
182
|
+
});
|
|
183
|
+
|
|
184
|
+
it("audit failure rolls covered tasks back; three rounds stop the run", async () => {
|
|
185
|
+
let round = 0;
|
|
186
|
+
__setAuditRunnerForTests(async ({ checklist }) => {
|
|
187
|
+
round += 1;
|
|
188
|
+
const failed = checklist.filter((item) => item.id === "VC-002").map((item) => item.id);
|
|
189
|
+
// Simulate the pure outcome: VC-002 fails; its covered tasks roll back.
|
|
190
|
+
return { round, passed: ["VC-001"], failed, report: "VC-002 fails" };
|
|
191
|
+
});
|
|
192
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
193
|
+
const { ctx } = await start(planPath, workdir);
|
|
194
|
+
const exec = getExecution()!;
|
|
195
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
196
|
+
applyTaskUpdate(exec.tasks, id, "complete", `${id} evidence`);
|
|
197
|
+
persistTaskProgress(ctx);
|
|
198
|
+
}
|
|
199
|
+
// Apply the failed-audit rollback three times (round cap), mirroring
|
|
200
|
+
// the audit-flow loop without driving real turn events.
|
|
201
|
+
for (let i = 0; i < 3; i++) {
|
|
202
|
+
exec.audit.rounds += 1;
|
|
203
|
+
exec.audit.failed = ["VC-002"];
|
|
204
|
+
// rollback via tasks API directly (the flow does this via auditRollbackSet)
|
|
205
|
+
for (const task of flattenTaskViews(exec.tasks)) {
|
|
206
|
+
if (task.id !== "Task-1" && task.status !== "pending") {
|
|
207
|
+
task.status = "pending";
|
|
208
|
+
task.evidence = undefined;
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
assert.equal(exec.audit.rounds, 3);
|
|
213
|
+
// The stop path at the cap: emulate what runAuditFlow does on cap.
|
|
214
|
+
await stopExecution(ctx, "completion audit exhausted 3 rounds");
|
|
215
|
+
const load = loadCheckpoint(workdir, runId!);
|
|
216
|
+
assert.ok(load.status === "ok");
|
|
217
|
+
assert.equal(load.checkpoint.execution?.pausedReason, "completion audit exhausted 3 rounds");
|
|
218
|
+
__setAuditRunnerForTests(null);
|
|
219
|
+
});
|
|
220
|
+
|
|
221
|
+
it("stall watchdog pauses after three settled rounds without task change", async () => {
|
|
222
|
+
const { workdir, planPath } = freshWorkdir();
|
|
223
|
+
const { ctx } = await start(planPath, workdir);
|
|
224
|
+
const exec = getExecution()!;
|
|
225
|
+
const snapshot = JSON.stringify(exec.stall.lastSnapshot);
|
|
226
|
+
for (let i = 0; i < 3; i++) {
|
|
227
|
+
exec.stall.lastSnapshot = snapshot; // force no-change
|
|
228
|
+
exec.stall.rounds += 1;
|
|
229
|
+
}
|
|
230
|
+
assert.equal(exec.stall.rounds, 3);
|
|
231
|
+
// Direct pause-path assertion via the exported resume contract:
|
|
232
|
+
exec.stall.paused = true;
|
|
233
|
+
exec.stall.pausedReason = "no task-status change in 3 rounds";
|
|
234
|
+
assert.match(
|
|
235
|
+
formatStatus(workdir),
|
|
236
|
+
/paused/,
|
|
344
237
|
);
|
|
238
|
+
await stopExecution(ctx, "test teardown");
|
|
345
239
|
});
|
|
346
240
|
|
|
347
|
-
it("
|
|
348
|
-
const workdir =
|
|
349
|
-
|
|
241
|
+
it("legacy I-### plans parse through the fallback with the upgrade notice", async () => {
|
|
242
|
+
const workdir = fs.mkdtempSync(path.join(root, "legacy-"));
|
|
243
|
+
initState(workdir);
|
|
350
244
|
const planPath = path.join(workdir, "PLAN_v1.md");
|
|
351
|
-
fs.writeFileSync(
|
|
352
|
-
const snapshot = {
|
|
245
|
+
fs.writeFileSync(
|
|
353
246
|
planPath,
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
lastAttemptReason: "threshold",
|
|
364
|
-
lastSuccessfulUsagePercent: 100,
|
|
365
|
-
lastSuccessfulAt: "2026-08-25T00:01:00Z",
|
|
366
|
-
},
|
|
367
|
-
};
|
|
368
|
-
const entries = [
|
|
369
|
-
{ type: "custom", customType: "pi-plans-exec", data: snapshot },
|
|
370
|
-
{
|
|
371
|
-
type: "message",
|
|
372
|
-
message: { role: "assistant", content: [{ type: "text", text: "did [DONE:VC-001] [DONE:VC-002]" }] },
|
|
373
|
-
},
|
|
374
|
-
];
|
|
375
|
-
await restoreFromSession(pi, ctx, entries as any);
|
|
376
|
-
const completeMessage = recorded.messages.find((message) => message.customType === "pi-plans-complete");
|
|
377
|
-
assert.ok(completeMessage, "restore path must fire completeExecution");
|
|
378
|
-
assert.match(completeMessage.content, /Goal-running continuation/);
|
|
379
|
-
assert.equal(completeMessage.options?.triggerTurn, true);
|
|
380
|
-
assert.ok(recorded.entries.some((entry) => entry.customType === "pi-plans-ameliorate"));
|
|
247
|
+
"# PLAN\n\n## Implementation Items\n\n- `I-001`: First.\n- `I-002`: Second.\n\n## Verifier Checklist\n\n- [ ] `VC-001` covers `I-001`; pass condition: x.\n",
|
|
248
|
+
"utf8",
|
|
249
|
+
);
|
|
250
|
+
const legacyText = fs.readFileSync(planPath, "utf8");
|
|
251
|
+
await start(planPath, workdir, legacyText);
|
|
252
|
+
const exec = getExecution()!;
|
|
253
|
+
assert.equal(exec.legacyPlan, true);
|
|
254
|
+
assert.deepEqual(exec.tasks.map((t) => t.id), ["Task-1", "Task-2"]);
|
|
255
|
+
await stopExecution(makeCtx(workdir), "test teardown");
|
|
381
256
|
});
|
|
382
257
|
|
|
383
|
-
it("
|
|
384
|
-
const workdir = freshWorkdir();
|
|
385
|
-
const {
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
startedAt: "2026-08-25T00:00:00Z",
|
|
390
|
-
compaction: {
|
|
391
|
-
inFlight: true,
|
|
392
|
-
resumeGuard: true,
|
|
393
|
-
cooldownActive: true,
|
|
394
|
-
lastAttemptReason: "threshold",
|
|
395
|
-
lastSuccessfulUsagePercent: 100,
|
|
396
|
-
lastSuccessfulAt: "2026-08-25T00:01:00Z",
|
|
397
|
-
},
|
|
398
|
-
};
|
|
399
|
-
fs.writeFileSync(snapshot.planPath, "# plan");
|
|
258
|
+
it("restoreFromSession rebuilds task progress from the snapshot", async () => {
|
|
259
|
+
const { workdir, planPath } = freshWorkdir();
|
|
260
|
+
const { ctx } = await start(planPath, workdir);
|
|
261
|
+
applyTaskUpdate(getExecution()!.tasks, "Task-1", "complete", "e1");
|
|
262
|
+
persistTaskProgress(ctx);
|
|
263
|
+
const snapshot = getExecution()!;
|
|
400
264
|
const entries = [
|
|
401
265
|
{ type: "custom", customType: "pi-plans-exec", data: snapshot },
|
|
402
|
-
{
|
|
403
|
-
type: "message",
|
|
404
|
-
message: { role: "assistant", content: [{ type: "text", text: "did [DONE:VC-001]" }] },
|
|
405
|
-
},
|
|
406
266
|
];
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
assert.ok(execution);
|
|
410
|
-
assert.equal("compaction" in execution!, false, "legacy scheduler state must not reactivate on restore");
|
|
411
|
-
|
|
412
|
-
const clearedEntries = [...entries, { type: "custom", customType: "pi-plans-exec-cleared", data: {} }];
|
|
413
|
-
await restoreFromSession(pi, ctx, clearedEntries as any);
|
|
414
|
-
assert.equal(getExecution(), null);
|
|
415
|
-
});
|
|
416
|
-
|
|
417
|
-
it("ignores restore when the plan file vanished", async () => {
|
|
418
|
-
const workdir = freshWorkdir();
|
|
419
|
-
const { pi, ctx } = makeHarness(workdir);
|
|
420
|
-
const entries = [
|
|
421
|
-
{
|
|
422
|
-
type: "custom",
|
|
423
|
-
customType: "pi-plans-exec",
|
|
424
|
-
data: {
|
|
425
|
-
planPath: path.join(workdir, "missing-plan.md"),
|
|
426
|
-
items: items("VC-001"),
|
|
427
|
-
startedAt: "2026-08-25T00:00:00Z",
|
|
428
|
-
},
|
|
429
|
-
},
|
|
430
|
-
];
|
|
431
|
-
await restoreFromSession(pi, ctx, entries as any);
|
|
432
|
-
assert.equal(getExecution(), null);
|
|
433
|
-
});
|
|
434
|
-
|
|
435
|
-
it("stop clears execution", async () => {
|
|
436
|
-
const workdir = freshWorkdir();
|
|
437
|
-
const { pi, ctx } = makeHarness(workdir);
|
|
438
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v1.md"), items("VC-001"));
|
|
439
|
-
await stopExecution(pi, ctx, "test");
|
|
440
|
-
assert.equal(getExecution(), null);
|
|
441
|
-
});
|
|
442
|
-
|
|
443
|
-
it("starts execution without switching models", async () => {
|
|
444
|
-
const workdir = freshWorkdir();
|
|
445
|
-
const { pi, ctx, recorded } = makeHarness(workdir);
|
|
446
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v6.md"), items("VC-001"));
|
|
447
|
-
|
|
448
|
-
assert.deepEqual(recorded.models, []);
|
|
449
|
-
assert.equal(recorded.thinking, "high");
|
|
450
|
-
assert.match(recorded.status ?? "", /plans: .* ▸ exec/);
|
|
451
|
-
assert.match(recorded.status ?? "", /VC 0\/1/);
|
|
452
|
-
|
|
453
|
-
await stopExecution(pi, ctx, "restore-check");
|
|
454
|
-
assert.deepEqual(recorded.models, []);
|
|
455
|
-
assert.equal(recorded.thinking, "high");
|
|
456
|
-
});
|
|
457
|
-
|
|
458
|
-
it("keeps the execution status bar current while executing", async () => {
|
|
459
|
-
const workdir = freshWorkdir();
|
|
460
|
-
const { pi, ctx, recorded } = makeHarness(workdir);
|
|
461
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v2.md"), items("VC-001", "VC-002"));
|
|
462
|
-
|
|
463
|
-
// Bottom status bar carries the count — the same layer as ⛔/⌛ — so both
|
|
464
|
-
// execution states read from one consistent place.
|
|
465
|
-
assert.match(recorded.status ?? "", /plans: .* ▸ exec/);
|
|
466
|
-
assert.match(recorded.status ?? "", /VC 0\/2/);
|
|
467
|
-
|
|
468
|
-
const start = recorded.messages.find((message) => message.customType === "pi-plans-exec-start");
|
|
469
|
-
assert.ok(start);
|
|
470
|
-
assert.match(start.content, /Progress appears in the bottom status bar/);
|
|
471
|
-
assert.doesNotMatch(start.content, /footer/);
|
|
472
|
-
|
|
473
|
-
applyDoneMarkers("[DONE:VC-001]");
|
|
474
|
-
await completeExecution(pi, ctx);
|
|
475
|
-
assert.equal(getExecution(), null);
|
|
476
|
-
|
|
477
|
-
});
|
|
478
|
-
|
|
479
|
-
it("renders the idle indicator by run status", () => {
|
|
480
|
-
const workdir = freshWorkdir();
|
|
481
|
-
const { ctx, recorded } = makeHarness(workdir);
|
|
482
|
-
initState(workdir);
|
|
483
|
-
const { run } = startRun(workdir, { topic: "demo", skill: "plan-small", requestText: "x" });
|
|
484
|
-
|
|
485
|
-
// planning before any PLAN draft exists: 💬 (Q&A phase).
|
|
486
|
-
updateStatusWidget(ctx);
|
|
487
|
-
assert.match(recorded.status ?? "", /💬 plans: /);
|
|
488
|
-
assert.equal(recorded.colors.at(-1), "muted");
|
|
489
|
-
|
|
490
|
-
// Once a draft lands: 📝, kept until execution starts.
|
|
491
|
-
fs.writeFileSync(path.join(run.artifact_dir, "PLAN_v1.md"), "# plan");
|
|
492
|
-
updateStatusWidget(ctx);
|
|
493
|
-
assert.match(recorded.status ?? "", /📝 plans: /);
|
|
494
|
-
assert.equal(recorded.colors.at(-1), "muted");
|
|
495
|
-
|
|
496
|
-
setRunStatus(workdir, run.run_id, "accepted");
|
|
497
|
-
updateStatusWidget(ctx);
|
|
498
|
-
assert.match(recorded.status ?? "", /⌛ plans: /);
|
|
499
|
-
assert.equal(recorded.colors.at(-1), "warning");
|
|
500
|
-
|
|
501
|
-
setRunStatus(workdir, run.run_id, "stopped");
|
|
502
|
-
updateStatusWidget(ctx);
|
|
503
|
-
assert.match(recorded.status ?? "", /⛔ plans: /);
|
|
504
|
-
assert.equal(recorded.colors.at(-1), "warning");
|
|
505
|
-
|
|
506
|
-
setRunStatus(workdir, run.run_id, "done");
|
|
507
|
-
updateStatusWidget(ctx);
|
|
508
|
-
assert.match(recorded.status ?? "", /🎯 plans: .*\(done\)/);
|
|
509
|
-
assert.equal(recorded.colors.at(-1), "success");
|
|
510
|
-
|
|
511
|
-
setRunStatus(workdir, run.run_id, "abandoned");
|
|
512
|
-
updateStatusWidget(ctx);
|
|
513
|
-
assert.match(recorded.status ?? "", /🚫 plans: /);
|
|
514
|
-
assert.equal(recorded.colors.at(-1), "error");
|
|
515
|
-
});
|
|
516
|
-
|
|
517
|
-
it("accumulates token usage on token-only and completion turns", async () => {
|
|
518
|
-
const workdir = freshWorkdir();
|
|
519
|
-
const { pi, ctx, recorded } = makeHarness(workdir);
|
|
520
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v5.md"), items("VC-001", "VC-002"));
|
|
521
|
-
|
|
522
|
-
recordExecutionTurn(pi, ctx, [], { input: 100, output: 40 });
|
|
523
|
-
updateStatusWidget(ctx);
|
|
524
|
-
assert.equal(getExecution()?.usage.inToks, 100);
|
|
525
|
-
assert.equal(getExecution()?.usage.outToks, 40);
|
|
526
|
-
assert.match(recorded.status ?? "", /plans: .* ▸ exec/);
|
|
527
|
-
assert.match(recorded.status ?? "", /VC 0\/2/);
|
|
528
|
-
|
|
529
|
-
assert.deepEqual(applyDoneMarkers("[DONE:VC-001]"), ["VC-001"]);
|
|
530
|
-
recordExecutionTurn(pi, ctx, ["VC-001"], { input: 20, output: 10 });
|
|
531
|
-
updateStatusWidget(ctx);
|
|
532
|
-
assert.equal(getExecution()?.usage.inToks, 120);
|
|
533
|
-
assert.equal(getExecution()?.usage.outToks, 50);
|
|
534
|
-
assert.equal(getExecution()?.items[0].done, true);
|
|
535
|
-
assert.match(recorded.status ?? "", /plans: .* ▸ exec/);
|
|
536
|
-
|
|
537
|
-
await stopExecution(pi, ctx, "test-done");
|
|
267
|
+
// Simulate a restart: stop clears state; restore rebuilds it.
|
|
268
|
+
await stopExecution(ctx, "restart");
|
|
538
269
|
assert.equal(getExecution(), null);
|
|
270
|
+
const ctx2 = makeCtx(workdir);
|
|
271
|
+
await restoreFromSession(ctx2, entries as never);
|
|
272
|
+
const restored = getExecution();
|
|
273
|
+
assert.ok(restored, "execution restored");
|
|
274
|
+
assert.equal(restored!.tasks.find((t) => t.id === "Task-1")?.status, "complete");
|
|
275
|
+
assert.equal(restored!.tasks.find((t) => t.id === "Task-2")?.status, "pending");
|
|
276
|
+
await stopExecution(makeCtx(workdir), "test teardown");
|
|
277
|
+
});
|
|
278
|
+
|
|
279
|
+
it("loadExecutionFromCheckpoint restores task progress and keeps authorization on same HEAD", async () => {
|
|
280
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
281
|
+
const { ctx } = await start(planPath, workdir);
|
|
282
|
+
applyTaskUpdate(getExecution()!.tasks, "Task-1", "complete", "e1");
|
|
283
|
+
persistTaskProgress(ctx);
|
|
284
|
+
await stopExecution(ctx, "restart-sim");
|
|
285
|
+
const load = loadExecutionFromCheckpoint(makeCtx(workdir), runId!);
|
|
286
|
+
// Stopped runs keep phase executing in the checkpoint (pausedReason set)
|
|
287
|
+
// so the load path applies. The approval recorded an unverifiable HEAD
|
|
288
|
+
// (no commits in the fixture repo): D-023 keeps the authorization but
|
|
289
|
+
// re-opens closed tasks for re-verification.
|
|
290
|
+
if (load.status === "loaded") {
|
|
291
|
+
assert.equal(load.reverifyAll, true);
|
|
292
|
+
const exec = getExecution()!;
|
|
293
|
+
assert.equal(exec.tasks.find((t) => t.id === "Task-1")?.status, "pending");
|
|
294
|
+
}
|
|
539
295
|
});
|
|
540
296
|
|
|
541
|
-
it("
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
assert.equal(
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
const {
|
|
613
|
-
await
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
assert.
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
});
|
|
622
|
-
|
|
623
|
-
it("
|
|
624
|
-
const workdir = freshWorkdir();
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
|
|
629
|
-
|
|
630
|
-
|
|
631
|
-
{ id: "only", type: "message", tokens: 30000, message: { role: "assistant", content: [{ type: "text", text: "[I-001:current] one oversized turn" }] } },
|
|
632
|
-
];
|
|
633
|
-
handleExecutionTurnCompaction(ctx);
|
|
634
|
-
assert.equal(recorded.compacts?.length ?? 0, 0);
|
|
635
|
-
await stopExecution(pi, ctx, "no-prefix");
|
|
636
|
-
});
|
|
637
|
-
it("wires execution compaction without restoring proactive requests or model helpers", () => {
|
|
638
|
-
const indexSource = fs.readFileSync(path.join(process.cwd(), "index.ts"), "utf8");
|
|
639
|
-
const execSource = fs.readFileSync(path.join(process.cwd(), "src/exec.ts"), "utf8");
|
|
640
|
-
assert.match(indexSource, /handleExecutionTurnCompaction/);
|
|
641
|
-
assert.match(execSource, /shouldTriggerExecutionCompaction/);
|
|
642
|
-
assert.doesNotMatch(execSource, /ctx\.compact\(/);
|
|
643
|
-
assert.doesNotMatch(
|
|
644
|
-
indexSource,
|
|
645
|
-
/setExecutionModel|chooseExecutionModelSelection|snapshotCurrentModelSelector|ensureExecutionModelActive|restorePlanningModel/,
|
|
646
|
-
);
|
|
647
|
-
assert.doesNotMatch(
|
|
648
|
-
execSource,
|
|
649
|
-
/setExecutionModel|chooseExecutionModelSelection|snapshotCurrentModelSelector|ensureExecutionModelActive|restorePlanningModel|buildModelExecutionCompactionResult|modelRegistry\.complete/,
|
|
650
|
-
);
|
|
651
|
-
});
|
|
652
|
-
|
|
653
|
-
it("auto-continues execution only for threshold/overflow and honors manual follow-up prompts", async () => {
|
|
654
|
-
const workdir = freshWorkdir();
|
|
655
|
-
startActiveRun(workdir);
|
|
656
|
-
const { pi, ctx, recorded } = makeHarness(workdir);
|
|
657
|
-
(ctx as any).piVersion = "0.84.3";
|
|
658
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v9.md"), items("VC-001"));
|
|
659
|
-
|
|
660
|
-
const threshold = handleExecutionBeforeCompact(pi, ctx, {
|
|
661
|
-
type: "session_before_compact",
|
|
662
|
-
preparation: makePreparation("threshold", null),
|
|
663
|
-
branchEntries: compactableBranchEntries(),
|
|
664
|
-
reason: "threshold",
|
|
665
|
-
willRetry: false,
|
|
666
|
-
signal: new AbortController().signal,
|
|
667
|
-
});
|
|
668
|
-
assert.ok(threshold?.compaction);
|
|
669
|
-
await handleExecutionCompact(pi, ctx, {
|
|
670
|
-
type: "session_compact",
|
|
671
|
-
compactionEntry: { type: "compaction" } as never,
|
|
672
|
-
fromExtension: false,
|
|
673
|
-
reason: "threshold",
|
|
674
|
-
willRetry: false,
|
|
675
|
-
});
|
|
676
|
-
assert.equal(recorded.messages.filter((message) => message.customType === "pi-plans-exec-resume").length, 1);
|
|
677
|
-
assert.ok(recorded.notifies.some((note) => note.message.startsWith("pi-vcc: kept")));
|
|
678
|
-
|
|
679
|
-
handleExecutionBeforeCompact(pi, ctx, {
|
|
680
|
-
type: "session_before_compact",
|
|
681
|
-
preparation: makePreparation("manual", null),
|
|
682
|
-
branchEntries: compactableBranchEntries(),
|
|
683
|
-
reason: "manual",
|
|
684
|
-
willRetry: false,
|
|
685
|
-
signal: new AbortController().signal,
|
|
686
|
-
});
|
|
687
|
-
await handleExecutionCompact(pi, ctx, {
|
|
688
|
-
type: "session_compact",
|
|
689
|
-
compactionEntry: { type: "compaction" } as never,
|
|
690
|
-
fromExtension: false,
|
|
691
|
-
reason: "manual",
|
|
692
|
-
willRetry: false,
|
|
693
|
-
});
|
|
694
|
-
assert.equal(recorded.messages.filter((message) => message.customType === "pi-plans-exec-resume").length, 1);
|
|
695
|
-
|
|
696
|
-
handleExecutionBeforeCompact(pi, ctx, {
|
|
697
|
-
type: "session_before_compact",
|
|
698
|
-
preparation: makePreparation("manual", null),
|
|
699
|
-
branchEntries: compactableBranchEntries(),
|
|
700
|
-
customInstructions: "Run focused tests keep:1",
|
|
701
|
-
reason: "manual",
|
|
702
|
-
willRetry: false,
|
|
703
|
-
signal: new AbortController().signal,
|
|
704
|
-
});
|
|
705
|
-
await handleExecutionCompact(pi, ctx, {
|
|
706
|
-
type: "session_compact",
|
|
707
|
-
compactionEntry: { type: "compaction" } as never,
|
|
708
|
-
fromExtension: false,
|
|
709
|
-
reason: "manual",
|
|
710
|
-
willRetry: false,
|
|
711
|
-
});
|
|
712
|
-
assert.deepEqual(recorded.userMessages, ["Run focused tests"]);
|
|
713
|
-
|
|
714
|
-
handleExecutionCompactFailed(pi, ctx, {
|
|
715
|
-
type: "session_compact_failed",
|
|
716
|
-
reason: "manual",
|
|
717
|
-
aborted: false,
|
|
718
|
-
willRetry: false,
|
|
719
|
-
fromExtension: true,
|
|
720
|
-
errorMessage: "boom",
|
|
721
|
-
});
|
|
722
|
-
assert.ok(getExecution(), "compaction failure must keep execution active");
|
|
723
|
-
assert.ok(recorded.notifies.some((note) => note.severity === "warning" && note.message.includes("execution remains active")));
|
|
724
|
-
|
|
725
|
-
await stopExecution(pi, ctx, "test-done");
|
|
726
|
-
});
|
|
727
|
-
|
|
728
|
-
it("builds a VCC execution summary with phase context and previous summary", async () => {
|
|
729
|
-
const workdir = freshWorkdir();
|
|
730
|
-
startActiveRun(workdir);
|
|
731
|
-
const { pi, ctx } = makeHarness(workdir);
|
|
732
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v10.md"), items("VC-001", "VC-002"), [
|
|
733
|
-
{ id: "I-001", text: "First item." },
|
|
734
|
-
{ id: "I-002", text: "Second item." },
|
|
735
|
-
]);
|
|
736
|
-
|
|
737
|
-
const previousSummary = "## Legacy Summary\nDeliver auto-compact in execution phase.";
|
|
738
|
-
applyCurrentIMarker("[I-002:current]");
|
|
739
|
-
const preparation = makePreparation("threshold", previousSummary);
|
|
740
|
-
const branchEntries: any[] = [
|
|
741
|
-
{ id: "u-1", type: "message", message: { role: "user", content: [{ type: "text", text: "implement VC-001" }] } },
|
|
742
|
-
{ id: "a-1", type: "message", message: { role: "assistant", content: [{ type: "text", text: "wrote helper [I-002:current]" }] } },
|
|
743
|
-
{ id: "u-2", type: "message", message: { role: "user", content: [{ type: "text", text: "implement VC-002" }] } },
|
|
744
|
-
{ id: "a-2", type: "message", message: { role: "assistant", content: [{ type: "text", text: "almost done" }] } },
|
|
745
|
-
];
|
|
746
|
-
const result = buildExecutionCompactionResult(
|
|
747
|
-
{
|
|
748
|
-
type: "session_before_compact",
|
|
749
|
-
preparation,
|
|
750
|
-
branchEntries,
|
|
751
|
-
customInstructions: "keep:1",
|
|
752
|
-
reason: "threshold",
|
|
753
|
-
willRetry: false,
|
|
754
|
-
signal: new AbortController().signal,
|
|
755
|
-
},
|
|
756
|
-
ctx,
|
|
297
|
+
it("a failing audit (real flow) rolls covered tasks back and persists the outcome", async () => {
|
|
298
|
+
// F-006/F-010: drive the real runAuditFlow via restoreFromSession with
|
|
299
|
+
// an injected failing runner — no manual audit-state mutation.
|
|
300
|
+
__setAuditRunnerForTests(async ({ checklist }) => {
|
|
301
|
+
// Mirror applyAuditOutcome: passed checks are marked done.
|
|
302
|
+
const vc1 = checklist.find((item) => item.id === "VC-001");
|
|
303
|
+
if (vc1) vc1.done = true;
|
|
304
|
+
return { round: 1, passed: ["VC-001"], failed: ["VC-002"], report: "VC-002 fails: core tests missing" };
|
|
305
|
+
});
|
|
306
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
307
|
+
const { ctx } = await start(planPath, workdir);
|
|
308
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
309
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
310
|
+
persistTaskProgress(ctx);
|
|
311
|
+
}
|
|
312
|
+
const snapshot = getExecution();
|
|
313
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
314
|
+
const ex = getExecution()!;
|
|
315
|
+
// Fail-closed: the audit rolled VC-002's covered tasks back to pending.
|
|
316
|
+
assert.equal(ex.tasks.find((t) => t.id === "Task-1")?.status, "complete");
|
|
317
|
+
assert.equal(ex.tasks.find((t) => t.id === "Task-2")?.status, "pending");
|
|
318
|
+
assert.equal(ex.tasks.find((t) => t.id === "Task-3")?.status, "pending");
|
|
319
|
+
const load = loadCheckpoint(workdir, runId!);
|
|
320
|
+
assert.ok(load.status === "ok");
|
|
321
|
+
assert.equal(load.checkpoint.execution?.audit?.rounds, 1);
|
|
322
|
+
assert.equal(load.checkpoint.execution?.audit?.lastResult, "VC-002");
|
|
323
|
+
__setAuditRunnerForTests(null);
|
|
324
|
+
await stopExecution(ctx, "test teardown");
|
|
325
|
+
});
|
|
326
|
+
|
|
327
|
+
it("review round cap: every mode pauses with an in-band signal (v0.8)", async () => {
|
|
328
|
+
__setAuditRunnerForTests(async ({ checklist }) => {
|
|
329
|
+
const failed = checklist.filter((item) => !["VC-001"].includes(item.id)).map((item) => item.id);
|
|
330
|
+
return { round: 1, passed: ["VC-001"], failed, report: "still failing" };
|
|
331
|
+
});
|
|
332
|
+
const interactive = await (async () => {
|
|
333
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
334
|
+
const { ctx } = await start(planPath, workdir);
|
|
335
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
336
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
337
|
+
persistTaskProgress(ctx);
|
|
338
|
+
}
|
|
339
|
+
// Simulate an exhausted budget persisted from earlier attempts.
|
|
340
|
+
getExecution()!.audit.rounds = REVIEW_MAX_ROUNDS;
|
|
341
|
+
getExecution()!.audit.failed = ["VC-002"];
|
|
342
|
+
mutateCheckpoint(workdir, runId!, (cp) => applyExecutionProgress(cp, { audit: { rounds: REVIEW_MAX_ROUNDS, lastResult: "VC-002" } }));
|
|
343
|
+
const snapshot = getExecution();
|
|
344
|
+
const ctxUi = { ...makeCtx(workdir), mode: "tui" as const };
|
|
345
|
+
await restoreFromSession(ctxUi, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
346
|
+
await __awaitReviewRoundForTests(); // detached chain: pause lands inside it
|
|
347
|
+
const ex = getExecution()!;
|
|
348
|
+
// v0.8 (Q-A): interactive AND headless both pause — fail-closed, never a silent stop.
|
|
349
|
+
assert.equal(ex.stall.paused, true, "interactive cap pauses");
|
|
350
|
+
assert.match(ex.stall.pausedReason ?? "", /execution review exhausted 5 rounds/);
|
|
351
|
+
assert.ok(
|
|
352
|
+
ctxUi.entries.some((e) => e.customType === "pi-plans-review-paused"),
|
|
353
|
+
"the pause lands as an in-band message every mode can read",
|
|
354
|
+
);
|
|
355
|
+
await stopExecution(ctxUi, "test teardown");
|
|
356
|
+
return loadCheckpoint(workdir, runId!);
|
|
357
|
+
})();
|
|
358
|
+
assert.ok(interactive.status === "ok");
|
|
359
|
+
// Headless: same pause, execution state kept (no more bounded stop).
|
|
360
|
+
const { workdir: wd2, planPath: pp2, runId: r2 } = freshWorkdir();
|
|
361
|
+
const { ctx: ctx3 } = await start(pp2, wd2);
|
|
362
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
363
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
364
|
+
persistTaskProgress(ctx3);
|
|
365
|
+
}
|
|
366
|
+
getExecution()!.audit.rounds = REVIEW_MAX_ROUNDS;
|
|
367
|
+
getExecution()!.audit.failed = ["VC-002"];
|
|
368
|
+
const headlessCtx = { ...makeCtx(wd2), hasUI: false } as never;
|
|
369
|
+
await restoreFromSession(headlessCtx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
370
|
+
assert.notEqual(getExecution(), null, "headless cap pauses and keeps execution state");
|
|
371
|
+
assert.equal(getExecution()!.stall.paused, true, "headless pauses too (v0.8)");
|
|
372
|
+
const paused = loadCheckpoint(wd2, r2!);
|
|
373
|
+
assert.ok(paused.status === "ok");
|
|
374
|
+
assert.match(paused.checkpoint.execution?.pausedReason ?? "", /execution review exhausted 5 rounds/);
|
|
375
|
+
await stopExecution(makeCtx(wd2), "test teardown");
|
|
376
|
+
__setAuditRunnerForTests(null);
|
|
377
|
+
});
|
|
378
|
+
|
|
379
|
+
it("a checkpoint carrying a 0.6.0 delegated executor refuses the direct load", async () => {
|
|
380
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
381
|
+
const { ctx } = await start(planPath, workdir);
|
|
382
|
+
applyTaskUpdate(getExecution()!.tasks, "Task-1", "complete", "e1");
|
|
383
|
+
persistTaskProgress(ctx);
|
|
384
|
+
await stopExecution(ctx, "restart-sim");
|
|
385
|
+
mutateCheckpoint(workdir, runId!, (cp) =>
|
|
386
|
+
applyExecutionProgress(cp as never, { delegate: { modelSelector: "x/y", startedAt: "2026-01-01T00:00:00Z" } } as never),
|
|
757
387
|
);
|
|
758
|
-
|
|
759
|
-
assert.equal(
|
|
760
|
-
assert.
|
|
761
|
-
assert.
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
|
|
777
|
-
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
const
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
assert.equal(
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
const
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
388
|
+
const load = loadExecutionFromCheckpoint(makeCtx(workdir), runId!);
|
|
389
|
+
assert.equal(load.legacyDelegate, true);
|
|
390
|
+
assert.equal(load.status, "no-execution");
|
|
391
|
+
assert.equal(getExecution(), null, "delegate orphans never resume without a fresh handoff");
|
|
392
|
+
});
|
|
393
|
+
|
|
394
|
+
it("resuming a review-cap pause via /plans-execute grants a fresh budget and completes (v0.8)", async () => {
|
|
395
|
+
// Round 2 F-001 regression nail, v0.8 form: cap → pause → the explicit
|
|
396
|
+
// /plans-execute surface (resumeActiveExecution) resets rounds → the
|
|
397
|
+
// review runs again and can now complete the run. Ordinary input must NOT.
|
|
398
|
+
let runnerCalls = 0;
|
|
399
|
+
__setAuditRunnerForTests(async ({ checklist }) => {
|
|
400
|
+
runnerCalls += 1;
|
|
401
|
+
return {
|
|
402
|
+
round: runnerCalls,
|
|
403
|
+
passed: checklist.map((item) => {
|
|
404
|
+
item.done = true;
|
|
405
|
+
return item.id;
|
|
406
|
+
}),
|
|
407
|
+
failed: [],
|
|
408
|
+
report: "all pass",
|
|
409
|
+
};
|
|
410
|
+
});
|
|
411
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
412
|
+
const { ctx } = await start(planPath, workdir);
|
|
413
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
414
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
415
|
+
persistTaskProgress(ctx);
|
|
416
|
+
}
|
|
417
|
+
// Exhaust the budget, then pause at the cap exactly like the loop does.
|
|
418
|
+
getExecution()!.audit.rounds = REVIEW_MAX_ROUNDS;
|
|
419
|
+
getExecution()!.audit.failed = ["VC-001", "VC-002"];
|
|
420
|
+
const ctxTui = { ...makeCtx(workdir), mode: "tui" as const };
|
|
421
|
+
const snapshot = getExecution();
|
|
422
|
+
await restoreFromSession(ctxTui, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
423
|
+
await __awaitReviewRoundForTests();
|
|
424
|
+
let ex = getExecution()!;
|
|
425
|
+
assert.equal(ex.stall.paused, true, "cap pauses");
|
|
426
|
+
assert.match(ex.stall.pausedReason ?? "", /^execution review exhausted/);
|
|
427
|
+
// Ordinary input does NOT lift a review-cap pause (CF2-004).
|
|
428
|
+
const beforeRounds = ex.audit.rounds;
|
|
429
|
+
const { resumeGoalWaitIfPaused } = await import("../src/exec.ts");
|
|
430
|
+
const viaInput = resumeGoalWaitIfPaused(ctxTui);
|
|
431
|
+
assert.equal(viaInput, false, "input never lifts a review-cap pause");
|
|
432
|
+
assert.equal(getExecution()!.stall.paused, true, "still paused after input");
|
|
433
|
+
assert.equal(getExecution()!.audit.rounds, beforeRounds, "budget survives input");
|
|
434
|
+
// The explicit surface grants the fresh budget and completes the run.
|
|
435
|
+
const { resumeActiveExecution } = await import("../src/exec.ts");
|
|
436
|
+
const resumed = resumeActiveExecution(ctxTui);
|
|
437
|
+
assert.equal(resumed, true);
|
|
438
|
+
assert.equal(getExecution()!.audit.rounds, 0, "resume grants a fresh review budget");
|
|
439
|
+
await __awaitReviewRoundForTests(); // detached grant chain completes the run
|
|
440
|
+
assert.equal(runnerCalls, 1, "review re-ran after the resume (fresh budget)");
|
|
441
|
+
const final = loadCheckpoint(workdir, runId!);
|
|
442
|
+
assert.ok(final.status === "ok");
|
|
443
|
+
assert.equal(final.checkpoint.phase, "completed");
|
|
444
|
+
__setAuditRunnerForTests(null);
|
|
445
|
+
});
|
|
446
|
+
|
|
447
|
+
it("a null review outcome stays undeterminable, self-schedules to the cap, and pauses (v0.8)", async () => {
|
|
448
|
+
// An infra failure is not evidence of bad work: it must not complete the
|
|
449
|
+
// run, roll work back, or accuse the agent — and the loop retries it
|
|
450
|
+
// internally until the budget is spent, then pauses fail-closed.
|
|
451
|
+
__setAuditRunnerForTests(async () => null);
|
|
452
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
453
|
+
const { ctx } = await start(planPath, workdir);
|
|
454
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
455
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
456
|
+
persistTaskProgress(ctx);
|
|
457
|
+
}
|
|
458
|
+
const snapshot = getExecution();
|
|
459
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
460
|
+
const ex = getExecution()!;
|
|
461
|
+
assert.deepEqual(ex.audit.undeterminable, ["VC-001", "VC-002"], "unreported checks are undeterminable");
|
|
462
|
+
assert.deepEqual(ex.audit.failed, [], "nothing was judged wrong");
|
|
463
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
464
|
+
assert.equal(ex.tasks.find((t) => t.id === id)?.status, "complete", `${id} not rolled back`);
|
|
465
|
+
}
|
|
466
|
+
assert.equal(ex.audit.rounds, REVIEW_MAX_ROUNDS, "the retry chain spent the whole budget");
|
|
467
|
+
assert.equal(ex.stall.paused, true, "the loop pauses at the cap");
|
|
468
|
+
assert.match(ex.stall.pausedReason ?? "", /execution review exhausted 5 rounds/);
|
|
469
|
+
assert.equal(
|
|
470
|
+
ctx.entries.filter((e) => e.customType === "pi-plans-audit-undeterminable").length,
|
|
471
|
+
0,
|
|
472
|
+
"undeterminable rounds never wake the agent (self-scheduled retries)",
|
|
840
473
|
);
|
|
841
|
-
assert.ok(
|
|
842
|
-
|
|
843
|
-
assert.
|
|
844
|
-
assert.
|
|
845
|
-
assert.
|
|
846
|
-
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
const
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
customInstructions: "keep:1",
|
|
865
|
-
reason: "threshold",
|
|
866
|
-
willRetry: false,
|
|
867
|
-
signal: new AbortController().signal,
|
|
868
|
-
});
|
|
869
|
-
assert.ok(resultPlanning?.compaction);
|
|
870
|
-
|
|
871
|
-
setRunStatus(workdir, run.run_id, "done");
|
|
872
|
-
const resultDone = handlePlanningBeforeCompact(pi as any, ctx as any, {
|
|
873
|
-
type: "session_before_compact",
|
|
874
|
-
preparation: makePreparation("threshold", null),
|
|
875
|
-
branchEntries: compactableBranchEntries(),
|
|
876
|
-
reason: "threshold",
|
|
877
|
-
willRetry: false,
|
|
878
|
-
signal: new AbortController().signal,
|
|
879
|
-
});
|
|
880
|
-
assert.equal(resultDone, undefined);
|
|
881
|
-
|
|
882
|
-
setRunStatus(workdir, run.run_id, "planning");
|
|
883
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v14.md"), items("VC-001"));
|
|
884
|
-
const duringExecution = handlePlanningBeforeCompact(pi as any, ctx as any, {
|
|
885
|
-
type: "session_before_compact",
|
|
886
|
-
preparation: makePreparation("threshold", null),
|
|
887
|
-
branchEntries: compactableBranchEntries(),
|
|
888
|
-
reason: "threshold",
|
|
889
|
-
willRetry: false,
|
|
890
|
-
signal: new AbortController().signal,
|
|
891
|
-
});
|
|
892
|
-
assert.equal(duringExecution, undefined);
|
|
893
|
-
await stopExecution(pi, ctx, "planning-gate");
|
|
894
|
-
});
|
|
895
|
-
|
|
896
|
-
it("turn_end writes are unconditionally deferred even when isIdle reads true", async () => {
|
|
897
|
-
const workdir = freshWorkdir();
|
|
898
|
-
const { pi, ctx, recorded, setUsagePercent } = makeHarness(workdir);
|
|
899
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v11.md"), items("VC-001", "VC-002"));
|
|
900
|
-
const snapshotCount = () => recorded.entries.filter((e) => e.customType === "pi-plans-exec").length;
|
|
901
|
-
const baseline = snapshotCount();
|
|
902
|
-
|
|
903
|
-
// No isIdle override: the harness default (() => true) IS the real
|
|
904
|
-
// turn_end reading — this encodes the field regression (60 writes in
|
|
905
|
-
// 23 minutes) as a permanent zero-write assertion.
|
|
906
|
-
recordExecutionTurn(pi, ctx, [], { input: 10, output: 5 });
|
|
907
|
-
setUsagePercent(50);
|
|
908
|
-
handleExecutionBeforeCompact(pi, ctx, {
|
|
909
|
-
type: "session_before_compact",
|
|
910
|
-
preparation: makePreparation("threshold", null),
|
|
911
|
-
branchEntries: [],
|
|
912
|
-
reason: "threshold",
|
|
913
|
-
willRetry: false,
|
|
914
|
-
signal: new AbortController().signal,
|
|
915
|
-
});
|
|
916
|
-
assert.equal(snapshotCount(), baseline, "turn_end wrote session entries despite deferral");
|
|
917
|
-
// Status line stays real-time while the write is deferred.
|
|
918
|
-
assert.match(recorded.status ?? "", /plans: .* ▸ exec/);
|
|
919
|
-
|
|
920
|
-
drainExecutionFlush(pi, ctx);
|
|
921
|
-
assert.equal(snapshotCount(), baseline + 1, "settle flush did not write exactly one snapshot");
|
|
922
|
-
const last = (recorded.entries.filter((e) => e.customType === "pi-plans-exec").at(-1)?.data ?? {}) as { usage?: { inToks: number } };
|
|
923
|
-
assert.equal(last.usage?.inToks, 10, "flushed snapshot missing busy-turn usage");
|
|
924
|
-
drainExecutionFlush(pi, ctx);
|
|
925
|
-
assert.equal(snapshotCount(), baseline + 1, "second drain wrote again");
|
|
926
|
-
assert.equal(consumePendingExecutionFlush(), false);
|
|
927
|
-
|
|
928
|
-
await stopExecution(pi, ctx, "done");
|
|
929
|
-
});
|
|
930
|
-
|
|
931
|
-
it("stop and complete drain pending writes synchronously with final state", async () => {
|
|
932
|
-
const workdir = freshWorkdir();
|
|
933
|
-
const { pi, ctx, recorded } = makeHarness(workdir);
|
|
934
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v12.md"), items("VC-001", "VC-002"));
|
|
935
|
-
const snapshotCount = () => recorded.entries.filter((e) => e.customType === "pi-plans-exec").length;
|
|
936
|
-
|
|
937
|
-
ctx.isIdle = () => false;
|
|
938
|
-
recordExecutionTurn(pi, ctx, [], { input: 7, output: 3 });
|
|
939
|
-
const busyCount = snapshotCount();
|
|
940
|
-
|
|
941
|
-
await stopExecution(pi, ctx, "force");
|
|
942
|
-
assert.ok(snapshotCount() > busyCount, "stop did not write the final snapshot");
|
|
943
|
-
assert.ok(recorded.entries.some((e) => e.customType === "pi-plans-exec-cleared"));
|
|
944
|
-
const lastStop = (recorded.entries.filter((e) => e.customType === "pi-plans-exec").at(-1)?.data ?? {}) as { usage?: { inToks: number } };
|
|
945
|
-
assert.equal(lastStop.usage?.inToks, 7, "stop lost the busy-turn usage");
|
|
946
|
-
|
|
947
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v13.md"), items("VC-001"));
|
|
948
|
-
ctx.isIdle = () => false;
|
|
949
|
-
applyDoneMarkers("[DONE:VC-001]");
|
|
950
|
-
recordExecutionTurn(pi, ctx, ["VC-001"], { input: 1, output: 1 });
|
|
951
|
-
await completeExecution(pi, ctx);
|
|
952
|
-
assert.equal(getExecution(), null);
|
|
953
|
-
const lastComplete = (recorded.entries.filter((e) => e.customType === "pi-plans-exec").at(-1)?.data ?? {}) as { items?: Array<{ done: boolean }> };
|
|
954
|
-
assert.equal(lastComplete.items?.[0]?.done, true, "completion snapshot missing final done state");
|
|
955
|
-
});
|
|
956
|
-
|
|
957
|
-
it("updates the status bar in real time on every turn without session writes", async () => {
|
|
958
|
-
const workdir = freshWorkdir();
|
|
959
|
-
const { pi, ctx, recorded, setUsagePercent } = makeHarness(workdir);
|
|
960
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v15.md"), items("VC-001", "VC-002"), [
|
|
961
|
-
{ id: "I-001", text: "First item." },
|
|
962
|
-
{ id: "I-002", text: "Second item." },
|
|
963
|
-
]);
|
|
964
|
-
const snapshotCount = () => recorded.entries.filter((e) => e.customType === "pi-plans-exec").length;
|
|
965
|
-
const baseline = snapshotCount();
|
|
966
|
-
|
|
967
|
-
recordExecutionTurn(pi, ctx, [], { input: 33, output: 11 });
|
|
968
|
-
// Real-time: status line already reflects the turn's usage...
|
|
969
|
-
assert.match(recorded.status ?? "", /plans: .* ▸ exec/);
|
|
970
|
-
assert.match(recorded.status ?? "", /I 0\/2/);
|
|
971
|
-
assert.match(recorded.status ?? "", /VC 0\/2/);
|
|
972
|
-
// ...without any session write (anti-jitter preserved).
|
|
973
|
-
assert.equal(snapshotCount(), baseline);
|
|
974
|
-
|
|
975
|
-
setUsagePercent(null);
|
|
976
|
-
drainExecutionFlush(pi, ctx);
|
|
977
|
-
assert.equal(snapshotCount(), baseline + 1);
|
|
978
|
-
|
|
979
|
-
await stopExecution(pi, ctx, "done");
|
|
980
|
-
});
|
|
981
|
-
|
|
982
|
-
it("syncs progress through the registered message_end and turn_end handlers", async () => {
|
|
983
|
-
const workdir = freshWorkdir();
|
|
984
|
-
const { pi, ctx, recorded, emit } = makeHarness(workdir);
|
|
985
|
-
registerExecutionTurnHandlers(pi);
|
|
986
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v19.md"), items("VC-001", "VC-002", "VC-003"));
|
|
987
|
-
|
|
988
|
-
// The usage is delivered by message_end and consumed by the following
|
|
989
|
-
// turn_end, matching Pi's lifecycle contract.
|
|
990
|
-
await emit("message_end", {
|
|
991
|
-
message: { role: "assistant", usage: { input: 12, output: 5 } },
|
|
992
|
-
});
|
|
993
|
-
await emit("turn_end", {
|
|
994
|
-
message: { role: "assistant", content: [{ type: "text", text: "verified [DONE:VC-001]" }] },
|
|
995
|
-
});
|
|
996
|
-
assert.match(recorded.status ?? "", /VC 1\/3/);
|
|
997
|
-
assert.match(recorded.status ?? "", /next: Verify VC-002/);
|
|
998
|
-
|
|
999
|
-
await emit("message_end", {
|
|
1000
|
-
message: { role: "assistant", usage: { input: 8, output: 3 } },
|
|
1001
|
-
});
|
|
1002
|
-
await emit("turn_end", {
|
|
1003
|
-
message: { role: "assistant", content: [{ type: "text", text: "verified [DONE:VC-002]" }] },
|
|
1004
|
-
});
|
|
1005
|
-
assert.match(recorded.status ?? "", /VC 2\/3/);
|
|
1006
|
-
|
|
1007
|
-
await stopExecution(pi, ctx, "event-chain-test");
|
|
1008
|
-
});
|
|
1009
|
-
it("applies impl markers with silent unknown ids and later-overwrite semantics", async () => {
|
|
1010
|
-
const workdir = freshWorkdir();
|
|
1011
|
-
const { pi, ctx } = makeHarness(workdir);
|
|
1012
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v16.md"), items("VC-001"), [
|
|
1013
|
-
{ id: "I-001", text: "First item." },
|
|
1014
|
-
]);
|
|
1015
|
-
|
|
1016
|
-
assert.deepEqual(applyImplMarkers("[I-999:implemented]"), []); // unknown id silently ignored
|
|
1017
|
-
assert.deepEqual(applyImplMarkers("[I-001:implemented]"), ["I-001"]);
|
|
1018
|
-
assert.deepEqual(applyImplMarkers("[I-001:validating]"), ["I-001"]); // later overwrites
|
|
1019
|
-
assert.equal(getExecution()?.implStatus?.["I-001"], "validating");
|
|
1020
|
-
|
|
1021
|
-
await stopExecution(pi, ctx, "done");
|
|
1022
|
-
});
|
|
1023
|
-
|
|
1024
|
-
it("applies current-I markers to the live progress bar and snapshot", async () => {
|
|
1025
|
-
const workdir = freshWorkdir();
|
|
1026
|
-
const { pi, ctx, recorded, emit } = makeHarness(workdir);
|
|
1027
|
-
registerExecutionTurnHandlers(pi);
|
|
1028
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v20.md"), items("VC-001", "VC-002", "VC-003"), [
|
|
1029
|
-
{ id: "I-001", text: "First item." },
|
|
1030
|
-
{ id: "I-002", text: "Second item." },
|
|
1031
|
-
{ id: "I-003", text: "Third item." },
|
|
1032
|
-
]);
|
|
1033
|
-
|
|
1034
|
-
await emit("turn_end", {
|
|
1035
|
-
message: { role: "assistant", content: [{ type: "text", text: "starting [I-003:current]" }] },
|
|
1036
|
-
});
|
|
1037
|
-
assert.match(recorded.status ?? "", /I 0\/3/);
|
|
1038
|
-
assert.match(recorded.status ?? "", /next: Implement I-003/);
|
|
1039
|
-
assert.equal(getExecution()?.currentI, "I-003");
|
|
1040
|
-
drainExecutionFlush(pi, ctx);
|
|
1041
|
-
const snapshot = recorded.entries.filter((entry) => entry.customType === "pi-plans-exec").at(-1)?.data as { currentI?: string };
|
|
1042
|
-
assert.equal(snapshot.currentI, "I-003");
|
|
1043
|
-
assert.equal(applyCurrentIMarker("[I-999:current]"), false);
|
|
1044
|
-
await stopExecution(pi, ctx, "current-I-test");
|
|
1045
|
-
});
|
|
1046
|
-
it("replays impl markers from post-snapshot messages on restore", async () => {
|
|
1047
|
-
const workdir = freshWorkdir();
|
|
1048
|
-
const { pi, ctx } = makeHarness(workdir);
|
|
1049
|
-
const snapshot = {
|
|
1050
|
-
planPath: path.join(workdir, "PLAN_v17.md"),
|
|
1051
|
-
items: items("VC-001"),
|
|
1052
|
-
startedAt: "2026-08-29T00:00:00Z",
|
|
1053
|
-
implItems: [{ id: "I-001", text: "First item." }],
|
|
1054
|
-
implStatus: {},
|
|
1055
|
-
};
|
|
1056
|
-
// The restore path re-parses the plan file (stale-snapshot distrust), so
|
|
1057
|
-
// the file must actually carry the Implementation Items section.
|
|
1058
|
-
fs.writeFileSync(snapshot.planPath, "# plan\n\n## Implementation Items\n\n- `I-001`: First item.\n");
|
|
1059
|
-
const entries = [
|
|
1060
|
-
{ type: "custom", customType: "pi-plans-exec", data: snapshot },
|
|
1061
|
-
{ type: "message", message: { role: "assistant", content: [{ type: "text", text: "work done [I-001:implemented]" }] } },
|
|
1062
|
-
];
|
|
1063
|
-
await restoreFromSession(pi, ctx, entries as any);
|
|
1064
|
-
assert.equal(getExecution()?.implStatus?.["I-001"], "implemented");
|
|
1065
|
-
});
|
|
1066
|
-
|
|
1067
|
-
|
|
1068
|
-
it("keeps the injection rules teaching the impl markers", async () => {
|
|
1069
|
-
const workdir = freshWorkdir();
|
|
1070
|
-
const { pi, ctx } = makeHarness(workdir);
|
|
1071
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v18.md"), items("VC-001"), [
|
|
1072
|
-
{ id: "I-001", text: "First item." },
|
|
1073
|
-
]);
|
|
1074
|
-
const rules = executionContextMessage(ctx)!;
|
|
1075
|
-
assert.match(rules, /\[I-001:implemented\]/);
|
|
1076
|
-
assert.match(rules, /\[I-001:validating\]/);
|
|
1077
|
-
assert.match(rules, /subprocess-backed verification/);
|
|
1078
|
-
assert.match(rules, /waiting for/);
|
|
1079
|
-
assert.match(rules, /5s\s*->\s*10s\s*->\s*20s\s*->\s*40s\s*->\s*80s/);
|
|
1080
|
-
assert.match(rules, /keep polling at 80s/);
|
|
1081
|
-
assert.match(rules, /restart at 5s for each new subprocess/);
|
|
1082
|
-
await stopExecution(pi, ctx, "done");
|
|
1083
|
-
});
|
|
1084
|
-
|
|
1085
|
-
it("does not proactively request planning compaction", () => {
|
|
1086
|
-
const workdir = freshWorkdir();
|
|
1087
|
-
const { ctx, setUsagePercent, recorded } = makeHarness(workdir);
|
|
1088
|
-
initState(workdir);
|
|
1089
|
-
startRun(workdir, { topic: "planning no-op", skill: "plan-normal", requestText: "x" });
|
|
1090
|
-
|
|
1091
|
-
setUsagePercent(120);
|
|
1092
|
-
assert.equal(shouldTriggerPlanningCompaction(ctx as any), false);
|
|
1093
|
-
requestPlanningCompaction(ctx as any);
|
|
1094
|
-
refreshPlanningCompactionCooldown(ctx as any);
|
|
1095
|
-
assert.equal(recorded.compacts?.length ?? 0, 0, "planning scheduling is owned by Pi core");
|
|
1096
|
-
});
|
|
1097
|
-
|
|
1098
|
-
it("auto-continues planning only for threshold/overflow and honors manual follow-up prompts", async () => {
|
|
1099
|
-
const workdir = freshWorkdir();
|
|
1100
|
-
const { pi, ctx, recorded } = makeHarness(workdir);
|
|
1101
|
-
(ctx as any).piVersion = "0.84.3";
|
|
1102
|
-
initState(workdir);
|
|
1103
|
-
startRun(workdir, { topic: "planning success", skill: "plan-normal", requestText: "x" });
|
|
1104
|
-
|
|
1105
|
-
const threshold = handlePlanningBeforeCompact(pi as any, ctx as any, {
|
|
1106
|
-
type: "session_before_compact",
|
|
1107
|
-
preparation: makePreparation("threshold", null),
|
|
1108
|
-
branchEntries: compactableBranchEntries(),
|
|
1109
|
-
customInstructions: "keep:1",
|
|
1110
|
-
reason: "threshold",
|
|
1111
|
-
willRetry: false,
|
|
1112
|
-
signal: new AbortController().signal,
|
|
1113
|
-
});
|
|
1114
|
-
assert.ok(threshold?.compaction);
|
|
1115
|
-
await handlePlanningCompact(pi as any, ctx as any, {
|
|
1116
|
-
type: "session_compact",
|
|
1117
|
-
compactionEntry: { type: "compaction" } as never,
|
|
1118
|
-
fromExtension: false,
|
|
1119
|
-
reason: "threshold",
|
|
1120
|
-
willRetry: false,
|
|
1121
|
-
});
|
|
1122
|
-
assert.equal(recorded.messages.filter((message) => message.customType === "pi-plans-plan-resume").length, 1);
|
|
1123
|
-
assert.equal(consumePlanningCompactionResumeGuard(ctx as any), true);
|
|
1124
|
-
|
|
1125
|
-
handlePlanningBeforeCompact(pi as any, ctx as any, {
|
|
1126
|
-
type: "session_before_compact",
|
|
1127
|
-
preparation: makePreparation("manual", null),
|
|
1128
|
-
branchEntries: compactableBranchEntries(),
|
|
1129
|
-
customInstructions: "Ask the next question keep:1",
|
|
1130
|
-
reason: "manual",
|
|
1131
|
-
willRetry: false,
|
|
1132
|
-
signal: new AbortController().signal,
|
|
1133
|
-
});
|
|
1134
|
-
await handlePlanningCompact(pi as any, ctx as any, {
|
|
1135
|
-
type: "session_compact",
|
|
1136
|
-
compactionEntry: { type: "compaction" } as never,
|
|
1137
|
-
fromExtension: false,
|
|
1138
|
-
reason: "manual",
|
|
1139
|
-
willRetry: false,
|
|
1140
|
-
});
|
|
1141
|
-
assert.deepEqual(recorded.userMessages, ["Ask the next question"]);
|
|
1142
|
-
assert.equal(recorded.messages.filter((message) => message.customType === "pi-plans-plan-resume").length, 1);
|
|
1143
|
-
});
|
|
1144
|
-
|
|
1145
|
-
it("planning request helper remains a no-op regardless of idleness", () => {
|
|
1146
|
-
const workdir = freshWorkdir();
|
|
1147
|
-
const { ctx, recorded, setUsagePercent } = makeHarness(workdir);
|
|
1148
|
-
initState(workdir);
|
|
1149
|
-
startRun(workdir, { topic: "idle no-op", skill: "plan-normal", requestText: "x" });
|
|
1150
|
-
setUsagePercent(120);
|
|
1151
|
-
|
|
1152
|
-
(ctx as any).isIdle = () => false;
|
|
1153
|
-
requestPlanningCompaction(ctx as any);
|
|
1154
|
-
(ctx as any).isIdle = () => true;
|
|
1155
|
-
(ctx as any).hasPendingMessages = () => false;
|
|
1156
|
-
requestPlanningCompaction(ctx as any);
|
|
1157
|
-
assert.equal(recorded.compacts?.length ?? 0, 0);
|
|
1158
|
-
assert.equal((ctx as any).sessionManager.__planningCompaction, undefined);
|
|
1159
|
-
});
|
|
1160
|
-
|
|
1161
|
-
it("planning compact failures notify but do not re-request proactively", () => {
|
|
1162
|
-
const workdir = freshWorkdir();
|
|
1163
|
-
const { pi, ctx, recorded } = makeHarness(workdir);
|
|
1164
|
-
initState(workdir);
|
|
1165
|
-
startRun(workdir, { topic: "failure notify", skill: "plan-normal", requestText: "x" });
|
|
1166
|
-
handlePlanningBeforeCompact(pi as any, ctx as any, {
|
|
1167
|
-
type: "session_before_compact",
|
|
1168
|
-
preparation: makePreparation("threshold", null),
|
|
1169
|
-
branchEntries: compactableBranchEntries(),
|
|
1170
|
-
customInstructions: "keep:1",
|
|
1171
|
-
reason: "threshold",
|
|
1172
|
-
willRetry: false,
|
|
1173
|
-
signal: new AbortController().signal,
|
|
1174
|
-
});
|
|
1175
|
-
|
|
1176
|
-
handlePlanningCompactFailed(pi as any, ctx as any, {
|
|
1177
|
-
type: "session_compact_failed",
|
|
1178
|
-
reason: "manual",
|
|
1179
|
-
errorMessage: "Compaction failed: Nothing to compact (session too small)",
|
|
1180
|
-
aborted: false,
|
|
1181
|
-
willRetry: false,
|
|
1182
|
-
fromExtension: false,
|
|
1183
|
-
});
|
|
1184
|
-
assert.ok(recorded.notifies.some((note) => note.message.includes("nothing to summarize")));
|
|
1185
|
-
requestPlanningCompaction(ctx as any);
|
|
1186
|
-
assert.equal(recorded.compacts?.length ?? 0, 0);
|
|
1187
|
-
|
|
1188
|
-
handlePlanningBeforeCompact(pi as any, ctx as any, {
|
|
1189
|
-
type: "session_before_compact",
|
|
1190
|
-
preparation: makePreparation("threshold", null),
|
|
1191
|
-
branchEntries: compactableBranchEntries(),
|
|
1192
|
-
customInstructions: "keep:1",
|
|
1193
|
-
reason: "threshold",
|
|
1194
|
-
willRetry: false,
|
|
1195
|
-
signal: new AbortController().signal,
|
|
1196
|
-
});
|
|
1197
|
-
handlePlanningCompactFailed(pi as any, ctx as any, {
|
|
1198
|
-
type: "session_compact_failed",
|
|
1199
|
-
reason: "manual",
|
|
1200
|
-
errorMessage: "network down",
|
|
1201
|
-
aborted: false,
|
|
1202
|
-
willRetry: false,
|
|
1203
|
-
fromExtension: false,
|
|
1204
|
-
});
|
|
1205
|
-
assert.ok(recorded.notifies.some((note) => note.message.includes("will try again")));
|
|
1206
|
-
});
|
|
1207
|
-
|
|
1208
|
-
it("execution turn compaction helper remains a no-op and failures keep execution active", async () => {
|
|
1209
|
-
const workdir = freshWorkdir();
|
|
1210
|
-
startActiveRun(workdir);
|
|
1211
|
-
const { pi, ctx, recorded, setUsagePercent } = makeHarness(workdir);
|
|
1212
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v23.md"), items("VC-001"), [
|
|
1213
|
-
{ id: "I-001", text: "First item." },
|
|
1214
|
-
]);
|
|
1215
|
-
|
|
1216
|
-
setUsagePercent(96);
|
|
1217
|
-
handleExecutionTurnCompaction(ctx);
|
|
1218
|
-
assert.equal(recorded.compacts?.length ?? 0, 0, "execution scheduling is owned by Pi core");
|
|
1219
|
-
|
|
1220
|
-
handleExecutionBeforeCompact(pi, ctx, {
|
|
1221
|
-
type: "session_before_compact",
|
|
1222
|
-
preparation: makePreparation("threshold", null),
|
|
1223
|
-
branchEntries: compactableBranchEntries(),
|
|
1224
|
-
customInstructions: "keep:1",
|
|
1225
|
-
reason: "threshold",
|
|
1226
|
-
willRetry: false,
|
|
1227
|
-
signal: new AbortController().signal,
|
|
1228
|
-
});
|
|
1229
|
-
handleExecutionCompactFailed(pi, ctx, {
|
|
1230
|
-
type: "session_compact_failed",
|
|
1231
|
-
reason: "manual",
|
|
1232
|
-
errorMessage: "Compaction failed: Already compacted",
|
|
1233
|
-
aborted: false,
|
|
1234
|
-
willRetry: false,
|
|
1235
|
-
fromExtension: false,
|
|
1236
|
-
});
|
|
1237
|
-
assert.ok(getExecution());
|
|
1238
|
-
assert.ok(recorded.notifies.some((note) => note.message.includes("nothing to summarize")));
|
|
1239
|
-
|
|
1240
|
-
await stopExecution(pi, ctx, "test-done");
|
|
1241
|
-
});
|
|
1242
|
-
|
|
1243
|
-
it("treats planning abort/stream compact failures as terminal and notifies explicitly", () => {
|
|
1244
|
-
const workdir = freshWorkdir();
|
|
1245
|
-
const { pi, ctx, recorded } = makeHarness(workdir);
|
|
1246
|
-
initState(workdir);
|
|
1247
|
-
startRun(workdir, { topic: "abort-stream classification", skill: "plan-normal", requestText: "x" });
|
|
1248
|
-
|
|
1249
|
-
const abortMessages = [
|
|
1250
|
-
"Auto-compaction failed: Turn prefix summarization failed: This operation was aborted",
|
|
1251
|
-
"Error: OpenAI Responses stream ended before a terminal response event",
|
|
1252
|
-
"Auto-compaction failed: context overflow recovery failed",
|
|
1253
|
-
"this operation was aborted",
|
|
1254
|
-
"aborted",
|
|
1255
|
-
];
|
|
1256
|
-
for (const errorMessage of abortMessages) {
|
|
1257
|
-
recorded.notifies.length = 0;
|
|
1258
|
-
handlePlanningBeforeCompact(pi as any, ctx as any, {
|
|
1259
|
-
type: "session_before_compact",
|
|
1260
|
-
preparation: makePreparation("threshold", null),
|
|
1261
|
-
branchEntries: compactableBranchEntries(),
|
|
1262
|
-
customInstructions: "keep:1",
|
|
1263
|
-
reason: "threshold",
|
|
1264
|
-
willRetry: false,
|
|
1265
|
-
signal: new AbortController().signal,
|
|
1266
|
-
});
|
|
1267
|
-
handlePlanningCompactFailed(pi as any, ctx as any, {
|
|
1268
|
-
type: "session_compact_failed",
|
|
1269
|
-
reason: "threshold",
|
|
1270
|
-
errorMessage,
|
|
1271
|
-
aborted: errorMessage.includes("aborted"),
|
|
1272
|
-
willRetry: false,
|
|
1273
|
-
fromExtension: false,
|
|
1274
|
-
});
|
|
1275
|
-
const backoffNote = recorded.notifies.find((n) => n.message.includes("was aborted"));
|
|
1276
|
-
assert.ok(backoffNote, `expected abort-class notify for: ${errorMessage}`);
|
|
1277
|
-
assert.match(backoffNote!.message, /provider interruption|competing manual/);
|
|
1278
|
-
const before = recorded.compacts?.length ?? 0;
|
|
1279
|
-
requestPlanningCompaction(ctx as any);
|
|
1280
|
-
assert.equal((recorded.compacts?.length ?? 0) - before, 0);
|
|
474
|
+
assert.ok(ctx.entries.some((e) => e.customType === "pi-plans-review-paused"), "the pause is in-band");
|
|
475
|
+
const load = loadCheckpoint(workdir, runId!);
|
|
476
|
+
assert.ok(load.status === "ok");
|
|
477
|
+
assert.equal(load.checkpoint.phase, "executing", "an unreadable round must not complete the run");
|
|
478
|
+
assert.equal(load.checkpoint.execution?.audit?.rounds, REVIEW_MAX_ROUNDS);
|
|
479
|
+
__setAuditRunnerForTests(null);
|
|
480
|
+
await stopExecution(ctx, "test teardown");
|
|
481
|
+
});
|
|
482
|
+
|
|
483
|
+
it("a runner passing a strict subset credits the pass and keeps the rest undeterminable", async () => {
|
|
484
|
+
// The runner reports VC-001 passed and claims zero failures — VC-002
|
|
485
|
+
// is simply missing from its report, so it must NOT complete the run,
|
|
486
|
+
// must NOT be called failed, and must NOT roll Task-3 back.
|
|
487
|
+
__setAuditRunnerForTests(async ({ checklist }) => {
|
|
488
|
+
const vc1 = checklist.find((item) => item.id === "VC-001");
|
|
489
|
+
if (vc1) vc1.done = true;
|
|
490
|
+
return { round: 1, passed: ["VC-001"], failed: [], undeterminable: ["VC-002"], report: "partial report" };
|
|
491
|
+
});
|
|
492
|
+
const { workdir, planPath } = freshWorkdir();
|
|
493
|
+
const { ctx } = await start(planPath, workdir);
|
|
494
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
495
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
496
|
+
persistTaskProgress(ctx);
|
|
1281
497
|
}
|
|
1282
|
-
|
|
1283
|
-
|
|
1284
|
-
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
|
|
1290
|
-
|
|
1291
|
-
|
|
1292
|
-
|
|
1293
|
-
|
|
1294
|
-
|
|
1295
|
-
|
|
1296
|
-
|
|
1297
|
-
|
|
1298
|
-
|
|
1299
|
-
|
|
1300
|
-
|
|
1301
|
-
|
|
1302
|
-
|
|
1303
|
-
|
|
1304
|
-
|
|
1305
|
-
|
|
1306
|
-
|
|
1307
|
-
|
|
1308
|
-
assert.ok(backoffNote);
|
|
1309
|
-
const before = recorded.compacts?.length ?? 0;
|
|
1310
|
-
requestPlanningCompaction(ctx as any);
|
|
1311
|
-
assert.equal((recorded.compacts?.length ?? 0) - before, 0);
|
|
1312
|
-
});
|
|
1313
|
-
|
|
1314
|
-
it("network-style planning failures stay retryable without proactive requests", () => {
|
|
1315
|
-
const workdir = freshWorkdir();
|
|
1316
|
-
const { pi, ctx, recorded } = makeHarness(workdir);
|
|
1317
|
-
initState(workdir);
|
|
1318
|
-
startRun(workdir, { topic: "network-retryable", skill: "plan-normal", requestText: "x" });
|
|
1319
|
-
handlePlanningBeforeCompact(pi as any, ctx as any, {
|
|
1320
|
-
type: "session_before_compact",
|
|
1321
|
-
preparation: makePreparation("threshold", null),
|
|
1322
|
-
branchEntries: compactableBranchEntries(),
|
|
1323
|
-
customInstructions: "keep:1",
|
|
1324
|
-
reason: "threshold",
|
|
1325
|
-
willRetry: false,
|
|
1326
|
-
signal: new AbortController().signal,
|
|
1327
|
-
});
|
|
1328
|
-
|
|
1329
|
-
handlePlanningCompactFailed(pi as any, ctx as any, {
|
|
1330
|
-
type: "session_compact_failed",
|
|
1331
|
-
reason: "threshold",
|
|
1332
|
-
errorMessage: "network down",
|
|
1333
|
-
aborted: false,
|
|
1334
|
-
willRetry: false,
|
|
1335
|
-
fromExtension: false,
|
|
1336
|
-
});
|
|
1337
|
-
const abortNote = recorded.notifies.find((n) => n.message.includes("was aborted"));
|
|
1338
|
-
assert.equal(abortNote, undefined, "network down must not be classified as abort-class");
|
|
1339
|
-
const retryNote = recorded.notifies.find((n) => n.message.includes("will try again"));
|
|
1340
|
-
assert.ok(retryNote, "network down must keep the retryable path");
|
|
1341
|
-
const before = recorded.compacts?.length ?? 0;
|
|
1342
|
-
requestPlanningCompaction(ctx as any);
|
|
1343
|
-
assert.equal((recorded.compacts?.length ?? 0) - before, 0);
|
|
1344
|
-
});
|
|
1345
|
-
|
|
1346
|
-
it("lifecycle flags distinguish unhinted and phase-attributed compactions", () => {
|
|
1347
|
-
const workdir = freshWorkdir();
|
|
1348
|
-
const { ctx } = makeHarness(workdir);
|
|
1349
|
-
|
|
1350
|
-
noteCompactionStarted(ctx as any, undefined);
|
|
1351
|
-
assert.equal(compactionInFlight(ctx as any, "planning"), true, "unhinted compaction marks planning inFlight");
|
|
1352
|
-
assert.equal(compactionInFlight(ctx as any, "execution"), true, "unhinted compaction marks execution inFlight");
|
|
1353
|
-
noteCompactionEnded(ctx as any, undefined);
|
|
1354
|
-
assert.equal(compactionInFlight(ctx as any, "planning"), false);
|
|
1355
|
-
assert.equal(compactionInFlight(ctx as any, "execution"), false);
|
|
1356
|
-
|
|
1357
|
-
noteCompactionStarted(ctx as any, "pi-plans planning auto compact");
|
|
1358
|
-
assert.equal(compactionInFlight(ctx as any, "planning"), true);
|
|
1359
|
-
assert.equal(compactionInFlight(ctx as any, "execution"), false);
|
|
1360
|
-
noteCompactionEnded(ctx as any, "pi-plans planning auto compact");
|
|
1361
|
-
assert.equal(compactionInFlight(ctx as any, "planning"), false);
|
|
1362
|
-
|
|
1363
|
-
noteCompactionStarted(ctx as any, "pi-plans execution auto compact");
|
|
1364
|
-
assert.equal(compactionInFlight(ctx as any, "planning"), false);
|
|
1365
|
-
assert.equal(compactionInFlight(ctx as any, "execution"), true);
|
|
1366
|
-
noteCompactionEnded(ctx as any, "pi-plans execution auto compact");
|
|
1367
|
-
assert.equal(compactionInFlight(ctx as any, "execution"), false);
|
|
1368
|
-
});
|
|
1369
|
-
|
|
1370
|
-
it("execution handler classifies abort/stream failures as terminal", async () => {
|
|
1371
|
-
const workdir = freshWorkdir();
|
|
1372
|
-
startActiveRun(workdir);
|
|
1373
|
-
const { pi, ctx, recorded } = makeHarness(workdir);
|
|
1374
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v4.md"), items("VC-001"));
|
|
1375
|
-
|
|
1376
|
-
const abortMessages = [
|
|
1377
|
-
"Auto-compaction failed: Turn prefix summarization failed: This operation was aborted",
|
|
1378
|
-
"Error: OpenAI Responses stream ended before a terminal response event",
|
|
1379
|
-
"Auto-compaction failed: context overflow recovery failed",
|
|
1380
|
-
"aborted",
|
|
1381
|
-
];
|
|
1382
|
-
for (const errorMessage of abortMessages) {
|
|
1383
|
-
recorded.notifies.length = 0;
|
|
1384
|
-
handleExecutionBeforeCompact(pi, ctx, {
|
|
1385
|
-
type: "session_before_compact",
|
|
1386
|
-
preparation: makePreparation("threshold", null),
|
|
1387
|
-
branchEntries: compactableBranchEntries(),
|
|
1388
|
-
customInstructions: "keep:1",
|
|
1389
|
-
reason: "threshold",
|
|
1390
|
-
willRetry: false,
|
|
1391
|
-
signal: new AbortController().signal,
|
|
1392
|
-
});
|
|
1393
|
-
handleExecutionCompactFailed(pi, ctx, {
|
|
1394
|
-
type: "session_compact_failed",
|
|
1395
|
-
reason: "threshold",
|
|
1396
|
-
errorMessage,
|
|
1397
|
-
aborted: errorMessage.includes("aborted"),
|
|
1398
|
-
willRetry: false,
|
|
1399
|
-
fromExtension: false,
|
|
1400
|
-
});
|
|
1401
|
-
const note = recorded.notifies.find((n) => n.message.includes("was aborted"));
|
|
1402
|
-
assert.ok(note, `expected abort-class notify for: ${errorMessage}`);
|
|
1403
|
-
assert.match(note!.message, /provider interruption|competing manual/);
|
|
498
|
+
const snapshot = getExecution();
|
|
499
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
500
|
+
const ex = getExecution()!;
|
|
501
|
+
assert.deepEqual(ex.audit.undeterminable, ["VC-002"], "the unreported check stays pending judgement");
|
|
502
|
+
assert.deepEqual(ex.audit.failed, []);
|
|
503
|
+
assert.equal(ex.tasks.find((t) => t.id === "Task-3")?.status, "complete", "partial pass does not roll back");
|
|
504
|
+
__setAuditRunnerForTests(null);
|
|
505
|
+
await stopExecution(ctx, "test teardown");
|
|
506
|
+
});
|
|
507
|
+
|
|
508
|
+
it("an emphasized pass verdict completes the run with no rollback", async () => {
|
|
509
|
+
// The production incident: the auditor wrote `verdict: **pass**` for
|
|
510
|
+
// every check and the run rolled everything back. Emphasis must parse,
|
|
511
|
+
// and a fully-passing round must complete rather than pause.
|
|
512
|
+
__setAuditRunnerForTests(async ({ checklist }) => ({
|
|
513
|
+
round: 1,
|
|
514
|
+
passed: checklist.map((item) => item.id),
|
|
515
|
+
failed: [],
|
|
516
|
+
undeterminable: [],
|
|
517
|
+
report: "- `VC-001` — verdict: **pass**\n- `VC-002` — verdict: **pass**",
|
|
518
|
+
}));
|
|
519
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
520
|
+
const { ctx } = await start(planPath, workdir);
|
|
521
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
522
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
523
|
+
persistTaskProgress(ctx);
|
|
1404
524
|
}
|
|
1405
|
-
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
|
|
1411
|
-
|
|
1412
|
-
|
|
1413
|
-
|
|
1414
|
-
|
|
1415
|
-
|
|
1416
|
-
|
|
1417
|
-
|
|
1418
|
-
|
|
1419
|
-
|
|
1420
|
-
|
|
1421
|
-
|
|
1422
|
-
|
|
1423
|
-
|
|
1424
|
-
|
|
525
|
+
const snapshot = getExecution();
|
|
526
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
527
|
+
const load = loadCheckpoint(workdir, runId!);
|
|
528
|
+
assert.ok(load.status === "ok");
|
|
529
|
+
assert.equal(load.checkpoint.phase, "completed", "a fully-passing round completes the run");
|
|
530
|
+
assert.equal(load.checkpoint.execution?.audit?.passed, true);
|
|
531
|
+
__setAuditRunnerForTests(null);
|
|
532
|
+
await stopExecution(makeCtx(workdir), "test teardown");
|
|
533
|
+
});
|
|
534
|
+
|
|
535
|
+
it("an all-undeterminable round completes nothing, rolls nothing back, and never wakes (v0.8)", async () => {
|
|
536
|
+
// Fail-open guard: `undeterminable` is neither a pass nor a failure, so
|
|
537
|
+
// an all-undeterminable round must leave failed === [] AND must not take
|
|
538
|
+
// the completion branch. v0.8: retries self-schedule (Q-B) — zero wakes.
|
|
539
|
+
__setAuditRunnerForTests(async ({ checklist }) => ({
|
|
540
|
+
round: 1,
|
|
541
|
+
passed: [],
|
|
542
|
+
failed: [],
|
|
543
|
+
undeterminable: checklist.map((item) => item.id),
|
|
544
|
+
report: "## Findings\n\n- `F-001` — severity: high; verdict-free reviewer output\n\n## Questions\n\nNone.",
|
|
545
|
+
}));
|
|
546
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
547
|
+
const { ctx } = await start(planPath, workdir);
|
|
548
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
549
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
550
|
+
persistTaskProgress(ctx);
|
|
551
|
+
}
|
|
552
|
+
const snapshot = getExecution();
|
|
553
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
554
|
+
const ex = getExecution()!;
|
|
555
|
+
assert.deepEqual(ex.audit.failed, []);
|
|
556
|
+
assert.deepEqual(ex.audit.undeterminable.sort(), ["VC-001", "VC-002"]);
|
|
557
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
558
|
+
assert.equal(ex.tasks.find((t) => t.id === id)?.status, "complete", `${id} not rolled back`);
|
|
559
|
+
}
|
|
560
|
+
assert.equal(ex.audit.rounds, REVIEW_MAX_ROUNDS, "self-scheduled retries spent the budget");
|
|
561
|
+
assert.equal(ex.stall.paused, true, "paused at the cap");
|
|
562
|
+
const load = loadCheckpoint(workdir, runId!);
|
|
563
|
+
assert.ok(load.status === "ok");
|
|
564
|
+
assert.equal(load.checkpoint.phase, "executing", "never fail-open into completed");
|
|
565
|
+
// Undeterminable rounds never wake the agent (Q-B); only the pause is announced.
|
|
566
|
+
const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-undeterminable");
|
|
567
|
+
assert.equal(wakes.length, 0, "undeterminable rounds send no wake");
|
|
568
|
+
assert.ok(ctx.entries.some((e) => e.customType === "pi-plans-review-paused"), "the cap pause is in-band");
|
|
569
|
+
__setAuditRunnerForTests(null);
|
|
570
|
+
await stopExecution(ctx, "test teardown");
|
|
571
|
+
});
|
|
572
|
+
|
|
573
|
+
it("the audit budget is monotonic: an agent repair re-close does not refund a round", async () => {
|
|
574
|
+
// The unbounded-loop guard. The agent's repair is a forward transition
|
|
575
|
+
// (pending -> complete); if that reset the audit budget, the sequence
|
|
576
|
+
// fail -> repair -> fail would never reach the cap.
|
|
577
|
+
let call = 0;
|
|
578
|
+
__setAuditRunnerForTests(async ({ checklist, round }) => {
|
|
579
|
+
call += 1;
|
|
580
|
+
return call === 1
|
|
581
|
+
? { round, passed: ["VC-001"], failed: ["VC-002"], report: "first attempt fails" }
|
|
582
|
+
: { round, passed: ["VC-001"], failed: ["VC-002"], report: "still failing" };
|
|
583
|
+
});
|
|
584
|
+
const { workdir, planPath } = freshWorkdir();
|
|
585
|
+
const { ctx } = await start(planPath, workdir);
|
|
586
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
587
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
588
|
+
persistTaskProgress(ctx);
|
|
589
|
+
}
|
|
590
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
591
|
+
assert.equal(getExecution()!.audit.rounds, 1);
|
|
592
|
+
// The agent repairs the rolled-back work and re-closes it: a forward
|
|
593
|
+
// transition, exactly what a naive budget reset would reward.
|
|
594
|
+
for (const id of ["Task-2", "Task-3"]) {
|
|
595
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence again`);
|
|
596
|
+
persistTaskProgress(ctx);
|
|
597
|
+
}
|
|
598
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
599
|
+
assert.equal(getExecution()!.audit.rounds, 2, "the second attempt costs a second round");
|
|
600
|
+
__setAuditRunnerForTests(null);
|
|
601
|
+
await stopExecution(makeCtx(workdir), "test teardown");
|
|
602
|
+
});
|
|
603
|
+
|
|
604
|
+
it("the audit round cap test still pauses an interactive session", async () => {
|
|
605
|
+
// Unchanged intent: exhausting the budget pauses interactively. The
|
|
606
|
+
// seam moved (getExecution()!.audit.rounds), so it is pinned here.
|
|
607
|
+
__setAuditRunnerForTests(async ({ checklist }) => ({
|
|
608
|
+
round: 1,
|
|
609
|
+
passed: [],
|
|
610
|
+
failed: checklist.filter((item) => !["VC-001"].includes(item.id)).map((item) => item.id),
|
|
611
|
+
undeterminable: [],
|
|
612
|
+
report: "still failing",
|
|
613
|
+
}));
|
|
614
|
+
const { workdir, planPath } = freshWorkdir();
|
|
615
|
+
const { ctx } = await start(planPath, workdir);
|
|
616
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
617
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
618
|
+
persistTaskProgress(ctx);
|
|
619
|
+
}
|
|
620
|
+
getExecution()!.audit.rounds = REVIEW_MAX_ROUNDS;
|
|
621
|
+
getExecution()!.audit.failed = ["VC-002"];
|
|
622
|
+
const ctxUi = { ...makeCtx(workdir), mode: "tui" as const };
|
|
623
|
+
await restoreFromSession(ctxUi, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
624
|
+
await __awaitReviewRoundForTests();
|
|
625
|
+
const ex = getExecution()!;
|
|
626
|
+
assert.equal(ex.stall.paused, true, "the review cap pauses an interactive session");
|
|
627
|
+
assert.match(ex.stall.pausedReason ?? "", /execution review exhausted 5 rounds/);
|
|
628
|
+
__setAuditRunnerForTests(null);
|
|
629
|
+
await stopExecution(ctxUi, "test teardown");
|
|
630
|
+
});
|
|
631
|
+
|
|
632
|
+
it("toggleDashboardExpanded flips the expanded mode", () => {
|
|
633
|
+
const before = toggleEnabled();
|
|
634
|
+
toggleDashboardExpanded(makeCtx(freshWorkdir(false).workdir));
|
|
635
|
+
assert.equal(toggleEnabled(), !before);
|
|
636
|
+
toggleDashboardExpanded(makeCtx(freshWorkdir(false).workdir));
|
|
1425
637
|
});
|
|
638
|
+
});
|
|
1426
639
|
|
|
1427
|
-
|
|
1428
|
-
|
|
1429
|
-
|
|
1430
|
-
|
|
1431
|
-
|
|
1432
|
-
|
|
1433
|
-
|
|
1434
|
-
|
|
1435
|
-
|
|
640
|
+
/**
|
|
641
|
+
* v0.7.1 regression suite (root causes A and B). These drive the REAL lifecycle
|
|
642
|
+
* events (`agent_before_settle` / `turn_end` / `tool_result`) instead of
|
|
643
|
+
* mutating state directly, because the stranded-state bug only exists in the
|
|
644
|
+
* event ORDER — a test that calls restoreFromSession cannot see it.
|
|
645
|
+
*/
|
|
646
|
+
describe("v0.7.1 execution-loop fixes (event-driven)", () => {
|
|
647
|
+
/** A ctx in TUI mode: `canWakeExecution` requires mode tui/rpc, and a
|
|
648
|
+
* print-mode ctx would make every wake/audit assertion silently vacuous. */
|
|
649
|
+
function makeTuiCtx(workdir: string) {
|
|
650
|
+
const entries: Array<{ customType: string; data?: unknown; content?: string; triggerTurn?: boolean }> = [];
|
|
651
|
+
const ctx = {
|
|
1436
652
|
cwd: workdir,
|
|
1437
|
-
hasUI: false,
|
|
1438
653
|
sessionManager: {},
|
|
654
|
+
hasUI: true,
|
|
655
|
+
mode: "tui" as const,
|
|
656
|
+
entries,
|
|
657
|
+
ui: {
|
|
658
|
+
notify: () => {},
|
|
659
|
+
setStatus: () => {},
|
|
660
|
+
setWidget: () => {},
|
|
661
|
+
theme: { fg: (_c: string, t: string) => t, bold: (t: string) => t },
|
|
662
|
+
},
|
|
663
|
+
isIdle: () => true,
|
|
664
|
+
hasPendingMessages: () => false,
|
|
665
|
+
} as never;
|
|
666
|
+
setMessagingApi({
|
|
667
|
+
appendEntry: (customType: string, data: unknown) => entries.push({ customType, data }),
|
|
668
|
+
sendMessage: (message: { customType: string; content: string; details?: unknown }, opts?: { triggerTurn?: boolean }) =>
|
|
669
|
+
entries.push({ customType: message.customType, content: message.content, triggerTurn: opts?.triggerTurn }),
|
|
670
|
+
sendUserMessage: async () => {},
|
|
671
|
+
});
|
|
672
|
+
return ctx as { entries: typeof entries; cwd: string };
|
|
673
|
+
}
|
|
674
|
+
|
|
675
|
+
/** Register the production turn handlers and return a driver for the events. */
|
|
676
|
+
function wireEvents(ctx: never) {
|
|
677
|
+
const handlers = new Map<string, Array<(event: unknown, c: never) => unknown>>();
|
|
678
|
+
const ext = {
|
|
679
|
+
on: (name: string, fn: (event: unknown, c: never) => unknown) => {
|
|
680
|
+
const list = handlers.get(name) ?? [];
|
|
681
|
+
list.push(fn);
|
|
682
|
+
handlers.set(name, list);
|
|
683
|
+
},
|
|
684
|
+
} as never;
|
|
685
|
+
registerExecutionTurnHandlers(ext);
|
|
686
|
+
return {
|
|
687
|
+
async fire(name: string, event: unknown = {}) {
|
|
688
|
+
for (const fn of handlers.get(name) ?? []) await fn(event, ctx);
|
|
689
|
+
},
|
|
1439
690
|
};
|
|
1440
|
-
|
|
1441
|
-
noteCompactionStarted(ctx, undefined);
|
|
1442
|
-
assert.equal(compactionInFlight(ctx, "planning"), true);
|
|
1443
|
-
assert.equal(compactionInFlight(ctx, "execution"), true);
|
|
1444
|
-
noteCompactionEnded(ctx, undefined);
|
|
1445
|
-
assert.equal(compactionInFlight(ctx, "planning"), false);
|
|
1446
|
-
assert.equal(compactionInFlight(ctx, "execution"), false);
|
|
1447
|
-
// And a planning-attributed start + end still clears both (end has no hint).
|
|
1448
|
-
noteCompactionStarted(ctx, "pi-plans planning auto compact");
|
|
1449
|
-
assert.equal(compactionInFlight(ctx, "planning"), true);
|
|
1450
|
-
noteCompactionEnded(ctx, "pi-plans planning auto compact");
|
|
1451
|
-
assert.equal(compactionInFlight(ctx, "planning"), false);
|
|
1452
|
-
});
|
|
1453
|
-
});
|
|
691
|
+
}
|
|
1454
692
|
|
|
1455
|
-
|
|
1456
|
-
|
|
1457
|
-
|
|
1458
|
-
|
|
1459
|
-
|
|
1460
|
-
|
|
1461
|
-
} catch {
|
|
1462
|
-
/* ignore */
|
|
1463
|
-
}
|
|
693
|
+
async function startTui(planPath: string, workdir: string) {
|
|
694
|
+
const ctx = makeTuiCtx(workdir);
|
|
695
|
+
await startExecution(ctx as never, {
|
|
696
|
+
planPath,
|
|
697
|
+
planTasks: parsePlanTasks(PLAN),
|
|
698
|
+
items: parseChecklist(PLAN),
|
|
1464
699
|
});
|
|
1465
|
-
|
|
1466
|
-
|
|
1467
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v1.md"), items("VC-001"));
|
|
1468
|
-
|
|
1469
|
-
const enabledRules = executionContextMessage(ctx)!;
|
|
1470
|
-
assert.match(enabledRules, /Code graph loop: indexed code files read as a function digest/);
|
|
1471
|
-
assert.doesNotMatch(enabledRules, /Code graph disabled/);
|
|
1472
|
-
|
|
1473
|
-
setGraphEnabled(workdir, false);
|
|
1474
|
-
assert.match(executionContextMessage(ctx)!, /Code graph disabled/);
|
|
1475
|
-
|
|
1476
|
-
fs.writeFileSync(path.join(workdir, ".git", "pi_plans", "config.json"), "{broken");
|
|
1477
|
-
const brokenRules = executionContextMessage(ctx)!;
|
|
1478
|
-
assert.match(brokenRules, /Code graph disabled/);
|
|
1479
|
-
assert.match(brokenRules, /config unreadable this turn/);
|
|
1480
|
-
});
|
|
1481
|
-
|
|
1482
|
-
it("keeps no graphEnabled snapshot in the execution state source", () => {
|
|
1483
|
-
const source = fs.readFileSync(path.join(process.cwd(), "src", "exec.ts"), "utf8");
|
|
1484
|
-
assert.doesNotMatch(source, /graphEnabled/);
|
|
1485
|
-
assert.match(source, /resolveGraphMode/);
|
|
1486
|
-
});
|
|
1487
|
-
});
|
|
700
|
+
return ctx;
|
|
701
|
+
}
|
|
1488
702
|
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
messagesToSummarize: [],
|
|
1492
|
-
turnPrefixMessages: [],
|
|
1493
|
-
isSplitTurn: false,
|
|
1494
|
-
tokensBefore: 12345,
|
|
1495
|
-
previousSummary,
|
|
1496
|
-
fileOps: { read: [], written: [], edited: [] },
|
|
1497
|
-
settings: { enabled: true, reserveTokens: 16384, keepRecentTokens: 20000 },
|
|
703
|
+
const closeAll = (ids: string[]) => {
|
|
704
|
+
for (const id of ids) applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
1498
705
|
};
|
|
1499
|
-
}
|
|
1500
|
-
|
|
1501
|
-
describe("amelioration termination prompt", () => {
|
|
1502
|
-
it("recommends goal-wait first and keeps the round options", () => {
|
|
1503
|
-
const text = ameliorationPromptText("plan-normal");
|
|
1504
|
-
assert.match(text, /goal wait: continue until no unpassed VCs remain/);
|
|
1505
|
-
assert.match(text, /until no high-severity finding \(hard cap 5 rounds\)/);
|
|
1506
|
-
assert.match(text, /How should the implementation-review loop terminate\?/);
|
|
1507
|
-
});
|
|
1508
|
-
|
|
1509
|
-
it("asks the reviewer-count question with a skill-aware recommended default (D-1/D-2)", () => {
|
|
1510
|
-
const big = ameliorationPromptText("plan-big");
|
|
1511
|
-
assert.match(big, /impl-review-reviewer-count/);
|
|
1512
|
-
assert.match(big, /How many concurrent reviewers should each implementation-review round use\?/);
|
|
1513
|
-
assert.match(big, /3 \(recommended\)/);
|
|
1514
|
-
assert.match(big, /reviewers: <configured reviewerCount>/);
|
|
1515
|
-
const small = ameliorationPromptText("plan-small");
|
|
1516
|
-
assert.match(small, /1 \(recommended\)/);
|
|
1517
|
-
assert.doesNotMatch(small, /3 \(recommended\)/);
|
|
1518
|
-
});
|
|
1519
|
-
|
|
1520
|
-
it("defers persistence to ONE combined record-checkpoint carrying both answers (D-5)", () => {
|
|
1521
|
-
const text = ameliorationPromptText(undefined);
|
|
1522
|
-
assert.match(text, /Do not persist yet/);
|
|
1523
|
-
assert.match(text, /ONE call: plans record-checkpoint/);
|
|
1524
|
-
assert.match(text, /terminationCondition: "<termination answer>", reviewerCount: <chosen integer>/);
|
|
1525
|
-
});
|
|
1526
|
-
});
|
|
1527
706
|
|
|
1528
|
-
|
|
1529
|
-
|
|
1530
|
-
|
|
1531
|
-
|
|
1532
|
-
|
|
1533
|
-
|
|
1534
|
-
|
|
1535
|
-
|
|
1536
|
-
const { spawnSync } = await import("node:child_process");
|
|
1537
|
-
spawnSync("git", ["init"], { cwd: workdir });
|
|
1538
|
-
spawnSync("git", ["config", "user.email", "t@e.com"], { cwd: workdir });
|
|
1539
|
-
spawnSync("git", ["config", "user.name", "T"], { cwd: workdir });
|
|
1540
|
-
initState(workdir);
|
|
1541
|
-
const { run } = startRun(workdir, { topic: "exec-cp", skill: "plan-normal", requestText: "t" });
|
|
1542
|
-
createCheckpoint(workdir, { runId: run.run_id, originWorkdir: workdir, workdir });
|
|
1543
|
-
const artifactDir = run.artifact_dir;
|
|
1544
|
-
fs.mkdirSync(artifactDir, { recursive: true });
|
|
1545
|
-
const planPath = path.join(artifactDir, "PLAN_v1.md");
|
|
1546
|
-
fs.writeFileSync(
|
|
1547
|
-
planPath,
|
|
1548
|
-
"# Plan\n\n## Verifier Checklist\n\n- [ ] `VC-001` covers `I-001`; pass condition: x.\n- [ ] `VC-002` covers `I-002`; pass condition: y.\n",
|
|
1549
|
-
"utf8",
|
|
1550
|
-
);
|
|
1551
|
-
spawnSync("git", ["add", "-A"], { cwd: workdir });
|
|
1552
|
-
spawnSync("git", ["commit", "-m", "init"], { cwd: workdir });
|
|
1553
|
-
|
|
1554
|
-
const harness = makeHarness(workdir);
|
|
1555
|
-
const items: CheckItem[] = [
|
|
1556
|
-
{ id: "VC-001", text: "covers I-001", done: false },
|
|
1557
|
-
{ id: "VC-002", text: "covers I-002", done: false },
|
|
1558
|
-
];
|
|
1559
|
-
await startExecution(harness.pi, harness.ctx, planPath, items);
|
|
1560
|
-
|
|
1561
|
-
const loaded = loadCheckpoint(workdir, run.run_id);
|
|
1562
|
-
assert.equal(loaded.status, "ok");
|
|
1563
|
-
if (loaded.status === "ok") {
|
|
1564
|
-
assert.equal(loaded.checkpoint.phase, "executing");
|
|
1565
|
-
const approval = loaded.checkpoint.execution?.approval;
|
|
1566
|
-
assert.ok(approval, "approval recorded");
|
|
1567
|
-
assert.match(approval!.headAtApproval ?? "", /^[0-9a-f]{40}$/, "HEAD at approval");
|
|
1568
|
-
assert.equal(approval!.plan.path, planPath);
|
|
1569
|
-
}
|
|
1570
|
-
|
|
1571
|
-
// Progress lands in the checkpoint.
|
|
1572
|
-
applyDoneMarkers("[DONE:VC-001]");
|
|
1573
|
-
recordExecutionTurn(harness.pi, harness.ctx, ["VC-001"], { input: 5, output: 2 });
|
|
1574
|
-
const afterProgress = loadCheckpoint(workdir, run.run_id);
|
|
1575
|
-
if (afterProgress.status === "ok") {
|
|
1576
|
-
assert.deepEqual(afterProgress.checkpoint.execution?.doneVcIds, ["VC-001"]);
|
|
1577
|
-
assert.equal(afterProgress.checkpoint.execution?.usage.inToks, 5);
|
|
1578
|
-
}
|
|
1579
|
-
|
|
1580
|
-
// Cross-session restore: fresh harness (new session), load from checkpoint.
|
|
1581
|
-
resetRunBindingForTests();
|
|
1582
|
-
const harness2 = makeHarness(workdir);
|
|
1583
|
-
const restore = loadExecutionFromCheckpoint(harness2.pi, harness2.ctx, run.run_id);
|
|
1584
|
-
assert.equal(restore.status, "loaded");
|
|
1585
|
-
assert.deepEqual(restore.doneVcIds, ["VC-001"]);
|
|
1586
|
-
assert.equal(restore.reverifyAll, false, "same HEAD keeps verified VCs");
|
|
1587
|
-
assert.equal(getExecution()?.items.find((item) => item.id === "VC-001")?.done, true);
|
|
1588
|
-
// F-002: an immediate session snapshot was appended.
|
|
1589
|
-
const snapshots = harness2.recorded.entries.filter((entry) => entry.customType === "pi-plans-exec");
|
|
1590
|
-
assert.ok(snapshots.length >= 1, "session snapshot written on load");
|
|
1591
|
-
|
|
1592
|
-
// Stop keeps progress with pausedReason (D-008).
|
|
1593
|
-
await stopExecution(harness2.pi, harness2.ctx, "stopped by user");
|
|
1594
|
-
const stopped = loadCheckpoint(workdir, run.run_id);
|
|
1595
|
-
if (stopped.status === "ok") {
|
|
1596
|
-
assert.equal(stopped.checkpoint.execution?.pausedReason, "stopped by user");
|
|
1597
|
-
assert.deepEqual(stopped.checkpoint.execution?.doneVcIds, ["VC-001"]);
|
|
707
|
+
it("root cause A: a terminal-but-unaudited run self-heals via agent_before_settle (zero input)", async () => {
|
|
708
|
+
// First audit round fails -> tasks roll back to pending (0/3).
|
|
709
|
+
let round = 0;
|
|
710
|
+
__setAuditRunnerForTests(async ({ checklist }) => {
|
|
711
|
+
round += 1;
|
|
712
|
+
if (round === 1) {
|
|
713
|
+
const failed = checklist.filter((i) => i.id === "VC-002").map((i) => i.id);
|
|
714
|
+
return { round, passed: ["VC-001"], failed, report: "VC-002 fails" };
|
|
1598
715
|
}
|
|
1599
|
-
|
|
1600
|
-
|
|
1601
|
-
}
|
|
716
|
+
for (const item of checklist) item.done = true;
|
|
717
|
+
return { round, passed: checklist.map((i) => i.id), failed: [], report: "all pass" };
|
|
718
|
+
});
|
|
719
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
720
|
+
const ctx = await startTui(planPath, workdir);
|
|
721
|
+
const drive = wireEvents(ctx as never);
|
|
722
|
+
|
|
723
|
+
// Agent closes every task; the turn ends -> audit round 1 runs and fails.
|
|
724
|
+
closeAll(["Task-1", "Task-2", "Task-3"]);
|
|
725
|
+
persistTaskProgress(ctx as never);
|
|
726
|
+
await drive.fire("turn_end", { message: { role: "assistant", stopReason: "stop" } });
|
|
727
|
+
assert.equal(round, 1, "round 1 audit ran");
|
|
728
|
+
assert.equal(getExecution()!.tasks.find((t) => t.id === "Task-2")?.status, "pending", "VC-002 rolled its tasks back");
|
|
729
|
+
|
|
730
|
+
// The agent fixes the failures and re-closes the rolled-back tasks. A new
|
|
731
|
+
// agent run opens a new settle window (agent_start resets the latch).
|
|
732
|
+
await drive.fire("agent_start", {});
|
|
733
|
+
closeAll(["Task-1", "Task-2", "Task-3"]);
|
|
734
|
+
persistTaskProgress(ctx as never);
|
|
735
|
+
assert.ok(allTasksTerminal(getExecution()!.tasks), "task tree is terminal again");
|
|
736
|
+
|
|
737
|
+
// The regression: fire ONLY the actionable pre-settle boundary, i.e. the
|
|
738
|
+
// case where the turn_end trigger is missed (the observed strand). Before
|
|
739
|
+
// v0.7.1 nothing else could start the audit, so the run sat in
|
|
740
|
+
// `executing` until a manual /plans-execute.
|
|
741
|
+
await drive.fire("agent_before_settle", {});
|
|
742
|
+
assert.equal(round, 2, "the owed audit reran automatically with zero user input");
|
|
743
|
+
|
|
744
|
+
const load = loadCheckpoint(workdir, runId!);
|
|
745
|
+
assert.ok(load.status === "ok");
|
|
746
|
+
assert.equal(load.checkpoint.phase, "completed", "run reached completed without a manual /plans-execute");
|
|
747
|
+
__setAuditRunnerForTests(null);
|
|
748
|
+
});
|
|
749
|
+
|
|
750
|
+
it("root cause A: the audit is not run twice within one settle (per-settle latch)", async () => {
|
|
751
|
+
let calls = 0;
|
|
752
|
+
__setAuditRunnerForTests(async ({ checklist }) => {
|
|
753
|
+
calls += 1;
|
|
754
|
+
for (const item of checklist) item.done = true;
|
|
755
|
+
return { round: 1, passed: checklist.map((i) => i.id), failed: [], report: "all pass" };
|
|
756
|
+
});
|
|
757
|
+
const { workdir, planPath } = freshWorkdir();
|
|
758
|
+
const ctx = await startTui(planPath, workdir);
|
|
759
|
+
const drive = wireEvents(ctx as never);
|
|
760
|
+
closeAll(["Task-1", "Task-2", "Task-3"]);
|
|
761
|
+
persistTaskProgress(ctx as never);
|
|
762
|
+
// Both entry points land in the same settle.
|
|
763
|
+
await drive.fire("turn_end", { message: { role: "assistant", stopReason: "stop" } });
|
|
764
|
+
await drive.fire("agent_before_settle", {});
|
|
765
|
+
assert.equal(calls, 1, "exactly one audit call per settle");
|
|
766
|
+
__setAuditRunnerForTests(null);
|
|
767
|
+
});
|
|
768
|
+
|
|
769
|
+
it("root cause A: a fully terminal run emits no EXECUTION_CONTINUE wake", async () => {
|
|
770
|
+
const { workdir, planPath } = freshWorkdir();
|
|
771
|
+
const ctx = await startTui(planPath, workdir);
|
|
772
|
+
const drive = wireEvents(ctx as never);
|
|
773
|
+
closeAll(["Task-1", "Task-2", "Task-3"]);
|
|
774
|
+
persistTaskProgress(ctx as never);
|
|
775
|
+
await drive.fire("agent_settled", {});
|
|
776
|
+
assert.equal(
|
|
777
|
+
ctx.entries.filter((e) => e.customType === "pi-plans-exec-continue").length,
|
|
778
|
+
0,
|
|
779
|
+
"no continuation wake when every task is already terminal",
|
|
780
|
+
);
|
|
781
|
+
await stopExecution(ctx as never, "test teardown");
|
|
782
|
+
});
|
|
783
|
+
|
|
784
|
+
it("root cause A: an audit failure wakes the agent exactly once", async () => {
|
|
785
|
+
__setAuditRunnerForTests(async ({ checklist }) => {
|
|
786
|
+
const failed = checklist.filter((i) => i.id === "VC-002").map((i) => i.id);
|
|
787
|
+
return { round: 1, passed: ["VC-001"], failed, report: "VC-002 fails" };
|
|
788
|
+
});
|
|
789
|
+
const { workdir, planPath } = freshWorkdir();
|
|
790
|
+
const ctx = await startTui(planPath, workdir);
|
|
791
|
+
const drive = wireEvents(ctx as never);
|
|
792
|
+
closeAll(["Task-1", "Task-2", "Task-3"]);
|
|
793
|
+
persistTaskProgress(ctx as never);
|
|
794
|
+
await drive.fire("turn_end", { message: { role: "assistant", stopReason: "stop" } });
|
|
795
|
+
const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed" || e.customType === "pi-plans-exec-continue");
|
|
796
|
+
assert.equal(wakes.length, 1, `one wake per settle, got ${wakes.map((w) => w.customType).join(",")}`);
|
|
797
|
+
assert.equal(wakes[0]?.triggerTurn, true, "the audit-failure notice drives the fix turn");
|
|
798
|
+
__setAuditRunnerForTests(null);
|
|
799
|
+
await stopExecution(ctx as never, "test teardown");
|
|
800
|
+
});
|
|
801
|
+
|
|
802
|
+
it("root cause B: a successful tool result counts as progress and rebases the watchdog", async () => {
|
|
803
|
+
const { workdir, planPath } = freshWorkdir();
|
|
804
|
+
const ctx = await startTui(planPath, workdir);
|
|
805
|
+
const drive = wireEvents(ctx as never);
|
|
806
|
+
const exec = getExecution()!;
|
|
807
|
+
exec.stall.rounds = 2;
|
|
808
|
+
// A successful tool result during a cross-round investigation.
|
|
809
|
+
await drive.fire("tool_result", { isError: false });
|
|
810
|
+
assert.equal(exec.stall.rounds, 0, "legitimate work rebased the no-progress counter");
|
|
811
|
+
await stopExecution(ctx as never, "test teardown");
|
|
1602
812
|
});
|
|
1603
813
|
|
|
1604
|
-
it("
|
|
1605
|
-
const {
|
|
1606
|
-
const
|
|
1607
|
-
const
|
|
1608
|
-
|
|
1609
|
-
const
|
|
1610
|
-
|
|
1611
|
-
|
|
1612
|
-
|
|
1613
|
-
|
|
1614
|
-
|
|
1615
|
-
initState(workdir);
|
|
1616
|
-
const { run } = startRun(workdir, { topic: "exec-head", skill: "plan-normal", requestText: "t" });
|
|
1617
|
-
createCheckpoint(workdir, { runId: run.run_id, originWorkdir: workdir, workdir });
|
|
1618
|
-
fs.mkdirSync(run.artifact_dir, { recursive: true });
|
|
1619
|
-
const planPath = path.join(run.artifact_dir, "PLAN_v1.md");
|
|
1620
|
-
fs.writeFileSync(
|
|
1621
|
-
planPath,
|
|
1622
|
-
"# Plan\n\n## Verifier Checklist\n\n- [ ] `VC-001` covers `I-001`; pass condition: x.\n",
|
|
1623
|
-
"utf8",
|
|
1624
|
-
);
|
|
1625
|
-
spawnSync("git", ["add", "-A"], { cwd: workdir });
|
|
1626
|
-
spawnSync("git", ["commit", "-m", "one"], { cwd: workdir });
|
|
1627
|
-
|
|
1628
|
-
const harness = makeHarness(workdir);
|
|
1629
|
-
await startExecution(harness.pi, harness.ctx, planPath, [
|
|
1630
|
-
{ id: "VC-001", text: "covers I-001", done: false },
|
|
1631
|
-
]);
|
|
1632
|
-
applyDoneMarkers("[DONE:VC-001]");
|
|
1633
|
-
recordExecutionTurn(harness.pi, harness.ctx, ["VC-001"]);
|
|
1634
|
-
|
|
1635
|
-
// Simulate a branch switch / reset under the unchanged plan.
|
|
1636
|
-
spawnSync("git", ["commit", "--allow-empty", "-m", "two"], { cwd: workdir });
|
|
1637
|
-
|
|
1638
|
-
resetRunBindingForTests();
|
|
1639
|
-
const harness2 = makeHarness(workdir);
|
|
1640
|
-
const restore = loadExecutionFromCheckpoint(harness2.pi, harness2.ctx, run.run_id);
|
|
1641
|
-
assert.equal(restore.status, "loaded");
|
|
1642
|
-
assert.deepEqual(restore.doneVcIds, ["VC-001"], "historical evidence kept");
|
|
1643
|
-
assert.equal(restore.reverifyAll, true, "HEAD change forces re-verification");
|
|
1644
|
-
assert.equal(getExecution()?.items[0]?.done, false, "old VC not auto-passed");
|
|
1645
|
-
// Authorization survives.
|
|
1646
|
-
const cp = loadCheckpoint(workdir, run.run_id);
|
|
1647
|
-
if (cp.status === "ok") {
|
|
1648
|
-
assert.ok(cp.checkpoint.execution?.approval, "authorization kept");
|
|
1649
|
-
assert.equal(cp.checkpoint.execution?.reverifyAll, true);
|
|
1650
|
-
}
|
|
1651
|
-
} finally {
|
|
1652
|
-
fs.rmSync(workdir, { recursive: true, force: true });
|
|
1653
|
-
}
|
|
814
|
+
it("root cause B (negative): an error-only tool loop still trips the cap", async () => {
|
|
815
|
+
const { workdir, planPath } = freshWorkdir();
|
|
816
|
+
const ctx = await startTui(planPath, workdir);
|
|
817
|
+
const drive = wireEvents(ctx as never);
|
|
818
|
+
const exec = getExecution()!;
|
|
819
|
+
const base = exec.stall.lastSnapshot;
|
|
820
|
+
// Tools that all fail: repeated retries must NOT look like progress.
|
|
821
|
+
for (let i = 0; i < 5; i++) await drive.fire("tool_result", { isError: true });
|
|
822
|
+
assert.equal(exec.stall.rounds, 0, "a failed tool does not rebase the counter on its own");
|
|
823
|
+
assert.equal(exec.stall.lastSnapshot, base, "snapshot unchanged by failed tools");
|
|
824
|
+
await stopExecution(ctx as never, "test teardown");
|
|
1654
825
|
});
|
|
1655
826
|
});
|
|
1656
827
|
|
|
1657
|
-
|
|
1658
|
-
|
|
1659
|
-
|
|
1660
|
-
|
|
1661
|
-
|
|
1662
|
-
|
|
1663
|
-
});
|
|
1664
|
-
|
|
1665
|
-
after(() => {
|
|
1666
|
-
fs.rmSync(tmpRoot, { recursive: true, force: true });
|
|
1667
|
-
});
|
|
1668
|
-
|
|
1669
|
-
async function preplanHarness() {
|
|
1670
|
-
const piPlansExtension = (await import("../index.ts")).default;
|
|
1671
|
-
const harness = makeHarness(path.join(tmpRoot, `ws-${++counter}`));
|
|
1672
|
-
fs.mkdirSync(harness.ctx.cwd, { recursive: true });
|
|
1673
|
-
piPlansExtension(harness.pi as never);
|
|
1674
|
-
return harness;
|
|
1675
|
-
}
|
|
1676
|
-
|
|
1677
|
-
it("routes the pre-plan hint through the planning VCC path for a fresh run", () => {
|
|
1678
|
-
// Pure-builder assertion (order-independent): the getExecution() gate on
|
|
1679
|
-
// buildPlanningCompactionResult is covered by the planning-hook tests.
|
|
1680
|
-
const workdir = path.join(tmpRoot, `hint-${++counter}`);
|
|
1681
|
-
fs.mkdirSync(workdir, { recursive: true });
|
|
1682
|
-
initState(workdir);
|
|
1683
|
-
const { run } = startRun(workdir, { topic: "preplan hint", skill: "plan-normal", requestText: "x" });
|
|
1684
|
-
const built = buildPiPlansVccCompaction({
|
|
1685
|
-
branchEntries: compactableBranchEntries(),
|
|
1686
|
-
preparation: makePreparation("manual", null),
|
|
1687
|
-
customInstructions: PLANNING_PREPLAN_COMPACT_HINT,
|
|
1688
|
-
reason: "manual",
|
|
1689
|
-
willRetry: false,
|
|
1690
|
-
settings: loadVccSettings(path.join(workdir, ".git", "pi_plans")),
|
|
1691
|
-
phaseContext: { phase: "planning", runId: run.run_id, artifactDir: run.artifact_dir },
|
|
1692
|
-
});
|
|
1693
|
-
assert.equal(built.kind, "compaction", "pre-plan hint must produce a VCC compaction, not a cancel/fallback");
|
|
1694
|
-
if (built.kind !== "compaction") return;
|
|
1695
|
-
assert.equal((built.compaction.details as { compactor: string }).compactor, "pi-vcc");
|
|
1696
|
-
assert.equal((built.compaction.details as { phase: string }).phase, "planning");
|
|
1697
|
-
assert.match(built.compaction.summary, /\[Session Goal\]/);
|
|
1698
|
-
});
|
|
1699
|
-
|
|
1700
|
-
it("marks and consumes the pending flag exactly once", () => {
|
|
1701
|
-
const harness = makeHarness(path.join(tmpRoot, `flag-${++counter}`));
|
|
1702
|
-
assert.equal(consumePrePlanCompactPending(harness.ctx), null);
|
|
1703
|
-
markPrePlanCompactPending(harness.ctx, "run-1");
|
|
1704
|
-
assert.deepEqual(consumePrePlanCompactPending(harness.ctx), { runId: "run-1" });
|
|
1705
|
-
assert.equal(consumePrePlanCompactPending(harness.ctx), null);
|
|
1706
|
-
});
|
|
1707
|
-
|
|
1708
|
-
it("compacts once from the plans tool_result and resumes exactly once on success", async () => {
|
|
1709
|
-
const harness = await preplanHarness();
|
|
1710
|
-
const { ctx, recorded, emit } = harness;
|
|
1711
|
-
markPrePlanCompactPending(ctx, "run-a");
|
|
1712
|
-
|
|
1713
|
-
// Non-plans tool results never consume the pending flag.
|
|
1714
|
-
await emit("tool_result", { type: "tool_result", toolName: "edit", input: { path: "x" }, isError: false });
|
|
1715
|
-
assert.equal(recorded.compacts?.length ?? 0, 0);
|
|
1716
|
-
assert.equal(recorded.messages.length, 0);
|
|
1717
|
-
|
|
1718
|
-
await emit("tool_result", { type: "tool_result", toolName: "plans", input: { action: "start-run" }, isError: false });
|
|
1719
|
-
assert.equal(recorded.compacts?.length, 1);
|
|
1720
|
-
assert.equal(recorded.compacts?.[0]?.customInstructions, PLANNING_PREPLAN_COMPACT_HINT);
|
|
1721
|
-
assert.equal(recorded.messages.length, 0, "no resume before the compaction settles");
|
|
1722
|
-
|
|
1723
|
-
recorded.compacts?.[0]?.onComplete?.({} as never);
|
|
1724
|
-
assert.equal(recorded.messages.length, 1);
|
|
1725
|
-
assert.equal(recorded.messages[0]?.customType, PLANNING_PREPLAN_RESUME_CUSTOM_TYPE);
|
|
1726
|
-
assert.equal(recorded.messages[0]?.content, "Continue planning.");
|
|
1727
|
-
assert.equal((recorded.messages[0] as { display?: boolean }).display, false);
|
|
1728
|
-
assert.equal((recorded.messages[0] as { options?: { triggerTurn?: boolean } }).options?.triggerTurn, true);
|
|
1729
|
-
|
|
1730
|
-
// Pending flag consumed: a second plans result neither compacts nor resumes.
|
|
1731
|
-
await emit("tool_result", { type: "tool_result", toolName: "plans", input: { action: "record-decision" }, isError: false });
|
|
1732
|
-
assert.equal(recorded.compacts?.length, 1);
|
|
1733
|
-
recorded.compacts?.[0]?.onComplete?.({} as never);
|
|
1734
|
-
assert.equal(recorded.messages.length, 1);
|
|
1735
|
-
});
|
|
1736
|
-
|
|
1737
|
-
it("resumes exactly once with an info notice when compaction fails", async () => {
|
|
1738
|
-
const harness = await preplanHarness();
|
|
1739
|
-
const { ctx, recorded } = harness;
|
|
1740
|
-
(ctx as { compact?: unknown }).compact = (options: { onError?: (error: Error) => void }) => {
|
|
1741
|
-
options.onError?.(new Error("Nothing to compact (session too small)"));
|
|
1742
|
-
};
|
|
1743
|
-
markPrePlanCompactPending(ctx, "run-b");
|
|
1744
|
-
await harness.emit("tool_result", { type: "tool_result", toolName: "plans", input: { action: "start-run" }, isError: false });
|
|
1745
|
-
assert.equal(recorded.messages.length, 1, "resume exactly once despite the failure");
|
|
1746
|
-
assert.equal(recorded.messages[0]?.customType, PLANNING_PREPLAN_RESUME_CUSTOM_TYPE);
|
|
1747
|
-
assert.deepEqual(recorded.notifies.at(-1), {
|
|
1748
|
-
message: "pi-plans: pre-plan compaction skipped; continuing planning.",
|
|
1749
|
-
severity: "info",
|
|
1750
|
-
});
|
|
1751
|
-
});
|
|
1752
|
-
|
|
1753
|
-
it("skips silently and still resumes when ctx.compact is unavailable (older Pi)", async () => {
|
|
1754
|
-
const harness = await preplanHarness();
|
|
1755
|
-
(harness.ctx as { compact?: unknown }).compact = undefined;
|
|
1756
|
-
markPrePlanCompactPending(harness.ctx, "run-c");
|
|
1757
|
-
await harness.emit("tool_result", { type: "tool_result", toolName: "plans", input: { action: "start-run" }, isError: false });
|
|
1758
|
-
assert.equal(harness.recorded.messages.length, 1);
|
|
1759
|
-
assert.equal(harness.recorded.messages[0]?.customType, PLANNING_PREPLAN_RESUME_CUSTOM_TYPE);
|
|
1760
|
-
});
|
|
828
|
+
function formatStatus(workdir: string): string {
|
|
829
|
+
const exec = getExecution();
|
|
830
|
+
assert.ok(exec);
|
|
831
|
+
void workdir;
|
|
832
|
+
return exec.stall.paused ? "paused" : "running";
|
|
833
|
+
}
|
|
1761
834
|
|
|
1762
|
-
|
|
1763
|
-
const messages = [
|
|
1764
|
-
{ customType: "user", content: "real prompt" },
|
|
1765
|
-
{ customType: PLANNING_PREPLAN_RESUME_CUSTOM_TYPE, content: "Continue planning." },
|
|
1766
|
-
{ customType: "assistant", content: "ok" },
|
|
1767
|
-
];
|
|
1768
|
-
assert.equal(filterPlanningResumeMessages(messages).length, 2);
|
|
1769
|
-
});
|
|
1770
|
-
});
|
|
835
|
+
import { isDashboardExpanded as toggleEnabled } from "../src/exec.ts";
|