pi-plans 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/AGENTS.md +58 -0
  2. package/CONTRIBUTING.md +8 -15
  3. package/README.md +39 -37
  4. package/agents/execution-reviewer.md +40 -0
  5. package/agents/reviewer.md +12 -3
  6. package/index.ts +55 -58
  7. package/package.json +2 -1
  8. package/references/pi-planning-workflow.md +50 -60
  9. package/references/plan-artifact-template.md +81 -60
  10. package/references/state-and-config.md +60 -44
  11. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  12. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  13. package/scripts/run-tests.ts +12 -1
  14. package/scripts/validate.ts +39 -11
  15. package/skills/debug-and-plan/SKILL.md +3 -3
  16. package/skills/plan-big/SKILL.md +3 -3
  17. package/skills/plan-normal/SKILL.md +3 -3
  18. package/skills/plan-small/SKILL.md +4 -4
  19. package/skills/plan-with-refs/SKILL.md +6 -6
  20. package/skills/planning/SKILL.md +1 -1
  21. package/src/ask-form.ts +4 -4
  22. package/src/auditor.ts +227 -0
  23. package/src/auto-approve.ts +1 -1
  24. package/src/autocomplete.ts +19 -17
  25. package/src/code-graph/commands.ts +8 -3
  26. package/src/code-graph/community.ts +1 -1
  27. package/src/code-graph/paths.ts +1 -1
  28. package/src/code-graph/watch.ts +2 -2
  29. package/src/compaction.ts +3 -3
  30. package/src/config-command.ts +146 -73
  31. package/src/dashboard.ts +303 -0
  32. package/src/exec.ts +1185 -924
  33. package/src/global-state.ts +304 -0
  34. package/src/guard.ts +18 -19
  35. package/src/messaging.ts +44 -0
  36. package/src/plan.ts +421 -112
  37. package/src/query-hook.ts +4 -4
  38. package/src/refine-prompts.ts +12 -70
  39. package/src/refine-ui-helpers.ts +24 -5
  40. package/src/refine-ui-state.ts +1 -1
  41. package/src/refine-ui.ts +19 -3
  42. package/src/resume-command.ts +45 -129
  43. package/src/resume.ts +5 -1
  44. package/src/role-panels.ts +542 -0
  45. package/src/run-context.ts +3 -10
  46. package/src/staleness.ts +53 -0
  47. package/src/state.ts +273 -72
  48. package/src/subagent.ts +19 -29
  49. package/src/task-tool.ts +100 -0
  50. package/src/tasks.ts +223 -0
  51. package/src/thinking-levels.ts +67 -0
  52. package/src/ui-language.ts +7 -54
  53. package/src/workflow-state.ts +76 -58
  54. package/tests/analyze-refs.test.ts +35 -18
  55. package/tests/ask-choice-schema.test.ts +0 -12
  56. package/tests/ask-choice.test.ts +2 -49
  57. package/tests/ask-form-tool.test.ts +4 -5
  58. package/tests/ask-form.test.ts +2 -2
  59. package/tests/auditor.test.ts +210 -0
  60. package/tests/auto-approve.test.ts +7 -10
  61. package/tests/autocomplete.test.ts +8 -11
  62. package/tests/code-graph-apply-action.test.ts +2 -2
  63. package/tests/code-graph-commands.test.ts +2 -2
  64. package/tests/code-graph-index.test.ts +2 -2
  65. package/tests/code-graph-loop.e2e.test.ts +1 -1
  66. package/tests/code-graph-mutations.test.ts +1 -1
  67. package/tests/code-graph-rollback.test.ts +1 -1
  68. package/tests/code-graph-v05.test.ts +2 -2
  69. package/tests/compaction.test.ts +1 -1
  70. package/tests/config-command.test.ts +103 -100
  71. package/tests/dashboard.test.ts +402 -0
  72. package/tests/exec-lifecycle.test.ts +181 -115
  73. package/tests/exec-panel-lifecycle.test.ts +106 -251
  74. package/tests/exec-review-loop.test.ts +331 -0
  75. package/tests/exec.test.ts +771 -1706
  76. package/tests/execute-plan.test.ts +44 -19
  77. package/tests/extension-load.test.ts +48 -0
  78. package/tests/global-state.test.ts +371 -0
  79. package/tests/graph-aware-file-tools.test.ts +5 -5
  80. package/tests/guard.test.ts +1 -1
  81. package/tests/multi-run.test.ts +3 -103
  82. package/tests/plan.test.ts +139 -62
  83. package/tests/plans.test.ts +7 -79
  84. package/tests/refine-prompts.test.ts +20 -71
  85. package/tests/refine-resume.test.ts +27 -22
  86. package/tests/refine-ui.test.ts +6 -15
  87. package/tests/resume-lifecycle.test.ts +41 -22
  88. package/tests/resume.test.ts +39 -81
  89. package/tests/role-panels.test.ts +391 -0
  90. package/tests/run-context.test.ts +1 -1
  91. package/tests/run-ownership.test.ts +1 -1
  92. package/tests/stale-ctx.test.ts +218 -0
  93. package/tests/staleness.test.ts +76 -0
  94. package/tests/state.test.ts +155 -32
  95. package/tests/subagent-thinking.test.ts +65 -0
  96. package/tests/subagent-usage.test.ts +1 -1
  97. package/tests/task-tool.test.ts +61 -0
  98. package/tests/tasks.test.ts +142 -0
  99. package/tests/thinking-levels.test.ts +77 -0
  100. package/tests/ui-language.test.ts +2 -17
  101. package/tests/workflow-state.test.ts +73 -90
  102. package/tools/analyze-refs.ts +67 -32
  103. package/tools/ask-choice.ts +7 -53
  104. package/tools/code-graph.ts +2 -2
  105. package/tools/execute-plan.ts +55 -99
  106. package/tools/graph-aware-file-tools.ts +4 -10
  107. package/tools/plans.ts +41 -67
  108. package/tools/refine.ts +101 -164
  109. package/agents/criticizer.md +0 -18
  110. package/agents/executor.md +0 -26
  111. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  112. package/src/panel.ts +0 -473
  113. package/src/termination-prompt.ts +0 -73
  114. package/tests/goal-wait.test.ts +0 -269
  115. package/tests/panel-i-zero.test.ts +0 -420
  116. package/tests/panel.test.ts +0 -355
package/src/tasks.ts ADDED
@@ -0,0 +1,223 @@
1
+ /**
2
+ * Task-tree runtime model (v0.6.1): the execution phase tracks plan tasks
3
+ * (`## Tasks`) as the unit of progress. Status flows in exclusively through
4
+ * the task status tool; completion of the whole run is gated by the
5
+ * independent execution reviewer over the plan's verification checks.
6
+ */
7
+
8
+ import type { CheckItem, PlanTasks, TaskNode, WaveEntry } from "./plan.ts";
9
+ import { extractTaskCoverage, flattenTasks, resolveTaskWaves } from "./plan.ts";
10
+
11
+ export type TaskStatus = "pending" | "complete" | "skipped";
12
+
13
+ export interface TaskProgress {
14
+ status: TaskStatus;
15
+ /** Evidence recorded with the completion/skip (tool call payload). */
16
+ evidence?: string;
17
+ /** Skip reason, present only for skipped tasks. */
18
+ skipReason?: string;
19
+ }
20
+
21
+ /** Runtime view of one task node: plan metadata merged with live status. */
22
+ export interface TaskView {
23
+ id: string;
24
+ title: string;
25
+ wave: number;
26
+ deps: string[];
27
+ files: string[];
28
+ status: TaskStatus;
29
+ evidence?: string;
30
+ skipReason?: string;
31
+ children: TaskView[];
32
+ }
33
+
34
+ /** Serializable progress map persisted in the run checkpoint. */
35
+ export type TaskProgressMap = Record<string, TaskProgress>;
36
+
37
+ /** Build the runtime task tree from parsed plan tasks + persisted progress. */
38
+ export function buildTaskView(planTasks: PlanTasks, progress?: TaskProgressMap): TaskView[] {
39
+ const waves = resolveTaskWaves(planTasks);
40
+ const build = (node: TaskNode): TaskView => {
41
+ const state = progress?.[node.id];
42
+ const children = node.children.map(build);
43
+ return {
44
+ id: node.id,
45
+ title: node.title,
46
+ wave: waves.get(node.id) ?? 1,
47
+ deps: node.deps,
48
+ files: node.files,
49
+ status: state?.status ?? "pending",
50
+ evidence: state?.evidence,
51
+ skipReason: state?.skipReason,
52
+ children,
53
+ };
54
+ };
55
+ return planTasks.tasks.map(build);
56
+ }
57
+
58
+ export function flattenTaskViews(tasks: TaskView[]): TaskView[] {
59
+ const out: TaskView[] = [];
60
+ for (const t of tasks) {
61
+ out.push(t);
62
+ out.push(...flattenTaskViews(t.children));
63
+ }
64
+ return out;
65
+ }
66
+
67
+ /** A task counts as terminal when it is skipped, or complete together with
68
+ * every child (parents close last; an open child reopens the parent for
69
+ * scheduling purposes even if the parent itself reported complete). */
70
+ export function taskIsTerminal(task: TaskView): boolean {
71
+ if (task.status === "skipped") return true;
72
+ if (task.status !== "complete") return false;
73
+ return task.children.every((child) => taskIsTerminal(child));
74
+ }
75
+
76
+ export function allTasksTerminal(tasks: TaskView[]): boolean {
77
+ return flattenTaskViews(tasks).length > 0 && flattenTaskViews(tasks).every((task) => taskIsTerminal(task));
78
+ }
79
+
80
+ /** Snapshot of statuses for persistence and stall detection. Every task gets
81
+ * a record: a rolled-back task must stay visible with the evidence from its
82
+ * previous attempt, otherwise a rollback erases the run's history and the
83
+ * agent can no longer tell "done and rolled back" from "never started". */
84
+ export function taskProgressMap(tasks: TaskView[]): TaskProgressMap {
85
+ const map: TaskProgressMap = {};
86
+ for (const task of flattenTaskViews(tasks)) {
87
+ map[task.id] = {
88
+ status: task.status,
89
+ evidence: task.evidence,
90
+ skipReason: task.skipReason,
91
+ };
92
+ }
93
+ return map;
94
+ }
95
+
96
+ /** Progress aggregation for the dashboard: done counts terminal tasks. */
97
+ export function taskProgress(tasks: TaskView[]): { done: number; total: number } {
98
+ const flat = flattenTaskViews(tasks);
99
+ return {
100
+ done: flat.filter((task) => taskIsTerminal(task)).length,
101
+ total: flat.length,
102
+ };
103
+ }
104
+
105
+ /** The current task: the first non-terminal task in wave order, then
106
+ * document order (top-level-first; subtasks close before their parent is
107
+ * re-listed). Deterministic; used for the ▸ anchor in the dashboard and
108
+ * the per-turn injection. */
109
+ export function currentTask(tasks: TaskView[]): TaskView | null {
110
+ const flat = flattenTaskViews(tasks);
111
+ const open = flat.filter((task) => !taskIsTerminal(task));
112
+ if (open.length === 0) return null;
113
+ open.sort((a, b) => (a.wave - b.wave) || (flat.indexOf(a) - flat.indexOf(b)));
114
+ return open[0] ?? null;
115
+ }
116
+
117
+ /** Tasks of one wave (top-level only — children belong to their parent). */
118
+ export function waveTasks(tasks: TaskView[], wave: number): TaskView[] {
119
+ return tasks.filter((task) => task.wave === wave);
120
+ }
121
+
122
+ export function maxWave(tasks: TaskView[]): number {
123
+ return flattenTaskViews(tasks).reduce((max, task) => Math.max(max, task.wave), 1);
124
+ }
125
+
126
+ /** Legal status transitions for the task status tool: pending may close
127
+ * (complete/skipped); closed states are immutable except through the
128
+ * completion-audit flow (never through the task tool). */
129
+ export function canTransition(task: TaskView, next: TaskStatus): boolean {
130
+ if (task.status === next) return false;
131
+ if (task.status === "pending") return next === "complete" || next === "skipped";
132
+ return false;
133
+ }
134
+
135
+ /** Rollback set for a failed verification check: every task in its covers
136
+ * clause (parents cascade to their children, skipped tasks reopen too).
137
+ * Returns the ids that actually reopen.
138
+ *
139
+ * The task's `evidence` is deliberately kept. It records what the previous
140
+ * attempt actually did, which is the one thing the agent cannot reconstruct
141
+ * once it reopens the task; re-reporting overwrites it. `skipReason` is
142
+ * cleared because it described a deliberate skip that the audit has now
143
+ * overturned. */
144
+ export function auditRollbackSet(
145
+ tasks: TaskView[],
146
+ checklist: CheckItem[],
147
+ vcId: string,
148
+ ): string[] {
149
+ const vc = checklist.find((item) => item.id === vcId);
150
+ if (!vc) return [];
151
+ const covered = new Set(extractTaskCoverage(vc.text));
152
+ const flat = flattenTaskViews(tasks);
153
+ const reopen: string[] = [];
154
+ const reopenNode = (node: TaskView): void => {
155
+ if (node.status !== "pending") {
156
+ node.status = "pending";
157
+ node.skipReason = undefined;
158
+ reopen.push(node.id);
159
+ }
160
+ for (const child of node.children) reopenNode(child);
161
+ };
162
+ for (const node of flat) {
163
+ if (covered.has(node.id)) reopenNode(node);
164
+ }
165
+ return reopen;
166
+ }
167
+
168
+ /** Checks that lose their satisfied state because a rollback reopened work
169
+ * they were verifying. Returns the ids whose `done` flag was cleared.
170
+ *
171
+ * A check can be presolved without an audit round (skipped-pass), or affirmed
172
+ * in an earlier round, and stay `done` while another check's rollback reopens
173
+ * one of its covered tasks. Left alone it renders as "check passed" next to a
174
+ * task that is open again with stale evidence. The caller owns this: it runs
175
+ * after the rollback set is known. */
176
+ export function invalidateChecksForRolledBackTasks(
177
+ checklist: CheckItem[],
178
+ tasks: TaskView[],
179
+ reopenedIds: string[],
180
+ ): string[] {
181
+ if (reopenedIds.length === 0) return [];
182
+ const reopened = new Set(reopenedIds);
183
+ const cleared: string[] = [];
184
+ for (const item of checklist) {
185
+ if (!item.done) continue;
186
+ const covered = extractTaskCoverage(item.text);
187
+ if (covered.some((id) => reopened.has(id))) {
188
+ item.done = false;
189
+ cleared.push(item.id);
190
+ }
191
+ }
192
+ return cleared;
193
+ }
194
+
195
+ /** Verification checks that cover no task are excluded from audit (their
196
+ * pass state cannot be derived from task statuses). */
197
+ export function auditableChecks(checklist: CheckItem[], tasks: TaskView[]): CheckItem[] {
198
+ const known = new Set(flattenTaskViews(tasks).map((task) => task.id));
199
+ return checklist.filter((item) => {
200
+ const covered = extractTaskCoverage(item.text);
201
+ return covered.length > 0 && covered.some((id) => known.has(id));
202
+ });
203
+ }
204
+
205
+ /** Checks with at least one skipped covered task whose remaining covered
206
+ * tasks are all complete (I-007: a skip whose siblings carry the delivery
207
+ * passes without an audit round). Pure-complete and pure-pending coverages
208
+ * are NOT presolved here — the former the audit affirms, the latter cannot
209
+ * arise once all tasks are terminal. */
210
+ export function skippedPassCheckIds(checklist: CheckItem[], tasks: TaskView[]): string[] {
211
+ const byId = new Map(flattenTaskViews(tasks).map((task) => [task.id, task]));
212
+ return auditableChecks(checklist, tasks)
213
+ .filter((item) => {
214
+ const covered = extractTaskCoverage(item.text).filter((id) => byId.has(id));
215
+ if (covered.length === 0) return false;
216
+ const statuses = covered.map((id) => byId.get(id)!.status);
217
+ return statuses.includes("skipped")
218
+ && statuses.every((status) => status === "skipped" || status === "complete");
219
+ })
220
+ .map((item) => item.id);
221
+ }
222
+
223
+ export type { PlanTasks, TaskNode, WaveEntry };
@@ -0,0 +1,67 @@
1
+ /**
2
+ * Thin thinking-level adapter over pi-ai's exported helpers.
3
+ *
4
+ * pi-ai already owns the level domain rules (`getSupportedThinkingLevels`:
5
+ * non-reasoning models only support "off"; a null mapping disables a level;
6
+ * xhigh/max appear only when explicitly mapped). Re-implementing them here
7
+ * would drift — this module only adds the pi-plans "default" sentinel and
8
+ * the stored-level resolution used by the reviewer spawn path (F-012).
9
+ *
10
+ * Semantics (decision 4 / F-009): `thinking_level: null` in the global
11
+ * reviewer config means DEFAULT — the child pi gets NO --thinking flag and
12
+ * resolves its own default chain (per-model settings → defaultThinkingLevel
13
+ * → medium, then model clamping). That is deliberately distinct from the
14
+ * explicit "off" level.
15
+ */
16
+
17
+ import { getSupportedThinkingLevels, type Model, type ModelThinkingLevel } from "@earendil-works/pi-ai";
18
+ import type { GlobalRoleConfig, ThinkingLevelValue } from "./global-state.ts";
19
+
20
+ /** Panel/label sentinel for `thinking_level: null` (omit --thinking). */
21
+ export const DEFAULT_LEVEL_SENTINEL = "default";
22
+
23
+ /** Levels a model actually supports, via pi-ai (single source of truth). */
24
+ export function levelsForModel(model: Pick<Model<any>, "reasoning" | "thinkingLevelMap">): ModelThinkingLevel[] {
25
+ return getSupportedThinkingLevels(model);
26
+ }
27
+
28
+ /** True when the stored level is supported by the model (spawn passes it
29
+ * through verbatim; unsupported stored levels are clamped by the child). */
30
+ export function isLevelSupported(
31
+ model: Pick<Model<any>, "reasoning" | "thinkingLevelMap">,
32
+ level: string | null,
33
+ ): boolean {
34
+ if (level === null) return true; // default: no flag, always valid
35
+ return levelsForModel(model).includes(level as ModelThinkingLevel);
36
+ }
37
+
38
+ /** Resolve the CLI argument for the stored level: null → omit the flag. */
39
+ export function spawnThinkingFlag(level: ThinkingLevelValue | null): string | null {
40
+ return level ?? null;
41
+ }
42
+
43
+ /** Display label for the reviewer overlay and subagents ledger. */
44
+ export function roleModelLabel(modelSelector: string, thinkingLevel: ThinkingLevelValue | null): string {
45
+ return `${modelSelector}:${thinkingLevel ?? DEFAULT_LEVEL_SENTINEL}`;
46
+ }
47
+
48
+ /** Human-facing one-liner for the default row / docs (F-009): the default
49
+ * is the child pi's own chain, NOT the current session's level. */
50
+ export const DEFAULT_LEVEL_DESCRIPTION =
51
+ "no --thinking flag: the child pi resolves its default (per-model settings → defaultThinkingLevel → medium)";
52
+
53
+ /** Spawn-view of a confirmed reviewer role: concrete model + optional level. */
54
+ export interface ResolvedReviewerSpawn {
55
+ modelSelector: string;
56
+ thinkingLevel: ThinkingLevelValue | null;
57
+ label: string;
58
+ }
59
+
60
+ export function resolveReviewerSpawn(role: Pick<GlobalRoleConfig, "model_selector" | "thinking_level">): ResolvedReviewerSpawn {
61
+ const modelSelector = role.model_selector ?? "";
62
+ return {
63
+ modelSelector,
64
+ thinkingLevel: role.thinking_level ?? null,
65
+ label: modelSelector ? roleModelLabel(modelSelector, role.thinking_level ?? null) : "unconfirmed",
66
+ };
67
+ }
@@ -1,10 +1,9 @@
1
1
  /**
2
2
  * Single source of truth for user-visible UI chrome language (issue #3).
3
3
  *
4
- * Workspace `language.tag` (`.git/pi_plans/config.json`) selects the chrome
5
- * strings for every user-visible surface: the batch form (src/ask-form.ts),
6
- * the refine/refs overlay footer (src/refine-ui.ts), the status panel
7
- * (src/panel.ts) and the execution status line (src/exec.ts).
4
+ * Workspace `language.tag` (`.git/pi-plans/config.json`) selects the chrome
5
+ * strings for every user-visible surface: the batch form (src/ask-form.ts)
6
+ * and the refine/refs overlay footer (src/refine-ui.ts).
8
7
  *
9
8
  * Mapping follows RFC 4647 primary-subtag fallback (zh-Hant-CN → zh-Hant →
10
9
  * zh), so every `zh*` tag renders the existing Simplified strings verbatim.
@@ -104,6 +103,8 @@ export interface RefineChrome {
104
103
  scroll: string;
105
104
  page: string;
106
105
  switchLane: string;
106
+ /** Auditor overlay only (v0.8): the shortcut that reopens the in-flight round. */
107
+ reopen: string;
107
108
  }
108
109
 
109
110
  const REFINE_CHROME: Record<UiLanguage, RefineChrome> = {
@@ -112,12 +113,14 @@ const REFINE_CHROME: Record<UiLanguage, RefineChrome> = {
112
113
  scroll: "↑/↓ 滚动",
113
114
  page: "PgUp/PgDn 翻页",
114
115
  switchLane: "Tab & Shift + Tab 切换 lane",
116
+ reopen: "Ctrl+Shift+R 重开评审面板",
115
117
  },
116
118
  en: {
117
119
  close: "Esc close",
118
120
  scroll: "↑/↓ scroll",
119
121
  page: "PgUp/PgDn page",
120
122
  switchLane: "Tab & Shift + Tab switch lane",
123
+ reopen: "Ctrl+Shift+R reopen review overlay",
121
124
  },
122
125
  };
123
126
 
@@ -125,54 +128,4 @@ export function refineChrome(lang: UiLanguage): RefineChrome {
125
128
  return REFINE_CHROME[lang];
126
129
  }
127
130
 
128
- // ---------------------------------------------------------------------------
129
- // Status panel chrome (src/panel.ts) — implWarning branch only.
130
- // ---------------------------------------------------------------------------
131
-
132
- export interface PanelChrome {
133
- badgeStatus(topic: string, vcDone: number, vcTotal: number): string;
134
- progressWarning(): string;
135
- summaryWarning(topic: string, vcDone: number, vcTotal: number, nextAction: string): string;
136
- }
137
-
138
- const PANEL_CHROME: Record<UiLanguage, PanelChrome> = {
139
- zh: {
140
- badgeStatus: (topic, vcDone, vcTotal) => `${topic} · ⚠ I 解析 0 项 · VC ${vcDone}/${vcTotal}`,
141
- progressWarning: () => `⚠ plan 格式:Implementation Items 解析 0 项(面板无法计 I 进度)`,
142
- summaryWarning: (topic, vcDone, vcTotal, nextAction) =>
143
- `plans: ${topic} ▸ ⚠ Implementation Items 解析 0 项 · VC ${vcDone}/${vcTotal} · next: ${nextAction}`,
144
- },
145
- en: {
146
- badgeStatus: (topic, vcDone, vcTotal) => `${topic} · ⚠ I parse 0 items · VC ${vcDone}/${vcTotal}`,
147
- progressWarning: () => `⚠ plan format: Implementation Items parsed 0 items (I progress cannot be counted)`,
148
- summaryWarning: (topic, vcDone, vcTotal, nextAction) =>
149
- `plans: ${topic} ▸ ⚠ Implementation Items parsed 0 items · VC ${vcDone}/${vcTotal} · next: ${nextAction}`,
150
- },
151
- };
152
-
153
- export function panelChrome(lang: UiLanguage): PanelChrome {
154
- return PANEL_CHROME[lang];
155
- }
156
-
157
- // ---------------------------------------------------------------------------
158
- // Execution status line chrome (src/exec.ts) — goal-wait segment.
159
- // ---------------------------------------------------------------------------
160
-
161
- export interface ExecChrome {
162
- goalWait(noProgressRounds: number, waitRounds: number): string;
163
- }
164
-
165
- const EXEC_CHROME: Record<UiLanguage, ExecChrome> = {
166
- zh: {
167
- goalWait: (noProgressRounds, waitRounds) =>
168
- ` · 🔁 goal-wait · 无进展 ${noProgressRounds}/3 · 等待 ${waitRounds}/6`,
169
- },
170
- en: {
171
- goalWait: (noProgressRounds, waitRounds) =>
172
- ` · 🔁 goal-wait · no progress ${noProgressRounds}/3 · waiting ${waitRounds}/6`,
173
- },
174
- };
175
131
 
176
- export function execChrome(lang: UiLanguage): ExecChrome {
177
- return EXEC_CHROME[lang];
178
- }
@@ -145,8 +145,19 @@ export interface ExecutionCheckpoint {
145
145
  /** True when this approval/progress was produced in a different (origin) worktree. */
146
146
  originWorktree?: string;
147
147
  /** v0.6.0: set while a delegated executor child owns the implementation;
148
- * stale after a restart (orphaned delegate — the child died with the parent). */
148
+ * stale after a restart (orphaned delegate — the child died with the parent).
149
+ * Removed with delegated execution in v0.6.1; read-tolerated on legacy checkpoints. */
149
150
  delegate?: { modelSelector: string; startedAt: string };
151
+ /** v0.6.1: task-tree progress (task id → status/evidence), the primary
152
+ * progress record. doneVcIds stays for legacy checkpoints and the final
153
+ * audit pass. */
154
+ tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
155
+ /** v0.7.1: watchdog round counter. Persisted so a session restart cannot
156
+ * silently hand a stalled run a fresh budget; optional so checkpoints
157
+ * written before this field keep loading. */
158
+ stallRounds?: number;
159
+ /** v0.6.1: completion-audit bookkeeping (rounds, last failed set, pass). */
160
+ audit?: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] };
150
161
  }
151
162
 
152
163
  export interface OwnerInfo {
@@ -454,7 +465,7 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
454
465
  const record = asRecord(value, label);
455
466
  rejectExtraKeys(
456
467
  record,
457
- new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate"]),
468
+ new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "stallRounds", "audit"]),
458
469
  label,
459
470
  );
460
471
  const execution: ExecutionCheckpoint = {
@@ -484,6 +495,36 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
484
495
  startedAt: asString(delegate.startedAt, `${label}.delegate.startedAt`),
485
496
  };
486
497
  }
498
+ if (record.tasks !== undefined && record.tasks !== null) {
499
+ const tasksRecord = asRecord(record.tasks, `${label}.tasks`);
500
+ const tasks: NonNullable<ExecutionCheckpoint["tasks"]> = {};
501
+ for (const [id, raw] of Object.entries(tasksRecord)) {
502
+ const entry = asRecord(raw, `${label}.tasks.${id}`);
503
+ rejectExtraKeys(entry, new Set(["status", "evidence", "skipReason"]), `${label}.tasks.${id}`);
504
+ const status = asEnum(entry.status, new Set(["pending", "complete", "skipped"]), `${label}.tasks.${id}.status`);
505
+ const item: { status: string; evidence?: string; skipReason?: string } = { status };
506
+ if (entry.evidence !== undefined) item.evidence = asString(entry.evidence, `${label}.tasks.${id}.evidence`);
507
+ if (entry.skipReason !== undefined) item.skipReason = asString(entry.skipReason, `${label}.tasks.${id}.skipReason`);
508
+ tasks[id] = item;
509
+ }
510
+ execution.tasks = tasks;
511
+ }
512
+ if (record.stallRounds !== undefined && record.stallRounds !== null) {
513
+ execution.stallRounds = asInt(record.stallRounds, `${label}.stallRounds`, 0);
514
+ }
515
+ if (record.audit !== undefined && record.audit !== null) {
516
+ const audit = asRecord(record.audit, `${label}.audit`);
517
+ rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed", "undeterminable"]), `${label}.audit`);
518
+ const parsed: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] } = {
519
+ rounds: asInt(audit.rounds, `${label}.audit.rounds`, 0),
520
+ };
521
+ if (audit.lastResult !== undefined) parsed.lastResult = asString(audit.lastResult, `${label}.audit.lastResult`);
522
+ if (audit.passed !== undefined) parsed.passed = asBool(audit.passed, `${label}.audit.passed`);
523
+ if (audit.undeterminable !== undefined) {
524
+ parsed.undeterminable = asStringArray(audit.undeterminable, `${label}.audit.undeterminable`);
525
+ }
526
+ execution.audit = parsed;
527
+ }
487
528
  return execution;
488
529
  }
489
530
 
@@ -1000,19 +1041,6 @@ export function applyReviewConsolidated(cp: WorkflowCheckpoint, roundId: string,
1000
1041
  return { ...cp, reviewRounds: cp.reviewRounds.map((entry) => (entry.roundId === roundId ? nextRound : entry)) };
1001
1042
  }
1002
1043
 
1003
- /** F-004: `completed` requires explicit evidence; approval cannot be forged by state writes. */
1004
- export function applyCompleted(cp: WorkflowCheckpoint, evidence: string): WorkflowCheckpoint {
1005
- if (cp.phase !== "implementation-review") {
1006
- throw new StateError(`cannot complete from phase "${cp.phase}"`);
1007
- }
1008
- const review = cp.implementationReview;
1009
- if (!review || review.terminationCondition === undefined) {
1010
- throw new StateError("cannot complete without a recorded termination condition");
1011
- }
1012
- if (evidence.trim() === "") throw new StateError("completion requires non-empty evidence");
1013
- return { ...cp, phase: "completed", nextAction: "none" };
1014
- }
1015
-
1016
1044
  export function applyExecutionApproved(cp: WorkflowCheckpoint, approval: ExecutionApproval): WorkflowCheckpoint {
1017
1045
  // F-007 (implementation review): terminal and review phases cannot approve
1018
1046
  // execution; a re-approval (stop/migration reset approval to null) is legal
@@ -1036,6 +1064,8 @@ export function applyExecutionApproved(cp: WorkflowCheckpoint, approval: Executi
1036
1064
  doneVcIds: [],
1037
1065
  implStatus: {},
1038
1066
  usage: { inToks: 0, outToks: 0 },
1067
+ tasks: {},
1068
+ audit: { rounds: 0 },
1039
1069
  },
1040
1070
  };
1041
1071
  }
@@ -1048,6 +1078,12 @@ export interface ExecutionProgressInput {
1048
1078
  pausedReason?: string | null;
1049
1079
  /** v0.6.0: set/clear the delegated-executor record; null clears it. */
1050
1080
  delegate?: { modelSelector: string; startedAt: string } | null;
1081
+ /** v0.6.1: task-tree progress snapshot (authoritative). */
1082
+ tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
1083
+ /** v0.6.1: completion-audit bookkeeping update. */
1084
+ audit?: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] };
1085
+ /** v0.7.1: watchdog budget counter, so a restart cannot refresh it. */
1086
+ stallRounds?: number;
1051
1087
  }
1052
1088
 
1053
1089
  export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: ExecutionProgressInput): WorkflowCheckpoint {
@@ -1066,6 +1102,9 @@ export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: Executi
1066
1102
  else if (progress.pausedReason !== undefined) execution.pausedReason = progress.pausedReason;
1067
1103
  if (progress.delegate === null) delete execution.delegate;
1068
1104
  else if (progress.delegate !== undefined) execution.delegate = progress.delegate;
1105
+ if (progress.tasks !== undefined) execution.tasks = progress.tasks;
1106
+ if (progress.stallRounds !== undefined) execution.stallRounds = progress.stallRounds;
1107
+ if (progress.audit !== undefined) execution.audit = progress.audit;
1069
1108
  return { ...cp, execution };
1070
1109
  }
1071
1110
 
@@ -1077,45 +1116,18 @@ export function applyExecutionHeadChanged(cp: WorkflowCheckpoint): WorkflowCheck
1077
1116
 
1078
1117
  export function applyExecutionCompleted(cp: WorkflowCheckpoint): WorkflowCheckpoint {
1079
1118
  if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
1119
+ // v0.6.1 (D-018): the post-execution amelioration loop is gone; a
1120
+ // completed audit passes the run straight to the terminal phase.
1080
1121
  return {
1081
1122
  ...cp,
1082
- phase: "implementation-review",
1083
- nextAction: "ask-question",
1084
- execution: { ...cp.execution, pausedReason: undefined },
1085
- };
1086
- }
1087
-
1088
- export function applyImplementationReviewConfigured(
1089
- cp: WorkflowCheckpoint,
1090
- terminationCondition: string,
1091
- reviewerCount?: number,
1092
- ): WorkflowCheckpoint {
1093
- if (cp.phase !== "implementation-review") throw new StateError("requires phase \"implementation-review\"");
1094
- if (cp.implementationReview?.terminationCondition !== undefined) {
1095
- throw new StateError("termination condition already configured; do not re-ask");
1096
- }
1097
- return {
1098
- ...cp,
1099
- implementationReview: {
1100
- terminationCondition,
1101
- reviewerCount,
1102
- completedRounds: cp.implementationReview?.completedRounds ?? 0,
1123
+ phase: "completed",
1124
+ nextAction: "none",
1125
+ execution: {
1126
+ ...cp.execution,
1127
+ pausedReason: undefined,
1128
+ delegate: undefined,
1129
+ audit: { ...(cp.execution.audit ?? { rounds: 0 }), passed: true },
1103
1130
  },
1104
- nextAction: "run-review",
1105
- };
1106
- }
1107
-
1108
- export function applyImplementationRoundFinished(cp: WorkflowCheckpoint): WorkflowCheckpoint {
1109
- if (cp.phase !== "implementation-review" || !cp.implementationReview) {
1110
- throw new StateError("requires phase \"implementation-review\"");
1111
- }
1112
- const review = cp.implementationReview;
1113
- if (review.currentRoundId === undefined) throw new StateError("no current round to finish");
1114
- const round = cp.reviewRounds.find((entry) => entry.roundId === review.currentRoundId);
1115
- if (!round || !round.consolidated) throw new StateError("current round is not consolidated");
1116
- return {
1117
- ...cp,
1118
- implementationReview: { ...review, completedRounds: review.completedRounds + 1, currentRoundId: undefined },
1119
1131
  };
1120
1132
  }
1121
1133
 
@@ -1135,12 +1147,16 @@ export function applyMigration(
1135
1147
  migration: { fromWorktree: cp.worktreeRoot, migratedAt: utcNow() },
1136
1148
  execution: cp.execution
1137
1149
  ? {
1138
- approval: null,
1139
- doneVcIds: [],
1140
- implStatus: {},
1141
- usage: cp.execution.usage,
1142
- originWorktree: cp.execution.originWorktree ?? cp.worktreeRoot,
1143
- }
1150
+ approval: null,
1151
+ doneVcIds: [],
1152
+ implStatus: {},
1153
+ usage: cp.execution.usage,
1154
+ originWorktree: cp.execution.originWorktree ?? cp.worktreeRoot,
1155
+ // v0.6.1: task progress and audit state do not survive a worktree
1156
+ // migration (same rule as VC validity).
1157
+ tasks: {},
1158
+ audit: { rounds: 0 },
1159
+ }
1144
1160
  : undefined,
1145
1161
  implementationReview: cp.implementationReview
1146
1162
  ? {
@@ -1164,7 +1180,9 @@ function migrationNextAction(cp: WorkflowCheckpoint): NextAction {
1164
1180
  export function applyExecutionStopped(cp: WorkflowCheckpoint, reason: string): WorkflowCheckpoint {
1165
1181
  if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
1166
1182
  const { delegate: _delegate, ...execution } = cp.execution;
1167
- return { ...cp, execution: { ...execution, pausedReason: reason } };
1183
+ // A stop also revokes any outstanding audit-rollback authorization.
1184
+ const audit = execution.audit ? { ...execution.audit } : undefined;
1185
+ return { ...cp, execution: { ...execution, audit, pausedReason: reason } };
1168
1186
  }
1169
1187
 
1170
1188
  // ---------------------------------------------------------------------------