pi-plans 0.5.7 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/CONTRIBUTING.md +126 -0
  2. package/README.md +49 -39
  3. package/agents/ref-analyst.md +7 -4
  4. package/agents/reviewer.md +12 -3
  5. package/index.ts +74 -40
  6. package/package.json +2 -1
  7. package/references/pi-planning-workflow.md +45 -58
  8. package/references/plan-artifact-template.md +71 -60
  9. package/references/state-and-config.md +63 -47
  10. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  11. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  12. package/scripts/run-tests.ts +12 -1
  13. package/scripts/validate.ts +22 -10
  14. package/skills/debug-and-plan/SKILL.md +4 -4
  15. package/skills/plan-big/SKILL.md +5 -5
  16. package/skills/plan-normal/SKILL.md +5 -5
  17. package/skills/plan-small/SKILL.md +5 -5
  18. package/skills/plan-with-refs/SKILL.md +8 -8
  19. package/skills/planning/SKILL.md +1 -1
  20. package/src/ask-form.ts +4 -4
  21. package/src/auditor.ts +126 -0
  22. package/src/auto-approve.ts +1 -1
  23. package/src/autocomplete.ts +19 -17
  24. package/src/code-graph/commands.ts +2 -2
  25. package/src/code-graph/community.ts +1 -1
  26. package/src/code-graph/paths.ts +1 -1
  27. package/src/code-graph/watch.ts +2 -2
  28. package/src/compaction.ts +3 -3
  29. package/src/config-command.ts +154 -76
  30. package/src/dashboard.ts +257 -0
  31. package/src/exec.ts +709 -705
  32. package/src/global-state.ts +304 -0
  33. package/src/guard.ts +16 -3
  34. package/src/messaging.ts +44 -0
  35. package/src/plan.ts +421 -112
  36. package/src/query-hook.ts +4 -4
  37. package/src/refine-prompts.ts +14 -72
  38. package/src/refine-ui-helpers.ts +24 -5
  39. package/src/refine-ui-state.ts +1 -1
  40. package/src/refine-ui.ts +1 -1
  41. package/src/resume-command.ts +40 -130
  42. package/src/resume.ts +15 -17
  43. package/src/role-panels.ts +542 -0
  44. package/src/run-context.ts +5 -4
  45. package/src/run-picker.ts +98 -0
  46. package/src/state.ts +380 -77
  47. package/src/subagent.ts +32 -1
  48. package/src/task-tool.ts +100 -0
  49. package/src/tasks.ts +189 -0
  50. package/src/thinking-levels.ts +67 -0
  51. package/src/ui-language.ts +3 -54
  52. package/src/workflow-state.ts +78 -57
  53. package/tests/analyze-refs.test.ts +35 -18
  54. package/tests/ask-choice-pros-cons.test.ts +147 -0
  55. package/tests/ask-choice-schema.test.ts +0 -12
  56. package/tests/ask-choice.test.ts +2 -49
  57. package/tests/ask-form-tool.test.ts +4 -5
  58. package/tests/ask-form.test.ts +2 -2
  59. package/tests/auditor.test.ts +111 -0
  60. package/tests/auto-approve.test.ts +7 -10
  61. package/tests/autocomplete.test.ts +8 -11
  62. package/tests/code-graph-apply-action.test.ts +2 -2
  63. package/tests/code-graph-commands.test.ts +2 -2
  64. package/tests/code-graph-index.test.ts +2 -2
  65. package/tests/code-graph-loop.e2e.test.ts +1 -1
  66. package/tests/code-graph-mutations.test.ts +1 -1
  67. package/tests/code-graph-rollback.test.ts +1 -1
  68. package/tests/code-graph-v05.test.ts +2 -2
  69. package/tests/compaction.test.ts +1 -1
  70. package/tests/config-command.test.ts +103 -100
  71. package/tests/dashboard.test.ts +268 -0
  72. package/tests/exec-lifecycle.test.ts +181 -115
  73. package/tests/exec-panel-lifecycle.test.ts +106 -251
  74. package/tests/exec.test.ts +617 -1706
  75. package/tests/execute-plan.test.ts +44 -19
  76. package/tests/extension-load.test.ts +48 -0
  77. package/tests/global-state.test.ts +371 -0
  78. package/tests/graph-aware-file-tools.test.ts +5 -5
  79. package/tests/guard.test.ts +1 -1
  80. package/tests/multi-run.test.ts +184 -0
  81. package/tests/plan.test.ts +139 -62
  82. package/tests/plans.test.ts +7 -79
  83. package/tests/refine-prompts.test.ts +20 -71
  84. package/tests/refine-resume.test.ts +27 -22
  85. package/tests/refine-ui.test.ts +6 -15
  86. package/tests/resume-lifecycle.test.ts +37 -22
  87. package/tests/resume.test.ts +43 -88
  88. package/tests/role-panels.test.ts +391 -0
  89. package/tests/run-context.test.ts +1 -1
  90. package/tests/run-ownership.test.ts +1 -1
  91. package/tests/stale-ctx.test.ts +218 -0
  92. package/tests/state.test.ts +151 -32
  93. package/tests/subagent-thinking.test.ts +65 -0
  94. package/tests/subagent-usage.test.ts +1 -1
  95. package/tests/task-tool.test.ts +61 -0
  96. package/tests/thinking-levels.test.ts +77 -0
  97. package/tests/ui-language.test.ts +2 -17
  98. package/tests/workflow-state.test.ts +17 -99
  99. package/tools/analyze-refs.ts +67 -32
  100. package/tools/ask-choice.ts +19 -49
  101. package/tools/code-graph.ts +2 -2
  102. package/tools/execute-plan.ts +63 -33
  103. package/tools/graph-aware-file-tools.ts +6 -4
  104. package/tools/plans.ts +40 -66
  105. package/tools/refine.ts +101 -164
  106. package/agents/criticizer.md +0 -18
  107. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  108. package/src/panel.ts +0 -473
  109. package/src/termination-prompt.ts +0 -73
  110. package/tests/goal-wait.test.ts +0 -269
  111. package/tests/panel-i-zero.test.ts +0 -420
  112. package/tests/panel.test.ts +0 -355
@@ -0,0 +1,100 @@
1
+ /**
2
+ * `plans_update_task` tool (v0.6.1): the single channel for execution-phase
3
+ * task status reporting. Each call validates the task exists and the
4
+ * transition is legal (pending → complete | skipped), persists progress to
5
+ * the run checkpoint, and refreshes the dashboard. Closed statuses are
6
+ * immutable here — rollbacks happen exclusively inside the completion-audit
7
+ * flow in src/exec.ts (runAuditFlow), never through this tool.
8
+ */
9
+
10
+ import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
11
+ import { Text } from "@earendil-works/pi-tui";
12
+ import { StringEnum } from "@earendil-works/pi-ai";
13
+ import { Type } from "typebox";
14
+ import { getExecution, persistTaskProgress, updateStatusWidget } from "./exec.ts";
15
+ import { canTransition, flattenTaskViews } from "./tasks.ts";
16
+ import { StateError } from "./state.ts";
17
+
18
+ const UpdateTaskParams = Type.Object({
19
+ taskId: Type.String({ description: "Task id from the plan (e.g. Task-3, Task-3.1)" }),
20
+ status: StringEnum(["complete", "skipped"] as const, { description: "New status for the task (pending is audit-only)" }),
21
+ evidence: Type.Optional(
22
+ Type.String({ description: "Evidence backing the completion (test command output, file paths, command result)" }),
23
+ ),
24
+ skipReason: Type.Optional(Type.String({ description: "Why the task is skipped (required for status=skipped)" })),
25
+ workdir: Type.Optional(Type.String({ description: "Target workspace; default current working directory" })),
26
+ });
27
+
28
+ export interface UpdateTaskResult {
29
+ ok: boolean;
30
+ taskId: string;
31
+ status: "complete" | "skipped";
32
+ message: string;
33
+ }
34
+
35
+ /** Core transition logic, exported for tests. */
36
+ export function applyTaskUpdate(
37
+ tasks: ReturnType<typeof flattenTaskViews> extends never ? never : import("./tasks.ts").TaskView[],
38
+ taskId: string,
39
+ status: "complete" | "skipped",
40
+ evidence?: string,
41
+ skipReason?: string,
42
+ ): { ok: boolean; message: string } {
43
+ const task = flattenTaskViews(tasks).find((candidate) => candidate.id === taskId);
44
+ if (!task) {
45
+ return { ok: false, message: `unknown task: ${taskId} (not in the plan's task tree)` };
46
+ }
47
+ if (status === "skipped" && !skipReason) {
48
+ return { ok: false, message: `skipping ${taskId} requires a skipReason` };
49
+ }
50
+ if (!canTransition(task, status)) {
51
+ return {
52
+ ok: false,
53
+ message: `${taskId} is already ${task.status}; closed tasks are immutable outside the audit rollback channel`,
54
+ };
55
+ }
56
+ task.status = status;
57
+ task.evidence = evidence;
58
+ task.skipReason = status === "skipped" ? skipReason : undefined;
59
+ return { ok: true, message: `${taskId} → ${status}` };
60
+ }
61
+
62
+ export function registerTaskStatusTool(ext: ExtensionAPI): void {
63
+ ext.registerTool({
64
+ name: "plans_update_task",
65
+ label: "Update task",
66
+ description:
67
+ 'Report execution progress for one task of the accepted plan: set status "complete" (with evidence) or "skipped" (with skipReason). Fails outside pi-plans execution mode. Statuses are immutable once set — the independent completion auditor handles any rollback.',
68
+ promptSnippet: "Report plan task completion",
69
+ parameters: UpdateTaskParams,
70
+
71
+ async execute(_toolCallId, params, _signal, _onUpdate, ctx: ExtensionContext) {
72
+ const execution = getExecution();
73
+ if (!execution) {
74
+ throw new StateError("no live pi-plans execution; task updates are only valid in execution mode");
75
+ }
76
+ const outcome = applyTaskUpdate(execution.tasks, params.taskId, params.status, params.evidence, params.skipReason);
77
+ if (!outcome.ok) {
78
+ throw new StateError(outcome.message);
79
+ }
80
+ persistTaskProgress(ctx);
81
+ updateStatusWidget(ctx);
82
+ const line = `✓ ${outcome.message}`;
83
+ return {
84
+ content: [{ type: "text", text: line }],
85
+ details: { taskId: params.taskId, status: params.status },
86
+ };
87
+ },
88
+
89
+ renderCall(args, theme) {
90
+ const parts = [theme.bold("task "), theme.fg("accent", String(args.taskId)), theme.fg("muted", ` → ${String(args.status)}`)];
91
+ return new Text(parts.join(""), 0, 0);
92
+ },
93
+
94
+ renderResult(result, _opts, theme) {
95
+ const text = result.content[0];
96
+ const raw = text?.type === "text" ? text.text : "";
97
+ return new Text(theme.fg("success", raw), 0, 0);
98
+ },
99
+ });
100
+ }
package/src/tasks.ts ADDED
@@ -0,0 +1,189 @@
1
+ /**
2
+ * Task-tree runtime model (v0.6.1): the execution phase tracks plan tasks
3
+ * (`## Tasks`) as the unit of progress. Status flows in exclusively through
4
+ * the task status tool; completion of the whole run is gated by the
5
+ * independent completion auditor over the plan's verification checks.
6
+ */
7
+
8
+ import type { CheckItem, PlanTasks, TaskNode, WaveEntry } from "./plan.ts";
9
+ import { extractTaskCoverage, flattenTasks, resolveTaskWaves } from "./plan.ts";
10
+
11
+ export type TaskStatus = "pending" | "complete" | "skipped";
12
+
13
+ export interface TaskProgress {
14
+ status: TaskStatus;
15
+ /** Evidence recorded with the completion/skip (tool call payload). */
16
+ evidence?: string;
17
+ /** Skip reason, present only for skipped tasks. */
18
+ skipReason?: string;
19
+ }
20
+
21
+ /** Runtime view of one task node: plan metadata merged with live status. */
22
+ export interface TaskView {
23
+ id: string;
24
+ title: string;
25
+ wave: number;
26
+ deps: string[];
27
+ files: string[];
28
+ status: TaskStatus;
29
+ evidence?: string;
30
+ skipReason?: string;
31
+ children: TaskView[];
32
+ }
33
+
34
+ /** Serializable progress map persisted in the run checkpoint. */
35
+ export type TaskProgressMap = Record<string, TaskProgress>;
36
+
37
+ /** Build the runtime task tree from parsed plan tasks + persisted progress. */
38
+ export function buildTaskView(planTasks: PlanTasks, progress?: TaskProgressMap): TaskView[] {
39
+ const waves = resolveTaskWaves(planTasks);
40
+ const build = (node: TaskNode): TaskView => {
41
+ const state = progress?.[node.id];
42
+ const children = node.children.map(build);
43
+ return {
44
+ id: node.id,
45
+ title: node.title,
46
+ wave: waves.get(node.id) ?? 1,
47
+ deps: node.deps,
48
+ files: node.files,
49
+ status: state?.status ?? "pending",
50
+ evidence: state?.evidence,
51
+ skipReason: state?.skipReason,
52
+ children,
53
+ };
54
+ };
55
+ return planTasks.tasks.map(build);
56
+ }
57
+
58
+ export function flattenTaskViews(tasks: TaskView[]): TaskView[] {
59
+ const out: TaskView[] = [];
60
+ for (const t of tasks) {
61
+ out.push(t);
62
+ out.push(...flattenTaskViews(t.children));
63
+ }
64
+ return out;
65
+ }
66
+
67
+ /** A task counts as terminal when it is skipped, or complete together with
68
+ * every child (parents close last; an open child reopens the parent for
69
+ * scheduling purposes even if the parent itself reported complete). */
70
+ export function taskIsTerminal(task: TaskView): boolean {
71
+ if (task.status === "skipped") return true;
72
+ if (task.status !== "complete") return false;
73
+ return task.children.every((child) => taskIsTerminal(child));
74
+ }
75
+
76
+ export function allTasksTerminal(tasks: TaskView[]): boolean {
77
+ return flattenTaskViews(tasks).length > 0 && flattenTaskViews(tasks).every((task) => taskIsTerminal(task));
78
+ }
79
+
80
+ /** Snapshot of statuses for persistence and stall detection. */
81
+ export function taskProgressMap(tasks: TaskView[]): TaskProgressMap {
82
+ const map: TaskProgressMap = {};
83
+ for (const task of flattenTaskViews(tasks)) {
84
+ if (task.status === "pending" && task.evidence === undefined && task.skipReason === undefined) continue;
85
+ map[task.id] = {
86
+ status: task.status,
87
+ evidence: task.evidence,
88
+ skipReason: task.skipReason,
89
+ };
90
+ }
91
+ return map;
92
+ }
93
+
94
+ /** Progress aggregation for the dashboard: done counts terminal tasks. */
95
+ export function taskProgress(tasks: TaskView[]): { done: number; total: number } {
96
+ const flat = flattenTaskViews(tasks);
97
+ return {
98
+ done: flat.filter((task) => taskIsTerminal(task)).length,
99
+ total: flat.length,
100
+ };
101
+ }
102
+
103
+ /** The current task: the first non-terminal task in wave order, then
104
+ * document order (top-level-first; subtasks close before their parent is
105
+ * re-listed). Deterministic; used for the ▸ anchor in the dashboard and
106
+ * the per-turn injection. */
107
+ export function currentTask(tasks: TaskView[]): TaskView | null {
108
+ const flat = flattenTaskViews(tasks);
109
+ const open = flat.filter((task) => !taskIsTerminal(task));
110
+ if (open.length === 0) return null;
111
+ open.sort((a, b) => (a.wave - b.wave) || (flat.indexOf(a) - flat.indexOf(b)));
112
+ return open[0] ?? null;
113
+ }
114
+
115
+ /** Tasks of one wave (top-level only — children belong to their parent). */
116
+ export function waveTasks(tasks: TaskView[], wave: number): TaskView[] {
117
+ return tasks.filter((task) => task.wave === wave);
118
+ }
119
+
120
+ export function maxWave(tasks: TaskView[]): number {
121
+ return flattenTaskViews(tasks).reduce((max, task) => Math.max(max, task.wave), 1);
122
+ }
123
+
124
+ /** Legal status transitions for the task status tool: pending may close
125
+ * (complete/skipped); closed states are immutable except through the
126
+ * completion-audit flow (never through the task tool). */
127
+ export function canTransition(task: TaskView, next: TaskStatus): boolean {
128
+ if (task.status === next) return false;
129
+ if (task.status === "pending") return next === "complete" || next === "skipped";
130
+ return false;
131
+ }
132
+
133
+ /** Rollback set for a failed verification check: every task in its covers
134
+ * clause (parents cascade to their children, skipped tasks reopen too).
135
+ * Returns the ids that actually reopen. */
136
+ export function auditRollbackSet(
137
+ tasks: TaskView[],
138
+ checklist: CheckItem[],
139
+ vcId: string,
140
+ ): string[] {
141
+ const vc = checklist.find((item) => item.id === vcId);
142
+ if (!vc) return [];
143
+ const covered = new Set(extractTaskCoverage(vc.text));
144
+ const flat = flattenTaskViews(tasks);
145
+ const reopen: string[] = [];
146
+ const reopenNode = (node: TaskView): void => {
147
+ if (node.status !== "pending") {
148
+ node.status = "pending";
149
+ node.evidence = undefined;
150
+ node.skipReason = undefined;
151
+ reopen.push(node.id);
152
+ }
153
+ for (const child of node.children) reopenNode(child);
154
+ };
155
+ for (const node of flat) {
156
+ if (covered.has(node.id)) reopenNode(node);
157
+ }
158
+ return reopen;
159
+ }
160
+
161
+ /** Verification checks that cover no task are excluded from audit (their
162
+ * pass state cannot be derived from task statuses). */
163
+ export function auditableChecks(checklist: CheckItem[], tasks: TaskView[]): CheckItem[] {
164
+ const known = new Set(flattenTaskViews(tasks).map((task) => task.id));
165
+ return checklist.filter((item) => {
166
+ const covered = extractTaskCoverage(item.text);
167
+ return covered.length > 0 && covered.some((id) => known.has(id));
168
+ });
169
+ }
170
+
171
+ /** Checks with at least one skipped covered task whose remaining covered
172
+ * tasks are all complete (I-007: a skip whose siblings carry the delivery
173
+ * passes without an audit round). Pure-complete and pure-pending coverages
174
+ * are NOT presolved here — the former the audit affirms, the latter cannot
175
+ * arise once all tasks are terminal. */
176
+ export function skippedPassCheckIds(checklist: CheckItem[], tasks: TaskView[]): string[] {
177
+ const byId = new Map(flattenTaskViews(tasks).map((task) => [task.id, task]));
178
+ return auditableChecks(checklist, tasks)
179
+ .filter((item) => {
180
+ const covered = extractTaskCoverage(item.text).filter((id) => byId.has(id));
181
+ if (covered.length === 0) return false;
182
+ const statuses = covered.map((id) => byId.get(id)!.status);
183
+ return statuses.includes("skipped")
184
+ && statuses.every((status) => status === "skipped" || status === "complete");
185
+ })
186
+ .map((item) => item.id);
187
+ }
188
+
189
+ export type { PlanTasks, TaskNode, WaveEntry };
@@ -0,0 +1,67 @@
1
+ /**
2
+ * Thin thinking-level adapter over pi-ai's exported helpers.
3
+ *
4
+ * pi-ai already owns the level domain rules (`getSupportedThinkingLevels`:
5
+ * non-reasoning models only support "off"; a null mapping disables a level;
6
+ * xhigh/max appear only when explicitly mapped). Re-implementing them here
7
+ * would drift — this module only adds the pi-plans "default" sentinel and
8
+ * the stored-level resolution used by the reviewer spawn path (F-012).
9
+ *
10
+ * Semantics (decision 4 / F-009): `thinking_level: null` in the global
11
+ * reviewer config means DEFAULT — the child pi gets NO --thinking flag and
12
+ * resolves its own default chain (per-model settings → defaultThinkingLevel
13
+ * → medium, then model clamping). That is deliberately distinct from the
14
+ * explicit "off" level.
15
+ */
16
+
17
+ import { getSupportedThinkingLevels, type Model, type ModelThinkingLevel } from "@earendil-works/pi-ai";
18
+ import type { GlobalRoleConfig, ThinkingLevelValue } from "./global-state.ts";
19
+
20
+ /** Panel/label sentinel for `thinking_level: null` (omit --thinking). */
21
+ export const DEFAULT_LEVEL_SENTINEL = "default";
22
+
23
+ /** Levels a model actually supports, via pi-ai (single source of truth). */
24
+ export function levelsForModel(model: Pick<Model<any>, "reasoning" | "thinkingLevelMap">): ModelThinkingLevel[] {
25
+ return getSupportedThinkingLevels(model);
26
+ }
27
+
28
+ /** True when the stored level is supported by the model (spawn passes it
29
+ * through verbatim; unsupported stored levels are clamped by the child). */
30
+ export function isLevelSupported(
31
+ model: Pick<Model<any>, "reasoning" | "thinkingLevelMap">,
32
+ level: string | null,
33
+ ): boolean {
34
+ if (level === null) return true; // default: no flag, always valid
35
+ return levelsForModel(model).includes(level as ModelThinkingLevel);
36
+ }
37
+
38
+ /** Resolve the CLI argument for the stored level: null → omit the flag. */
39
+ export function spawnThinkingFlag(level: ThinkingLevelValue | null): string | null {
40
+ return level ?? null;
41
+ }
42
+
43
+ /** Display label for the reviewer overlay and subagents ledger. */
44
+ export function roleModelLabel(modelSelector: string, thinkingLevel: ThinkingLevelValue | null): string {
45
+ return `${modelSelector}:${thinkingLevel ?? DEFAULT_LEVEL_SENTINEL}`;
46
+ }
47
+
48
+ /** Human-facing one-liner for the default row / docs (F-009): the default
49
+ * is the child pi's own chain, NOT the current session's level. */
50
+ export const DEFAULT_LEVEL_DESCRIPTION =
51
+ "no --thinking flag: the child pi resolves its default (per-model settings → defaultThinkingLevel → medium)";
52
+
53
+ /** Spawn-view of a confirmed reviewer role: concrete model + optional level. */
54
+ export interface ResolvedReviewerSpawn {
55
+ modelSelector: string;
56
+ thinkingLevel: ThinkingLevelValue | null;
57
+ label: string;
58
+ }
59
+
60
+ export function resolveReviewerSpawn(role: Pick<GlobalRoleConfig, "model_selector" | "thinking_level">): ResolvedReviewerSpawn {
61
+ const modelSelector = role.model_selector ?? "";
62
+ return {
63
+ modelSelector,
64
+ thinkingLevel: role.thinking_level ?? null,
65
+ label: modelSelector ? roleModelLabel(modelSelector, role.thinking_level ?? null) : "unconfirmed",
66
+ };
67
+ }
@@ -1,10 +1,9 @@
1
1
  /**
2
2
  * Single source of truth for user-visible UI chrome language (issue #3).
3
3
  *
4
- * Workspace `language.tag` (`.git/pi_plans/config.json`) selects the chrome
5
- * strings for every user-visible surface: the batch form (src/ask-form.ts),
6
- * the refine/refs overlay footer (src/refine-ui.ts), the status panel
7
- * (src/panel.ts) and the execution status line (src/exec.ts).
4
+ * Workspace `language.tag` (`.git/pi-plans/config.json`) selects the chrome
5
+ * strings for every user-visible surface: the batch form (src/ask-form.ts)
6
+ * and the refine/refs overlay footer (src/refine-ui.ts).
8
7
  *
9
8
  * Mapping follows RFC 4647 primary-subtag fallback (zh-Hant-CN → zh-Hant →
10
9
  * zh), so every `zh*` tag renders the existing Simplified strings verbatim.
@@ -125,54 +124,4 @@ export function refineChrome(lang: UiLanguage): RefineChrome {
125
124
  return REFINE_CHROME[lang];
126
125
  }
127
126
 
128
- // ---------------------------------------------------------------------------
129
- // Status panel chrome (src/panel.ts) — implWarning branch only.
130
- // ---------------------------------------------------------------------------
131
-
132
- export interface PanelChrome {
133
- badgeStatus(topic: string, vcDone: number, vcTotal: number): string;
134
- progressWarning(): string;
135
- summaryWarning(topic: string, vcDone: number, vcTotal: number, nextAction: string): string;
136
- }
137
-
138
- const PANEL_CHROME: Record<UiLanguage, PanelChrome> = {
139
- zh: {
140
- badgeStatus: (topic, vcDone, vcTotal) => `${topic} · ⚠ I 解析 0 项 · VC ${vcDone}/${vcTotal}`,
141
- progressWarning: () => `⚠ plan 格式:Implementation Items 解析 0 项(面板无法计 I 进度)`,
142
- summaryWarning: (topic, vcDone, vcTotal, nextAction) =>
143
- `plans: ${topic} ▸ ⚠ Implementation Items 解析 0 项 · VC ${vcDone}/${vcTotal} · next: ${nextAction}`,
144
- },
145
- en: {
146
- badgeStatus: (topic, vcDone, vcTotal) => `${topic} · ⚠ I parse 0 items · VC ${vcDone}/${vcTotal}`,
147
- progressWarning: () => `⚠ plan format: Implementation Items parsed 0 items (I progress cannot be counted)`,
148
- summaryWarning: (topic, vcDone, vcTotal, nextAction) =>
149
- `plans: ${topic} ▸ ⚠ Implementation Items parsed 0 items · VC ${vcDone}/${vcTotal} · next: ${nextAction}`,
150
- },
151
- };
152
-
153
- export function panelChrome(lang: UiLanguage): PanelChrome {
154
- return PANEL_CHROME[lang];
155
- }
156
-
157
- // ---------------------------------------------------------------------------
158
- // Execution status line chrome (src/exec.ts) — goal-wait segment.
159
- // ---------------------------------------------------------------------------
160
-
161
- export interface ExecChrome {
162
- goalWait(noProgressRounds: number, waitRounds: number): string;
163
- }
164
-
165
- const EXEC_CHROME: Record<UiLanguage, ExecChrome> = {
166
- zh: {
167
- goalWait: (noProgressRounds, waitRounds) =>
168
- ` · 🔁 goal-wait · 无进展 ${noProgressRounds}/3 · 等待 ${waitRounds}/6`,
169
- },
170
- en: {
171
- goalWait: (noProgressRounds, waitRounds) =>
172
- ` · 🔁 goal-wait · no progress ${noProgressRounds}/3 · waiting ${waitRounds}/6`,
173
- },
174
- };
175
127
 
176
- export function execChrome(lang: UiLanguage): ExecChrome {
177
- return EXEC_CHROME[lang];
178
- }
@@ -144,6 +144,16 @@ export interface ExecutionCheckpoint {
144
144
  reverifyAll?: boolean;
145
145
  /** True when this approval/progress was produced in a different (origin) worktree. */
146
146
  originWorktree?: string;
147
+ /** v0.6.0: set while a delegated executor child owns the implementation;
148
+ * stale after a restart (orphaned delegate — the child died with the parent).
149
+ * Removed with delegated execution in v0.6.1; read-tolerated on legacy checkpoints. */
150
+ delegate?: { modelSelector: string; startedAt: string };
151
+ /** v0.6.1: task-tree progress (task id → status/evidence), the primary
152
+ * progress record. doneVcIds stays for legacy checkpoints and the final
153
+ * audit pass. */
154
+ tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
155
+ /** v0.6.1: completion-audit bookkeeping (rounds, last failed set, pass). */
156
+ audit?: { rounds: number; lastResult?: string; passed?: boolean };
147
157
  }
148
158
 
149
159
  export interface OwnerInfo {
@@ -451,7 +461,7 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
451
461
  const record = asRecord(value, label);
452
462
  rejectExtraKeys(
453
463
  record,
454
- new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree"]),
464
+ new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "audit"]),
455
465
  label,
456
466
  );
457
467
  const execution: ExecutionCheckpoint = {
@@ -473,6 +483,38 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
473
483
  if (record.pausedReason !== undefined) execution.pausedReason = asString(record.pausedReason, `${label}.pausedReason`);
474
484
  if (record.reverifyAll !== undefined) execution.reverifyAll = asBool(record.reverifyAll, `${label}.reverifyAll`);
475
485
  if (record.originWorktree !== undefined) execution.originWorktree = asString(record.originWorktree, `${label}.originWorktree`);
486
+ if (record.delegate !== undefined && record.delegate !== null) {
487
+ const delegate = asRecord(record.delegate, `${label}.delegate`);
488
+ rejectExtraKeys(delegate, new Set(["modelSelector", "startedAt"]), `${label}.delegate`);
489
+ execution.delegate = {
490
+ modelSelector: asString(delegate.modelSelector, `${label}.delegate.modelSelector`),
491
+ startedAt: asString(delegate.startedAt, `${label}.delegate.startedAt`),
492
+ };
493
+ }
494
+ if (record.tasks !== undefined && record.tasks !== null) {
495
+ const tasksRecord = asRecord(record.tasks, `${label}.tasks`);
496
+ const tasks: NonNullable<ExecutionCheckpoint["tasks"]> = {};
497
+ for (const [id, raw] of Object.entries(tasksRecord)) {
498
+ const entry = asRecord(raw, `${label}.tasks.${id}`);
499
+ rejectExtraKeys(entry, new Set(["status", "evidence", "skipReason"]), `${label}.tasks.${id}`);
500
+ const status = asEnum(entry.status, new Set(["pending", "complete", "skipped"]), `${label}.tasks.${id}.status`);
501
+ const item: { status: string; evidence?: string; skipReason?: string } = { status };
502
+ if (entry.evidence !== undefined) item.evidence = asString(entry.evidence, `${label}.tasks.${id}.evidence`);
503
+ if (entry.skipReason !== undefined) item.skipReason = asString(entry.skipReason, `${label}.tasks.${id}.skipReason`);
504
+ tasks[id] = item;
505
+ }
506
+ execution.tasks = tasks;
507
+ }
508
+ if (record.audit !== undefined && record.audit !== null) {
509
+ const audit = asRecord(record.audit, `${label}.audit`);
510
+ rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed"]), `${label}.audit`);
511
+ const parsed: { rounds: number; lastResult?: string; passed?: boolean } = {
512
+ rounds: asInt(audit.rounds, `${label}.audit.rounds`, 0),
513
+ };
514
+ if (audit.lastResult !== undefined) parsed.lastResult = asString(audit.lastResult, `${label}.audit.lastResult`);
515
+ if (audit.passed !== undefined) parsed.passed = asBool(audit.passed, `${label}.audit.passed`);
516
+ execution.audit = parsed;
517
+ }
476
518
  return execution;
477
519
  }
478
520
 
@@ -989,19 +1031,6 @@ export function applyReviewConsolidated(cp: WorkflowCheckpoint, roundId: string,
989
1031
  return { ...cp, reviewRounds: cp.reviewRounds.map((entry) => (entry.roundId === roundId ? nextRound : entry)) };
990
1032
  }
991
1033
 
992
- /** F-004: `completed` requires explicit evidence; approval cannot be forged by state writes. */
993
- export function applyCompleted(cp: WorkflowCheckpoint, evidence: string): WorkflowCheckpoint {
994
- if (cp.phase !== "implementation-review") {
995
- throw new StateError(`cannot complete from phase "${cp.phase}"`);
996
- }
997
- const review = cp.implementationReview;
998
- if (!review || review.terminationCondition === undefined) {
999
- throw new StateError("cannot complete without a recorded termination condition");
1000
- }
1001
- if (evidence.trim() === "") throw new StateError("completion requires non-empty evidence");
1002
- return { ...cp, phase: "completed", nextAction: "none" };
1003
- }
1004
-
1005
1034
  export function applyExecutionApproved(cp: WorkflowCheckpoint, approval: ExecutionApproval): WorkflowCheckpoint {
1006
1035
  // F-007 (implementation review): terminal and review phases cannot approve
1007
1036
  // execution; a re-approval (stop/migration reset approval to null) is legal
@@ -1025,6 +1054,8 @@ export function applyExecutionApproved(cp: WorkflowCheckpoint, approval: Executi
1025
1054
  doneVcIds: [],
1026
1055
  implStatus: {},
1027
1056
  usage: { inToks: 0, outToks: 0 },
1057
+ tasks: {},
1058
+ audit: { rounds: 0 },
1028
1059
  },
1029
1060
  };
1030
1061
  }
@@ -1035,6 +1066,12 @@ export interface ExecutionProgressInput {
1035
1066
  currentI?: string;
1036
1067
  usage?: { inToks: number; outToks: number };
1037
1068
  pausedReason?: string | null;
1069
+ /** v0.6.0: set/clear the delegated-executor record; null clears it. */
1070
+ delegate?: { modelSelector: string; startedAt: string } | null;
1071
+ /** v0.6.1: task-tree progress snapshot (authoritative). */
1072
+ tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
1073
+ /** v0.6.1: completion-audit bookkeeping update. */
1074
+ audit?: { rounds: number; lastResult?: string; passed?: boolean };
1038
1075
  }
1039
1076
 
1040
1077
  export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: ExecutionProgressInput): WorkflowCheckpoint {
@@ -1051,6 +1088,10 @@ export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: Executi
1051
1088
  }
1052
1089
  if (progress.pausedReason === null) delete execution.pausedReason;
1053
1090
  else if (progress.pausedReason !== undefined) execution.pausedReason = progress.pausedReason;
1091
+ if (progress.delegate === null) delete execution.delegate;
1092
+ else if (progress.delegate !== undefined) execution.delegate = progress.delegate;
1093
+ if (progress.tasks !== undefined) execution.tasks = progress.tasks;
1094
+ if (progress.audit !== undefined) execution.audit = progress.audit;
1054
1095
  return { ...cp, execution };
1055
1096
  }
1056
1097
 
@@ -1062,45 +1103,18 @@ export function applyExecutionHeadChanged(cp: WorkflowCheckpoint): WorkflowCheck
1062
1103
 
1063
1104
  export function applyExecutionCompleted(cp: WorkflowCheckpoint): WorkflowCheckpoint {
1064
1105
  if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
1106
+ // v0.6.1 (D-018): the post-execution amelioration loop is gone; a
1107
+ // completed audit passes the run straight to the terminal phase.
1065
1108
  return {
1066
1109
  ...cp,
1067
- phase: "implementation-review",
1068
- nextAction: "ask-question",
1069
- execution: { ...cp.execution, pausedReason: undefined },
1070
- };
1071
- }
1072
-
1073
- export function applyImplementationReviewConfigured(
1074
- cp: WorkflowCheckpoint,
1075
- terminationCondition: string,
1076
- reviewerCount?: number,
1077
- ): WorkflowCheckpoint {
1078
- if (cp.phase !== "implementation-review") throw new StateError("requires phase \"implementation-review\"");
1079
- if (cp.implementationReview?.terminationCondition !== undefined) {
1080
- throw new StateError("termination condition already configured; do not re-ask");
1081
- }
1082
- return {
1083
- ...cp,
1084
- implementationReview: {
1085
- terminationCondition,
1086
- reviewerCount,
1087
- completedRounds: cp.implementationReview?.completedRounds ?? 0,
1110
+ phase: "completed",
1111
+ nextAction: "none",
1112
+ execution: {
1113
+ ...cp.execution,
1114
+ pausedReason: undefined,
1115
+ delegate: undefined,
1116
+ audit: { ...(cp.execution.audit ?? { rounds: 0 }), passed: true },
1088
1117
  },
1089
- nextAction: "run-review",
1090
- };
1091
- }
1092
-
1093
- export function applyImplementationRoundFinished(cp: WorkflowCheckpoint): WorkflowCheckpoint {
1094
- if (cp.phase !== "implementation-review" || !cp.implementationReview) {
1095
- throw new StateError("requires phase \"implementation-review\"");
1096
- }
1097
- const review = cp.implementationReview;
1098
- if (review.currentRoundId === undefined) throw new StateError("no current round to finish");
1099
- const round = cp.reviewRounds.find((entry) => entry.roundId === review.currentRoundId);
1100
- if (!round || !round.consolidated) throw new StateError("current round is not consolidated");
1101
- return {
1102
- ...cp,
1103
- implementationReview: { ...review, completedRounds: review.completedRounds + 1, currentRoundId: undefined },
1104
1118
  };
1105
1119
  }
1106
1120
 
@@ -1120,12 +1134,16 @@ export function applyMigration(
1120
1134
  migration: { fromWorktree: cp.worktreeRoot, migratedAt: utcNow() },
1121
1135
  execution: cp.execution
1122
1136
  ? {
1123
- approval: null,
1124
- doneVcIds: [],
1125
- implStatus: {},
1126
- usage: cp.execution.usage,
1127
- originWorktree: cp.execution.originWorktree ?? cp.worktreeRoot,
1128
- }
1137
+ approval: null,
1138
+ doneVcIds: [],
1139
+ implStatus: {},
1140
+ usage: cp.execution.usage,
1141
+ originWorktree: cp.execution.originWorktree ?? cp.worktreeRoot,
1142
+ // v0.6.1: task progress and audit state do not survive a worktree
1143
+ // migration (same rule as VC validity).
1144
+ tasks: {},
1145
+ audit: { rounds: 0 },
1146
+ }
1129
1147
  : undefined,
1130
1148
  implementationReview: cp.implementationReview
1131
1149
  ? {
@@ -1148,7 +1166,10 @@ function migrationNextAction(cp: WorkflowCheckpoint): NextAction {
1148
1166
  /** Mark a paused stop without erasing the last phase (D-008). */
1149
1167
  export function applyExecutionStopped(cp: WorkflowCheckpoint, reason: string): WorkflowCheckpoint {
1150
1168
  if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
1151
- return { ...cp, execution: { ...cp.execution, pausedReason: reason } };
1169
+ const { delegate: _delegate, ...execution } = cp.execution;
1170
+ // A stop also revokes any outstanding audit-rollback authorization.
1171
+ const audit = execution.audit ? { ...execution.audit } : undefined;
1172
+ return { ...cp, execution: { ...execution, audit, pausedReason: reason } };
1152
1173
  }
1153
1174
 
1154
1175
  // ---------------------------------------------------------------------------