pi-plans 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. package/CONTRIBUTING.md +3 -3
  2. package/README.md +39 -37
  3. package/agents/reviewer.md +12 -3
  4. package/index.ts +42 -35
  5. package/package.json +1 -1
  6. package/references/pi-planning-workflow.md +44 -60
  7. package/references/plan-artifact-template.md +71 -60
  8. package/references/state-and-config.md +59 -43
  9. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  10. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  11. package/scripts/run-tests.ts +12 -1
  12. package/scripts/validate.ts +20 -9
  13. package/skills/debug-and-plan/SKILL.md +3 -3
  14. package/skills/plan-big/SKILL.md +3 -3
  15. package/skills/plan-normal/SKILL.md +3 -3
  16. package/skills/plan-small/SKILL.md +4 -4
  17. package/skills/plan-with-refs/SKILL.md +6 -6
  18. package/skills/planning/SKILL.md +1 -1
  19. package/src/ask-form.ts +4 -4
  20. package/src/auditor.ts +126 -0
  21. package/src/auto-approve.ts +1 -1
  22. package/src/autocomplete.ts +19 -17
  23. package/src/code-graph/commands.ts +2 -2
  24. package/src/code-graph/community.ts +1 -1
  25. package/src/code-graph/paths.ts +1 -1
  26. package/src/code-graph/watch.ts +2 -2
  27. package/src/compaction.ts +3 -3
  28. package/src/config-command.ts +146 -73
  29. package/src/dashboard.ts +257 -0
  30. package/src/exec.ts +692 -919
  31. package/src/global-state.ts +304 -0
  32. package/src/guard.ts +18 -19
  33. package/src/messaging.ts +44 -0
  34. package/src/plan.ts +421 -112
  35. package/src/query-hook.ts +4 -4
  36. package/src/refine-prompts.ts +12 -70
  37. package/src/refine-ui-helpers.ts +24 -5
  38. package/src/refine-ui-state.ts +1 -1
  39. package/src/refine-ui.ts +1 -1
  40. package/src/resume-command.ts +34 -128
  41. package/src/role-panels.ts +542 -0
  42. package/src/run-context.ts +3 -10
  43. package/src/state.ts +272 -72
  44. package/src/subagent.ts +19 -29
  45. package/src/task-tool.ts +100 -0
  46. package/src/tasks.ts +189 -0
  47. package/src/thinking-levels.ts +67 -0
  48. package/src/ui-language.ts +3 -54
  49. package/src/workflow-state.ts +63 -58
  50. package/tests/analyze-refs.test.ts +35 -18
  51. package/tests/ask-choice-schema.test.ts +0 -12
  52. package/tests/ask-choice.test.ts +2 -49
  53. package/tests/ask-form-tool.test.ts +4 -5
  54. package/tests/ask-form.test.ts +2 -2
  55. package/tests/auditor.test.ts +111 -0
  56. package/tests/auto-approve.test.ts +7 -10
  57. package/tests/autocomplete.test.ts +8 -11
  58. package/tests/code-graph-apply-action.test.ts +2 -2
  59. package/tests/code-graph-commands.test.ts +2 -2
  60. package/tests/code-graph-index.test.ts +2 -2
  61. package/tests/code-graph-loop.e2e.test.ts +1 -1
  62. package/tests/code-graph-mutations.test.ts +1 -1
  63. package/tests/code-graph-rollback.test.ts +1 -1
  64. package/tests/code-graph-v05.test.ts +2 -2
  65. package/tests/compaction.test.ts +1 -1
  66. package/tests/config-command.test.ts +103 -100
  67. package/tests/dashboard.test.ts +268 -0
  68. package/tests/exec-lifecycle.test.ts +181 -115
  69. package/tests/exec-panel-lifecycle.test.ts +106 -251
  70. package/tests/exec.test.ts +617 -1706
  71. package/tests/execute-plan.test.ts +44 -19
  72. package/tests/extension-load.test.ts +48 -0
  73. package/tests/global-state.test.ts +371 -0
  74. package/tests/graph-aware-file-tools.test.ts +5 -5
  75. package/tests/guard.test.ts +1 -1
  76. package/tests/multi-run.test.ts +3 -103
  77. package/tests/plan.test.ts +139 -62
  78. package/tests/plans.test.ts +7 -79
  79. package/tests/refine-prompts.test.ts +20 -71
  80. package/tests/refine-resume.test.ts +27 -22
  81. package/tests/refine-ui.test.ts +6 -15
  82. package/tests/resume-lifecycle.test.ts +37 -22
  83. package/tests/resume.test.ts +33 -81
  84. package/tests/role-panels.test.ts +391 -0
  85. package/tests/run-context.test.ts +1 -1
  86. package/tests/run-ownership.test.ts +1 -1
  87. package/tests/stale-ctx.test.ts +218 -0
  88. package/tests/state.test.ts +151 -32
  89. package/tests/subagent-thinking.test.ts +65 -0
  90. package/tests/subagent-usage.test.ts +1 -1
  91. package/tests/task-tool.test.ts +61 -0
  92. package/tests/thinking-levels.test.ts +77 -0
  93. package/tests/ui-language.test.ts +2 -17
  94. package/tests/workflow-state.test.ts +17 -99
  95. package/tools/analyze-refs.ts +67 -32
  96. package/tools/ask-choice.ts +7 -53
  97. package/tools/code-graph.ts +2 -2
  98. package/tools/execute-plan.ts +48 -99
  99. package/tools/graph-aware-file-tools.ts +4 -10
  100. package/tools/plans.ts +40 -66
  101. package/tools/refine.ts +101 -164
  102. package/agents/criticizer.md +0 -18
  103. package/agents/executor.md +0 -26
  104. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  105. package/src/panel.ts +0 -473
  106. package/src/termination-prompt.ts +0 -73
  107. package/tests/goal-wait.test.ts +0 -269
  108. package/tests/panel-i-zero.test.ts +0 -420
  109. package/tests/panel.test.ts +0 -355
package/src/tasks.ts ADDED
@@ -0,0 +1,189 @@
1
+ /**
2
+ * Task-tree runtime model (v0.6.1): the execution phase tracks plan tasks
3
+ * (`## Tasks`) as the unit of progress. Status flows in exclusively through
4
+ * the task status tool; completion of the whole run is gated by the
5
+ * independent completion auditor over the plan's verification checks.
6
+ */
7
+
8
+ import type { CheckItem, PlanTasks, TaskNode, WaveEntry } from "./plan.ts";
9
+ import { extractTaskCoverage, flattenTasks, resolveTaskWaves } from "./plan.ts";
10
+
11
+ export type TaskStatus = "pending" | "complete" | "skipped";
12
+
13
+ export interface TaskProgress {
14
+ status: TaskStatus;
15
+ /** Evidence recorded with the completion/skip (tool call payload). */
16
+ evidence?: string;
17
+ /** Skip reason, present only for skipped tasks. */
18
+ skipReason?: string;
19
+ }
20
+
21
+ /** Runtime view of one task node: plan metadata merged with live status. */
22
+ export interface TaskView {
23
+ id: string;
24
+ title: string;
25
+ wave: number;
26
+ deps: string[];
27
+ files: string[];
28
+ status: TaskStatus;
29
+ evidence?: string;
30
+ skipReason?: string;
31
+ children: TaskView[];
32
+ }
33
+
34
+ /** Serializable progress map persisted in the run checkpoint. */
35
+ export type TaskProgressMap = Record<string, TaskProgress>;
36
+
37
+ /** Build the runtime task tree from parsed plan tasks + persisted progress. */
38
+ export function buildTaskView(planTasks: PlanTasks, progress?: TaskProgressMap): TaskView[] {
39
+ const waves = resolveTaskWaves(planTasks);
40
+ const build = (node: TaskNode): TaskView => {
41
+ const state = progress?.[node.id];
42
+ const children = node.children.map(build);
43
+ return {
44
+ id: node.id,
45
+ title: node.title,
46
+ wave: waves.get(node.id) ?? 1,
47
+ deps: node.deps,
48
+ files: node.files,
49
+ status: state?.status ?? "pending",
50
+ evidence: state?.evidence,
51
+ skipReason: state?.skipReason,
52
+ children,
53
+ };
54
+ };
55
+ return planTasks.tasks.map(build);
56
+ }
57
+
58
+ export function flattenTaskViews(tasks: TaskView[]): TaskView[] {
59
+ const out: TaskView[] = [];
60
+ for (const t of tasks) {
61
+ out.push(t);
62
+ out.push(...flattenTaskViews(t.children));
63
+ }
64
+ return out;
65
+ }
66
+
67
+ /** A task counts as terminal when it is skipped, or complete together with
68
+ * every child (parents close last; an open child reopens the parent for
69
+ * scheduling purposes even if the parent itself reported complete). */
70
+ export function taskIsTerminal(task: TaskView): boolean {
71
+ if (task.status === "skipped") return true;
72
+ if (task.status !== "complete") return false;
73
+ return task.children.every((child) => taskIsTerminal(child));
74
+ }
75
+
76
+ export function allTasksTerminal(tasks: TaskView[]): boolean {
77
+ return flattenTaskViews(tasks).length > 0 && flattenTaskViews(tasks).every((task) => taskIsTerminal(task));
78
+ }
79
+
80
+ /** Snapshot of statuses for persistence and stall detection. */
81
+ export function taskProgressMap(tasks: TaskView[]): TaskProgressMap {
82
+ const map: TaskProgressMap = {};
83
+ for (const task of flattenTaskViews(tasks)) {
84
+ if (task.status === "pending" && task.evidence === undefined && task.skipReason === undefined) continue;
85
+ map[task.id] = {
86
+ status: task.status,
87
+ evidence: task.evidence,
88
+ skipReason: task.skipReason,
89
+ };
90
+ }
91
+ return map;
92
+ }
93
+
94
+ /** Progress aggregation for the dashboard: done counts terminal tasks. */
95
+ export function taskProgress(tasks: TaskView[]): { done: number; total: number } {
96
+ const flat = flattenTaskViews(tasks);
97
+ return {
98
+ done: flat.filter((task) => taskIsTerminal(task)).length,
99
+ total: flat.length,
100
+ };
101
+ }
102
+
103
+ /** The current task: the first non-terminal task in wave order, then
104
+ * document order (top-level-first; subtasks close before their parent is
105
+ * re-listed). Deterministic; used for the ▸ anchor in the dashboard and
106
+ * the per-turn injection. */
107
+ export function currentTask(tasks: TaskView[]): TaskView | null {
108
+ const flat = flattenTaskViews(tasks);
109
+ const open = flat.filter((task) => !taskIsTerminal(task));
110
+ if (open.length === 0) return null;
111
+ open.sort((a, b) => (a.wave - b.wave) || (flat.indexOf(a) - flat.indexOf(b)));
112
+ return open[0] ?? null;
113
+ }
114
+
115
+ /** Tasks of one wave (top-level only — children belong to their parent). */
116
+ export function waveTasks(tasks: TaskView[], wave: number): TaskView[] {
117
+ return tasks.filter((task) => task.wave === wave);
118
+ }
119
+
120
+ export function maxWave(tasks: TaskView[]): number {
121
+ return flattenTaskViews(tasks).reduce((max, task) => Math.max(max, task.wave), 1);
122
+ }
123
+
124
+ /** Legal status transitions for the task status tool: pending may close
125
+ * (complete/skipped); closed states are immutable except through the
126
+ * completion-audit flow (never through the task tool). */
127
+ export function canTransition(task: TaskView, next: TaskStatus): boolean {
128
+ if (task.status === next) return false;
129
+ if (task.status === "pending") return next === "complete" || next === "skipped";
130
+ return false;
131
+ }
132
+
133
+ /** Rollback set for a failed verification check: every task in its covers
134
+ * clause (parents cascade to their children, skipped tasks reopen too).
135
+ * Returns the ids that actually reopen. */
136
+ export function auditRollbackSet(
137
+ tasks: TaskView[],
138
+ checklist: CheckItem[],
139
+ vcId: string,
140
+ ): string[] {
141
+ const vc = checklist.find((item) => item.id === vcId);
142
+ if (!vc) return [];
143
+ const covered = new Set(extractTaskCoverage(vc.text));
144
+ const flat = flattenTaskViews(tasks);
145
+ const reopen: string[] = [];
146
+ const reopenNode = (node: TaskView): void => {
147
+ if (node.status !== "pending") {
148
+ node.status = "pending";
149
+ node.evidence = undefined;
150
+ node.skipReason = undefined;
151
+ reopen.push(node.id);
152
+ }
153
+ for (const child of node.children) reopenNode(child);
154
+ };
155
+ for (const node of flat) {
156
+ if (covered.has(node.id)) reopenNode(node);
157
+ }
158
+ return reopen;
159
+ }
160
+
161
+ /** Verification checks that cover no task are excluded from audit (their
162
+ * pass state cannot be derived from task statuses). */
163
+ export function auditableChecks(checklist: CheckItem[], tasks: TaskView[]): CheckItem[] {
164
+ const known = new Set(flattenTaskViews(tasks).map((task) => task.id));
165
+ return checklist.filter((item) => {
166
+ const covered = extractTaskCoverage(item.text);
167
+ return covered.length > 0 && covered.some((id) => known.has(id));
168
+ });
169
+ }
170
+
171
+ /** Checks with at least one skipped covered task whose remaining covered
172
+ * tasks are all complete (I-007: a skip whose siblings carry the delivery
173
+ * passes without an audit round). Pure-complete and pure-pending coverages
174
+ * are NOT presolved here — the former the audit affirms, the latter cannot
175
+ * arise once all tasks are terminal. */
176
+ export function skippedPassCheckIds(checklist: CheckItem[], tasks: TaskView[]): string[] {
177
+ const byId = new Map(flattenTaskViews(tasks).map((task) => [task.id, task]));
178
+ return auditableChecks(checklist, tasks)
179
+ .filter((item) => {
180
+ const covered = extractTaskCoverage(item.text).filter((id) => byId.has(id));
181
+ if (covered.length === 0) return false;
182
+ const statuses = covered.map((id) => byId.get(id)!.status);
183
+ return statuses.includes("skipped")
184
+ && statuses.every((status) => status === "skipped" || status === "complete");
185
+ })
186
+ .map((item) => item.id);
187
+ }
188
+
189
+ export type { PlanTasks, TaskNode, WaveEntry };
@@ -0,0 +1,67 @@
1
+ /**
2
+ * Thin thinking-level adapter over pi-ai's exported helpers.
3
+ *
4
+ * pi-ai already owns the level domain rules (`getSupportedThinkingLevels`:
5
+ * non-reasoning models only support "off"; a null mapping disables a level;
6
+ * xhigh/max appear only when explicitly mapped). Re-implementing them here
7
+ * would drift — this module only adds the pi-plans "default" sentinel and
8
+ * the stored-level resolution used by the reviewer spawn path (F-012).
9
+ *
10
+ * Semantics (decision 4 / F-009): `thinking_level: null` in the global
11
+ * reviewer config means DEFAULT — the child pi gets NO --thinking flag and
12
+ * resolves its own default chain (per-model settings → defaultThinkingLevel
13
+ * → medium, then model clamping). That is deliberately distinct from the
14
+ * explicit "off" level.
15
+ */
16
+
17
+ import { getSupportedThinkingLevels, type Model, type ModelThinkingLevel } from "@earendil-works/pi-ai";
18
+ import type { GlobalRoleConfig, ThinkingLevelValue } from "./global-state.ts";
19
+
20
+ /** Panel/label sentinel for `thinking_level: null` (omit --thinking). */
21
+ export const DEFAULT_LEVEL_SENTINEL = "default";
22
+
23
+ /** Levels a model actually supports, via pi-ai (single source of truth). */
24
+ export function levelsForModel(model: Pick<Model<any>, "reasoning" | "thinkingLevelMap">): ModelThinkingLevel[] {
25
+ return getSupportedThinkingLevels(model);
26
+ }
27
+
28
+ /** True when the stored level is supported by the model (spawn passes it
29
+ * through verbatim; unsupported stored levels are clamped by the child). */
30
+ export function isLevelSupported(
31
+ model: Pick<Model<any>, "reasoning" | "thinkingLevelMap">,
32
+ level: string | null,
33
+ ): boolean {
34
+ if (level === null) return true; // default: no flag, always valid
35
+ return levelsForModel(model).includes(level as ModelThinkingLevel);
36
+ }
37
+
38
+ /** Resolve the CLI argument for the stored level: null → omit the flag. */
39
+ export function spawnThinkingFlag(level: ThinkingLevelValue | null): string | null {
40
+ return level ?? null;
41
+ }
42
+
43
+ /** Display label for the reviewer overlay and subagents ledger. */
44
+ export function roleModelLabel(modelSelector: string, thinkingLevel: ThinkingLevelValue | null): string {
45
+ return `${modelSelector}:${thinkingLevel ?? DEFAULT_LEVEL_SENTINEL}`;
46
+ }
47
+
48
+ /** Human-facing one-liner for the default row / docs (F-009): the default
49
+ * is the child pi's own chain, NOT the current session's level. */
50
+ export const DEFAULT_LEVEL_DESCRIPTION =
51
+ "no --thinking flag: the child pi resolves its default (per-model settings → defaultThinkingLevel → medium)";
52
+
53
+ /** Spawn-view of a confirmed reviewer role: concrete model + optional level. */
54
+ export interface ResolvedReviewerSpawn {
55
+ modelSelector: string;
56
+ thinkingLevel: ThinkingLevelValue | null;
57
+ label: string;
58
+ }
59
+
60
+ export function resolveReviewerSpawn(role: Pick<GlobalRoleConfig, "model_selector" | "thinking_level">): ResolvedReviewerSpawn {
61
+ const modelSelector = role.model_selector ?? "";
62
+ return {
63
+ modelSelector,
64
+ thinkingLevel: role.thinking_level ?? null,
65
+ label: modelSelector ? roleModelLabel(modelSelector, role.thinking_level ?? null) : "unconfirmed",
66
+ };
67
+ }
@@ -1,10 +1,9 @@
1
1
  /**
2
2
  * Single source of truth for user-visible UI chrome language (issue #3).
3
3
  *
4
- * Workspace `language.tag` (`.git/pi_plans/config.json`) selects the chrome
5
- * strings for every user-visible surface: the batch form (src/ask-form.ts),
6
- * the refine/refs overlay footer (src/refine-ui.ts), the status panel
7
- * (src/panel.ts) and the execution status line (src/exec.ts).
4
+ * Workspace `language.tag` (`.git/pi-plans/config.json`) selects the chrome
5
+ * strings for every user-visible surface: the batch form (src/ask-form.ts)
6
+ * and the refine/refs overlay footer (src/refine-ui.ts).
8
7
  *
9
8
  * Mapping follows RFC 4647 primary-subtag fallback (zh-Hant-CN → zh-Hant →
10
9
  * zh), so every `zh*` tag renders the existing Simplified strings verbatim.
@@ -125,54 +124,4 @@ export function refineChrome(lang: UiLanguage): RefineChrome {
125
124
  return REFINE_CHROME[lang];
126
125
  }
127
126
 
128
- // ---------------------------------------------------------------------------
129
- // Status panel chrome (src/panel.ts) — implWarning branch only.
130
- // ---------------------------------------------------------------------------
131
-
132
- export interface PanelChrome {
133
- badgeStatus(topic: string, vcDone: number, vcTotal: number): string;
134
- progressWarning(): string;
135
- summaryWarning(topic: string, vcDone: number, vcTotal: number, nextAction: string): string;
136
- }
137
-
138
- const PANEL_CHROME: Record<UiLanguage, PanelChrome> = {
139
- zh: {
140
- badgeStatus: (topic, vcDone, vcTotal) => `${topic} · ⚠ I 解析 0 项 · VC ${vcDone}/${vcTotal}`,
141
- progressWarning: () => `⚠ plan 格式:Implementation Items 解析 0 项(面板无法计 I 进度)`,
142
- summaryWarning: (topic, vcDone, vcTotal, nextAction) =>
143
- `plans: ${topic} ▸ ⚠ Implementation Items 解析 0 项 · VC ${vcDone}/${vcTotal} · next: ${nextAction}`,
144
- },
145
- en: {
146
- badgeStatus: (topic, vcDone, vcTotal) => `${topic} · ⚠ I parse 0 items · VC ${vcDone}/${vcTotal}`,
147
- progressWarning: () => `⚠ plan format: Implementation Items parsed 0 items (I progress cannot be counted)`,
148
- summaryWarning: (topic, vcDone, vcTotal, nextAction) =>
149
- `plans: ${topic} ▸ ⚠ Implementation Items parsed 0 items · VC ${vcDone}/${vcTotal} · next: ${nextAction}`,
150
- },
151
- };
152
-
153
- export function panelChrome(lang: UiLanguage): PanelChrome {
154
- return PANEL_CHROME[lang];
155
- }
156
-
157
- // ---------------------------------------------------------------------------
158
- // Execution status line chrome (src/exec.ts) — goal-wait segment.
159
- // ---------------------------------------------------------------------------
160
-
161
- export interface ExecChrome {
162
- goalWait(noProgressRounds: number, waitRounds: number): string;
163
- }
164
-
165
- const EXEC_CHROME: Record<UiLanguage, ExecChrome> = {
166
- zh: {
167
- goalWait: (noProgressRounds, waitRounds) =>
168
- ` · 🔁 goal-wait · 无进展 ${noProgressRounds}/3 · 等待 ${waitRounds}/6`,
169
- },
170
- en: {
171
- goalWait: (noProgressRounds, waitRounds) =>
172
- ` · 🔁 goal-wait · no progress ${noProgressRounds}/3 · waiting ${waitRounds}/6`,
173
- },
174
- };
175
127
 
176
- export function execChrome(lang: UiLanguage): ExecChrome {
177
- return EXEC_CHROME[lang];
178
- }
@@ -145,8 +145,15 @@ export interface ExecutionCheckpoint {
145
145
  /** True when this approval/progress was produced in a different (origin) worktree. */
146
146
  originWorktree?: string;
147
147
  /** v0.6.0: set while a delegated executor child owns the implementation;
148
- * stale after a restart (orphaned delegate — the child died with the parent). */
148
+ * stale after a restart (orphaned delegate — the child died with the parent).
149
+ * Removed with delegated execution in v0.6.1; read-tolerated on legacy checkpoints. */
149
150
  delegate?: { modelSelector: string; startedAt: string };
151
+ /** v0.6.1: task-tree progress (task id → status/evidence), the primary
152
+ * progress record. doneVcIds stays for legacy checkpoints and the final
153
+ * audit pass. */
154
+ tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
155
+ /** v0.6.1: completion-audit bookkeeping (rounds, last failed set, pass). */
156
+ audit?: { rounds: number; lastResult?: string; passed?: boolean };
150
157
  }
151
158
 
152
159
  export interface OwnerInfo {
@@ -454,7 +461,7 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
454
461
  const record = asRecord(value, label);
455
462
  rejectExtraKeys(
456
463
  record,
457
- new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate"]),
464
+ new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "audit"]),
458
465
  label,
459
466
  );
460
467
  const execution: ExecutionCheckpoint = {
@@ -484,6 +491,30 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
484
491
  startedAt: asString(delegate.startedAt, `${label}.delegate.startedAt`),
485
492
  };
486
493
  }
494
+ if (record.tasks !== undefined && record.tasks !== null) {
495
+ const tasksRecord = asRecord(record.tasks, `${label}.tasks`);
496
+ const tasks: NonNullable<ExecutionCheckpoint["tasks"]> = {};
497
+ for (const [id, raw] of Object.entries(tasksRecord)) {
498
+ const entry = asRecord(raw, `${label}.tasks.${id}`);
499
+ rejectExtraKeys(entry, new Set(["status", "evidence", "skipReason"]), `${label}.tasks.${id}`);
500
+ const status = asEnum(entry.status, new Set(["pending", "complete", "skipped"]), `${label}.tasks.${id}.status`);
501
+ const item: { status: string; evidence?: string; skipReason?: string } = { status };
502
+ if (entry.evidence !== undefined) item.evidence = asString(entry.evidence, `${label}.tasks.${id}.evidence`);
503
+ if (entry.skipReason !== undefined) item.skipReason = asString(entry.skipReason, `${label}.tasks.${id}.skipReason`);
504
+ tasks[id] = item;
505
+ }
506
+ execution.tasks = tasks;
507
+ }
508
+ if (record.audit !== undefined && record.audit !== null) {
509
+ const audit = asRecord(record.audit, `${label}.audit`);
510
+ rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed"]), `${label}.audit`);
511
+ const parsed: { rounds: number; lastResult?: string; passed?: boolean } = {
512
+ rounds: asInt(audit.rounds, `${label}.audit.rounds`, 0),
513
+ };
514
+ if (audit.lastResult !== undefined) parsed.lastResult = asString(audit.lastResult, `${label}.audit.lastResult`);
515
+ if (audit.passed !== undefined) parsed.passed = asBool(audit.passed, `${label}.audit.passed`);
516
+ execution.audit = parsed;
517
+ }
487
518
  return execution;
488
519
  }
489
520
 
@@ -1000,19 +1031,6 @@ export function applyReviewConsolidated(cp: WorkflowCheckpoint, roundId: string,
1000
1031
  return { ...cp, reviewRounds: cp.reviewRounds.map((entry) => (entry.roundId === roundId ? nextRound : entry)) };
1001
1032
  }
1002
1033
 
1003
- /** F-004: `completed` requires explicit evidence; approval cannot be forged by state writes. */
1004
- export function applyCompleted(cp: WorkflowCheckpoint, evidence: string): WorkflowCheckpoint {
1005
- if (cp.phase !== "implementation-review") {
1006
- throw new StateError(`cannot complete from phase "${cp.phase}"`);
1007
- }
1008
- const review = cp.implementationReview;
1009
- if (!review || review.terminationCondition === undefined) {
1010
- throw new StateError("cannot complete without a recorded termination condition");
1011
- }
1012
- if (evidence.trim() === "") throw new StateError("completion requires non-empty evidence");
1013
- return { ...cp, phase: "completed", nextAction: "none" };
1014
- }
1015
-
1016
1034
  export function applyExecutionApproved(cp: WorkflowCheckpoint, approval: ExecutionApproval): WorkflowCheckpoint {
1017
1035
  // F-007 (implementation review): terminal and review phases cannot approve
1018
1036
  // execution; a re-approval (stop/migration reset approval to null) is legal
@@ -1036,6 +1054,8 @@ export function applyExecutionApproved(cp: WorkflowCheckpoint, approval: Executi
1036
1054
  doneVcIds: [],
1037
1055
  implStatus: {},
1038
1056
  usage: { inToks: 0, outToks: 0 },
1057
+ tasks: {},
1058
+ audit: { rounds: 0 },
1039
1059
  },
1040
1060
  };
1041
1061
  }
@@ -1048,6 +1068,10 @@ export interface ExecutionProgressInput {
1048
1068
  pausedReason?: string | null;
1049
1069
  /** v0.6.0: set/clear the delegated-executor record; null clears it. */
1050
1070
  delegate?: { modelSelector: string; startedAt: string } | null;
1071
+ /** v0.6.1: task-tree progress snapshot (authoritative). */
1072
+ tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
1073
+ /** v0.6.1: completion-audit bookkeeping update. */
1074
+ audit?: { rounds: number; lastResult?: string; passed?: boolean };
1051
1075
  }
1052
1076
 
1053
1077
  export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: ExecutionProgressInput): WorkflowCheckpoint {
@@ -1066,6 +1090,8 @@ export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: Executi
1066
1090
  else if (progress.pausedReason !== undefined) execution.pausedReason = progress.pausedReason;
1067
1091
  if (progress.delegate === null) delete execution.delegate;
1068
1092
  else if (progress.delegate !== undefined) execution.delegate = progress.delegate;
1093
+ if (progress.tasks !== undefined) execution.tasks = progress.tasks;
1094
+ if (progress.audit !== undefined) execution.audit = progress.audit;
1069
1095
  return { ...cp, execution };
1070
1096
  }
1071
1097
 
@@ -1077,45 +1103,18 @@ export function applyExecutionHeadChanged(cp: WorkflowCheckpoint): WorkflowCheck
1077
1103
 
1078
1104
  export function applyExecutionCompleted(cp: WorkflowCheckpoint): WorkflowCheckpoint {
1079
1105
  if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
1106
+ // v0.6.1 (D-018): the post-execution amelioration loop is gone; a
1107
+ // completed audit passes the run straight to the terminal phase.
1080
1108
  return {
1081
1109
  ...cp,
1082
- phase: "implementation-review",
1083
- nextAction: "ask-question",
1084
- execution: { ...cp.execution, pausedReason: undefined },
1085
- };
1086
- }
1087
-
1088
- export function applyImplementationReviewConfigured(
1089
- cp: WorkflowCheckpoint,
1090
- terminationCondition: string,
1091
- reviewerCount?: number,
1092
- ): WorkflowCheckpoint {
1093
- if (cp.phase !== "implementation-review") throw new StateError("requires phase \"implementation-review\"");
1094
- if (cp.implementationReview?.terminationCondition !== undefined) {
1095
- throw new StateError("termination condition already configured; do not re-ask");
1096
- }
1097
- return {
1098
- ...cp,
1099
- implementationReview: {
1100
- terminationCondition,
1101
- reviewerCount,
1102
- completedRounds: cp.implementationReview?.completedRounds ?? 0,
1110
+ phase: "completed",
1111
+ nextAction: "none",
1112
+ execution: {
1113
+ ...cp.execution,
1114
+ pausedReason: undefined,
1115
+ delegate: undefined,
1116
+ audit: { ...(cp.execution.audit ?? { rounds: 0 }), passed: true },
1103
1117
  },
1104
- nextAction: "run-review",
1105
- };
1106
- }
1107
-
1108
- export function applyImplementationRoundFinished(cp: WorkflowCheckpoint): WorkflowCheckpoint {
1109
- if (cp.phase !== "implementation-review" || !cp.implementationReview) {
1110
- throw new StateError("requires phase \"implementation-review\"");
1111
- }
1112
- const review = cp.implementationReview;
1113
- if (review.currentRoundId === undefined) throw new StateError("no current round to finish");
1114
- const round = cp.reviewRounds.find((entry) => entry.roundId === review.currentRoundId);
1115
- if (!round || !round.consolidated) throw new StateError("current round is not consolidated");
1116
- return {
1117
- ...cp,
1118
- implementationReview: { ...review, completedRounds: review.completedRounds + 1, currentRoundId: undefined },
1119
1118
  };
1120
1119
  }
1121
1120
 
@@ -1135,12 +1134,16 @@ export function applyMigration(
1135
1134
  migration: { fromWorktree: cp.worktreeRoot, migratedAt: utcNow() },
1136
1135
  execution: cp.execution
1137
1136
  ? {
1138
- approval: null,
1139
- doneVcIds: [],
1140
- implStatus: {},
1141
- usage: cp.execution.usage,
1142
- originWorktree: cp.execution.originWorktree ?? cp.worktreeRoot,
1143
- }
1137
+ approval: null,
1138
+ doneVcIds: [],
1139
+ implStatus: {},
1140
+ usage: cp.execution.usage,
1141
+ originWorktree: cp.execution.originWorktree ?? cp.worktreeRoot,
1142
+ // v0.6.1: task progress and audit state do not survive a worktree
1143
+ // migration (same rule as VC validity).
1144
+ tasks: {},
1145
+ audit: { rounds: 0 },
1146
+ }
1144
1147
  : undefined,
1145
1148
  implementationReview: cp.implementationReview
1146
1149
  ? {
@@ -1164,7 +1167,9 @@ function migrationNextAction(cp: WorkflowCheckpoint): NextAction {
1164
1167
  export function applyExecutionStopped(cp: WorkflowCheckpoint, reason: string): WorkflowCheckpoint {
1165
1168
  if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
1166
1169
  const { delegate: _delegate, ...execution } = cp.execution;
1167
- return { ...cp, execution: { ...execution, pausedReason: reason } };
1170
+ // A stop also revokes any outstanding audit-rollback authorization.
1171
+ const audit = execution.audit ? { ...execution.audit } : undefined;
1172
+ return { ...cp, execution: { ...execution, audit, pausedReason: reason } };
1168
1173
  }
1169
1174
 
1170
1175
  // ---------------------------------------------------------------------------
@@ -12,13 +12,22 @@ import { initState, setRole, setLanguage, startRun, readActive } from "../src/st
12
12
  const ROOT = path.dirname(path.dirname(url.fileURLToPath(import.meta.url)));
13
13
 
14
14
  let tmpRoot: string;
15
+ let globalDir: string;
16
+ let previousGlobalDir: string | undefined;
15
17
 
16
18
  before(() => {
17
19
  tmpRoot = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-analyze-refs-"));
20
+ // Isolate the global reviewer config (F-002): never touch ~/.pi/pi-plans.
21
+ previousGlobalDir = process.env.PI_PLANS_GLOBAL_DIR;
22
+ globalDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-global-analyze-refs-"));
23
+ process.env.PI_PLANS_GLOBAL_DIR = globalDir;
18
24
  });
19
25
 
20
26
  after(() => {
27
+ if (previousGlobalDir === undefined) delete process.env.PI_PLANS_GLOBAL_DIR;
28
+ else process.env.PI_PLANS_GLOBAL_DIR = previousGlobalDir;
21
29
  fs.rmSync(tmpRoot, { recursive: true, force: true });
30
+ fs.rmSync(globalDir, { recursive: true, force: true });
22
31
  });
23
32
 
24
33
  function mkWorkdir(name: string): string {
@@ -102,7 +111,7 @@ function subagentLines(workdir: string): Array<any> {
102
111
  .map((line) => JSON.parse(line));
103
112
  }
104
113
 
105
- describe("analyze_refs gates", () => {
114
+ describe("analyze_refs gates", () => {
106
115
  it("refuses when no pi-plans state exists", async () => {
107
116
  const workdir = mkWorkdir("gates-no-state");
108
117
  const tool = loadTool();
@@ -112,35 +121,43 @@ describe("analyze_refs gates", () => {
112
121
  );
113
122
  });
114
123
 
115
- it("refuses with role-setting guidance when reviewer mode is invalid", async () => {
116
- const workdir = mkWorkdir("gates-bad-mode");
124
+ it("refuses with embedded text guidance when the reviewer model is unconfirmed (headless)", async () => {
125
+ const workdir = mkWorkdir("gates-unconfirmed");
117
126
  initState(workdir);
118
- const configPath = path.join(workdir, ".git", "pi_plans", "config.json");
119
- const config = JSON.parse(fs.readFileSync(configPath, "utf8"));
120
- config.reviewer.mode = "bogus";
121
- fs.writeFileSync(configPath, `${JSON.stringify(config, null, "\t")}\n`, "utf8");
122
127
  const tool = loadTool();
123
128
  await assert.rejects(
124
129
  tool.execute("c1", { refs: [{ id: "ref-1", localPath: "." }] }, undefined, undefined, headlessCtx(workdir)),
125
- /reviewer role mode is missing or invalid/,
130
+ /model was never confirmed/,
126
131
  );
127
132
  });
128
133
 
129
- it("refuses current-session reviewer mode with a switch-to-delegated message", async () => {
134
+ it("ignores the reviewer mode entirely (Q-4=B): current-session proceeds once a model is confirmed", async () => {
130
135
  const workdir = mkWorkdir("gates-current-session");
131
136
  initState(workdir);
132
- setRole(workdir, { role: "reviewer", mode: "current-session" });
133
- const tool = loadTool();
134
- await assert.rejects(
135
- tool.execute("c1", { refs: [{ id: "ref-1", localPath: "." }] }, undefined, undefined, headlessCtx(workdir)),
136
- /current-session.*delegated-subagent/s,
137
+ setRole(workdir, { role: "reviewer", mode: "current-session", modelSelector: "fake/model", confirmed: true });
138
+ const refDir = path.join(workdir, "refs", "solo");
139
+ fs.mkdirSync(refDir, { recursive: true });
140
+ const restore = withFakePi(
141
+ fakePiScript(
142
+ `emit({ type: "message_end", message: { role: "assistant", model: "fake/model", content: [{ type: "text", text: "OK" }] } });`,
143
+ ),
137
144
  );
145
+ const tool = loadTool();
146
+ try {
147
+ const result = await tool.execute("c1", { refs: [{ id: "ref-1", localPath: refDir }] }, undefined, undefined, headlessCtx(workdir));
148
+ assert.match(result.content[0]!.text, /mode is current-session, but analyze_refs always spawns/);
149
+ assert.match(result.content[0]!.text, /OK/);
150
+ } finally {
151
+ restore();
152
+ }
138
153
  });
139
154
 
140
- it("refuses with confirmation guidance when the reviewer model is unconfirmed", async () => {
141
- const workdir = mkWorkdir("gates-unconfirmed");
155
+ it("requires a confirmed model even in current-session mode (analysis always spawns)", async () => {
156
+ const workdir = mkWorkdir("gates-current-session-unconfirmed");
142
157
  initState(workdir);
143
- setRole(workdir, { role: "reviewer", mode: "delegated-subagent" });
158
+ // 'inherit' fully resets selector AND confirmation — the prior test's
159
+ // confirmed selector must not carry over (the global config is shared).
160
+ setRole(workdir, { role: "reviewer", mode: "current-session", modelSelector: "inherit" });
144
161
  const tool = loadTool();
145
162
  await assert.rejects(
146
163
  tool.execute("c1", { refs: [{ id: "ref-1", localPath: "." }] }, undefined, undefined, headlessCtx(workdir)),
@@ -310,7 +327,7 @@ describe("analyze_refs fanout", () => {
310
327
  const result = await tool.execute("c1", { refs: [{ id: "ref-1", localPath: refDir }] }, undefined, undefined, headlessCtx(workdir));
311
328
  assert.match(result.content[0]!.text, /### pi-plans-refs-adhoc-ref-1/);
312
329
  assert.equal(readActive(workdir), null);
313
- const ledger = path.join(workdir, ".git", "pi_plans", "runs");
330
+ const ledger = path.join(workdir, ".git", "pi-plans", "runs");
314
331
  const runs = fs.existsSync(ledger) ? fs.readdirSync(ledger) : [];
315
332
  const spawnFiles = runs.flatMap((run) =>
316
333
  fs.existsSync(path.join(ledger, run, "subagents.jsonl")) ? [fs.readFileSync(path.join(ledger, run, "subagents.jsonl"), "utf8")] : [],
@@ -157,18 +157,6 @@ describe("VC-2 · hardening does not wound legitimate single-question shapes", (
157
157
  assert.equal(Value.Check(AskChoiceParams as never, wellFormed), true);
158
158
  });
159
159
 
160
- it("single question with trailing + autoComplete:false passes Value.Check", () => {
161
- const wellFormed = {
162
- question: "实现评审循环如何终止?",
163
- options: [
164
- { label: "goal wait:直到无未通过 VC", recommended: true },
165
- { label: "直到无高危发现(硬帽 5 轮)" },
166
- ],
167
- trailing: "auto-refine-loop",
168
- autoComplete: false,
169
- };
170
- assert.equal(Value.Check(AskChoiceParams as never, wellFormed), true);
171
- });
172
160
  });
173
161
 
174
162
  describe("VC-3 · zero-recommended batches reject loudly with zero side effects", () => {