pi-plans 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +3 -3
- package/README.md +39 -37
- package/agents/reviewer.md +12 -3
- package/index.ts +42 -35
- package/package.json +1 -1
- package/references/pi-planning-workflow.md +44 -60
- package/references/plan-artifact-template.md +71 -60
- package/references/state-and-config.md +59 -43
- package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
- package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
- package/scripts/run-tests.ts +12 -1
- package/scripts/validate.ts +20 -9
- package/skills/debug-and-plan/SKILL.md +3 -3
- package/skills/plan-big/SKILL.md +3 -3
- package/skills/plan-normal/SKILL.md +3 -3
- package/skills/plan-small/SKILL.md +4 -4
- package/skills/plan-with-refs/SKILL.md +6 -6
- package/skills/planning/SKILL.md +1 -1
- package/src/ask-form.ts +4 -4
- package/src/auditor.ts +126 -0
- package/src/auto-approve.ts +1 -1
- package/src/autocomplete.ts +19 -17
- package/src/code-graph/commands.ts +2 -2
- package/src/code-graph/community.ts +1 -1
- package/src/code-graph/paths.ts +1 -1
- package/src/code-graph/watch.ts +2 -2
- package/src/compaction.ts +3 -3
- package/src/config-command.ts +146 -73
- package/src/dashboard.ts +257 -0
- package/src/exec.ts +692 -919
- package/src/global-state.ts +304 -0
- package/src/guard.ts +18 -19
- package/src/messaging.ts +44 -0
- package/src/plan.ts +421 -112
- package/src/query-hook.ts +4 -4
- package/src/refine-prompts.ts +12 -70
- package/src/refine-ui-helpers.ts +24 -5
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +1 -1
- package/src/resume-command.ts +34 -128
- package/src/role-panels.ts +542 -0
- package/src/run-context.ts +3 -10
- package/src/state.ts +272 -72
- package/src/subagent.ts +19 -29
- package/src/task-tool.ts +100 -0
- package/src/tasks.ts +189 -0
- package/src/thinking-levels.ts +67 -0
- package/src/ui-language.ts +3 -54
- package/src/workflow-state.ts +63 -58
- package/tests/analyze-refs.test.ts +35 -18
- package/tests/ask-choice-schema.test.ts +0 -12
- package/tests/ask-choice.test.ts +2 -49
- package/tests/ask-form-tool.test.ts +4 -5
- package/tests/ask-form.test.ts +2 -2
- package/tests/auditor.test.ts +111 -0
- package/tests/auto-approve.test.ts +7 -10
- package/tests/autocomplete.test.ts +8 -11
- package/tests/code-graph-apply-action.test.ts +2 -2
- package/tests/code-graph-commands.test.ts +2 -2
- package/tests/code-graph-index.test.ts +2 -2
- package/tests/code-graph-loop.e2e.test.ts +1 -1
- package/tests/code-graph-mutations.test.ts +1 -1
- package/tests/code-graph-rollback.test.ts +1 -1
- package/tests/code-graph-v05.test.ts +2 -2
- package/tests/compaction.test.ts +1 -1
- package/tests/config-command.test.ts +103 -100
- package/tests/dashboard.test.ts +268 -0
- package/tests/exec-lifecycle.test.ts +181 -115
- package/tests/exec-panel-lifecycle.test.ts +106 -251
- package/tests/exec.test.ts +617 -1706
- package/tests/execute-plan.test.ts +44 -19
- package/tests/extension-load.test.ts +48 -0
- package/tests/global-state.test.ts +371 -0
- package/tests/graph-aware-file-tools.test.ts +5 -5
- package/tests/guard.test.ts +1 -1
- package/tests/multi-run.test.ts +3 -103
- package/tests/plan.test.ts +139 -62
- package/tests/plans.test.ts +7 -79
- package/tests/refine-prompts.test.ts +20 -71
- package/tests/refine-resume.test.ts +27 -22
- package/tests/refine-ui.test.ts +6 -15
- package/tests/resume-lifecycle.test.ts +37 -22
- package/tests/resume.test.ts +33 -81
- package/tests/role-panels.test.ts +391 -0
- package/tests/run-context.test.ts +1 -1
- package/tests/run-ownership.test.ts +1 -1
- package/tests/stale-ctx.test.ts +218 -0
- package/tests/state.test.ts +151 -32
- package/tests/subagent-thinking.test.ts +65 -0
- package/tests/subagent-usage.test.ts +1 -1
- package/tests/task-tool.test.ts +61 -0
- package/tests/thinking-levels.test.ts +77 -0
- package/tests/ui-language.test.ts +2 -17
- package/tests/workflow-state.test.ts +17 -99
- package/tools/analyze-refs.ts +67 -32
- package/tools/ask-choice.ts +7 -53
- package/tools/code-graph.ts +2 -2
- package/tools/execute-plan.ts +48 -99
- package/tools/graph-aware-file-tools.ts +4 -10
- package/tools/plans.ts +40 -66
- package/tools/refine.ts +101 -164
- package/agents/criticizer.md +0 -18
- package/agents/executor.md +0 -26
- package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
- package/src/panel.ts +0 -473
- package/src/termination-prompt.ts +0 -73
- package/tests/goal-wait.test.ts +0 -269
- package/tests/panel-i-zero.test.ts +0 -420
- package/tests/panel.test.ts +0 -355
package/src/tasks.ts
ADDED
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Task-tree runtime model (v0.6.1): the execution phase tracks plan tasks
|
|
3
|
+
* (`## Tasks`) as the unit of progress. Status flows in exclusively through
|
|
4
|
+
* the task status tool; completion of the whole run is gated by the
|
|
5
|
+
* independent completion auditor over the plan's verification checks.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import type { CheckItem, PlanTasks, TaskNode, WaveEntry } from "./plan.ts";
|
|
9
|
+
import { extractTaskCoverage, flattenTasks, resolveTaskWaves } from "./plan.ts";
|
|
10
|
+
|
|
11
|
+
export type TaskStatus = "pending" | "complete" | "skipped";
|
|
12
|
+
|
|
13
|
+
export interface TaskProgress {
|
|
14
|
+
status: TaskStatus;
|
|
15
|
+
/** Evidence recorded with the completion/skip (tool call payload). */
|
|
16
|
+
evidence?: string;
|
|
17
|
+
/** Skip reason, present only for skipped tasks. */
|
|
18
|
+
skipReason?: string;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/** Runtime view of one task node: plan metadata merged with live status. */
|
|
22
|
+
export interface TaskView {
|
|
23
|
+
id: string;
|
|
24
|
+
title: string;
|
|
25
|
+
wave: number;
|
|
26
|
+
deps: string[];
|
|
27
|
+
files: string[];
|
|
28
|
+
status: TaskStatus;
|
|
29
|
+
evidence?: string;
|
|
30
|
+
skipReason?: string;
|
|
31
|
+
children: TaskView[];
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** Serializable progress map persisted in the run checkpoint. */
|
|
35
|
+
export type TaskProgressMap = Record<string, TaskProgress>;
|
|
36
|
+
|
|
37
|
+
/** Build the runtime task tree from parsed plan tasks + persisted progress. */
|
|
38
|
+
export function buildTaskView(planTasks: PlanTasks, progress?: TaskProgressMap): TaskView[] {
|
|
39
|
+
const waves = resolveTaskWaves(planTasks);
|
|
40
|
+
const build = (node: TaskNode): TaskView => {
|
|
41
|
+
const state = progress?.[node.id];
|
|
42
|
+
const children = node.children.map(build);
|
|
43
|
+
return {
|
|
44
|
+
id: node.id,
|
|
45
|
+
title: node.title,
|
|
46
|
+
wave: waves.get(node.id) ?? 1,
|
|
47
|
+
deps: node.deps,
|
|
48
|
+
files: node.files,
|
|
49
|
+
status: state?.status ?? "pending",
|
|
50
|
+
evidence: state?.evidence,
|
|
51
|
+
skipReason: state?.skipReason,
|
|
52
|
+
children,
|
|
53
|
+
};
|
|
54
|
+
};
|
|
55
|
+
return planTasks.tasks.map(build);
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export function flattenTaskViews(tasks: TaskView[]): TaskView[] {
|
|
59
|
+
const out: TaskView[] = [];
|
|
60
|
+
for (const t of tasks) {
|
|
61
|
+
out.push(t);
|
|
62
|
+
out.push(...flattenTaskViews(t.children));
|
|
63
|
+
}
|
|
64
|
+
return out;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** A task counts as terminal when it is skipped, or complete together with
|
|
68
|
+
* every child (parents close last; an open child reopens the parent for
|
|
69
|
+
* scheduling purposes even if the parent itself reported complete). */
|
|
70
|
+
export function taskIsTerminal(task: TaskView): boolean {
|
|
71
|
+
if (task.status === "skipped") return true;
|
|
72
|
+
if (task.status !== "complete") return false;
|
|
73
|
+
return task.children.every((child) => taskIsTerminal(child));
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
export function allTasksTerminal(tasks: TaskView[]): boolean {
|
|
77
|
+
return flattenTaskViews(tasks).length > 0 && flattenTaskViews(tasks).every((task) => taskIsTerminal(task));
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/** Snapshot of statuses for persistence and stall detection. */
|
|
81
|
+
export function taskProgressMap(tasks: TaskView[]): TaskProgressMap {
|
|
82
|
+
const map: TaskProgressMap = {};
|
|
83
|
+
for (const task of flattenTaskViews(tasks)) {
|
|
84
|
+
if (task.status === "pending" && task.evidence === undefined && task.skipReason === undefined) continue;
|
|
85
|
+
map[task.id] = {
|
|
86
|
+
status: task.status,
|
|
87
|
+
evidence: task.evidence,
|
|
88
|
+
skipReason: task.skipReason,
|
|
89
|
+
};
|
|
90
|
+
}
|
|
91
|
+
return map;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/** Progress aggregation for the dashboard: done counts terminal tasks. */
|
|
95
|
+
export function taskProgress(tasks: TaskView[]): { done: number; total: number } {
|
|
96
|
+
const flat = flattenTaskViews(tasks);
|
|
97
|
+
return {
|
|
98
|
+
done: flat.filter((task) => taskIsTerminal(task)).length,
|
|
99
|
+
total: flat.length,
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/** The current task: the first non-terminal task in wave order, then
|
|
104
|
+
* document order (top-level-first; subtasks close before their parent is
|
|
105
|
+
* re-listed). Deterministic; used for the ▸ anchor in the dashboard and
|
|
106
|
+
* the per-turn injection. */
|
|
107
|
+
export function currentTask(tasks: TaskView[]): TaskView | null {
|
|
108
|
+
const flat = flattenTaskViews(tasks);
|
|
109
|
+
const open = flat.filter((task) => !taskIsTerminal(task));
|
|
110
|
+
if (open.length === 0) return null;
|
|
111
|
+
open.sort((a, b) => (a.wave - b.wave) || (flat.indexOf(a) - flat.indexOf(b)));
|
|
112
|
+
return open[0] ?? null;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/** Tasks of one wave (top-level only — children belong to their parent). */
|
|
116
|
+
export function waveTasks(tasks: TaskView[], wave: number): TaskView[] {
|
|
117
|
+
return tasks.filter((task) => task.wave === wave);
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
export function maxWave(tasks: TaskView[]): number {
|
|
121
|
+
return flattenTaskViews(tasks).reduce((max, task) => Math.max(max, task.wave), 1);
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/** Legal status transitions for the task status tool: pending may close
|
|
125
|
+
* (complete/skipped); closed states are immutable except through the
|
|
126
|
+
* completion-audit flow (never through the task tool). */
|
|
127
|
+
export function canTransition(task: TaskView, next: TaskStatus): boolean {
|
|
128
|
+
if (task.status === next) return false;
|
|
129
|
+
if (task.status === "pending") return next === "complete" || next === "skipped";
|
|
130
|
+
return false;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/** Rollback set for a failed verification check: every task in its covers
|
|
134
|
+
* clause (parents cascade to their children, skipped tasks reopen too).
|
|
135
|
+
* Returns the ids that actually reopen. */
|
|
136
|
+
export function auditRollbackSet(
|
|
137
|
+
tasks: TaskView[],
|
|
138
|
+
checklist: CheckItem[],
|
|
139
|
+
vcId: string,
|
|
140
|
+
): string[] {
|
|
141
|
+
const vc = checklist.find((item) => item.id === vcId);
|
|
142
|
+
if (!vc) return [];
|
|
143
|
+
const covered = new Set(extractTaskCoverage(vc.text));
|
|
144
|
+
const flat = flattenTaskViews(tasks);
|
|
145
|
+
const reopen: string[] = [];
|
|
146
|
+
const reopenNode = (node: TaskView): void => {
|
|
147
|
+
if (node.status !== "pending") {
|
|
148
|
+
node.status = "pending";
|
|
149
|
+
node.evidence = undefined;
|
|
150
|
+
node.skipReason = undefined;
|
|
151
|
+
reopen.push(node.id);
|
|
152
|
+
}
|
|
153
|
+
for (const child of node.children) reopenNode(child);
|
|
154
|
+
};
|
|
155
|
+
for (const node of flat) {
|
|
156
|
+
if (covered.has(node.id)) reopenNode(node);
|
|
157
|
+
}
|
|
158
|
+
return reopen;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
/** Verification checks that cover no task are excluded from audit (their
|
|
162
|
+
* pass state cannot be derived from task statuses). */
|
|
163
|
+
export function auditableChecks(checklist: CheckItem[], tasks: TaskView[]): CheckItem[] {
|
|
164
|
+
const known = new Set(flattenTaskViews(tasks).map((task) => task.id));
|
|
165
|
+
return checklist.filter((item) => {
|
|
166
|
+
const covered = extractTaskCoverage(item.text);
|
|
167
|
+
return covered.length > 0 && covered.some((id) => known.has(id));
|
|
168
|
+
});
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
/** Checks with at least one skipped covered task whose remaining covered
|
|
172
|
+
* tasks are all complete (I-007: a skip whose siblings carry the delivery
|
|
173
|
+
* passes without an audit round). Pure-complete and pure-pending coverages
|
|
174
|
+
* are NOT presolved here — the former the audit affirms, the latter cannot
|
|
175
|
+
* arise once all tasks are terminal. */
|
|
176
|
+
export function skippedPassCheckIds(checklist: CheckItem[], tasks: TaskView[]): string[] {
|
|
177
|
+
const byId = new Map(flattenTaskViews(tasks).map((task) => [task.id, task]));
|
|
178
|
+
return auditableChecks(checklist, tasks)
|
|
179
|
+
.filter((item) => {
|
|
180
|
+
const covered = extractTaskCoverage(item.text).filter((id) => byId.has(id));
|
|
181
|
+
if (covered.length === 0) return false;
|
|
182
|
+
const statuses = covered.map((id) => byId.get(id)!.status);
|
|
183
|
+
return statuses.includes("skipped")
|
|
184
|
+
&& statuses.every((status) => status === "skipped" || status === "complete");
|
|
185
|
+
})
|
|
186
|
+
.map((item) => item.id);
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
export type { PlanTasks, TaskNode, WaveEntry };
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Thin thinking-level adapter over pi-ai's exported helpers.
|
|
3
|
+
*
|
|
4
|
+
* pi-ai already owns the level domain rules (`getSupportedThinkingLevels`:
|
|
5
|
+
* non-reasoning models only support "off"; a null mapping disables a level;
|
|
6
|
+
* xhigh/max appear only when explicitly mapped). Re-implementing them here
|
|
7
|
+
* would drift — this module only adds the pi-plans "default" sentinel and
|
|
8
|
+
* the stored-level resolution used by the reviewer spawn path (F-012).
|
|
9
|
+
*
|
|
10
|
+
* Semantics (decision 4 / F-009): `thinking_level: null` in the global
|
|
11
|
+
* reviewer config means DEFAULT — the child pi gets NO --thinking flag and
|
|
12
|
+
* resolves its own default chain (per-model settings → defaultThinkingLevel
|
|
13
|
+
* → medium, then model clamping). That is deliberately distinct from the
|
|
14
|
+
* explicit "off" level.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { getSupportedThinkingLevels, type Model, type ModelThinkingLevel } from "@earendil-works/pi-ai";
|
|
18
|
+
import type { GlobalRoleConfig, ThinkingLevelValue } from "./global-state.ts";
|
|
19
|
+
|
|
20
|
+
/** Panel/label sentinel for `thinking_level: null` (omit --thinking). */
|
|
21
|
+
export const DEFAULT_LEVEL_SENTINEL = "default";
|
|
22
|
+
|
|
23
|
+
/** Levels a model actually supports, via pi-ai (single source of truth). */
|
|
24
|
+
export function levelsForModel(model: Pick<Model<any>, "reasoning" | "thinkingLevelMap">): ModelThinkingLevel[] {
|
|
25
|
+
return getSupportedThinkingLevels(model);
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** True when the stored level is supported by the model (spawn passes it
|
|
29
|
+
* through verbatim; unsupported stored levels are clamped by the child). */
|
|
30
|
+
export function isLevelSupported(
|
|
31
|
+
model: Pick<Model<any>, "reasoning" | "thinkingLevelMap">,
|
|
32
|
+
level: string | null,
|
|
33
|
+
): boolean {
|
|
34
|
+
if (level === null) return true; // default: no flag, always valid
|
|
35
|
+
return levelsForModel(model).includes(level as ModelThinkingLevel);
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/** Resolve the CLI argument for the stored level: null → omit the flag. */
|
|
39
|
+
export function spawnThinkingFlag(level: ThinkingLevelValue | null): string | null {
|
|
40
|
+
return level ?? null;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/** Display label for the reviewer overlay and subagents ledger. */
|
|
44
|
+
export function roleModelLabel(modelSelector: string, thinkingLevel: ThinkingLevelValue | null): string {
|
|
45
|
+
return `${modelSelector}:${thinkingLevel ?? DEFAULT_LEVEL_SENTINEL}`;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/** Human-facing one-liner for the default row / docs (F-009): the default
|
|
49
|
+
* is the child pi's own chain, NOT the current session's level. */
|
|
50
|
+
export const DEFAULT_LEVEL_DESCRIPTION =
|
|
51
|
+
"no --thinking flag: the child pi resolves its default (per-model settings → defaultThinkingLevel → medium)";
|
|
52
|
+
|
|
53
|
+
/** Spawn-view of a confirmed reviewer role: concrete model + optional level. */
|
|
54
|
+
export interface ResolvedReviewerSpawn {
|
|
55
|
+
modelSelector: string;
|
|
56
|
+
thinkingLevel: ThinkingLevelValue | null;
|
|
57
|
+
label: string;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export function resolveReviewerSpawn(role: Pick<GlobalRoleConfig, "model_selector" | "thinking_level">): ResolvedReviewerSpawn {
|
|
61
|
+
const modelSelector = role.model_selector ?? "";
|
|
62
|
+
return {
|
|
63
|
+
modelSelector,
|
|
64
|
+
thinkingLevel: role.thinking_level ?? null,
|
|
65
|
+
label: modelSelector ? roleModelLabel(modelSelector, role.thinking_level ?? null) : "unconfirmed",
|
|
66
|
+
};
|
|
67
|
+
}
|
package/src/ui-language.ts
CHANGED
|
@@ -1,10 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Single source of truth for user-visible UI chrome language (issue #3).
|
|
3
3
|
*
|
|
4
|
-
* Workspace `language.tag` (`.git/
|
|
5
|
-
* strings for every user-visible surface: the batch form (src/ask-form.ts)
|
|
6
|
-
* the refine/refs overlay footer (src/refine-ui.ts)
|
|
7
|
-
* (src/panel.ts) and the execution status line (src/exec.ts).
|
|
4
|
+
* Workspace `language.tag` (`.git/pi-plans/config.json`) selects the chrome
|
|
5
|
+
* strings for every user-visible surface: the batch form (src/ask-form.ts)
|
|
6
|
+
* and the refine/refs overlay footer (src/refine-ui.ts).
|
|
8
7
|
*
|
|
9
8
|
* Mapping follows RFC 4647 primary-subtag fallback (zh-Hant-CN → zh-Hant →
|
|
10
9
|
* zh), so every `zh*` tag renders the existing Simplified strings verbatim.
|
|
@@ -125,54 +124,4 @@ export function refineChrome(lang: UiLanguage): RefineChrome {
|
|
|
125
124
|
return REFINE_CHROME[lang];
|
|
126
125
|
}
|
|
127
126
|
|
|
128
|
-
// ---------------------------------------------------------------------------
|
|
129
|
-
// Status panel chrome (src/panel.ts) — implWarning branch only.
|
|
130
|
-
// ---------------------------------------------------------------------------
|
|
131
|
-
|
|
132
|
-
export interface PanelChrome {
|
|
133
|
-
badgeStatus(topic: string, vcDone: number, vcTotal: number): string;
|
|
134
|
-
progressWarning(): string;
|
|
135
|
-
summaryWarning(topic: string, vcDone: number, vcTotal: number, nextAction: string): string;
|
|
136
|
-
}
|
|
137
|
-
|
|
138
|
-
const PANEL_CHROME: Record<UiLanguage, PanelChrome> = {
|
|
139
|
-
zh: {
|
|
140
|
-
badgeStatus: (topic, vcDone, vcTotal) => `${topic} · ⚠ I 解析 0 项 · VC ${vcDone}/${vcTotal}`,
|
|
141
|
-
progressWarning: () => `⚠ plan 格式:Implementation Items 解析 0 项(面板无法计 I 进度)`,
|
|
142
|
-
summaryWarning: (topic, vcDone, vcTotal, nextAction) =>
|
|
143
|
-
`plans: ${topic} ▸ ⚠ Implementation Items 解析 0 项 · VC ${vcDone}/${vcTotal} · next: ${nextAction}`,
|
|
144
|
-
},
|
|
145
|
-
en: {
|
|
146
|
-
badgeStatus: (topic, vcDone, vcTotal) => `${topic} · ⚠ I parse 0 items · VC ${vcDone}/${vcTotal}`,
|
|
147
|
-
progressWarning: () => `⚠ plan format: Implementation Items parsed 0 items (I progress cannot be counted)`,
|
|
148
|
-
summaryWarning: (topic, vcDone, vcTotal, nextAction) =>
|
|
149
|
-
`plans: ${topic} ▸ ⚠ Implementation Items parsed 0 items · VC ${vcDone}/${vcTotal} · next: ${nextAction}`,
|
|
150
|
-
},
|
|
151
|
-
};
|
|
152
|
-
|
|
153
|
-
export function panelChrome(lang: UiLanguage): PanelChrome {
|
|
154
|
-
return PANEL_CHROME[lang];
|
|
155
|
-
}
|
|
156
|
-
|
|
157
|
-
// ---------------------------------------------------------------------------
|
|
158
|
-
// Execution status line chrome (src/exec.ts) — goal-wait segment.
|
|
159
|
-
// ---------------------------------------------------------------------------
|
|
160
|
-
|
|
161
|
-
export interface ExecChrome {
|
|
162
|
-
goalWait(noProgressRounds: number, waitRounds: number): string;
|
|
163
|
-
}
|
|
164
|
-
|
|
165
|
-
const EXEC_CHROME: Record<UiLanguage, ExecChrome> = {
|
|
166
|
-
zh: {
|
|
167
|
-
goalWait: (noProgressRounds, waitRounds) =>
|
|
168
|
-
` · 🔁 goal-wait · 无进展 ${noProgressRounds}/3 · 等待 ${waitRounds}/6`,
|
|
169
|
-
},
|
|
170
|
-
en: {
|
|
171
|
-
goalWait: (noProgressRounds, waitRounds) =>
|
|
172
|
-
` · 🔁 goal-wait · no progress ${noProgressRounds}/3 · waiting ${waitRounds}/6`,
|
|
173
|
-
},
|
|
174
|
-
};
|
|
175
127
|
|
|
176
|
-
export function execChrome(lang: UiLanguage): ExecChrome {
|
|
177
|
-
return EXEC_CHROME[lang];
|
|
178
|
-
}
|
package/src/workflow-state.ts
CHANGED
|
@@ -145,8 +145,15 @@ export interface ExecutionCheckpoint {
|
|
|
145
145
|
/** True when this approval/progress was produced in a different (origin) worktree. */
|
|
146
146
|
originWorktree?: string;
|
|
147
147
|
/** v0.6.0: set while a delegated executor child owns the implementation;
|
|
148
|
-
* stale after a restart (orphaned delegate — the child died with the parent).
|
|
148
|
+
* stale after a restart (orphaned delegate — the child died with the parent).
|
|
149
|
+
* Removed with delegated execution in v0.6.1; read-tolerated on legacy checkpoints. */
|
|
149
150
|
delegate?: { modelSelector: string; startedAt: string };
|
|
151
|
+
/** v0.6.1: task-tree progress (task id → status/evidence), the primary
|
|
152
|
+
* progress record. doneVcIds stays for legacy checkpoints and the final
|
|
153
|
+
* audit pass. */
|
|
154
|
+
tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
|
|
155
|
+
/** v0.6.1: completion-audit bookkeeping (rounds, last failed set, pass). */
|
|
156
|
+
audit?: { rounds: number; lastResult?: string; passed?: boolean };
|
|
150
157
|
}
|
|
151
158
|
|
|
152
159
|
export interface OwnerInfo {
|
|
@@ -454,7 +461,7 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
454
461
|
const record = asRecord(value, label);
|
|
455
462
|
rejectExtraKeys(
|
|
456
463
|
record,
|
|
457
|
-
new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate"]),
|
|
464
|
+
new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "audit"]),
|
|
458
465
|
label,
|
|
459
466
|
);
|
|
460
467
|
const execution: ExecutionCheckpoint = {
|
|
@@ -484,6 +491,30 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
484
491
|
startedAt: asString(delegate.startedAt, `${label}.delegate.startedAt`),
|
|
485
492
|
};
|
|
486
493
|
}
|
|
494
|
+
if (record.tasks !== undefined && record.tasks !== null) {
|
|
495
|
+
const tasksRecord = asRecord(record.tasks, `${label}.tasks`);
|
|
496
|
+
const tasks: NonNullable<ExecutionCheckpoint["tasks"]> = {};
|
|
497
|
+
for (const [id, raw] of Object.entries(tasksRecord)) {
|
|
498
|
+
const entry = asRecord(raw, `${label}.tasks.${id}`);
|
|
499
|
+
rejectExtraKeys(entry, new Set(["status", "evidence", "skipReason"]), `${label}.tasks.${id}`);
|
|
500
|
+
const status = asEnum(entry.status, new Set(["pending", "complete", "skipped"]), `${label}.tasks.${id}.status`);
|
|
501
|
+
const item: { status: string; evidence?: string; skipReason?: string } = { status };
|
|
502
|
+
if (entry.evidence !== undefined) item.evidence = asString(entry.evidence, `${label}.tasks.${id}.evidence`);
|
|
503
|
+
if (entry.skipReason !== undefined) item.skipReason = asString(entry.skipReason, `${label}.tasks.${id}.skipReason`);
|
|
504
|
+
tasks[id] = item;
|
|
505
|
+
}
|
|
506
|
+
execution.tasks = tasks;
|
|
507
|
+
}
|
|
508
|
+
if (record.audit !== undefined && record.audit !== null) {
|
|
509
|
+
const audit = asRecord(record.audit, `${label}.audit`);
|
|
510
|
+
rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed"]), `${label}.audit`);
|
|
511
|
+
const parsed: { rounds: number; lastResult?: string; passed?: boolean } = {
|
|
512
|
+
rounds: asInt(audit.rounds, `${label}.audit.rounds`, 0),
|
|
513
|
+
};
|
|
514
|
+
if (audit.lastResult !== undefined) parsed.lastResult = asString(audit.lastResult, `${label}.audit.lastResult`);
|
|
515
|
+
if (audit.passed !== undefined) parsed.passed = asBool(audit.passed, `${label}.audit.passed`);
|
|
516
|
+
execution.audit = parsed;
|
|
517
|
+
}
|
|
487
518
|
return execution;
|
|
488
519
|
}
|
|
489
520
|
|
|
@@ -1000,19 +1031,6 @@ export function applyReviewConsolidated(cp: WorkflowCheckpoint, roundId: string,
|
|
|
1000
1031
|
return { ...cp, reviewRounds: cp.reviewRounds.map((entry) => (entry.roundId === roundId ? nextRound : entry)) };
|
|
1001
1032
|
}
|
|
1002
1033
|
|
|
1003
|
-
/** F-004: `completed` requires explicit evidence; approval cannot be forged by state writes. */
|
|
1004
|
-
export function applyCompleted(cp: WorkflowCheckpoint, evidence: string): WorkflowCheckpoint {
|
|
1005
|
-
if (cp.phase !== "implementation-review") {
|
|
1006
|
-
throw new StateError(`cannot complete from phase "${cp.phase}"`);
|
|
1007
|
-
}
|
|
1008
|
-
const review = cp.implementationReview;
|
|
1009
|
-
if (!review || review.terminationCondition === undefined) {
|
|
1010
|
-
throw new StateError("cannot complete without a recorded termination condition");
|
|
1011
|
-
}
|
|
1012
|
-
if (evidence.trim() === "") throw new StateError("completion requires non-empty evidence");
|
|
1013
|
-
return { ...cp, phase: "completed", nextAction: "none" };
|
|
1014
|
-
}
|
|
1015
|
-
|
|
1016
1034
|
export function applyExecutionApproved(cp: WorkflowCheckpoint, approval: ExecutionApproval): WorkflowCheckpoint {
|
|
1017
1035
|
// F-007 (implementation review): terminal and review phases cannot approve
|
|
1018
1036
|
// execution; a re-approval (stop/migration reset approval to null) is legal
|
|
@@ -1036,6 +1054,8 @@ export function applyExecutionApproved(cp: WorkflowCheckpoint, approval: Executi
|
|
|
1036
1054
|
doneVcIds: [],
|
|
1037
1055
|
implStatus: {},
|
|
1038
1056
|
usage: { inToks: 0, outToks: 0 },
|
|
1057
|
+
tasks: {},
|
|
1058
|
+
audit: { rounds: 0 },
|
|
1039
1059
|
},
|
|
1040
1060
|
};
|
|
1041
1061
|
}
|
|
@@ -1048,6 +1068,10 @@ export interface ExecutionProgressInput {
|
|
|
1048
1068
|
pausedReason?: string | null;
|
|
1049
1069
|
/** v0.6.0: set/clear the delegated-executor record; null clears it. */
|
|
1050
1070
|
delegate?: { modelSelector: string; startedAt: string } | null;
|
|
1071
|
+
/** v0.6.1: task-tree progress snapshot (authoritative). */
|
|
1072
|
+
tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
|
|
1073
|
+
/** v0.6.1: completion-audit bookkeeping update. */
|
|
1074
|
+
audit?: { rounds: number; lastResult?: string; passed?: boolean };
|
|
1051
1075
|
}
|
|
1052
1076
|
|
|
1053
1077
|
export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: ExecutionProgressInput): WorkflowCheckpoint {
|
|
@@ -1066,6 +1090,8 @@ export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: Executi
|
|
|
1066
1090
|
else if (progress.pausedReason !== undefined) execution.pausedReason = progress.pausedReason;
|
|
1067
1091
|
if (progress.delegate === null) delete execution.delegate;
|
|
1068
1092
|
else if (progress.delegate !== undefined) execution.delegate = progress.delegate;
|
|
1093
|
+
if (progress.tasks !== undefined) execution.tasks = progress.tasks;
|
|
1094
|
+
if (progress.audit !== undefined) execution.audit = progress.audit;
|
|
1069
1095
|
return { ...cp, execution };
|
|
1070
1096
|
}
|
|
1071
1097
|
|
|
@@ -1077,45 +1103,18 @@ export function applyExecutionHeadChanged(cp: WorkflowCheckpoint): WorkflowCheck
|
|
|
1077
1103
|
|
|
1078
1104
|
export function applyExecutionCompleted(cp: WorkflowCheckpoint): WorkflowCheckpoint {
|
|
1079
1105
|
if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
|
|
1106
|
+
// v0.6.1 (D-018): the post-execution amelioration loop is gone; a
|
|
1107
|
+
// completed audit passes the run straight to the terminal phase.
|
|
1080
1108
|
return {
|
|
1081
1109
|
...cp,
|
|
1082
|
-
phase: "
|
|
1083
|
-
nextAction: "
|
|
1084
|
-
execution: {
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
cp: WorkflowCheckpoint,
|
|
1090
|
-
terminationCondition: string,
|
|
1091
|
-
reviewerCount?: number,
|
|
1092
|
-
): WorkflowCheckpoint {
|
|
1093
|
-
if (cp.phase !== "implementation-review") throw new StateError("requires phase \"implementation-review\"");
|
|
1094
|
-
if (cp.implementationReview?.terminationCondition !== undefined) {
|
|
1095
|
-
throw new StateError("termination condition already configured; do not re-ask");
|
|
1096
|
-
}
|
|
1097
|
-
return {
|
|
1098
|
-
...cp,
|
|
1099
|
-
implementationReview: {
|
|
1100
|
-
terminationCondition,
|
|
1101
|
-
reviewerCount,
|
|
1102
|
-
completedRounds: cp.implementationReview?.completedRounds ?? 0,
|
|
1110
|
+
phase: "completed",
|
|
1111
|
+
nextAction: "none",
|
|
1112
|
+
execution: {
|
|
1113
|
+
...cp.execution,
|
|
1114
|
+
pausedReason: undefined,
|
|
1115
|
+
delegate: undefined,
|
|
1116
|
+
audit: { ...(cp.execution.audit ?? { rounds: 0 }), passed: true },
|
|
1103
1117
|
},
|
|
1104
|
-
nextAction: "run-review",
|
|
1105
|
-
};
|
|
1106
|
-
}
|
|
1107
|
-
|
|
1108
|
-
export function applyImplementationRoundFinished(cp: WorkflowCheckpoint): WorkflowCheckpoint {
|
|
1109
|
-
if (cp.phase !== "implementation-review" || !cp.implementationReview) {
|
|
1110
|
-
throw new StateError("requires phase \"implementation-review\"");
|
|
1111
|
-
}
|
|
1112
|
-
const review = cp.implementationReview;
|
|
1113
|
-
if (review.currentRoundId === undefined) throw new StateError("no current round to finish");
|
|
1114
|
-
const round = cp.reviewRounds.find((entry) => entry.roundId === review.currentRoundId);
|
|
1115
|
-
if (!round || !round.consolidated) throw new StateError("current round is not consolidated");
|
|
1116
|
-
return {
|
|
1117
|
-
...cp,
|
|
1118
|
-
implementationReview: { ...review, completedRounds: review.completedRounds + 1, currentRoundId: undefined },
|
|
1119
1118
|
};
|
|
1120
1119
|
}
|
|
1121
1120
|
|
|
@@ -1135,12 +1134,16 @@ export function applyMigration(
|
|
|
1135
1134
|
migration: { fromWorktree: cp.worktreeRoot, migratedAt: utcNow() },
|
|
1136
1135
|
execution: cp.execution
|
|
1137
1136
|
? {
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1137
|
+
approval: null,
|
|
1138
|
+
doneVcIds: [],
|
|
1139
|
+
implStatus: {},
|
|
1140
|
+
usage: cp.execution.usage,
|
|
1141
|
+
originWorktree: cp.execution.originWorktree ?? cp.worktreeRoot,
|
|
1142
|
+
// v0.6.1: task progress and audit state do not survive a worktree
|
|
1143
|
+
// migration (same rule as VC validity).
|
|
1144
|
+
tasks: {},
|
|
1145
|
+
audit: { rounds: 0 },
|
|
1146
|
+
}
|
|
1144
1147
|
: undefined,
|
|
1145
1148
|
implementationReview: cp.implementationReview
|
|
1146
1149
|
? {
|
|
@@ -1164,7 +1167,9 @@ function migrationNextAction(cp: WorkflowCheckpoint): NextAction {
|
|
|
1164
1167
|
export function applyExecutionStopped(cp: WorkflowCheckpoint, reason: string): WorkflowCheckpoint {
|
|
1165
1168
|
if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
|
|
1166
1169
|
const { delegate: _delegate, ...execution } = cp.execution;
|
|
1167
|
-
|
|
1170
|
+
// A stop also revokes any outstanding audit-rollback authorization.
|
|
1171
|
+
const audit = execution.audit ? { ...execution.audit } : undefined;
|
|
1172
|
+
return { ...cp, execution: { ...execution, audit, pausedReason: reason } };
|
|
1168
1173
|
}
|
|
1169
1174
|
|
|
1170
1175
|
// ---------------------------------------------------------------------------
|
|
@@ -12,13 +12,22 @@ import { initState, setRole, setLanguage, startRun, readActive } from "../src/st
|
|
|
12
12
|
const ROOT = path.dirname(path.dirname(url.fileURLToPath(import.meta.url)));
|
|
13
13
|
|
|
14
14
|
let tmpRoot: string;
|
|
15
|
+
let globalDir: string;
|
|
16
|
+
let previousGlobalDir: string | undefined;
|
|
15
17
|
|
|
16
18
|
before(() => {
|
|
17
19
|
tmpRoot = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-analyze-refs-"));
|
|
20
|
+
// Isolate the global reviewer config (F-002): never touch ~/.pi/pi-plans.
|
|
21
|
+
previousGlobalDir = process.env.PI_PLANS_GLOBAL_DIR;
|
|
22
|
+
globalDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-global-analyze-refs-"));
|
|
23
|
+
process.env.PI_PLANS_GLOBAL_DIR = globalDir;
|
|
18
24
|
});
|
|
19
25
|
|
|
20
26
|
after(() => {
|
|
27
|
+
if (previousGlobalDir === undefined) delete process.env.PI_PLANS_GLOBAL_DIR;
|
|
28
|
+
else process.env.PI_PLANS_GLOBAL_DIR = previousGlobalDir;
|
|
21
29
|
fs.rmSync(tmpRoot, { recursive: true, force: true });
|
|
30
|
+
fs.rmSync(globalDir, { recursive: true, force: true });
|
|
22
31
|
});
|
|
23
32
|
|
|
24
33
|
function mkWorkdir(name: string): string {
|
|
@@ -102,7 +111,7 @@ function subagentLines(workdir: string): Array<any> {
|
|
|
102
111
|
.map((line) => JSON.parse(line));
|
|
103
112
|
}
|
|
104
113
|
|
|
105
|
-
describe("analyze_refs gates", () => {
|
|
114
|
+
describe("analyze_refs gates", () => {
|
|
106
115
|
it("refuses when no pi-plans state exists", async () => {
|
|
107
116
|
const workdir = mkWorkdir("gates-no-state");
|
|
108
117
|
const tool = loadTool();
|
|
@@ -112,35 +121,43 @@ describe("analyze_refs gates", () => {
|
|
|
112
121
|
);
|
|
113
122
|
});
|
|
114
123
|
|
|
115
|
-
it("refuses with
|
|
116
|
-
const workdir = mkWorkdir("gates-
|
|
124
|
+
it("refuses with embedded text guidance when the reviewer model is unconfirmed (headless)", async () => {
|
|
125
|
+
const workdir = mkWorkdir("gates-unconfirmed");
|
|
117
126
|
initState(workdir);
|
|
118
|
-
const configPath = path.join(workdir, ".git", "pi_plans", "config.json");
|
|
119
|
-
const config = JSON.parse(fs.readFileSync(configPath, "utf8"));
|
|
120
|
-
config.reviewer.mode = "bogus";
|
|
121
|
-
fs.writeFileSync(configPath, `${JSON.stringify(config, null, "\t")}\n`, "utf8");
|
|
122
127
|
const tool = loadTool();
|
|
123
128
|
await assert.rejects(
|
|
124
129
|
tool.execute("c1", { refs: [{ id: "ref-1", localPath: "." }] }, undefined, undefined, headlessCtx(workdir)),
|
|
125
|
-
/
|
|
130
|
+
/model was never confirmed/,
|
|
126
131
|
);
|
|
127
132
|
});
|
|
128
133
|
|
|
129
|
-
it("
|
|
134
|
+
it("ignores the reviewer mode entirely (Q-4=B): current-session proceeds once a model is confirmed", async () => {
|
|
130
135
|
const workdir = mkWorkdir("gates-current-session");
|
|
131
136
|
initState(workdir);
|
|
132
|
-
setRole(workdir, { role: "reviewer", mode: "current-session" });
|
|
133
|
-
const
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
+
setRole(workdir, { role: "reviewer", mode: "current-session", modelSelector: "fake/model", confirmed: true });
|
|
138
|
+
const refDir = path.join(workdir, "refs", "solo");
|
|
139
|
+
fs.mkdirSync(refDir, { recursive: true });
|
|
140
|
+
const restore = withFakePi(
|
|
141
|
+
fakePiScript(
|
|
142
|
+
`emit({ type: "message_end", message: { role: "assistant", model: "fake/model", content: [{ type: "text", text: "OK" }] } });`,
|
|
143
|
+
),
|
|
137
144
|
);
|
|
145
|
+
const tool = loadTool();
|
|
146
|
+
try {
|
|
147
|
+
const result = await tool.execute("c1", { refs: [{ id: "ref-1", localPath: refDir }] }, undefined, undefined, headlessCtx(workdir));
|
|
148
|
+
assert.match(result.content[0]!.text, /mode is current-session, but analyze_refs always spawns/);
|
|
149
|
+
assert.match(result.content[0]!.text, /OK/);
|
|
150
|
+
} finally {
|
|
151
|
+
restore();
|
|
152
|
+
}
|
|
138
153
|
});
|
|
139
154
|
|
|
140
|
-
it("
|
|
141
|
-
const workdir = mkWorkdir("gates-unconfirmed");
|
|
155
|
+
it("requires a confirmed model even in current-session mode (analysis always spawns)", async () => {
|
|
156
|
+
const workdir = mkWorkdir("gates-current-session-unconfirmed");
|
|
142
157
|
initState(workdir);
|
|
143
|
-
|
|
158
|
+
// 'inherit' fully resets selector AND confirmation — the prior test's
|
|
159
|
+
// confirmed selector must not carry over (the global config is shared).
|
|
160
|
+
setRole(workdir, { role: "reviewer", mode: "current-session", modelSelector: "inherit" });
|
|
144
161
|
const tool = loadTool();
|
|
145
162
|
await assert.rejects(
|
|
146
163
|
tool.execute("c1", { refs: [{ id: "ref-1", localPath: "." }] }, undefined, undefined, headlessCtx(workdir)),
|
|
@@ -310,7 +327,7 @@ describe("analyze_refs fanout", () => {
|
|
|
310
327
|
const result = await tool.execute("c1", { refs: [{ id: "ref-1", localPath: refDir }] }, undefined, undefined, headlessCtx(workdir));
|
|
311
328
|
assert.match(result.content[0]!.text, /### pi-plans-refs-adhoc-ref-1/);
|
|
312
329
|
assert.equal(readActive(workdir), null);
|
|
313
|
-
const ledger = path.join(workdir, ".git", "
|
|
330
|
+
const ledger = path.join(workdir, ".git", "pi-plans", "runs");
|
|
314
331
|
const runs = fs.existsSync(ledger) ? fs.readdirSync(ledger) : [];
|
|
315
332
|
const spawnFiles = runs.flatMap((run) =>
|
|
316
333
|
fs.existsSync(path.join(ledger, run, "subagents.jsonl")) ? [fs.readFileSync(path.join(ledger, run, "subagents.jsonl"), "utf8")] : [],
|
|
@@ -157,18 +157,6 @@ describe("VC-2 · hardening does not wound legitimate single-question shapes", (
|
|
|
157
157
|
assert.equal(Value.Check(AskChoiceParams as never, wellFormed), true);
|
|
158
158
|
});
|
|
159
159
|
|
|
160
|
-
it("single question with trailing + autoComplete:false passes Value.Check", () => {
|
|
161
|
-
const wellFormed = {
|
|
162
|
-
question: "实现评审循环如何终止?",
|
|
163
|
-
options: [
|
|
164
|
-
{ label: "goal wait:直到无未通过 VC", recommended: true },
|
|
165
|
-
{ label: "直到无高危发现(硬帽 5 轮)" },
|
|
166
|
-
],
|
|
167
|
-
trailing: "auto-refine-loop",
|
|
168
|
-
autoComplete: false,
|
|
169
|
-
};
|
|
170
|
-
assert.equal(Value.Check(AskChoiceParams as never, wellFormed), true);
|
|
171
|
-
});
|
|
172
160
|
});
|
|
173
161
|
|
|
174
162
|
describe("VC-3 · zero-recommended batches reject loudly with zero side effects", () => {
|