pi-plans 0.6.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +58 -0
- package/CONTRIBUTING.md +8 -15
- package/README.md +39 -37
- package/agents/execution-reviewer.md +40 -0
- package/agents/reviewer.md +12 -3
- package/index.ts +55 -58
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +50 -60
- package/references/plan-artifact-template.md +81 -60
- package/references/state-and-config.md +60 -44
- package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
- package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
- package/scripts/run-tests.ts +12 -1
- package/scripts/validate.ts +39 -11
- package/skills/debug-and-plan/SKILL.md +3 -3
- package/skills/plan-big/SKILL.md +3 -3
- package/skills/plan-normal/SKILL.md +3 -3
- package/skills/plan-small/SKILL.md +4 -4
- package/skills/plan-with-refs/SKILL.md +6 -6
- package/skills/planning/SKILL.md +1 -1
- package/src/ask-form.ts +4 -4
- package/src/auditor.ts +227 -0
- package/src/auto-approve.ts +1 -1
- package/src/autocomplete.ts +19 -17
- package/src/code-graph/commands.ts +8 -3
- package/src/code-graph/community.ts +1 -1
- package/src/code-graph/paths.ts +1 -1
- package/src/code-graph/watch.ts +2 -2
- package/src/compaction.ts +3 -3
- package/src/config-command.ts +146 -73
- package/src/dashboard.ts +303 -0
- package/src/exec.ts +1185 -924
- package/src/global-state.ts +304 -0
- package/src/guard.ts +18 -19
- package/src/messaging.ts +44 -0
- package/src/plan.ts +421 -112
- package/src/query-hook.ts +4 -4
- package/src/refine-prompts.ts +12 -70
- package/src/refine-ui-helpers.ts +24 -5
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +19 -3
- package/src/resume-command.ts +45 -129
- package/src/resume.ts +5 -1
- package/src/role-panels.ts +542 -0
- package/src/run-context.ts +3 -10
- package/src/staleness.ts +53 -0
- package/src/state.ts +273 -72
- package/src/subagent.ts +19 -29
- package/src/task-tool.ts +100 -0
- package/src/tasks.ts +223 -0
- package/src/thinking-levels.ts +67 -0
- package/src/ui-language.ts +7 -54
- package/src/workflow-state.ts +76 -58
- package/tests/analyze-refs.test.ts +35 -18
- package/tests/ask-choice-schema.test.ts +0 -12
- package/tests/ask-choice.test.ts +2 -49
- package/tests/ask-form-tool.test.ts +4 -5
- package/tests/ask-form.test.ts +2 -2
- package/tests/auditor.test.ts +210 -0
- package/tests/auto-approve.test.ts +7 -10
- package/tests/autocomplete.test.ts +8 -11
- package/tests/code-graph-apply-action.test.ts +2 -2
- package/tests/code-graph-commands.test.ts +2 -2
- package/tests/code-graph-index.test.ts +2 -2
- package/tests/code-graph-loop.e2e.test.ts +1 -1
- package/tests/code-graph-mutations.test.ts +1 -1
- package/tests/code-graph-rollback.test.ts +1 -1
- package/tests/code-graph-v05.test.ts +2 -2
- package/tests/compaction.test.ts +1 -1
- package/tests/config-command.test.ts +103 -100
- package/tests/dashboard.test.ts +402 -0
- package/tests/exec-lifecycle.test.ts +181 -115
- package/tests/exec-panel-lifecycle.test.ts +106 -251
- package/tests/exec-review-loop.test.ts +331 -0
- package/tests/exec.test.ts +771 -1706
- package/tests/execute-plan.test.ts +44 -19
- package/tests/extension-load.test.ts +48 -0
- package/tests/global-state.test.ts +371 -0
- package/tests/graph-aware-file-tools.test.ts +5 -5
- package/tests/guard.test.ts +1 -1
- package/tests/multi-run.test.ts +3 -103
- package/tests/plan.test.ts +139 -62
- package/tests/plans.test.ts +7 -79
- package/tests/refine-prompts.test.ts +20 -71
- package/tests/refine-resume.test.ts +27 -22
- package/tests/refine-ui.test.ts +6 -15
- package/tests/resume-lifecycle.test.ts +41 -22
- package/tests/resume.test.ts +39 -81
- package/tests/role-panels.test.ts +391 -0
- package/tests/run-context.test.ts +1 -1
- package/tests/run-ownership.test.ts +1 -1
- package/tests/stale-ctx.test.ts +218 -0
- package/tests/staleness.test.ts +76 -0
- package/tests/state.test.ts +155 -32
- package/tests/subagent-thinking.test.ts +65 -0
- package/tests/subagent-usage.test.ts +1 -1
- package/tests/task-tool.test.ts +61 -0
- package/tests/tasks.test.ts +142 -0
- package/tests/thinking-levels.test.ts +77 -0
- package/tests/ui-language.test.ts +2 -17
- package/tests/workflow-state.test.ts +73 -90
- package/tools/analyze-refs.ts +67 -32
- package/tools/ask-choice.ts +7 -53
- package/tools/code-graph.ts +2 -2
- package/tools/execute-plan.ts +55 -99
- package/tools/graph-aware-file-tools.ts +4 -10
- package/tools/plans.ts +41 -67
- package/tools/refine.ts +101 -164
- package/agents/criticizer.md +0 -18
- package/agents/executor.md +0 -26
- package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
- package/src/panel.ts +0 -473
- package/src/termination-prompt.ts +0 -73
- package/tests/goal-wait.test.ts +0 -269
- package/tests/panel-i-zero.test.ts +0 -420
- package/tests/panel.test.ts +0 -355
package/src/tasks.ts
ADDED
|
@@ -0,0 +1,223 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Task-tree runtime model (v0.6.1): the execution phase tracks plan tasks
|
|
3
|
+
* (`## Tasks`) as the unit of progress. Status flows in exclusively through
|
|
4
|
+
* the task status tool; completion of the whole run is gated by the
|
|
5
|
+
* independent execution reviewer over the plan's verification checks.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import type { CheckItem, PlanTasks, TaskNode, WaveEntry } from "./plan.ts";
|
|
9
|
+
import { extractTaskCoverage, flattenTasks, resolveTaskWaves } from "./plan.ts";
|
|
10
|
+
|
|
11
|
+
export type TaskStatus = "pending" | "complete" | "skipped";
|
|
12
|
+
|
|
13
|
+
export interface TaskProgress {
|
|
14
|
+
status: TaskStatus;
|
|
15
|
+
/** Evidence recorded with the completion/skip (tool call payload). */
|
|
16
|
+
evidence?: string;
|
|
17
|
+
/** Skip reason, present only for skipped tasks. */
|
|
18
|
+
skipReason?: string;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/** Runtime view of one task node: plan metadata merged with live status. */
|
|
22
|
+
export interface TaskView {
|
|
23
|
+
id: string;
|
|
24
|
+
title: string;
|
|
25
|
+
wave: number;
|
|
26
|
+
deps: string[];
|
|
27
|
+
files: string[];
|
|
28
|
+
status: TaskStatus;
|
|
29
|
+
evidence?: string;
|
|
30
|
+
skipReason?: string;
|
|
31
|
+
children: TaskView[];
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** Serializable progress map persisted in the run checkpoint. */
|
|
35
|
+
export type TaskProgressMap = Record<string, TaskProgress>;
|
|
36
|
+
|
|
37
|
+
/** Build the runtime task tree from parsed plan tasks + persisted progress. */
|
|
38
|
+
export function buildTaskView(planTasks: PlanTasks, progress?: TaskProgressMap): TaskView[] {
|
|
39
|
+
const waves = resolveTaskWaves(planTasks);
|
|
40
|
+
const build = (node: TaskNode): TaskView => {
|
|
41
|
+
const state = progress?.[node.id];
|
|
42
|
+
const children = node.children.map(build);
|
|
43
|
+
return {
|
|
44
|
+
id: node.id,
|
|
45
|
+
title: node.title,
|
|
46
|
+
wave: waves.get(node.id) ?? 1,
|
|
47
|
+
deps: node.deps,
|
|
48
|
+
files: node.files,
|
|
49
|
+
status: state?.status ?? "pending",
|
|
50
|
+
evidence: state?.evidence,
|
|
51
|
+
skipReason: state?.skipReason,
|
|
52
|
+
children,
|
|
53
|
+
};
|
|
54
|
+
};
|
|
55
|
+
return planTasks.tasks.map(build);
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export function flattenTaskViews(tasks: TaskView[]): TaskView[] {
|
|
59
|
+
const out: TaskView[] = [];
|
|
60
|
+
for (const t of tasks) {
|
|
61
|
+
out.push(t);
|
|
62
|
+
out.push(...flattenTaskViews(t.children));
|
|
63
|
+
}
|
|
64
|
+
return out;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** A task counts as terminal when it is skipped, or complete together with
|
|
68
|
+
* every child (parents close last; an open child reopens the parent for
|
|
69
|
+
* scheduling purposes even if the parent itself reported complete). */
|
|
70
|
+
export function taskIsTerminal(task: TaskView): boolean {
|
|
71
|
+
if (task.status === "skipped") return true;
|
|
72
|
+
if (task.status !== "complete") return false;
|
|
73
|
+
return task.children.every((child) => taskIsTerminal(child));
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
export function allTasksTerminal(tasks: TaskView[]): boolean {
|
|
77
|
+
return flattenTaskViews(tasks).length > 0 && flattenTaskViews(tasks).every((task) => taskIsTerminal(task));
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/** Snapshot of statuses for persistence and stall detection. Every task gets
|
|
81
|
+
* a record: a rolled-back task must stay visible with the evidence from its
|
|
82
|
+
* previous attempt, otherwise a rollback erases the run's history and the
|
|
83
|
+
* agent can no longer tell "done and rolled back" from "never started". */
|
|
84
|
+
export function taskProgressMap(tasks: TaskView[]): TaskProgressMap {
|
|
85
|
+
const map: TaskProgressMap = {};
|
|
86
|
+
for (const task of flattenTaskViews(tasks)) {
|
|
87
|
+
map[task.id] = {
|
|
88
|
+
status: task.status,
|
|
89
|
+
evidence: task.evidence,
|
|
90
|
+
skipReason: task.skipReason,
|
|
91
|
+
};
|
|
92
|
+
}
|
|
93
|
+
return map;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/** Progress aggregation for the dashboard: done counts terminal tasks. */
|
|
97
|
+
export function taskProgress(tasks: TaskView[]): { done: number; total: number } {
|
|
98
|
+
const flat = flattenTaskViews(tasks);
|
|
99
|
+
return {
|
|
100
|
+
done: flat.filter((task) => taskIsTerminal(task)).length,
|
|
101
|
+
total: flat.length,
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** The current task: the first non-terminal task in wave order, then
|
|
106
|
+
* document order (top-level-first; subtasks close before their parent is
|
|
107
|
+
* re-listed). Deterministic; used for the ▸ anchor in the dashboard and
|
|
108
|
+
* the per-turn injection. */
|
|
109
|
+
export function currentTask(tasks: TaskView[]): TaskView | null {
|
|
110
|
+
const flat = flattenTaskViews(tasks);
|
|
111
|
+
const open = flat.filter((task) => !taskIsTerminal(task));
|
|
112
|
+
if (open.length === 0) return null;
|
|
113
|
+
open.sort((a, b) => (a.wave - b.wave) || (flat.indexOf(a) - flat.indexOf(b)));
|
|
114
|
+
return open[0] ?? null;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/** Tasks of one wave (top-level only — children belong to their parent). */
|
|
118
|
+
export function waveTasks(tasks: TaskView[], wave: number): TaskView[] {
|
|
119
|
+
return tasks.filter((task) => task.wave === wave);
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
export function maxWave(tasks: TaskView[]): number {
|
|
123
|
+
return flattenTaskViews(tasks).reduce((max, task) => Math.max(max, task.wave), 1);
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/** Legal status transitions for the task status tool: pending may close
|
|
127
|
+
* (complete/skipped); closed states are immutable except through the
|
|
128
|
+
* completion-audit flow (never through the task tool). */
|
|
129
|
+
export function canTransition(task: TaskView, next: TaskStatus): boolean {
|
|
130
|
+
if (task.status === next) return false;
|
|
131
|
+
if (task.status === "pending") return next === "complete" || next === "skipped";
|
|
132
|
+
return false;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/** Rollback set for a failed verification check: every task in its covers
|
|
136
|
+
* clause (parents cascade to their children, skipped tasks reopen too).
|
|
137
|
+
* Returns the ids that actually reopen.
|
|
138
|
+
*
|
|
139
|
+
* The task's `evidence` is deliberately kept. It records what the previous
|
|
140
|
+
* attempt actually did, which is the one thing the agent cannot reconstruct
|
|
141
|
+
* once it reopens the task; re-reporting overwrites it. `skipReason` is
|
|
142
|
+
* cleared because it described a deliberate skip that the audit has now
|
|
143
|
+
* overturned. */
|
|
144
|
+
export function auditRollbackSet(
|
|
145
|
+
tasks: TaskView[],
|
|
146
|
+
checklist: CheckItem[],
|
|
147
|
+
vcId: string,
|
|
148
|
+
): string[] {
|
|
149
|
+
const vc = checklist.find((item) => item.id === vcId);
|
|
150
|
+
if (!vc) return [];
|
|
151
|
+
const covered = new Set(extractTaskCoverage(vc.text));
|
|
152
|
+
const flat = flattenTaskViews(tasks);
|
|
153
|
+
const reopen: string[] = [];
|
|
154
|
+
const reopenNode = (node: TaskView): void => {
|
|
155
|
+
if (node.status !== "pending") {
|
|
156
|
+
node.status = "pending";
|
|
157
|
+
node.skipReason = undefined;
|
|
158
|
+
reopen.push(node.id);
|
|
159
|
+
}
|
|
160
|
+
for (const child of node.children) reopenNode(child);
|
|
161
|
+
};
|
|
162
|
+
for (const node of flat) {
|
|
163
|
+
if (covered.has(node.id)) reopenNode(node);
|
|
164
|
+
}
|
|
165
|
+
return reopen;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/** Checks that lose their satisfied state because a rollback reopened work
|
|
169
|
+
* they were verifying. Returns the ids whose `done` flag was cleared.
|
|
170
|
+
*
|
|
171
|
+
* A check can be presolved without an audit round (skipped-pass), or affirmed
|
|
172
|
+
* in an earlier round, and stay `done` while another check's rollback reopens
|
|
173
|
+
* one of its covered tasks. Left alone it renders as "check passed" next to a
|
|
174
|
+
* task that is open again with stale evidence. The caller owns this: it runs
|
|
175
|
+
* after the rollback set is known. */
|
|
176
|
+
export function invalidateChecksForRolledBackTasks(
|
|
177
|
+
checklist: CheckItem[],
|
|
178
|
+
tasks: TaskView[],
|
|
179
|
+
reopenedIds: string[],
|
|
180
|
+
): string[] {
|
|
181
|
+
if (reopenedIds.length === 0) return [];
|
|
182
|
+
const reopened = new Set(reopenedIds);
|
|
183
|
+
const cleared: string[] = [];
|
|
184
|
+
for (const item of checklist) {
|
|
185
|
+
if (!item.done) continue;
|
|
186
|
+
const covered = extractTaskCoverage(item.text);
|
|
187
|
+
if (covered.some((id) => reopened.has(id))) {
|
|
188
|
+
item.done = false;
|
|
189
|
+
cleared.push(item.id);
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
return cleared;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/** Verification checks that cover no task are excluded from audit (their
|
|
196
|
+
* pass state cannot be derived from task statuses). */
|
|
197
|
+
export function auditableChecks(checklist: CheckItem[], tasks: TaskView[]): CheckItem[] {
|
|
198
|
+
const known = new Set(flattenTaskViews(tasks).map((task) => task.id));
|
|
199
|
+
return checklist.filter((item) => {
|
|
200
|
+
const covered = extractTaskCoverage(item.text);
|
|
201
|
+
return covered.length > 0 && covered.some((id) => known.has(id));
|
|
202
|
+
});
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/** Checks with at least one skipped covered task whose remaining covered
|
|
206
|
+
* tasks are all complete (I-007: a skip whose siblings carry the delivery
|
|
207
|
+
* passes without an audit round). Pure-complete and pure-pending coverages
|
|
208
|
+
* are NOT presolved here — the former the audit affirms, the latter cannot
|
|
209
|
+
* arise once all tasks are terminal. */
|
|
210
|
+
export function skippedPassCheckIds(checklist: CheckItem[], tasks: TaskView[]): string[] {
|
|
211
|
+
const byId = new Map(flattenTaskViews(tasks).map((task) => [task.id, task]));
|
|
212
|
+
return auditableChecks(checklist, tasks)
|
|
213
|
+
.filter((item) => {
|
|
214
|
+
const covered = extractTaskCoverage(item.text).filter((id) => byId.has(id));
|
|
215
|
+
if (covered.length === 0) return false;
|
|
216
|
+
const statuses = covered.map((id) => byId.get(id)!.status);
|
|
217
|
+
return statuses.includes("skipped")
|
|
218
|
+
&& statuses.every((status) => status === "skipped" || status === "complete");
|
|
219
|
+
})
|
|
220
|
+
.map((item) => item.id);
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
export type { PlanTasks, TaskNode, WaveEntry };
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Thin thinking-level adapter over pi-ai's exported helpers.
|
|
3
|
+
*
|
|
4
|
+
* pi-ai already owns the level domain rules (`getSupportedThinkingLevels`:
|
|
5
|
+
* non-reasoning models only support "off"; a null mapping disables a level;
|
|
6
|
+
* xhigh/max appear only when explicitly mapped). Re-implementing them here
|
|
7
|
+
* would drift — this module only adds the pi-plans "default" sentinel and
|
|
8
|
+
* the stored-level resolution used by the reviewer spawn path (F-012).
|
|
9
|
+
*
|
|
10
|
+
* Semantics (decision 4 / F-009): `thinking_level: null` in the global
|
|
11
|
+
* reviewer config means DEFAULT — the child pi gets NO --thinking flag and
|
|
12
|
+
* resolves its own default chain (per-model settings → defaultThinkingLevel
|
|
13
|
+
* → medium, then model clamping). That is deliberately distinct from the
|
|
14
|
+
* explicit "off" level.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { getSupportedThinkingLevels, type Model, type ModelThinkingLevel } from "@earendil-works/pi-ai";
|
|
18
|
+
import type { GlobalRoleConfig, ThinkingLevelValue } from "./global-state.ts";
|
|
19
|
+
|
|
20
|
+
/** Panel/label sentinel for `thinking_level: null` (omit --thinking). */
|
|
21
|
+
export const DEFAULT_LEVEL_SENTINEL = "default";
|
|
22
|
+
|
|
23
|
+
/** Levels a model actually supports, via pi-ai (single source of truth). */
|
|
24
|
+
export function levelsForModel(model: Pick<Model<any>, "reasoning" | "thinkingLevelMap">): ModelThinkingLevel[] {
|
|
25
|
+
return getSupportedThinkingLevels(model);
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** True when the stored level is supported by the model (spawn passes it
|
|
29
|
+
* through verbatim; unsupported stored levels are clamped by the child). */
|
|
30
|
+
export function isLevelSupported(
|
|
31
|
+
model: Pick<Model<any>, "reasoning" | "thinkingLevelMap">,
|
|
32
|
+
level: string | null,
|
|
33
|
+
): boolean {
|
|
34
|
+
if (level === null) return true; // default: no flag, always valid
|
|
35
|
+
return levelsForModel(model).includes(level as ModelThinkingLevel);
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/** Resolve the CLI argument for the stored level: null → omit the flag. */
|
|
39
|
+
export function spawnThinkingFlag(level: ThinkingLevelValue | null): string | null {
|
|
40
|
+
return level ?? null;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/** Display label for the reviewer overlay and subagents ledger. */
|
|
44
|
+
export function roleModelLabel(modelSelector: string, thinkingLevel: ThinkingLevelValue | null): string {
|
|
45
|
+
return `${modelSelector}:${thinkingLevel ?? DEFAULT_LEVEL_SENTINEL}`;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/** Human-facing one-liner for the default row / docs (F-009): the default
|
|
49
|
+
* is the child pi's own chain, NOT the current session's level. */
|
|
50
|
+
export const DEFAULT_LEVEL_DESCRIPTION =
|
|
51
|
+
"no --thinking flag: the child pi resolves its default (per-model settings → defaultThinkingLevel → medium)";
|
|
52
|
+
|
|
53
|
+
/** Spawn-view of a confirmed reviewer role: concrete model + optional level. */
|
|
54
|
+
export interface ResolvedReviewerSpawn {
|
|
55
|
+
modelSelector: string;
|
|
56
|
+
thinkingLevel: ThinkingLevelValue | null;
|
|
57
|
+
label: string;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export function resolveReviewerSpawn(role: Pick<GlobalRoleConfig, "model_selector" | "thinking_level">): ResolvedReviewerSpawn {
|
|
61
|
+
const modelSelector = role.model_selector ?? "";
|
|
62
|
+
return {
|
|
63
|
+
modelSelector,
|
|
64
|
+
thinkingLevel: role.thinking_level ?? null,
|
|
65
|
+
label: modelSelector ? roleModelLabel(modelSelector, role.thinking_level ?? null) : "unconfirmed",
|
|
66
|
+
};
|
|
67
|
+
}
|
package/src/ui-language.ts
CHANGED
|
@@ -1,10 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Single source of truth for user-visible UI chrome language (issue #3).
|
|
3
3
|
*
|
|
4
|
-
* Workspace `language.tag` (`.git/
|
|
5
|
-
* strings for every user-visible surface: the batch form (src/ask-form.ts)
|
|
6
|
-
* the refine/refs overlay footer (src/refine-ui.ts)
|
|
7
|
-
* (src/panel.ts) and the execution status line (src/exec.ts).
|
|
4
|
+
* Workspace `language.tag` (`.git/pi-plans/config.json`) selects the chrome
|
|
5
|
+
* strings for every user-visible surface: the batch form (src/ask-form.ts)
|
|
6
|
+
* and the refine/refs overlay footer (src/refine-ui.ts).
|
|
8
7
|
*
|
|
9
8
|
* Mapping follows RFC 4647 primary-subtag fallback (zh-Hant-CN → zh-Hant →
|
|
10
9
|
* zh), so every `zh*` tag renders the existing Simplified strings verbatim.
|
|
@@ -104,6 +103,8 @@ export interface RefineChrome {
|
|
|
104
103
|
scroll: string;
|
|
105
104
|
page: string;
|
|
106
105
|
switchLane: string;
|
|
106
|
+
/** Auditor overlay only (v0.8): the shortcut that reopens the in-flight round. */
|
|
107
|
+
reopen: string;
|
|
107
108
|
}
|
|
108
109
|
|
|
109
110
|
const REFINE_CHROME: Record<UiLanguage, RefineChrome> = {
|
|
@@ -112,12 +113,14 @@ const REFINE_CHROME: Record<UiLanguage, RefineChrome> = {
|
|
|
112
113
|
scroll: "↑/↓ 滚动",
|
|
113
114
|
page: "PgUp/PgDn 翻页",
|
|
114
115
|
switchLane: "Tab & Shift + Tab 切换 lane",
|
|
116
|
+
reopen: "Ctrl+Shift+R 重开评审面板",
|
|
115
117
|
},
|
|
116
118
|
en: {
|
|
117
119
|
close: "Esc close",
|
|
118
120
|
scroll: "↑/↓ scroll",
|
|
119
121
|
page: "PgUp/PgDn page",
|
|
120
122
|
switchLane: "Tab & Shift + Tab switch lane",
|
|
123
|
+
reopen: "Ctrl+Shift+R reopen review overlay",
|
|
121
124
|
},
|
|
122
125
|
};
|
|
123
126
|
|
|
@@ -125,54 +128,4 @@ export function refineChrome(lang: UiLanguage): RefineChrome {
|
|
|
125
128
|
return REFINE_CHROME[lang];
|
|
126
129
|
}
|
|
127
130
|
|
|
128
|
-
// ---------------------------------------------------------------------------
|
|
129
|
-
// Status panel chrome (src/panel.ts) — implWarning branch only.
|
|
130
|
-
// ---------------------------------------------------------------------------
|
|
131
|
-
|
|
132
|
-
export interface PanelChrome {
|
|
133
|
-
badgeStatus(topic: string, vcDone: number, vcTotal: number): string;
|
|
134
|
-
progressWarning(): string;
|
|
135
|
-
summaryWarning(topic: string, vcDone: number, vcTotal: number, nextAction: string): string;
|
|
136
|
-
}
|
|
137
|
-
|
|
138
|
-
const PANEL_CHROME: Record<UiLanguage, PanelChrome> = {
|
|
139
|
-
zh: {
|
|
140
|
-
badgeStatus: (topic, vcDone, vcTotal) => `${topic} · ⚠ I 解析 0 项 · VC ${vcDone}/${vcTotal}`,
|
|
141
|
-
progressWarning: () => `⚠ plan 格式:Implementation Items 解析 0 项(面板无法计 I 进度)`,
|
|
142
|
-
summaryWarning: (topic, vcDone, vcTotal, nextAction) =>
|
|
143
|
-
`plans: ${topic} ▸ ⚠ Implementation Items 解析 0 项 · VC ${vcDone}/${vcTotal} · next: ${nextAction}`,
|
|
144
|
-
},
|
|
145
|
-
en: {
|
|
146
|
-
badgeStatus: (topic, vcDone, vcTotal) => `${topic} · ⚠ I parse 0 items · VC ${vcDone}/${vcTotal}`,
|
|
147
|
-
progressWarning: () => `⚠ plan format: Implementation Items parsed 0 items (I progress cannot be counted)`,
|
|
148
|
-
summaryWarning: (topic, vcDone, vcTotal, nextAction) =>
|
|
149
|
-
`plans: ${topic} ▸ ⚠ Implementation Items parsed 0 items · VC ${vcDone}/${vcTotal} · next: ${nextAction}`,
|
|
150
|
-
},
|
|
151
|
-
};
|
|
152
|
-
|
|
153
|
-
export function panelChrome(lang: UiLanguage): PanelChrome {
|
|
154
|
-
return PANEL_CHROME[lang];
|
|
155
|
-
}
|
|
156
|
-
|
|
157
|
-
// ---------------------------------------------------------------------------
|
|
158
|
-
// Execution status line chrome (src/exec.ts) — goal-wait segment.
|
|
159
|
-
// ---------------------------------------------------------------------------
|
|
160
|
-
|
|
161
|
-
export interface ExecChrome {
|
|
162
|
-
goalWait(noProgressRounds: number, waitRounds: number): string;
|
|
163
|
-
}
|
|
164
|
-
|
|
165
|
-
const EXEC_CHROME: Record<UiLanguage, ExecChrome> = {
|
|
166
|
-
zh: {
|
|
167
|
-
goalWait: (noProgressRounds, waitRounds) =>
|
|
168
|
-
` · 🔁 goal-wait · 无进展 ${noProgressRounds}/3 · 等待 ${waitRounds}/6`,
|
|
169
|
-
},
|
|
170
|
-
en: {
|
|
171
|
-
goalWait: (noProgressRounds, waitRounds) =>
|
|
172
|
-
` · 🔁 goal-wait · no progress ${noProgressRounds}/3 · waiting ${waitRounds}/6`,
|
|
173
|
-
},
|
|
174
|
-
};
|
|
175
131
|
|
|
176
|
-
export function execChrome(lang: UiLanguage): ExecChrome {
|
|
177
|
-
return EXEC_CHROME[lang];
|
|
178
|
-
}
|
package/src/workflow-state.ts
CHANGED
|
@@ -145,8 +145,19 @@ export interface ExecutionCheckpoint {
|
|
|
145
145
|
/** True when this approval/progress was produced in a different (origin) worktree. */
|
|
146
146
|
originWorktree?: string;
|
|
147
147
|
/** v0.6.0: set while a delegated executor child owns the implementation;
|
|
148
|
-
* stale after a restart (orphaned delegate — the child died with the parent).
|
|
148
|
+
* stale after a restart (orphaned delegate — the child died with the parent).
|
|
149
|
+
* Removed with delegated execution in v0.6.1; read-tolerated on legacy checkpoints. */
|
|
149
150
|
delegate?: { modelSelector: string; startedAt: string };
|
|
151
|
+
/** v0.6.1: task-tree progress (task id → status/evidence), the primary
|
|
152
|
+
* progress record. doneVcIds stays for legacy checkpoints and the final
|
|
153
|
+
* audit pass. */
|
|
154
|
+
tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
|
|
155
|
+
/** v0.7.1: watchdog round counter. Persisted so a session restart cannot
|
|
156
|
+
* silently hand a stalled run a fresh budget; optional so checkpoints
|
|
157
|
+
* written before this field keep loading. */
|
|
158
|
+
stallRounds?: number;
|
|
159
|
+
/** v0.6.1: completion-audit bookkeeping (rounds, last failed set, pass). */
|
|
160
|
+
audit?: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] };
|
|
150
161
|
}
|
|
151
162
|
|
|
152
163
|
export interface OwnerInfo {
|
|
@@ -454,7 +465,7 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
454
465
|
const record = asRecord(value, label);
|
|
455
466
|
rejectExtraKeys(
|
|
456
467
|
record,
|
|
457
|
-
new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate"]),
|
|
468
|
+
new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "stallRounds", "audit"]),
|
|
458
469
|
label,
|
|
459
470
|
);
|
|
460
471
|
const execution: ExecutionCheckpoint = {
|
|
@@ -484,6 +495,36 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
484
495
|
startedAt: asString(delegate.startedAt, `${label}.delegate.startedAt`),
|
|
485
496
|
};
|
|
486
497
|
}
|
|
498
|
+
if (record.tasks !== undefined && record.tasks !== null) {
|
|
499
|
+
const tasksRecord = asRecord(record.tasks, `${label}.tasks`);
|
|
500
|
+
const tasks: NonNullable<ExecutionCheckpoint["tasks"]> = {};
|
|
501
|
+
for (const [id, raw] of Object.entries(tasksRecord)) {
|
|
502
|
+
const entry = asRecord(raw, `${label}.tasks.${id}`);
|
|
503
|
+
rejectExtraKeys(entry, new Set(["status", "evidence", "skipReason"]), `${label}.tasks.${id}`);
|
|
504
|
+
const status = asEnum(entry.status, new Set(["pending", "complete", "skipped"]), `${label}.tasks.${id}.status`);
|
|
505
|
+
const item: { status: string; evidence?: string; skipReason?: string } = { status };
|
|
506
|
+
if (entry.evidence !== undefined) item.evidence = asString(entry.evidence, `${label}.tasks.${id}.evidence`);
|
|
507
|
+
if (entry.skipReason !== undefined) item.skipReason = asString(entry.skipReason, `${label}.tasks.${id}.skipReason`);
|
|
508
|
+
tasks[id] = item;
|
|
509
|
+
}
|
|
510
|
+
execution.tasks = tasks;
|
|
511
|
+
}
|
|
512
|
+
if (record.stallRounds !== undefined && record.stallRounds !== null) {
|
|
513
|
+
execution.stallRounds = asInt(record.stallRounds, `${label}.stallRounds`, 0);
|
|
514
|
+
}
|
|
515
|
+
if (record.audit !== undefined && record.audit !== null) {
|
|
516
|
+
const audit = asRecord(record.audit, `${label}.audit`);
|
|
517
|
+
rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed", "undeterminable"]), `${label}.audit`);
|
|
518
|
+
const parsed: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] } = {
|
|
519
|
+
rounds: asInt(audit.rounds, `${label}.audit.rounds`, 0),
|
|
520
|
+
};
|
|
521
|
+
if (audit.lastResult !== undefined) parsed.lastResult = asString(audit.lastResult, `${label}.audit.lastResult`);
|
|
522
|
+
if (audit.passed !== undefined) parsed.passed = asBool(audit.passed, `${label}.audit.passed`);
|
|
523
|
+
if (audit.undeterminable !== undefined) {
|
|
524
|
+
parsed.undeterminable = asStringArray(audit.undeterminable, `${label}.audit.undeterminable`);
|
|
525
|
+
}
|
|
526
|
+
execution.audit = parsed;
|
|
527
|
+
}
|
|
487
528
|
return execution;
|
|
488
529
|
}
|
|
489
530
|
|
|
@@ -1000,19 +1041,6 @@ export function applyReviewConsolidated(cp: WorkflowCheckpoint, roundId: string,
|
|
|
1000
1041
|
return { ...cp, reviewRounds: cp.reviewRounds.map((entry) => (entry.roundId === roundId ? nextRound : entry)) };
|
|
1001
1042
|
}
|
|
1002
1043
|
|
|
1003
|
-
/** F-004: `completed` requires explicit evidence; approval cannot be forged by state writes. */
|
|
1004
|
-
export function applyCompleted(cp: WorkflowCheckpoint, evidence: string): WorkflowCheckpoint {
|
|
1005
|
-
if (cp.phase !== "implementation-review") {
|
|
1006
|
-
throw new StateError(`cannot complete from phase "${cp.phase}"`);
|
|
1007
|
-
}
|
|
1008
|
-
const review = cp.implementationReview;
|
|
1009
|
-
if (!review || review.terminationCondition === undefined) {
|
|
1010
|
-
throw new StateError("cannot complete without a recorded termination condition");
|
|
1011
|
-
}
|
|
1012
|
-
if (evidence.trim() === "") throw new StateError("completion requires non-empty evidence");
|
|
1013
|
-
return { ...cp, phase: "completed", nextAction: "none" };
|
|
1014
|
-
}
|
|
1015
|
-
|
|
1016
1044
|
export function applyExecutionApproved(cp: WorkflowCheckpoint, approval: ExecutionApproval): WorkflowCheckpoint {
|
|
1017
1045
|
// F-007 (implementation review): terminal and review phases cannot approve
|
|
1018
1046
|
// execution; a re-approval (stop/migration reset approval to null) is legal
|
|
@@ -1036,6 +1064,8 @@ export function applyExecutionApproved(cp: WorkflowCheckpoint, approval: Executi
|
|
|
1036
1064
|
doneVcIds: [],
|
|
1037
1065
|
implStatus: {},
|
|
1038
1066
|
usage: { inToks: 0, outToks: 0 },
|
|
1067
|
+
tasks: {},
|
|
1068
|
+
audit: { rounds: 0 },
|
|
1039
1069
|
},
|
|
1040
1070
|
};
|
|
1041
1071
|
}
|
|
@@ -1048,6 +1078,12 @@ export interface ExecutionProgressInput {
|
|
|
1048
1078
|
pausedReason?: string | null;
|
|
1049
1079
|
/** v0.6.0: set/clear the delegated-executor record; null clears it. */
|
|
1050
1080
|
delegate?: { modelSelector: string; startedAt: string } | null;
|
|
1081
|
+
/** v0.6.1: task-tree progress snapshot (authoritative). */
|
|
1082
|
+
tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
|
|
1083
|
+
/** v0.6.1: completion-audit bookkeeping update. */
|
|
1084
|
+
audit?: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] };
|
|
1085
|
+
/** v0.7.1: watchdog budget counter, so a restart cannot refresh it. */
|
|
1086
|
+
stallRounds?: number;
|
|
1051
1087
|
}
|
|
1052
1088
|
|
|
1053
1089
|
export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: ExecutionProgressInput): WorkflowCheckpoint {
|
|
@@ -1066,6 +1102,9 @@ export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: Executi
|
|
|
1066
1102
|
else if (progress.pausedReason !== undefined) execution.pausedReason = progress.pausedReason;
|
|
1067
1103
|
if (progress.delegate === null) delete execution.delegate;
|
|
1068
1104
|
else if (progress.delegate !== undefined) execution.delegate = progress.delegate;
|
|
1105
|
+
if (progress.tasks !== undefined) execution.tasks = progress.tasks;
|
|
1106
|
+
if (progress.stallRounds !== undefined) execution.stallRounds = progress.stallRounds;
|
|
1107
|
+
if (progress.audit !== undefined) execution.audit = progress.audit;
|
|
1069
1108
|
return { ...cp, execution };
|
|
1070
1109
|
}
|
|
1071
1110
|
|
|
@@ -1077,45 +1116,18 @@ export function applyExecutionHeadChanged(cp: WorkflowCheckpoint): WorkflowCheck
|
|
|
1077
1116
|
|
|
1078
1117
|
export function applyExecutionCompleted(cp: WorkflowCheckpoint): WorkflowCheckpoint {
|
|
1079
1118
|
if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
|
|
1119
|
+
// v0.6.1 (D-018): the post-execution amelioration loop is gone; a
|
|
1120
|
+
// completed audit passes the run straight to the terminal phase.
|
|
1080
1121
|
return {
|
|
1081
1122
|
...cp,
|
|
1082
|
-
phase: "
|
|
1083
|
-
nextAction: "
|
|
1084
|
-
execution: {
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
cp: WorkflowCheckpoint,
|
|
1090
|
-
terminationCondition: string,
|
|
1091
|
-
reviewerCount?: number,
|
|
1092
|
-
): WorkflowCheckpoint {
|
|
1093
|
-
if (cp.phase !== "implementation-review") throw new StateError("requires phase \"implementation-review\"");
|
|
1094
|
-
if (cp.implementationReview?.terminationCondition !== undefined) {
|
|
1095
|
-
throw new StateError("termination condition already configured; do not re-ask");
|
|
1096
|
-
}
|
|
1097
|
-
return {
|
|
1098
|
-
...cp,
|
|
1099
|
-
implementationReview: {
|
|
1100
|
-
terminationCondition,
|
|
1101
|
-
reviewerCount,
|
|
1102
|
-
completedRounds: cp.implementationReview?.completedRounds ?? 0,
|
|
1123
|
+
phase: "completed",
|
|
1124
|
+
nextAction: "none",
|
|
1125
|
+
execution: {
|
|
1126
|
+
...cp.execution,
|
|
1127
|
+
pausedReason: undefined,
|
|
1128
|
+
delegate: undefined,
|
|
1129
|
+
audit: { ...(cp.execution.audit ?? { rounds: 0 }), passed: true },
|
|
1103
1130
|
},
|
|
1104
|
-
nextAction: "run-review",
|
|
1105
|
-
};
|
|
1106
|
-
}
|
|
1107
|
-
|
|
1108
|
-
export function applyImplementationRoundFinished(cp: WorkflowCheckpoint): WorkflowCheckpoint {
|
|
1109
|
-
if (cp.phase !== "implementation-review" || !cp.implementationReview) {
|
|
1110
|
-
throw new StateError("requires phase \"implementation-review\"");
|
|
1111
|
-
}
|
|
1112
|
-
const review = cp.implementationReview;
|
|
1113
|
-
if (review.currentRoundId === undefined) throw new StateError("no current round to finish");
|
|
1114
|
-
const round = cp.reviewRounds.find((entry) => entry.roundId === review.currentRoundId);
|
|
1115
|
-
if (!round || !round.consolidated) throw new StateError("current round is not consolidated");
|
|
1116
|
-
return {
|
|
1117
|
-
...cp,
|
|
1118
|
-
implementationReview: { ...review, completedRounds: review.completedRounds + 1, currentRoundId: undefined },
|
|
1119
1131
|
};
|
|
1120
1132
|
}
|
|
1121
1133
|
|
|
@@ -1135,12 +1147,16 @@ export function applyMigration(
|
|
|
1135
1147
|
migration: { fromWorktree: cp.worktreeRoot, migratedAt: utcNow() },
|
|
1136
1148
|
execution: cp.execution
|
|
1137
1149
|
? {
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1150
|
+
approval: null,
|
|
1151
|
+
doneVcIds: [],
|
|
1152
|
+
implStatus: {},
|
|
1153
|
+
usage: cp.execution.usage,
|
|
1154
|
+
originWorktree: cp.execution.originWorktree ?? cp.worktreeRoot,
|
|
1155
|
+
// v0.6.1: task progress and audit state do not survive a worktree
|
|
1156
|
+
// migration (same rule as VC validity).
|
|
1157
|
+
tasks: {},
|
|
1158
|
+
audit: { rounds: 0 },
|
|
1159
|
+
}
|
|
1144
1160
|
: undefined,
|
|
1145
1161
|
implementationReview: cp.implementationReview
|
|
1146
1162
|
? {
|
|
@@ -1164,7 +1180,9 @@ function migrationNextAction(cp: WorkflowCheckpoint): NextAction {
|
|
|
1164
1180
|
export function applyExecutionStopped(cp: WorkflowCheckpoint, reason: string): WorkflowCheckpoint {
|
|
1165
1181
|
if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
|
|
1166
1182
|
const { delegate: _delegate, ...execution } = cp.execution;
|
|
1167
|
-
|
|
1183
|
+
// A stop also revokes any outstanding audit-rollback authorization.
|
|
1184
|
+
const audit = execution.audit ? { ...execution.audit } : undefined;
|
|
1185
|
+
return { ...cp, execution: { ...execution, audit, pausedReason: reason } };
|
|
1168
1186
|
}
|
|
1169
1187
|
|
|
1170
1188
|
// ---------------------------------------------------------------------------
|