pi-gauntlet 5.5.4 → 5.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/extensions/phase-tracker.test.ts +21 -0
- package/extensions/phase-tracker.ts +4 -1
- package/extensions/plan-tracker.test.ts +211 -9
- package/extensions/plan-tracker.ts +95 -87
- package/package.json +1 -1
- package/skills/brainstorming/SKILL.md +22 -3
- package/skills/check-delivery/SKILL.md +6 -5
- package/skills/gatekeep-pr/SKILL.md +12 -8
- package/skills/subagent-driven-development/SKILL.md +2 -2
- package/skills/verification-before-completion/reference/conformance-check.md +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,16 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## v5.6.1 - 2026-09-17
|
|
4
|
+
|
|
5
|
+
- `brainstorming`: the questionary looks up questions the code, docs, or issue tracker already answer instead of asking; every question it does ask ends with `Recommendation: <answer> - <why>` (including the ask to accept a corrected fact); before approaches, a plain-prose premise note in chat names which design-dependent claims the sources support, disprove (with the corrected fact and where), or leave unverified - a contradicted claim stops the flow until the user accepts the correction or explicitly overrides, with the outcome recorded in the draft. One checklist pointer, one Red Flags entry. (#32)
|
|
6
|
+
- AGENTS core v4: exactness via the example, no inline SHAs, no spiderweb replies.
|
|
7
|
+
|
|
8
|
+
## v5.6.0 - 2026-09-16
|
|
9
|
+
|
|
10
|
+
- `plan-tracker`: pending-suffix rule enforced at runtime - an `update` that would leave a `pending` task ahead of a started or finished one is rejected with a message naming every offending task, the legal fixes, and the current snapshot; state is unchanged and widget replay / `phase_tracker` ignore the rejection. `update` never sets `pending`. New terminal status `skipped` (`⊘`, not applicable in this run, counted done). `init` accepts `{ name, status }` elements (mixable with strings) to recreate a list with known statuses. Registered with sequential execution. Counts are labelled `done` (`complete + skipped`).
|
|
11
|
+
- `phase-tracker`: `implement` auto-completes when every task is `complete` or `skipped`.
|
|
12
|
+
- `check-delivery`, `gatekeep-pr`, `subagent-driven-development`, `brainstorming`: tracker usage aligned with the rule - a skipped stage is `skipped`, gatekeep-pr inits four stages and adds claims/review/consent progressively, wave fan-out marks `in_progress` in increasing index order, the amendment path re-inits with statuses instead of setting tasks back to `pending`.
|
|
13
|
+
|
|
3
14
|
## v5.5.4 - 2026-09-14
|
|
4
15
|
|
|
5
16
|
- `phase-tracker`: new `phase_tracker` action `grant_fix_rounds` records an explicit human approval of N more conformance fix rounds (reason quotes the human), accepted only while the cap block is live; each qualifying implementer wave spends one granted round, credits replay with the session and reset with the audit latch. The cap-block message now leads with that action and names the exact `.pi/settings.json` for the `enforce: false` last resort (applies without restart, whole-block precedence).
|
|
@@ -370,6 +370,27 @@ test("all-complete plan activity auto-completes an active implement phase", asyn
|
|
|
370
370
|
assert.equal(status.details.phases.implement.status, "complete");
|
|
371
371
|
});
|
|
372
372
|
|
|
373
|
+
test("plan activity auto-completes implement with complete+skipped, not with failed, not from an errored result", async () => {
|
|
374
|
+
const implementStatus = async (tasks: { status: string }[], isError = false) => {
|
|
375
|
+
const h = harness({ branch: implementBranch() });
|
|
376
|
+
await h.emit("session_start");
|
|
377
|
+
await h.emitEvent("tool_execution_end", {
|
|
378
|
+
toolName: "plan_tracker",
|
|
379
|
+
isError,
|
|
380
|
+
result: { details: { tasks } },
|
|
381
|
+
});
|
|
382
|
+
const tool = h.tools.find((t) => t.name === "phase_tracker")!;
|
|
383
|
+
const status = (await tool.execute("t1", { action: "status" }, undefined, undefined, h.ctx)) as {
|
|
384
|
+
details: { phases: { implement: { status: string } } };
|
|
385
|
+
};
|
|
386
|
+
return status.details.phases.implement.status;
|
|
387
|
+
};
|
|
388
|
+
assert.equal(await implementStatus([{ status: "complete" }, { status: "skipped" }]), "complete");
|
|
389
|
+
assert.equal(await implementStatus([{ status: "skipped" }, { status: "skipped" }]), "complete");
|
|
390
|
+
assert.equal(await implementStatus([{ status: "complete" }, { status: "failed" }]), "in_progress");
|
|
391
|
+
assert.equal(await implementStatus([{ status: "complete" }, { status: "skipped" }], true), "in_progress");
|
|
392
|
+
});
|
|
393
|
+
|
|
373
394
|
test("cold implement and verify completions ignore unfinished snapshots", async () => {
|
|
374
395
|
for (const phase of ["implement", "verify"] as const) {
|
|
375
396
|
const h = harness({
|
|
@@ -417,7 +417,10 @@ export default function (pi: ExtensionAPI) {
|
|
|
417
417
|
// skills; plan-tracker tracks tasks independently.
|
|
418
418
|
const applyPlanActivity = (tasks?: { status: string }[]) => {
|
|
419
419
|
if (!tasks || tasks.length === 0) return;
|
|
420
|
-
if (
|
|
420
|
+
if (
|
|
421
|
+
phases.implement.status === "in_progress" &&
|
|
422
|
+
tasks.every((t) => t.status === "complete" || t.status === "skipped")
|
|
423
|
+
) {
|
|
421
424
|
phases = { ...phases, implement: { status: "complete" } };
|
|
422
425
|
firedGuards.clear();
|
|
423
426
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import assert from "node:assert/strict";
|
|
2
2
|
import { test } from "node:test";
|
|
3
|
-
import registerPlanTracker from "./plan-tracker.ts";
|
|
3
|
+
import registerPlanTracker, { validateSnapshot } from "./plan-tracker.ts";
|
|
4
4
|
|
|
5
5
|
type ToolResult = {
|
|
6
6
|
content: { type: string; text: string }[];
|
|
@@ -10,6 +10,9 @@ type ToolResult = {
|
|
|
10
10
|
function harness(branch: unknown[] = []) {
|
|
11
11
|
const tools: {
|
|
12
12
|
name: string;
|
|
13
|
+
description: string;
|
|
14
|
+
executionMode?: string;
|
|
15
|
+
parameters: any;
|
|
13
16
|
execute: (...args: any[]) => unknown;
|
|
14
17
|
renderResult: (result: unknown, options: unknown, theme: unknown) => { text?: string };
|
|
15
18
|
}[] = [];
|
|
@@ -39,7 +42,16 @@ function harness(branch: unknown[] = []) {
|
|
|
39
42
|
const fire = async (event: string) => {
|
|
40
43
|
for (const h of handlers) if (h.event === event) await h.handler({}, ctx);
|
|
41
44
|
};
|
|
42
|
-
return {
|
|
45
|
+
return {
|
|
46
|
+
call,
|
|
47
|
+
fire,
|
|
48
|
+
tool: () => tools[0],
|
|
49
|
+
theme,
|
|
50
|
+
widget: () => widgetText,
|
|
51
|
+
parameters: () => tools[0].parameters,
|
|
52
|
+
description: () => tools[0].description,
|
|
53
|
+
executionMode: () => tools[0].executionMode,
|
|
54
|
+
};
|
|
43
55
|
}
|
|
44
56
|
|
|
45
57
|
test("add appends pending tasks and preserves existing statuses", async () => {
|
|
@@ -95,7 +107,7 @@ test("update to failed round-trips and is excluded from complete count", async (
|
|
|
95
107
|
res.details.tasks.map((t) => [t.name, t.status]),
|
|
96
108
|
[["a", "complete"], ["b", "failed"], ["c", "pending"]],
|
|
97
109
|
);
|
|
98
|
-
assert.match(res.content[0].text, /1\/3
|
|
110
|
+
assert.match(res.content[0].text, /1\/3 done/);
|
|
99
111
|
assert.match(res.content[0].text, /1 failed/);
|
|
100
112
|
});
|
|
101
113
|
|
|
@@ -140,11 +152,11 @@ test("widget renders failed as \u2717, keeps failed out of complete count and cu
|
|
|
140
152
|
test("renderResult status path shows \u2717 for failed and excludes it from complete", async () => {
|
|
141
153
|
const { call, tool, theme } = harness();
|
|
142
154
|
await call({ action: "init", tasks: ["a", "b"] });
|
|
143
|
-
await call({ action: "update", index:
|
|
155
|
+
await call({ action: "update", index: 0, status: "failed" });
|
|
144
156
|
const res = await call({ action: "status" });
|
|
145
157
|
const rendered = tool().renderResult(res as any, {}, theme as any);
|
|
146
158
|
const text = (rendered as any).text as string;
|
|
147
|
-
assert.match(text, /0\/2
|
|
159
|
+
assert.match(text, /0\/2 done/);
|
|
148
160
|
assert.match(text, /\u2717/);
|
|
149
161
|
});
|
|
150
162
|
|
|
@@ -156,7 +168,7 @@ test("renderResult status header appends failed count when a task has failed", a
|
|
|
156
168
|
const res = await call({ action: "status" });
|
|
157
169
|
const rendered = tool().renderResult(res as any, {}, theme as any);
|
|
158
170
|
const text = (rendered as any).text as string;
|
|
159
|
-
assert.match(text, /1\/3
|
|
171
|
+
assert.match(text, /1\/3 done, 1 failed/);
|
|
160
172
|
});
|
|
161
173
|
|
|
162
174
|
test("renderResult status header omits failed count when no task has failed", async () => {
|
|
@@ -166,7 +178,7 @@ test("renderResult status header omits failed count when no task has failed", as
|
|
|
166
178
|
const res = await call({ action: "status" });
|
|
167
179
|
const rendered = tool().renderResult(res as any, {}, theme as any);
|
|
168
180
|
const text = (rendered as any).text as string;
|
|
169
|
-
assert.match(text, /^1\/2
|
|
181
|
+
assert.match(text, /^1\/2 done\n/);
|
|
170
182
|
assert.doesNotMatch(text, /failed/);
|
|
171
183
|
});
|
|
172
184
|
|
|
@@ -177,7 +189,7 @@ test("renderResult update case appends failed count when a task has failed", asy
|
|
|
177
189
|
const res = await call({ action: "update", index: 1, status: "failed" });
|
|
178
190
|
const rendered = tool().renderResult(res as any, {}, theme as any);
|
|
179
191
|
const text = (rendered as any).text as string;
|
|
180
|
-
assert.match(text, /^\u2713 Updated \(1\/3
|
|
192
|
+
assert.match(text, /^\u2713 Updated \(1\/3 done, 1 failed\)$/);
|
|
181
193
|
});
|
|
182
194
|
|
|
183
195
|
test("renderResult update case omits failed count when no task has failed", async () => {
|
|
@@ -186,6 +198,196 @@ test("renderResult update case omits failed count when no task has failed", asyn
|
|
|
186
198
|
const res = await call({ action: "update", index: 0, status: "complete" });
|
|
187
199
|
const rendered = tool().renderResult(res as any, {}, theme as any);
|
|
188
200
|
const text = (rendered as any).text as string;
|
|
189
|
-
assert.match(text, /^\u2713 Updated \(1\/2
|
|
201
|
+
assert.match(text, /^\u2713 Updated \(1\/2 done\)$/);
|
|
190
202
|
assert.doesNotMatch(text, /failed/);
|
|
191
203
|
});
|
|
204
|
+
|
|
205
|
+
const S = (code: string) =>
|
|
206
|
+
[...code].map((c, i) => ({
|
|
207
|
+
name: `t${i}`,
|
|
208
|
+
status: ({ o: "pending", ">": "in_progress", k: "complete", x: "failed", "/": "skipped" } as const)[c as "o" | ">" | "k" | "x" | "/"],
|
|
209
|
+
}));
|
|
210
|
+
|
|
211
|
+
test("validateSnapshot: pending must trail every non-pending task", () => {
|
|
212
|
+
for (const code of ["kkkoo", "kk>>oo", "kk>kk", "x>o", "ooo", "///", "", "k"]) {
|
|
213
|
+
assert.deepEqual(validateSnapshot(S(code)), [], code);
|
|
214
|
+
}
|
|
215
|
+
assert.deepEqual(validateSnapshot(S("okoo")), [0]);
|
|
216
|
+
assert.deepEqual(validateSnapshot(S("kkk>ookkkk")), [4, 5]);
|
|
217
|
+
assert.deepEqual(validateSnapshot(S("o>")), [0]);
|
|
218
|
+
assert.deepEqual(validateSnapshot(S("ook")), [0, 1]);
|
|
219
|
+
});
|
|
220
|
+
|
|
221
|
+
test("update past a pending index is rejected with every offender, fixes, and the current snapshot", async () => {
|
|
222
|
+
const { call } = harness();
|
|
223
|
+
await call({ action: "init", tasks: ["gather", "resolve evidence", "claim-check", "provision worktree", "review", "verify claims"] });
|
|
224
|
+
for (const i of [0, 1, 2]) await call({ action: "update", index: i, status: "complete" });
|
|
225
|
+
const res = await call({ action: "update", index: 5, status: "complete" });
|
|
226
|
+
assert.equal(res.details.error, "pending-suffix violation: 3,4");
|
|
227
|
+
assert.deepEqual(
|
|
228
|
+
res.details.tasks.map((t) => t.status),
|
|
229
|
+
["complete", "complete", "complete", "pending", "pending", "pending"],
|
|
230
|
+
);
|
|
231
|
+
const text = res.content[0].text;
|
|
232
|
+
assert.equal(
|
|
233
|
+
text,
|
|
234
|
+
[
|
|
235
|
+
'Error: cannot set task 5 "verify claims" to complete: pending tasks precede it: 3 "provision worktree", 4 "review".',
|
|
236
|
+
"Rule: pending tasks must trail every started or finished task.",
|
|
237
|
+
"Fix first, then retry: update each listed task to in_progress (working on it now), complete, failed, or skipped (not applicable in this run); or, if the list order itself is wrong, re-init with {name, status}[] keeping every task and its true status so the untouched ones trail.",
|
|
238
|
+
"Never clear, drop tasks, or record a status the work has not actually reached.",
|
|
239
|
+
"Current: 0 gather=complete, 1 resolve evidence=complete, 2 claim-check=complete, 3 provision worktree=pending, 4 review=pending, 5 verify claims=pending",
|
|
240
|
+
].join("\n"),
|
|
241
|
+
);
|
|
242
|
+
const after = await call({ action: "status" });
|
|
243
|
+
assert.deepEqual(after.details.tasks.map((t) => t.status), ["complete", "complete", "complete", "pending", "pending", "pending"]);
|
|
244
|
+
});
|
|
245
|
+
|
|
246
|
+
test("update -> pending is rejected in execute and names reopen and re-init", async () => {
|
|
247
|
+
const { call } = harness();
|
|
248
|
+
await call({ action: "init", tasks: ["a", "b", "claim-check"] });
|
|
249
|
+
for (const i of [0, 1, 2]) await call({ action: "update", index: i, status: "complete" });
|
|
250
|
+
const res = await call({ action: "update", index: 2, status: "pending" });
|
|
251
|
+
assert.equal(res.details.error, "update cannot set pending");
|
|
252
|
+
assert.deepEqual(res.details.tasks.map((t) => t.status), ["complete", "complete", "complete"]);
|
|
253
|
+
assert.equal(
|
|
254
|
+
res.content[0].text,
|
|
255
|
+
[
|
|
256
|
+
'Error: cannot set task 2 "claim-check" to pending: update never sets pending; a task is pending only from init/add.',
|
|
257
|
+
"To redo it, update it to in_progress. To recreate the whole list, re-init with {name, status}[].",
|
|
258
|
+
"Current: 0 a=complete, 1 b=complete, 2 claim-check=complete",
|
|
259
|
+
].join("\n"),
|
|
260
|
+
);
|
|
261
|
+
});
|
|
262
|
+
|
|
263
|
+
test("update at index 0 / first pending index, direct jumps, and reopens are accepted", async () => {
|
|
264
|
+
const { call } = harness();
|
|
265
|
+
await call({ action: "init", tasks: ["a", "b", "c", "d"] });
|
|
266
|
+
assert.equal((await call({ action: "update", index: 0, status: "complete" })).details.error, undefined);
|
|
267
|
+
assert.equal((await call({ action: "update", index: 1, status: "failed" })).details.error, undefined);
|
|
268
|
+
assert.equal((await call({ action: "update", index: 2, status: "skipped" })).details.error, undefined);
|
|
269
|
+
assert.equal((await call({ action: "update", index: 0, status: "in_progress" })).details.error, undefined);
|
|
270
|
+
assert.equal((await call({ action: "update", index: 1, status: "in_progress" })).details.error, undefined);
|
|
271
|
+
const res = await call({ action: "status" });
|
|
272
|
+
assert.deepEqual(res.details.tasks.map((t) => t.status), ["in_progress", "in_progress", "skipped", "pending"]);
|
|
273
|
+
});
|
|
274
|
+
|
|
275
|
+
test("wave fan-out: in_progress in increasing order succeeds, out of order is rejected naming earlier pending indices", async () => {
|
|
276
|
+
const ok = harness();
|
|
277
|
+
await ok.call({ action: "init", tasks: ["a", "b", "c"] });
|
|
278
|
+
for (const i of [0, 1, 2]) assert.equal((await ok.call({ action: "update", index: i, status: "in_progress" })).details.error, undefined);
|
|
279
|
+
const bad = harness();
|
|
280
|
+
await bad.call({ action: "init", tasks: ["a", "b", "c"] });
|
|
281
|
+
const res = await bad.call({ action: "update", index: 2, status: "in_progress" });
|
|
282
|
+
assert.equal(res.details.error, "pending-suffix violation: 0,1");
|
|
283
|
+
assert.match(res.content[0].text, /pending tasks precede it: 0 "a", 1 "b"\./);
|
|
284
|
+
});
|
|
285
|
+
|
|
286
|
+
test("init: strings, objects, and mixed forms normalize; ordering violation rejects the whole list", async () => {
|
|
287
|
+
const { call } = harness();
|
|
288
|
+
const strings = await call({ action: "init", tasks: ["a", "b"] });
|
|
289
|
+
assert.deepEqual(strings.details.tasks, [{ name: "a", status: "pending" }, { name: "b", status: "pending" }]);
|
|
290
|
+
const objects = await call({ action: "init", tasks: [{ name: "a", status: "complete" }, { name: "b", status: "skipped" }, "c"] });
|
|
291
|
+
assert.equal(objects.details.error, undefined);
|
|
292
|
+
assert.deepEqual(objects.details.tasks, [
|
|
293
|
+
{ name: "a", status: "complete" },
|
|
294
|
+
{ name: "b", status: "skipped" },
|
|
295
|
+
{ name: "c", status: "pending" },
|
|
296
|
+
]);
|
|
297
|
+
const bad = await call({ action: "init", tasks: [{ name: "a", status: "complete" }, "b", { name: "c", status: "complete" }] });
|
|
298
|
+
assert.equal(bad.details.error, "pending-suffix violation: 1");
|
|
299
|
+
assert.equal(
|
|
300
|
+
bad.content[0].text,
|
|
301
|
+
[
|
|
302
|
+
'Error: cannot init: pending tasks precede started or finished ones: 1 "b".',
|
|
303
|
+
"Rule: pending tasks must trail every started or finished task.",
|
|
304
|
+
"Reorder the list or restate those statuses truthfully, then retry.",
|
|
305
|
+
"Proposed: 0 a=complete, 1 b=pending, 2 c=complete",
|
|
306
|
+
].join("\n"),
|
|
307
|
+
);
|
|
308
|
+
assert.doesNotMatch(bad.content[0].text, /Current:/);
|
|
309
|
+
assert.deepEqual(bad.details.tasks.map((t) => t.status), ["complete", "skipped", "pending"]);
|
|
310
|
+
});
|
|
311
|
+
|
|
312
|
+
test("init ordering violation with no prior plan leaves no plan", async () => {
|
|
313
|
+
const { call, widget } = harness();
|
|
314
|
+
const bad = await call({ action: "init", tasks: ["a", { name: "b", status: "complete" }] });
|
|
315
|
+
assert.equal(bad.details.error, "pending-suffix violation: 0");
|
|
316
|
+
assert.deepEqual(bad.details.tasks, []);
|
|
317
|
+
assert.equal(widget(), undefined);
|
|
318
|
+
});
|
|
319
|
+
|
|
320
|
+
test("skipped round-trips, replays, renders \u2298, counts as done, and is never current", async () => {
|
|
321
|
+
const { call, widget, tool, theme } = harness();
|
|
322
|
+
await call({ action: "init", tasks: ["a", "b", "c"] });
|
|
323
|
+
await call({ action: "update", index: 0, status: "complete" });
|
|
324
|
+
const res = await call({ action: "update", index: 1, status: "skipped" });
|
|
325
|
+
assert.equal(res.details.error, undefined);
|
|
326
|
+
assert.match(res.content[0].text, /Plan: 2\/3 done \(1 pending, 1 skipped\)/);
|
|
327
|
+
assert.match(res.content[0].text, /\u2298 \[1\] b/);
|
|
328
|
+
const w = widget()!;
|
|
329
|
+
assert.match(w, /\u2298/);
|
|
330
|
+
assert.match(w, /\(2\/3\)/);
|
|
331
|
+
assert.match(w, /c$/);
|
|
332
|
+
const rendered = (tool().renderResult(res as any, {}, theme as any) as any).text as string;
|
|
333
|
+
assert.match(rendered, /^\u2713 Updated \(2\/3 done, 1 skipped\)$/);
|
|
334
|
+
const status = await call({ action: "status" });
|
|
335
|
+
const statusText = (tool().renderResult(status as any, {}, theme as any) as any).text as string;
|
|
336
|
+
assert.match(statusText, /^2\/3 done, 1 skipped\n/);
|
|
337
|
+
|
|
338
|
+
const replay = harness([{ type: "message", message: { role: "toolResult", toolName: "plan_tracker", details: res.details } }]);
|
|
339
|
+
await replay.fire("session_start");
|
|
340
|
+
const after = await replay.call({ action: "status" });
|
|
341
|
+
assert.deepEqual(after.details.tasks.map((t) => t.status), ["complete", "skipped", "pending"]);
|
|
342
|
+
});
|
|
343
|
+
|
|
344
|
+
test("all tasks skipped renders (M/M)", async () => {
|
|
345
|
+
const { call, widget } = harness();
|
|
346
|
+
await call({ action: "init", tasks: ["a", "b"] });
|
|
347
|
+
await call({ action: "update", index: 0, status: "skipped" });
|
|
348
|
+
await call({ action: "update", index: 1, status: "skipped" });
|
|
349
|
+
assert.match(widget()!, /\(2\/2\)/);
|
|
350
|
+
});
|
|
351
|
+
|
|
352
|
+
test("legacy-invalid snapshot: first update is rejected listing legacy offenders; truthful object-form re-init repairs it", async () => {
|
|
353
|
+
const branch = [
|
|
354
|
+
{
|
|
355
|
+
type: "message",
|
|
356
|
+
message: {
|
|
357
|
+
role: "toolResult",
|
|
358
|
+
toolName: "plan_tracker",
|
|
359
|
+
details: {
|
|
360
|
+
action: "update",
|
|
361
|
+
tasks: [
|
|
362
|
+
{ name: "a", status: "pending" },
|
|
363
|
+
{ name: "b", status: "pending" },
|
|
364
|
+
{ name: "c", status: "complete" },
|
|
365
|
+
],
|
|
366
|
+
},
|
|
367
|
+
},
|
|
368
|
+
},
|
|
369
|
+
];
|
|
370
|
+
const { call, fire } = harness(branch);
|
|
371
|
+
await fire("session_start");
|
|
372
|
+
const rejected = await call({ action: "update", index: 0, status: "complete" });
|
|
373
|
+
assert.equal(rejected.details.error, "pending-suffix violation: 1");
|
|
374
|
+
assert.deepEqual(rejected.details.tasks.map((t) => t.status), ["pending", "pending", "complete"]);
|
|
375
|
+
const repaired = await call({
|
|
376
|
+
action: "init",
|
|
377
|
+
tasks: [{ name: "a", status: "complete" }, { name: "b", status: "skipped" }, { name: "c", status: "complete" }],
|
|
378
|
+
});
|
|
379
|
+
assert.equal(repaired.details.error, undefined);
|
|
380
|
+
assert.deepEqual(repaired.details.tasks.map((t) => t.status), ["complete", "skipped", "complete"]);
|
|
381
|
+
});
|
|
382
|
+
|
|
383
|
+
test("registration: sequential, tasks schema is Array<Union<String, Object>>, description carries the contract", () => {
|
|
384
|
+
const { executionMode, parameters, description } = harness();
|
|
385
|
+
assert.equal(executionMode(), "sequential");
|
|
386
|
+
const tasksSchema = parameters().args[0].tasks.args[0];
|
|
387
|
+
assert.equal(tasksSchema.kind, "Array");
|
|
388
|
+
assert.equal(tasksSchema.args[0].kind, "Union");
|
|
389
|
+
assert.deepEqual(tasksSchema.args[0].args[0].map((s: { kind: string }) => s.kind), ["String", "Object"]);
|
|
390
|
+
assert.match(description(), /skipped/);
|
|
391
|
+
assert.match(description(), /pending tasks must trail every started or finished task/);
|
|
392
|
+
assert.match(description(), /update never sets pending/);
|
|
393
|
+
});
|
|
@@ -11,87 +11,68 @@ import type { ExtensionAPI, ExtensionContext, Theme } from "@earendil-works/pi-c
|
|
|
11
11
|
import { Text } from "@earendil-works/pi-tui";
|
|
12
12
|
import { type Static, Type } from "@sinclair/typebox";
|
|
13
13
|
|
|
14
|
-
|
|
14
|
+
const TASK_STATUSES = ["pending", "in_progress", "complete", "failed", "skipped"] as const;
|
|
15
|
+
type TaskStatus = (typeof TASK_STATUSES)[number];
|
|
15
16
|
|
|
16
|
-
interface Task {
|
|
17
|
-
|
|
18
|
-
status: TaskStatus;
|
|
19
|
-
}
|
|
20
|
-
|
|
21
|
-
interface PlanTrackerDetails {
|
|
22
|
-
action: "init" | "add" | "update" | "status" | "clear";
|
|
23
|
-
tasks: Task[];
|
|
24
|
-
error?: string;
|
|
25
|
-
}
|
|
17
|
+
interface Task { name: string; status: TaskStatus; }
|
|
18
|
+
interface PlanTrackerDetails { action: "init" | "add" | "update" | "status" | "clear"; tasks: Task[]; error?: string; }
|
|
26
19
|
|
|
27
20
|
const PlanTrackerParams = Type.Object({
|
|
28
|
-
action: StringEnum(["init", "add", "update", "status", "clear"] as const, {
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
),
|
|
36
|
-
index: Type.Optional(
|
|
37
|
-
Type.Integer({
|
|
38
|
-
minimum: 0,
|
|
39
|
-
description: "Task index, 0-based (for update)",
|
|
40
|
-
}),
|
|
41
|
-
),
|
|
42
|
-
status: Type.Optional(
|
|
43
|
-
StringEnum(["pending", "in_progress", "complete", "failed"] as const, {
|
|
44
|
-
description: "New status (for update); failed is terminal-negative (ran and did not pass)",
|
|
45
|
-
}),
|
|
46
|
-
),
|
|
21
|
+
action: StringEnum(["init", "add", "update", "status", "clear"] as const, { description: "Action to perform" }),
|
|
22
|
+
tasks: Type.Optional(Type.Array(Type.Union([
|
|
23
|
+
Type.String(),
|
|
24
|
+
Type.Object({ name: Type.String(), status: StringEnum(TASK_STATUSES, { description: "Status to recreate the task with (init only)" }) }),
|
|
25
|
+
]), { description: "Tasks for init and add. A string is a pending task. init also accepts { name, status } to recreate a list with known statuses (fresh session, amendment re-init); the whole list must satisfy the pending-suffix rule." })),
|
|
26
|
+
index: Type.Optional(Type.Integer({ minimum: 0, description: "Task index, 0-based (for update)" })),
|
|
27
|
+
status: Type.Optional(StringEnum(TASK_STATUSES, { description: "New status (for update). failed is terminal-negative (ran and did not pass); skipped is terminal, not applicable in this run, counted as done. pending is set only by init/add; to redo a task, reopen it as in_progress, or re-init with {name, status}[] to recreate a whole list." })),
|
|
47
28
|
});
|
|
48
29
|
|
|
49
30
|
export type PlanTrackerInput = Static<typeof PlanTrackerParams>;
|
|
50
31
|
|
|
51
|
-
|
|
52
|
-
|
|
32
|
+
/** Indices of every pending task that has a non-pending task after it (empty = valid snapshot). */
|
|
33
|
+
export function validateSnapshot(tasks: Task[]): number[] {
|
|
34
|
+
const offenders: number[] = [];
|
|
35
|
+
let seenNonPending = false;
|
|
36
|
+
for (let i = tasks.length - 1; i >= 0; i--) {
|
|
37
|
+
if (tasks[i].status !== "pending") seenNonPending = true;
|
|
38
|
+
else if (seenNonPending) offenders.unshift(i);
|
|
39
|
+
}
|
|
40
|
+
return offenders;
|
|
41
|
+
}
|
|
53
42
|
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
return theme.fg("error", "✗");
|
|
64
|
-
default:
|
|
65
|
-
return theme.fg("dim", "○");
|
|
66
|
-
}
|
|
67
|
-
})
|
|
68
|
-
.join("");
|
|
43
|
+
const RULE_LINE = "Rule: pending tasks must trail every started or finished task.";
|
|
44
|
+
const snapshotLine = (tasks: Task[]): string => tasks.map((t, i) => `${i} ${t.name}=${t.status}`).join(", ");
|
|
45
|
+
const offenderList = (tasks: Task[], indices: number[]): string => indices.map((i) => `${i} "${tasks[i].name}"`).join(", ");
|
|
46
|
+
const normalizeTask = (t: string | { name: string; status: TaskStatus }): Task => typeof t === "string" ? { name: t, status: "pending" } : { name: t.name, status: t.status };
|
|
47
|
+
|
|
48
|
+
const glyph = (status: TaskStatus, theme?: Theme): string => {
|
|
49
|
+
const [color, icon]: [string, string] = status === "complete" ? ["success", "✓"] : status === "in_progress" ? ["warning", "→"] : status === "failed" ? ["error", "✗"] : status === "skipped" ? ["dim", "⊘"] : ["dim", "○"];
|
|
50
|
+
return theme ? theme.fg(color as Parameters<Theme["fg"]>[0], icon) : icon;
|
|
51
|
+
};
|
|
69
52
|
|
|
70
|
-
|
|
53
|
+
const counts = (tasks: Task[]) => {
|
|
54
|
+
const by = (s: TaskStatus) => tasks.filter((t) => t.status === s).length;
|
|
55
|
+
const failed = by("failed");
|
|
56
|
+
const skipped = by("skipped");
|
|
57
|
+
return { done: by("complete") + skipped, inProgress: by("in_progress"), pending: by("pending"), failed, skipped, suffix: `${failed > 0 ? `, ${failed} failed` : ""}${skipped > 0 ? `, ${skipped} skipped` : ""}` };
|
|
58
|
+
};
|
|
59
|
+
|
|
60
|
+
function formatWidget(tasks: Task[], theme: Theme): string {
|
|
61
|
+
if (tasks.length === 0) return "";
|
|
62
|
+
const { done } = counts(tasks);
|
|
63
|
+
const icons = tasks.map((t) => glyph(t.status, theme)).join("");
|
|
71
64
|
const current = tasks.find((t) => t.status === "in_progress") ?? tasks.find((t) => t.status === "pending");
|
|
72
65
|
const currentName = current ? ` ${current.name}` : "";
|
|
73
|
-
|
|
74
|
-
return `${theme.fg("muted", "Tasks:")} ${icons} ${theme.fg("muted", `(${complete}/${tasks.length})`)}${currentName}`;
|
|
66
|
+
return `${theme.fg("muted", "Tasks:")} ${icons} ${theme.fg("muted", `(${done}/${tasks.length})`)}${currentName}`;
|
|
75
67
|
}
|
|
76
68
|
|
|
77
69
|
function formatStatus(tasks: Task[]): string {
|
|
78
70
|
if (tasks.length === 0) return "No plan active.";
|
|
79
|
-
|
|
80
|
-
const
|
|
81
|
-
const
|
|
82
|
-
const
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
const lines: string[] = [];
|
|
86
|
-
lines.push(
|
|
87
|
-
`Plan: ${complete}/${tasks.length} complete (${inProgress} in progress, ${pending} pending, ${failed} failed)`,
|
|
88
|
-
);
|
|
89
|
-
lines.push("");
|
|
90
|
-
for (let i = 0; i < tasks.length; i++) {
|
|
91
|
-
const t = tasks[i];
|
|
92
|
-
const icon = t.status === "complete" ? "✓" : t.status === "in_progress" ? "→" : t.status === "failed" ? "✗" : "○";
|
|
93
|
-
lines.push(` ${icon} [${i}] ${t.name}`);
|
|
94
|
-
}
|
|
71
|
+
const c = counts(tasks);
|
|
72
|
+
const groups = [[c.inProgress, "in progress"], [c.pending, "pending"], [c.failed, "failed"], [c.skipped, "skipped"]] as const;
|
|
73
|
+
const detail = groups.filter(([n]) => n > 0).map(([n, label]) => `${n} ${label}`);
|
|
74
|
+
const lines: string[] = [`Plan: ${c.done}/${tasks.length} done${detail.length > 0 ? ` (${detail.join(", ")})` : ""}`, ""];
|
|
75
|
+
for (let i = 0; i < tasks.length; i++) lines.push(` ${glyph(tasks[i].status)} [${i}] ${tasks[i].name}`);
|
|
95
76
|
return lines.join("\n");
|
|
96
77
|
}
|
|
97
78
|
|
|
@@ -134,8 +115,9 @@ export default function (pi: ExtensionAPI) {
|
|
|
134
115
|
name: "plan_tracker",
|
|
135
116
|
label: "Plan Tracker",
|
|
136
117
|
description:
|
|
137
|
-
"Track progress while EXECUTING an implementation plan (the implement phase), a verify-phase conformance fix wave, or another bounded gate checklist (e.g. pre-merge PR verification). Statuses: pending, in_progress, complete, failed (terminal-negative:
|
|
118
|
+
"Track progress while EXECUTING an implementation plan (the implement phase), a verify-phase conformance fix wave, or another bounded gate checklist (e.g. pre-merge PR verification). Statuses: pending (not yet touched), in_progress (started; several at once is fine), complete, failed (terminal-negative: ran and did not pass; never counted done), skipped (terminal: not applicable in this run; counted done). Rule: pending tasks must trail every started or finished task - an update that would leave a pending task ahead of a non-pending one is rejected with the fix. update never sets pending; init and add do. Actions: init (set task list; elements are task-name strings or { name, status } to recreate a list with known statuses), add (append tasks as pending; existing statuses preserved), update (change task status), status (show current state), clear (remove plan). Do NOT use for brainstorming, research, or planning checklists: those phases are open-ended and a bounded task list misrepresents them as a fixed N-step process.",
|
|
138
119
|
parameters: PlanTrackerParams,
|
|
120
|
+
executionMode: "sequential",
|
|
139
121
|
|
|
140
122
|
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
|
141
123
|
switch (params.action) {
|
|
@@ -150,7 +132,20 @@ export default function (pi: ExtensionAPI) {
|
|
|
150
132
|
} as PlanTrackerDetails,
|
|
151
133
|
};
|
|
152
134
|
}
|
|
153
|
-
|
|
135
|
+
const proposed = params.tasks.map(normalizeTask);
|
|
136
|
+
const offenders = validateSnapshot(proposed);
|
|
137
|
+
if (offenders.length > 0) {
|
|
138
|
+
return {
|
|
139
|
+
content: [{ type: "text", text: [
|
|
140
|
+
`Error: cannot init: pending tasks precede started or finished ones: ${offenderList(proposed, offenders)}.`,
|
|
141
|
+
RULE_LINE,
|
|
142
|
+
"Reorder the list or restate those statuses truthfully, then retry.",
|
|
143
|
+
`Proposed: ${snapshotLine(proposed)}`,
|
|
144
|
+
].join("\n") }],
|
|
145
|
+
details: { action: "init", tasks: tasks.map((t) => ({ ...t })), error: `pending-suffix violation: ${offenders.join(",")}` } as PlanTrackerDetails,
|
|
146
|
+
};
|
|
147
|
+
}
|
|
148
|
+
tasks = proposed;
|
|
154
149
|
updateWidget(ctx);
|
|
155
150
|
return {
|
|
156
151
|
content: [
|
|
@@ -174,7 +169,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
174
169
|
} as PlanTrackerDetails,
|
|
175
170
|
};
|
|
176
171
|
}
|
|
177
|
-
tasks.push(...params.tasks.map((
|
|
172
|
+
tasks.push(...params.tasks.map((t) => ({ name: normalizeTask(t).name, status: "pending" as TaskStatus })));
|
|
178
173
|
updateWidget(ctx);
|
|
179
174
|
return {
|
|
180
175
|
content: [
|
|
@@ -223,7 +218,32 @@ export default function (pi: ExtensionAPI) {
|
|
|
223
218
|
} as PlanTrackerDetails,
|
|
224
219
|
};
|
|
225
220
|
}
|
|
226
|
-
tasks[params.index]
|
|
221
|
+
const target = tasks[params.index];
|
|
222
|
+
if (params.status === "pending") {
|
|
223
|
+
return {
|
|
224
|
+
content: [{ type: "text", text: [
|
|
225
|
+
`Error: cannot set task ${params.index} "${target.name}" to pending: update never sets pending; a task is pending only from init/add.`,
|
|
226
|
+
"To redo it, update it to in_progress. To recreate the whole list, re-init with {name, status}[].",
|
|
227
|
+
`Current: ${snapshotLine(tasks)}`,
|
|
228
|
+
].join("\n") }],
|
|
229
|
+
details: { action: "update", tasks: tasks.map((t) => ({ ...t })), error: "update cannot set pending" } as PlanTrackerDetails,
|
|
230
|
+
};
|
|
231
|
+
}
|
|
232
|
+
const candidate = tasks.map((t, i) => (i === params.index ? { ...t, status: params.status as TaskStatus } : { ...t }));
|
|
233
|
+
const offenders = validateSnapshot(candidate);
|
|
234
|
+
if (offenders.length > 0) {
|
|
235
|
+
return {
|
|
236
|
+
content: [{ type: "text", text: [
|
|
237
|
+
`Error: cannot set task ${params.index} "${target.name}" to ${params.status}: pending tasks precede it: ${offenderList(tasks, offenders)}.`,
|
|
238
|
+
RULE_LINE,
|
|
239
|
+
"Fix first, then retry: update each listed task to in_progress (working on it now), complete, failed, or skipped (not applicable in this run); or, if the list order itself is wrong, re-init with {name, status}[] keeping every task and its true status so the untouched ones trail.",
|
|
240
|
+
"Never clear, drop tasks, or record a status the work has not actually reached.",
|
|
241
|
+
`Current: ${snapshotLine(tasks)}`,
|
|
242
|
+
].join("\n") }],
|
|
243
|
+
details: { action: "update", tasks: tasks.map((t) => ({ ...t })), error: `pending-suffix violation: ${offenders.join(",")}` } as PlanTrackerDetails,
|
|
244
|
+
};
|
|
245
|
+
}
|
|
246
|
+
tasks = candidate;
|
|
227
247
|
updateWidget(ctx);
|
|
228
248
|
return {
|
|
229
249
|
content: [
|
|
@@ -309,12 +329,9 @@ export default function (pi: ExtensionAPI) {
|
|
|
309
329
|
0,
|
|
310
330
|
);
|
|
311
331
|
case "update": {
|
|
312
|
-
const
|
|
313
|
-
const failed = taskList.filter((t) => t.status === "failed").length;
|
|
314
|
-
const suffix = failed > 0 ? `, ${failed} failed` : "";
|
|
332
|
+
const c = counts(taskList);
|
|
315
333
|
return new Text(
|
|
316
|
-
theme.fg("success", "✓ ") +
|
|
317
|
-
theme.fg("muted", `Updated (${complete}/${taskList.length} complete${suffix})`),
|
|
334
|
+
theme.fg("success", "✓ ") + theme.fg("muted", `Updated (${c.done}/${taskList.length} done${c.suffix})`),
|
|
318
335
|
0,
|
|
319
336
|
0,
|
|
320
337
|
);
|
|
@@ -323,19 +340,10 @@ export default function (pi: ExtensionAPI) {
|
|
|
323
340
|
if (taskList.length === 0) {
|
|
324
341
|
return new Text(theme.fg("dim", "No plan active"), 0, 0);
|
|
325
342
|
}
|
|
326
|
-
const
|
|
327
|
-
|
|
328
|
-
const suffix = failed > 0 ? `, ${failed} failed` : "";
|
|
329
|
-
let text = theme.fg("muted", `${complete}/${taskList.length} complete${suffix}`);
|
|
343
|
+
const c = counts(taskList);
|
|
344
|
+
let text = theme.fg("muted", `${c.done}/${taskList.length} done${c.suffix}`);
|
|
330
345
|
for (const t of taskList) {
|
|
331
|
-
const icon =
|
|
332
|
-
t.status === "complete"
|
|
333
|
-
? theme.fg("success", "✓")
|
|
334
|
-
: t.status === "in_progress"
|
|
335
|
-
? theme.fg("warning", "→")
|
|
336
|
-
: t.status === "failed"
|
|
337
|
-
? theme.fg("error", "✗")
|
|
338
|
-
: theme.fg("dim", "○");
|
|
346
|
+
const icon = glyph(t.status, theme);
|
|
339
347
|
text += `\n${icon} ${theme.fg("muted", t.name)}`;
|
|
340
348
|
}
|
|
341
349
|
return new Text(text, 0, 0);
|
package/package.json
CHANGED
|
@@ -61,7 +61,8 @@ Work through the items below **in order**. This is your own checklist to follow,
|
|
|
61
61
|
substep. Unconditional, foreground, no user interaction — the next thing the user
|
|
62
62
|
sees is a questionary question. This produces the draft; the step below consumes it.
|
|
63
63
|
4. **Understand the idea against the draft** — `Read` the draft, verify load-bearing
|
|
64
|
-
claims against real code, ask questions one at a time, append citable findings
|
|
64
|
+
claims against real code, ask questions one at a time, append citable findings;
|
|
65
|
+
then state the chat premise note (section 3) before approaches
|
|
65
66
|
5. **Propose 2-3 approaches** — with trade-offs and a recommendation
|
|
66
67
|
6. **Present the design** — in two rounds, one approval each
|
|
67
68
|
7. **Write the spec** — to `doc/specs/` (see [Filename Convention](#filename-convention)); then mark any known superseded predecessor(s) per [Marking superseded specs](#marking-superseded-specs), at the exact-order position defined in [Spec Self-Review](#spec-self-review-before-user-review-gate)
|
|
@@ -143,13 +144,30 @@ path.
|
|
|
143
144
|
section starts that answer; confirm it before designing from scratch.
|
|
144
145
|
- Ask questions **one at a time** to refine the idea. Prefer multiple-choice; one
|
|
145
146
|
question per message. Focus on: purpose, constraints, success criteria, who/what
|
|
146
|
-
it touches.
|
|
147
|
+
it touches. Before asking, check whether the code, the docs, or the issue tracker
|
|
148
|
+
already answer it - if so, look it up instead of asking (dispatch a subagent when
|
|
149
|
+
the lookup is costly), and ask only what no source can answer. Every question you
|
|
150
|
+
ask, including asking the user to accept a corrected fact, ends with the line
|
|
151
|
+
`Recommendation: <answer> - <why>` - the answer, then " - ", then the reason.
|
|
147
152
|
- **Append bar:** append to the draft's `## Appended during questionary` only
|
|
148
153
|
findings the spec will cite — schema shapes, hard constraints, ticket-vs-code
|
|
149
154
|
contradictions, user answers that changed scope. Not a log of every grep.
|
|
150
155
|
(Appending uses `edit`; the `edit` prohibition in the spec-writing step applies
|
|
151
156
|
only there.)
|
|
152
157
|
|
|
158
|
+
Before proposing approaches, state the premise note in chat: what the design depends
|
|
159
|
+
on, which of those claims the sources support and where you saw it (a file and line,
|
|
160
|
+
a doc, a ticket), which they disprove - give the corrected fact and where you found
|
|
161
|
+
it - and which remain unverified, naming the lookup you tried. Write it as a note a
|
|
162
|
+
person can act on: full sentences, no status-keyword lists, no template; when the
|
|
163
|
+
design depends on no claims at all, one sentence saying so is enough. An unverified claim is not a stop - it enters the
|
|
164
|
+
spec as an Open Question or a stated assumption. If a claim the design depends on was
|
|
165
|
+
contradicted, the note is your next message and it ends by asking the user to accept
|
|
166
|
+
the corrected fact or explicitly override it; nothing else continues - no other
|
|
167
|
+
questions, no approaches - until they answer, and the outcome is recorded in the draft's
|
|
168
|
+
`## Appended during questionary` so spec-writing carries it into `## Problem` or the relevant `## Design`
|
|
169
|
+
decision.
|
|
170
|
+
|
|
153
171
|
### 4. Explore approaches
|
|
154
172
|
|
|
155
173
|
- Propose 2-3 different approaches with trade-offs.
|
|
@@ -354,7 +372,7 @@ Execute this section in place from any later phase. Do not invoke `/skill:brains
|
|
|
354
372
|
|
|
355
373
|
1. Edit the spec. Show `git --no-pager diff -- <spec path>` and one line of impact (affected plan tasks / waves, or "no plan yet").
|
|
356
374
|
2. Wait for approval. Change request -> revise, re-show.
|
|
357
|
-
3. No plan yet -> commit the spec; continue. Plan exists -> update affected anchors and tasks: `plan_tracker` `add` for new tasks
|
|
375
|
+
3. No plan yet -> commit the spec; continue. Plan exists -> update affected anchors and tasks: `plan_tracker` `add` for new tasks; anchor-changed completed tasks are reopened as `in_progress` and re-run the task loop (`update` never sets `pending`). A removed task is deleted from the plan; then re-`init` the tracker with `{ name, status }` elements: preserved tasks keep their order and statuses, reopened tasks are `in_progress` in place, every still-`pending` task (including newly added ones, whatever wave label they carry) trails the non-pending ones, removed tasks are the only deletions (the only permitted `init` after handoff; never `clear`). Re-run `plan_check` until it passes, commit spec + plan together; continue. A task reopened while `verify` or `ship` is in progress: `phase_tracker({ action: "skip", phase: "<current>", reason: "amendment reopened Task N" })`, then `phase_tracker({ action: "start", phase: "implement", force: true })`; later phases re-enter with `force: true` and rerun in full.
|
|
358
376
|
|
|
359
377
|
Redraw test: the diff changes the problem statement, adds or removes a component, or moves a component boundary -> redraw. A change inside one component (a persistence mechanism, a worker's HTTP client, dropping a fallback and its task) -> amend. State the call in the same message as the diff; the user overrides either way.
|
|
360
378
|
|
|
@@ -382,6 +400,7 @@ Redraw: keep the worktree and the approved spec file. `plan_tracker({ action: "c
|
|
|
382
400
|
- Running, deploying, or validating the proposed change before approval.
|
|
383
401
|
- Proceeding to `/skill:writing-plans` before the user approves the spec, or invoking `/skill:brainstorming` to amend an approved spec.
|
|
384
402
|
- Writing a replacement spec without the known predecessor's banner, or offering a multi-spec split that fails `../shape-ticket/reference/split-axes.md`.
|
|
403
|
+
- Proposing approaches while a contradicted claim is unresolved.
|
|
385
404
|
|
|
386
405
|
## Project overrides
|
|
387
406
|
|
|
@@ -125,11 +125,12 @@ configured) all emitted for manual execution, none auto-posted.
|
|
|
125
125
|
stage plus one task per AC - status mappings below apply only after that
|
|
126
126
|
init: pass / `satisfied` / `not externally observable` /
|
|
127
127
|
`unverified: no delivery target` / `allowed gap` / `proposed descope` ->
|
|
128
|
-
`complete`; failed stage / `unexplained gap` -> `failed`; skipped stage 2 ->
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
128
|
+
`complete`; failed stage / `unexplained gap` -> `failed`; skipped stage 2 ->
|
|
129
|
+
`skipped` (rendered ⊘, counted as done - never `complete` under a renamed
|
|
130
|
+
title). AC tasks follow the four stage tasks; record each AC verdict while
|
|
131
|
+
stage 3 is `in_progress` - the tracker rejects a verdict recorded behind a
|
|
132
|
+
still-pending stage. Optional-degrading: a native task list, or no tracking
|
|
133
|
+
at all, on harnesses without `plan_tracker`; absence is never a hard stop.
|
|
133
134
|
|
|
134
135
|
**Stage 0 - Pre-flight.** Fetch the ticket and all comments. Check the
|
|
135
136
|
current tracker status first: if it is already in a terminal/done state,
|
|
@@ -83,14 +83,18 @@ configuration.
|
|
|
83
83
|
|
|
84
84
|
## Progress tracking
|
|
85
85
|
|
|
86
|
-
Use `plan_tracker`, never `phase_tracker`. Init with the
|
|
87
|
-
`provision worktree`, `resolve evidence`, `claim-check
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
(shown crossed, error color)
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
86
|
+
Use `plan_tracker`, never `phase_tracker`. Init with the first four stages:
|
|
87
|
+
`gather`, `provision worktree`, `resolve evidence`, `claim-check`. While
|
|
88
|
+
`claim-check` is `in_progress`, `add` one task per material claim as the
|
|
89
|
+
Verifier enumerates them and record each verdict: a matched claim ->
|
|
90
|
+
`complete`, a contradicted claim -> `failed` (shown crossed, error color).
|
|
91
|
+
Once every claim is terminal and `claim-check` is closed, `add` `review` and
|
|
92
|
+
`consent menu` and continue. Stages are never inited ahead of the claims:
|
|
93
|
+
the tracker rejects a verdict recorded behind a still-pending stage. A
|
|
94
|
+
failed stage or claim stays `failed` while the skill stops at the menu -
|
|
95
|
+
never marked complete to move on. On a harness without the `plan_tracker`
|
|
96
|
+
tool: fall back to a plain checklist (or skip if none is available);
|
|
97
|
+
functionality is unchanged either way.
|
|
94
98
|
|
|
95
99
|
## Assessment
|
|
96
100
|
|
|
@@ -171,9 +171,9 @@ Auto-selected at handoff by `writing-plans` (any wave with ≥2 tasks) when the
|
|
|
171
171
|
**Progress tracking (`plan_tracker`).** `plan_tracker` is a flat list with no native group concept, so waves are *encoded*, not modeled:
|
|
172
172
|
|
|
173
173
|
- **Consume, preserve, recover only if absent:** consume the wave-ordered list initialized at writing-plans handoff; indices are positional and stable, so never re-init on continuation or mid-run. Only direct recovery with no tracker initializes the full plan list once before dispatch.
|
|
174
|
-
- **Wave fan-out → `in_progress`:** unconditionally mark every task index in the wave `in_progress` before dispatch. Multiple simultaneous entries are expected (sequential mode has one).
|
|
174
|
+
- **Wave fan-out → `in_progress`:** unconditionally mark every task index in the wave `in_progress` before dispatch, in increasing index order (the tracker validates each call against the previous one and rejects a start while an earlier index is still `pending`). Multiple simultaneous entries are expected (sequential mode has one).
|
|
175
175
|
- **Wave commit → `complete`:** after the wave's gate passes and it commits, unconditionally mark all those same indices `complete`. `complete` = durably committed, so a task in conflict fallback stays `in_progress` until its wave commits.
|
|
176
|
-
- **Lifecycle per task:** `pending → in_progress (wave fan-out) → complete (wave commit)
|
|
176
|
+
- **Lifecycle per task:** `pending → in_progress (wave fan-out) → complete (wave commit)`; `failed` when a task ran and did not pass; `skipped` when the plan drops it as not applicable (terminal, counted done). A rejected `update` names the earlier pending tasks and the legal fixes - record those tasks' true state, never `clear` or re-`init` to move on.
|
|
177
177
|
- **Widget caveat (known, deliberately unfixed).** The persistent `plan_tracker` widget's icon strip (`○ → ✓`) and `(c/total)` count reflect every task, but its trailing *name* shows only the **first** `in_progress` task. In parallel mode the icon strip and the `status` action are the full in-flight view; a richer multi-task widget is a separate extension change, out of scope (YAGNI).
|
|
178
178
|
- **Sequential mode:** consume the same existing full list, one `in_progress` index at a time; wave prefixes are harmless.
|
|
179
179
|
|
|
@@ -143,7 +143,7 @@ prerequisites hold.
|
|
|
143
143
|
|
|
144
144
|
Per round:
|
|
145
145
|
|
|
146
|
-
1. **Synchronize gap tasks** — append only a genuinely new gap that is entering remediation, named `Gn: <gap origin clause verbatim, truncated>`; never `init`. Find existing gaps by their exact `Gn:` prefix and reuse that index even if origin wording changes. Carried-OPEN inventory-only gaps add nothing. Before dispatch, mark every remediated gap's existing index `in_progress
|
|
146
|
+
1. **Synchronize gap tasks** — append only a genuinely new gap that is entering remediation, named `Gn: <gap origin clause verbatim, truncated>`; never `init`. Find existing gaps by their exact `Gn:` prefix and reuse that index even if origin wording changes. Carried-OPEN inventory-only gaps add nothing. Before dispatch, mark every remediated gap's existing index `in_progress`, in increasing index order (the tracker rejects a start while a lower index is still `pending`); a re-audit needing more work reopens that same `Gn` index. The lifecycle traces `[T1,T2]`, then `[T1,T2,G1]`, then `[T1,T2,G1,G2]`; no test-retry or review-round wrapper task.
|
|
147
147
|
2. **Fix wave** — select gaps greedily in `Gn` order: take each `fix` gap unless a gap it `conflicts` with (per the report's `Parallel-safe:` line) is already taken; the certificate's `disjoint` grouping is ignored, and a certificate still malformed after the one re-ask means every gap `conflicts` with every other. Held gaps carry to the next round; the wave is never empty while an eligible `fix` gap exists. Dispatch **one** call - `subagent({ context: "fresh", async: false, tasks: [...] })` - with one `implementer` task per selected gap (`worktree: true`, `cwd` = the conformance worktree, task = the gap block verbatim with `touched-files` as the ownership boundary). A single gap is a one-task `tasks` call; a lone `agent: "implementer"` call never appears in this loop. For an `UNAUTHORIZED` `fix` gap
|
|
148
148
|
whose `evidence` opens with the over-spec provenance (`spec "<section>" - "<clause>" (over-spec)`), the orchestrator adds the spec path to that gap's `touched-files` before dispatch,
|
|
149
149
|
so the implementer deletes the surface **and** the clause/AC line in the same
|