pi-gauntlet 5.5.3 → 5.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/extensions/phase-tracker.test.ts +319 -3
- package/extensions/phase-tracker.ts +80 -9
- package/extensions/plan-tracker.test.ts +211 -9
- package/extensions/plan-tracker.ts +95 -87
- package/extensions/test-support/pi-stubs.mjs +9 -9
- package/package.json +1 -1
- package/skills/brainstorming/SKILL.md +1 -1
- package/skills/check-delivery/SKILL.md +6 -5
- package/skills/gatekeep-pr/SKILL.md +12 -8
- package/skills/subagent-driven-development/SKILL.md +2 -2
- package/skills/verification-before-completion/reference/conformance-check.md +11 -6
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,17 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## v5.6.0 - 2026-09-16
|
|
4
|
+
|
|
5
|
+
- `plan-tracker`: pending-suffix rule enforced at runtime - an `update` that would leave a `pending` task ahead of a started or finished one is rejected with a message naming every offending task, the legal fixes, and the current snapshot; state is unchanged and widget replay / `phase_tracker` ignore the rejection. `update` never sets `pending`. New terminal status `skipped` (`⊘`, not applicable in this run, counted done). `init` accepts `{ name, status }` elements (mixable with strings) to recreate a list with known statuses. Registered with sequential execution. Counts are labelled `done` (`complete + skipped`).
|
|
6
|
+
- `phase-tracker`: `implement` auto-completes when every task is `complete` or `skipped`.
|
|
7
|
+
- `check-delivery`, `gatekeep-pr`, `subagent-driven-development`, `brainstorming`: tracker usage aligned with the rule - a skipped stage is `skipped`, gatekeep-pr inits four stages and adds claims/review/consent progressively, wave fan-out marks `in_progress` in increasing index order, the amendment path re-inits with statuses instead of setting tasks back to `pending`.
|
|
8
|
+
|
|
9
|
+
## v5.5.4 - 2026-09-14
|
|
10
|
+
|
|
11
|
+
- `phase-tracker`: new `phase_tracker` action `grant_fix_rounds` records an explicit human approval of N more conformance fix rounds (reason quotes the human), accepted only while the cap block is live; each qualifying implementer wave spends one granted round, credits replay with the session and reset with the audit latch. The cap-block message now leads with that action and names the exact `.pi/settings.json` for the `enforce: false` last resort (applies without restart, whole-block precedence).
|
|
12
|
+
- `gauntlet_setting` is registered with sequential execution, so a settings write and a verifying read batched in one message run in order.
|
|
13
|
+
- `conformance-check.md`: an explicit human approval re-enters the fix loop via `grant_fix_rounds`; without it escalation stays terminal.
|
|
14
|
+
|
|
3
15
|
## v5.5.3 - 2026-09-13
|
|
4
16
|
|
|
5
17
|
- `code-reviewer`: verdict is a stated function of severity - a Critical or Moderate finding means `FIX_FIRST`, Minor-only and clean reports mean `SHIP`, matching the orchestrating skills. (#30)
|
|
@@ -58,7 +58,7 @@ const resumedBranch = (rest: Partial<Record<Phase, Status>>) => [
|
|
|
58
58
|
|
|
59
59
|
function harness(options: { cwd?: string; branch?: unknown[]; idle?: boolean; beforeSettled?: (setIdle: (idle: boolean) => void) => void; sendThrows?: boolean; model?: { provider: string; id: string }; thinkingLevel?: string } = {}) {
|
|
60
60
|
const handlers = new Map<string, ((event: unknown, ctx: unknown) => unknown)[]>();
|
|
61
|
-
const tools: { name: string; execute: (...args: any[]) => unknown }[] = [];
|
|
61
|
+
const tools: { name: string; executionMode?: string; parameters?: any; execute: (...args: any[]) => unknown }[] = [];
|
|
62
62
|
const sent: { message: any; options: any }[] = [];
|
|
63
63
|
let idle = options.idle ?? true;
|
|
64
64
|
let branch = options.branch ?? [];
|
|
@@ -76,7 +76,7 @@ function harness(options: { cwd?: string; branch?: unknown[]; idle?: boolean; be
|
|
|
76
76
|
registered.push(handler);
|
|
77
77
|
handlers.set(event, registered);
|
|
78
78
|
},
|
|
79
|
-
registerTool(tool: { name: string; executionMode?: string; execute: (...args: any[]) => unknown }) {
|
|
79
|
+
registerTool(tool: { name: string; executionMode?: string; parameters?: any; execute: (...args: any[]) => unknown }) {
|
|
80
80
|
tools.push(tool);
|
|
81
81
|
},
|
|
82
82
|
sendMessage(message: unknown, sendOptions: unknown) {
|
|
@@ -370,6 +370,27 @@ test("all-complete plan activity auto-completes an active implement phase", asyn
|
|
|
370
370
|
assert.equal(status.details.phases.implement.status, "complete");
|
|
371
371
|
});
|
|
372
372
|
|
|
373
|
+
test("plan activity auto-completes implement with complete+skipped, not with failed, not from an errored result", async () => {
|
|
374
|
+
const implementStatus = async (tasks: { status: string }[], isError = false) => {
|
|
375
|
+
const h = harness({ branch: implementBranch() });
|
|
376
|
+
await h.emit("session_start");
|
|
377
|
+
await h.emitEvent("tool_execution_end", {
|
|
378
|
+
toolName: "plan_tracker",
|
|
379
|
+
isError,
|
|
380
|
+
result: { details: { tasks } },
|
|
381
|
+
});
|
|
382
|
+
const tool = h.tools.find((t) => t.name === "phase_tracker")!;
|
|
383
|
+
const status = (await tool.execute("t1", { action: "status" }, undefined, undefined, h.ctx)) as {
|
|
384
|
+
details: { phases: { implement: { status: string } } };
|
|
385
|
+
};
|
|
386
|
+
return status.details.phases.implement.status;
|
|
387
|
+
};
|
|
388
|
+
assert.equal(await implementStatus([{ status: "complete" }, { status: "skipped" }]), "complete");
|
|
389
|
+
assert.equal(await implementStatus([{ status: "skipped" }, { status: "skipped" }]), "complete");
|
|
390
|
+
assert.equal(await implementStatus([{ status: "complete" }, { status: "failed" }]), "in_progress");
|
|
391
|
+
assert.equal(await implementStatus([{ status: "complete" }, { status: "skipped" }], true), "in_progress");
|
|
392
|
+
});
|
|
393
|
+
|
|
373
394
|
test("cold implement and verify completions ignore unfinished snapshots", async () => {
|
|
374
395
|
for (const phase of ["implement", "verify"] as const) {
|
|
375
396
|
const h = harness({
|
|
@@ -1275,7 +1296,7 @@ test("cap guard: default 3 waves pass, the fourth is blocked with the escalation
|
|
|
1275
1296
|
}
|
|
1276
1297
|
const blocked = await firstCallResult(h, "c4", implementerWave(1));
|
|
1277
1298
|
assert.equal(blocked?.block, true);
|
|
1278
|
-
assert.match(blocked?.reason ?? "", /fix round
|
|
1299
|
+
assert.match(blocked?.reason ?? "", /3 fix round\(s\) used against a cap of 3 \(granted rounds included\); escalate to the human/);
|
|
1279
1300
|
});
|
|
1280
1301
|
|
|
1281
1302
|
test("cap guard: maxFixRounds 0 blocks the first wave; 1 blocks the second; a chain implementer is cap-checked and counted", async () => {
|
|
@@ -1422,3 +1443,298 @@ test("replay: two implementer waves after the audit in verify restore fixRounds
|
|
|
1422
1443
|
await inShip.emit("session_start");
|
|
1423
1444
|
assert.equal(await firstCallResult(inShip, "c1", implementerWave(1)), undefined);
|
|
1424
1445
|
});
|
|
1446
|
+
|
|
1447
|
+
// --- Human overrule of the fix-round cap (spec 2026-09-14-fix-round-human-overrule) ---
|
|
1448
|
+
|
|
1449
|
+
const grantResult = (rounds: number, reason = "human approved") => ({
|
|
1450
|
+
type: "message",
|
|
1451
|
+
message: { role: "toolResult", toolName: "phase_tracker", details: {
|
|
1452
|
+
action: "grant_fix_rounds", rounds, reason,
|
|
1453
|
+
phases: phases({ brainstorm: "complete", plan: "complete", implement: "complete", verify: "in_progress" }),
|
|
1454
|
+
} },
|
|
1455
|
+
});
|
|
1456
|
+
|
|
1457
|
+
const grant = async (h: ReturnType<typeof harness>, id: string, input: Record<string, unknown>) =>
|
|
1458
|
+
(await h.tools.find((t) => t.name === "phase_tracker")!.execute(id, { action: "grant_fix_rounds", ...input }, undefined, undefined, h.ctx)) as {
|
|
1459
|
+
content: { type: string; text: string }[];
|
|
1460
|
+
details: { action: string; rounds?: number; reason?: string; error?: string };
|
|
1461
|
+
};
|
|
1462
|
+
|
|
1463
|
+
const exhaust = async (h: ReturnType<typeof harness>, rounds: number, prefix = "x") => {
|
|
1464
|
+
for (let i = 1; i <= rounds; i++) {
|
|
1465
|
+
assert.equal(await firstCallResult(h, `${prefix}${i}`, implementerWave(1)), undefined);
|
|
1466
|
+
await h.emitEvent("tool_result", okWave(`${prefix}${i}`));
|
|
1467
|
+
}
|
|
1468
|
+
};
|
|
1469
|
+
|
|
1470
|
+
test("grant: funds extra waves and rejects another grant while credits remain", async () => {
|
|
1471
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { enforce: true } } });
|
|
1472
|
+
await h.emit("session_start");
|
|
1473
|
+
await exhaust(h, 3);
|
|
1474
|
+
const res = await grant(h, "g1", { rounds: 2, reason: "I'm approving 2 more rounds" });
|
|
1475
|
+
assert.equal(res.details.error, undefined);
|
|
1476
|
+
assert.deepEqual([res.details.rounds, res.details.reason], [2, "I'm approving 2 more rounds"]);
|
|
1477
|
+
assert.equal((await grant(h, "g2", { rounds: 1, reason: "more" })).details.error, "grant_fix_rounds: 2 granted round(s) still unused; spend them before granting more");
|
|
1478
|
+
await exhaust(h, 2, "c");
|
|
1479
|
+
const blocked = await firstCallResult(h, "c3", implementerWave(1));
|
|
1480
|
+
assert.equal(blocked?.block, true);
|
|
1481
|
+
assert.match(blocked?.reason ?? "", /5 fix round\(s\) used against a cap of 3 \(granted rounds included\)/);
|
|
1482
|
+
});
|
|
1483
|
+
|
|
1484
|
+
test("grant: schema declares rounds as integer 1..MAX_SAFE_INTEGER; missing rounds or empty reason error without state change", async () => {
|
|
1485
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1486
|
+
await h.emit("session_start");
|
|
1487
|
+
const params = h.tools.find((t) => t.name === "phase_tracker")!.parameters;
|
|
1488
|
+
const roundOptions = params.args[0].rounds.args[0].args[0];
|
|
1489
|
+
assert.equal(params.args[0].rounds.args[0].kind, "Integer");
|
|
1490
|
+
assert.deepEqual([roundOptions.minimum, roundOptions.maximum], [1, Number.MAX_SAFE_INTEGER]);
|
|
1491
|
+
assert.ok(params.args[0].action.values.includes("grant_fix_rounds"));
|
|
1492
|
+
|
|
1493
|
+
const noRounds = await grant(h, "g1", { reason: "ok" });
|
|
1494
|
+
assert.equal(noRounds.details.error, "grant_fix_rounds requires rounds: a positive integer");
|
|
1495
|
+
assert.equal(noRounds.details.rounds, undefined);
|
|
1496
|
+
const noReason = await grant(h, "g2", { rounds: 1, reason: " " });
|
|
1497
|
+
assert.equal(noReason.details.error, "grant_fix_rounds requires reason: the human's approval, quoted");
|
|
1498
|
+
assert.equal(noReason.details.rounds, undefined);
|
|
1499
|
+
assert.equal((await firstCallResult(h, "c1", implementerWave(1)))?.block, true, "no credit was granted");
|
|
1500
|
+
});
|
|
1501
|
+
|
|
1502
|
+
test("grant: rejected when no cap block is live - before the audit, in implement, below the cap, enforce off, and in a child", async () => {
|
|
1503
|
+
const expectNotLive = async (h: ReturnType<typeof harness>, used: number, cap: number) => {
|
|
1504
|
+
const res = await grant(h, "g", { rounds: 1, reason: "ok" });
|
|
1505
|
+
assert.equal(res.details.error, `grant_fix_rounds: no fix-round cap block is active (${used} used, cap ${cap}); nothing to overrule`);
|
|
1506
|
+
};
|
|
1507
|
+
|
|
1508
|
+
const preAudit = harness({ cwd: tempCwd({ piGauntlet: { closureReview: { maxFixRounds: 0 } } }), branch: verifyBranch() });
|
|
1509
|
+
await preAudit.emit("session_start");
|
|
1510
|
+
await expectNotLive(preAudit, 0, 0);
|
|
1511
|
+
|
|
1512
|
+
const implement = harness({ cwd: tempCwd({ piGauntlet: { closureReview: { maxFixRounds: 0 } } }), branch: implementBranch([subagentResult(["conformance-reviewer"])]) });
|
|
1513
|
+
await implement.emit("session_start");
|
|
1514
|
+
await expectNotLive(implement, 0, 0);
|
|
1515
|
+
|
|
1516
|
+
const below = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 3 } } });
|
|
1517
|
+
await below.emit("session_start");
|
|
1518
|
+
await exhaust(below, 2);
|
|
1519
|
+
await expectNotLive(below, 2, 3);
|
|
1520
|
+
|
|
1521
|
+
const off = guardedHarness({ piGauntlet: { closureReview: { enforce: false, maxFixRounds: 0 } } });
|
|
1522
|
+
await off.emit("session_start");
|
|
1523
|
+
await expectNotLive(off, 0, 0);
|
|
1524
|
+
|
|
1525
|
+
const priorDepth = process.env.PI_SUBAGENT_DEPTH;
|
|
1526
|
+
process.env.PI_SUBAGENT_DEPTH = "1";
|
|
1527
|
+
let child: ReturnType<typeof harness>;
|
|
1528
|
+
try {
|
|
1529
|
+
child = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1530
|
+
} finally {
|
|
1531
|
+
if (priorDepth === undefined) delete process.env.PI_SUBAGENT_DEPTH;
|
|
1532
|
+
else process.env.PI_SUBAGENT_DEPTH = priorDepth;
|
|
1533
|
+
}
|
|
1534
|
+
await child.emit("session_start");
|
|
1535
|
+
await expectNotLive(child, 0, 0);
|
|
1536
|
+
});
|
|
1537
|
+
|
|
1538
|
+
test("grant: only qualifying synchronous parent task waves spend credits", async () => {
|
|
1539
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1540
|
+
await h.emit("session_start");
|
|
1541
|
+
await grant(h, "g", { rounds: 1, reason: "ok" });
|
|
1542
|
+
await h.emitEvent("tool_result", waveResult("e1", [{ agent: "code-reviewer", exitCode: 0 }]));
|
|
1543
|
+
await h.emitEvent("tool_result", waveResult("e2", [{ agent: "implementer", exitCode: 0 }], true));
|
|
1544
|
+
await h.emitEvent("tool_result", waveResult("e3", []));
|
|
1545
|
+
assert.equal((await firstCallResult(h, "a", { ...implementerWave(1), async: true }))?.block, true);
|
|
1546
|
+
assert.equal((await firstCallResult(h, "l", loneImplementer()))?.block, true);
|
|
1547
|
+
assert.equal(await firstCallResult(h, "c", implementerWave(1)), undefined);
|
|
1548
|
+
await h.emitEvent("tool_result", okWave("c"));
|
|
1549
|
+
assert.equal((await firstCallResult(h, "d", implementerWave(1)))?.block, true);
|
|
1550
|
+
|
|
1551
|
+
// In a child (PI_SUBAGENT_DEPTH >= 1, fixed at registration) both the cap gate and observeFixWave
|
|
1552
|
+
// are dormant, so credit consumption has no public effect; the observable contract is dormancy.
|
|
1553
|
+
const priorDepth = process.env.PI_SUBAGENT_DEPTH;
|
|
1554
|
+
process.env.PI_SUBAGENT_DEPTH = "1";
|
|
1555
|
+
let child: ReturnType<typeof harness>;
|
|
1556
|
+
try {
|
|
1557
|
+
child = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } }, [grantResult(1), subagentResult(["implementer"])]);
|
|
1558
|
+
} finally {
|
|
1559
|
+
if (priorDepth === undefined) delete process.env.PI_SUBAGENT_DEPTH;
|
|
1560
|
+
else process.env.PI_SUBAGENT_DEPTH = priorDepth;
|
|
1561
|
+
}
|
|
1562
|
+
await child.emit("session_start");
|
|
1563
|
+
assert.equal(await firstCallResult(child, "child-wave", implementerWave(1)), undefined, "child session: cap gate and observer are dormant, so the replayed grant and implementer result have no observable effect");
|
|
1564
|
+
});
|
|
1565
|
+
|
|
1566
|
+
test("grant replay restores and spends the recorded pool independent of current cap; rejected grants restore no credit", async () => {
|
|
1567
|
+
const trail = [subagentResult(["implementer"]), subagentResult(["implementer"]), subagentResult(["implementer"]), grantResult(2), subagentResult(["implementer"])];
|
|
1568
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 3 } } }, trail);
|
|
1569
|
+
await h.emit("session_start");
|
|
1570
|
+
assert.equal(
|
|
1571
|
+
(await grant(h, "g", { rounds: 1, reason: "more" })).details.error,
|
|
1572
|
+
"grant_fix_rounds: 1 granted round(s) still unused; spend them before granting more",
|
|
1573
|
+
);
|
|
1574
|
+
assert.equal(await firstCallResult(h, "c", implementerWave(1)), undefined);
|
|
1575
|
+
await h.emitEvent("tool_result", okWave("c"));
|
|
1576
|
+
const blocked = await firstCallResult(h, "d", implementerWave(1));
|
|
1577
|
+
assert.equal(blocked?.block, true);
|
|
1578
|
+
assert.match(blocked?.reason ?? "", /5 fix round\(s\) used against a cap of 3 \(granted rounds included\)/);
|
|
1579
|
+
|
|
1580
|
+
// Same trail at cap 5: replay yields fixRounds = 4 (cap not live yet); lowering the live cap
|
|
1581
|
+
// exposes the one replayed credit, then wave 5 passes on the counter and wave 6 blocks.
|
|
1582
|
+
const capChanged = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 5 } } }, trail);
|
|
1583
|
+
await capChanged.emit("session_start");
|
|
1584
|
+
assert.equal(
|
|
1585
|
+
(await grant(capChanged, "g", { rounds: 1, reason: "more" })).details.error,
|
|
1586
|
+
"grant_fix_rounds: no fix-round cap block is active (4 used, cap 5); nothing to overrule",
|
|
1587
|
+
"replayed fixRounds = 4 regardless of cap on disk",
|
|
1588
|
+
);
|
|
1589
|
+
// Lower the cap on disk without spending a wave: the block goes live at 4 >= 3 and the grant
|
|
1590
|
+
// probe now reads the replayed pool - exactly one credit, independent of the cap at replay time.
|
|
1591
|
+
writeFileSync(join(capChanged.ctx.cwd, ".pi", "settings.json"), JSON.stringify({ piGauntlet: { closureReview: { maxFixRounds: 3 } } }));
|
|
1592
|
+
assert.equal(
|
|
1593
|
+
(await grant(capChanged, "g2", { rounds: 1, reason: "more" })).details.error,
|
|
1594
|
+
"grant_fix_rounds: 1 granted round(s) still unused; spend them before granting more",
|
|
1595
|
+
"same credits regardless of cap on disk",
|
|
1596
|
+
);
|
|
1597
|
+
writeFileSync(join(capChanged.ctx.cwd, ".pi", "settings.json"), JSON.stringify({ piGauntlet: { closureReview: { maxFixRounds: 5 } } }));
|
|
1598
|
+
assert.equal(await firstCallResult(capChanged, "c", implementerWave(1)), undefined);
|
|
1599
|
+
await capChanged.emitEvent("tool_result", okWave("c"));
|
|
1600
|
+
const blockedAt5 = await firstCallResult(capChanged, "d", implementerWave(1));
|
|
1601
|
+
assert.equal(blockedAt5?.block, true);
|
|
1602
|
+
assert.match(blockedAt5?.reason ?? "", /5 fix round\(s\) used against a cap of 5 \(granted rounds included\)/);
|
|
1603
|
+
|
|
1604
|
+
const rejected = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } }, [{
|
|
1605
|
+
type: "message",
|
|
1606
|
+
message: {
|
|
1607
|
+
role: "toolResult",
|
|
1608
|
+
toolName: "phase_tracker",
|
|
1609
|
+
details: { action: "grant_fix_rounds", error: "grant_fix_rounds requires rounds: a positive integer", phases: phases({ verify: "in_progress" }) },
|
|
1610
|
+
},
|
|
1611
|
+
}]);
|
|
1612
|
+
await rejected.emit("session_start");
|
|
1613
|
+
assert.equal((await firstCallResult(rejected, "c1", implementerWave(1)))?.block, true, "no credits from a rejected grant");
|
|
1614
|
+
});
|
|
1615
|
+
|
|
1616
|
+
test("grant reset: start implement and reset zero credits (live and replay); start verify --force keeps them", async () => {
|
|
1617
|
+
const settings = { piGauntlet: { closureReview: { maxFixRounds: 0 }, flowGuards: { enforce: false } } };
|
|
1618
|
+
|
|
1619
|
+
const force = guardedHarness(settings, [grantResult(1)]);
|
|
1620
|
+
await force.emit("session_start");
|
|
1621
|
+
await force.tools.find((t) => t.name === "phase_tracker")!.execute("p", { action: "start", phase: "verify", force: true }, undefined, undefined, force.ctx);
|
|
1622
|
+
assert.equal(await firstCallResult(force, "c", implementerWave(1)), undefined);
|
|
1623
|
+
|
|
1624
|
+
const liveImpl = guardedHarness(settings);
|
|
1625
|
+
await liveImpl.emit("session_start");
|
|
1626
|
+
await grant(liveImpl, "g", { rounds: 1, reason: "ok" });
|
|
1627
|
+
const tool = liveImpl.tools.find((t) => t.name === "phase_tracker")!;
|
|
1628
|
+
for (const [id, input] of [
|
|
1629
|
+
["p1", { action: "skip", phase: "verify", reason: "amendment" }],
|
|
1630
|
+
["p2", { action: "start", phase: "implement", force: true }],
|
|
1631
|
+
["p3", { action: "complete", phase: "implement" }],
|
|
1632
|
+
["p4", { action: "start", phase: "verify", force: true }],
|
|
1633
|
+
] as const) {
|
|
1634
|
+
assert.equal(((await tool.execute(id, input, undefined, undefined, liveImpl.ctx)) as { details: { error?: string } }).details.error, undefined);
|
|
1635
|
+
}
|
|
1636
|
+
await liveImpl.emitEvent("tool_result", waveResult("audit", [{ agent: "conformance-reviewer", exitCode: 0 }]));
|
|
1637
|
+
assert.equal((await firstCallResult(liveImpl, "c1", implementerWave(1)))?.block, true, "credits zeroed by start implement (cap 0 blocks)");
|
|
1638
|
+
|
|
1639
|
+
const reset = guardedHarness(settings);
|
|
1640
|
+
await reset.emit("session_start");
|
|
1641
|
+
await grant(reset, "g", { rounds: 1, reason: "ok" });
|
|
1642
|
+
const resetTool = reset.tools.find((t) => t.name === "phase_tracker")!;
|
|
1643
|
+
await resetTool.execute("p", { action: "reset" }, undefined, undefined, reset.ctx);
|
|
1644
|
+
assert.match((await grant(reset, "g2", { rounds: 1, reason: "ok" })).details.error ?? "", /no fix-round cap block is active/);
|
|
1645
|
+
for (const [id, input] of [
|
|
1646
|
+
["p1", { action: "start", phase: "brainstorm" }],
|
|
1647
|
+
["p2", { action: "complete", phase: "brainstorm" }],
|
|
1648
|
+
["p3", { action: "start", phase: "plan" }],
|
|
1649
|
+
["p4", { action: "complete", phase: "plan" }],
|
|
1650
|
+
["p5", { action: "start", phase: "implement" }],
|
|
1651
|
+
["p6", { action: "complete", phase: "implement" }],
|
|
1652
|
+
["p7", { action: "start", phase: "verify" }],
|
|
1653
|
+
] as const) {
|
|
1654
|
+
assert.equal(((await resetTool.execute(id, input, undefined, undefined, reset.ctx)) as { details: { error?: string } }).details.error, undefined);
|
|
1655
|
+
}
|
|
1656
|
+
await reset.emitEvent("tool_result", waveResult("audit", [{ agent: "conformance-reviewer", exitCode: 0 }]));
|
|
1657
|
+
assert.equal((await firstCallResult(reset, "c1", implementerWave(1)))?.block, true, "credits zeroed by reset (cap 0 blocks)");
|
|
1658
|
+
|
|
1659
|
+
const replayImpl = harness({
|
|
1660
|
+
cwd: tempCwd(settings),
|
|
1661
|
+
branch: verifyBranch([
|
|
1662
|
+
subagentResult(["conformance-reviewer"]),
|
|
1663
|
+
grantResult(1),
|
|
1664
|
+
phaseResult("skip", phases({ brainstorm: "complete", plan: "complete", implement: "complete", verify: "skipped" })),
|
|
1665
|
+
phaseResult("start", phases({ brainstorm: "complete", plan: "complete", implement: "in_progress", verify: "skipped" })),
|
|
1666
|
+
phaseResult("complete", phases({ brainstorm: "complete", plan: "complete", implement: "complete", verify: "skipped" })),
|
|
1667
|
+
phaseResult("start", phases({ brainstorm: "complete", plan: "complete", implement: "complete", verify: "in_progress" })),
|
|
1668
|
+
subagentResult(["conformance-reviewer"]),
|
|
1669
|
+
]),
|
|
1670
|
+
});
|
|
1671
|
+
await replayImpl.emit("session_start");
|
|
1672
|
+
assert.equal((await firstCallResult(replayImpl, "c1", implementerWave(1)))?.block, true, "replayed start implement zeroed credits");
|
|
1673
|
+
|
|
1674
|
+
const replayReset = harness({
|
|
1675
|
+
cwd: tempCwd(settings),
|
|
1676
|
+
branch: verifyBranch([
|
|
1677
|
+
subagentResult(["conformance-reviewer"]),
|
|
1678
|
+
grantResult(1),
|
|
1679
|
+
phaseResult("reset", phases({})),
|
|
1680
|
+
phaseResult("start", phases({ brainstorm: "in_progress" })),
|
|
1681
|
+
phaseResult("complete", phases({ brainstorm: "complete" })),
|
|
1682
|
+
phaseResult("start", phases({ brainstorm: "complete", plan: "in_progress" })),
|
|
1683
|
+
phaseResult("complete", phases({ brainstorm: "complete", plan: "complete" })),
|
|
1684
|
+
phaseResult("start", phases({ brainstorm: "complete", plan: "complete", implement: "in_progress" })),
|
|
1685
|
+
phaseResult("complete", phases({ brainstorm: "complete", plan: "complete", implement: "complete" })),
|
|
1686
|
+
phaseResult("start", phases({ brainstorm: "complete", plan: "complete", implement: "complete", verify: "in_progress" })),
|
|
1687
|
+
subagentResult(["conformance-reviewer"]),
|
|
1688
|
+
]),
|
|
1689
|
+
});
|
|
1690
|
+
await replayReset.emit("session_start");
|
|
1691
|
+
assert.equal((await firstCallResult(replayReset, "c1", implementerWave(1)))?.block, true, "replayed reset zeroed credits");
|
|
1692
|
+
});
|
|
1693
|
+
|
|
1694
|
+
test("grant: block reason gives action, settings path, and live-read guidance", async () => {
|
|
1695
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1696
|
+
await h.emit("session_start");
|
|
1697
|
+
const reason = (await firstCallResult(h, "c", implementerWave(1)))?.reason ?? "";
|
|
1698
|
+
assert.match(reason, /phase_tracker\(\{ action: "grant_fix_rounds"/);
|
|
1699
|
+
assert.ok(reason.includes(join(h.ctx.cwd, ".pi", "settings.json")));
|
|
1700
|
+
assert.match(reason, /no restart.*restate model and maxFixRounds/s);
|
|
1701
|
+
});
|
|
1702
|
+
|
|
1703
|
+
test("gauntlet_setting registration requests sequential execution", () => {
|
|
1704
|
+
assert.equal(harness().tools.find((t) => t.name === "gauntlet_setting")!.executionMode, "sequential");
|
|
1705
|
+
});
|
|
1706
|
+
|
|
1707
|
+
test("grant renderResult shows count and reason", async () => {
|
|
1708
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1709
|
+
await h.emit("session_start");
|
|
1710
|
+
const res = await grant(h, "g", { rounds: 2, reason: "go on" });
|
|
1711
|
+
const tool = h.tools.find((t) => t.name === "phase_tracker")! as any;
|
|
1712
|
+
const rendered = tool.renderResult(res, {}, { fg: (_c: string, s: string) => s, bold: (s: string) => s });
|
|
1713
|
+
assert.match(JSON.stringify(rendered), /2.*go on/);
|
|
1714
|
+
});
|
|
1715
|
+
|
|
1716
|
+
test("grant: child sessions cannot overrule even a zero cap", async () => {
|
|
1717
|
+
const priorDepth = process.env.PI_SUBAGENT_DEPTH;
|
|
1718
|
+
process.env.PI_SUBAGENT_DEPTH = "1";
|
|
1719
|
+
let child: ReturnType<typeof harness>;
|
|
1720
|
+
try {
|
|
1721
|
+
child = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1722
|
+
} finally {
|
|
1723
|
+
if (priorDepth === undefined) delete process.env.PI_SUBAGENT_DEPTH;
|
|
1724
|
+
else process.env.PI_SUBAGENT_DEPTH = priorDepth;
|
|
1725
|
+
}
|
|
1726
|
+
await child.emit("session_start");
|
|
1727
|
+
assert.match((await grant(child, "g", { rounds: 1, reason: "ok" })).details.error ?? "", /no fix-round cap block is active/);
|
|
1728
|
+
});
|
|
1729
|
+
|
|
1730
|
+
test("grant: accepts MAX_SAFE_INTEGER and current on-disk cap controls whether the block is live", async () => {
|
|
1731
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1732
|
+
await h.emit("session_start");
|
|
1733
|
+
const result = await grant(h, "g", { rounds: Number.MAX_SAFE_INTEGER, reason: "approved" });
|
|
1734
|
+
assert.equal(result.details.rounds, Number.MAX_SAFE_INTEGER);
|
|
1735
|
+
|
|
1736
|
+
const changed = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1737
|
+
await changed.emit("session_start");
|
|
1738
|
+
writeFileSync(join(changed.ctx.cwd, ".pi", "settings.json"), JSON.stringify({ piGauntlet: { closureReview: { maxFixRounds: 1 } } }));
|
|
1739
|
+
assert.match((await grant(changed, "g", { rounds: 1, reason: "ok" })).details.error ?? "", /0 used, cap 1/);
|
|
1740
|
+
});
|
|
@@ -56,9 +56,11 @@ interface PhaseState {
|
|
|
56
56
|
type PhaseMap = Record<Phase, PhaseState>;
|
|
57
57
|
|
|
58
58
|
interface PhaseTrackerDetails {
|
|
59
|
-
action: "start" | "complete" | "skip" | "status" | "reset" | "substep";
|
|
59
|
+
action: "start" | "complete" | "skip" | "status" | "reset" | "substep" | "grant_fix_rounds";
|
|
60
60
|
phases: PhaseMap;
|
|
61
61
|
error?: string;
|
|
62
|
+
rounds?: number;
|
|
63
|
+
reason?: string;
|
|
62
64
|
}
|
|
63
65
|
|
|
64
66
|
interface PlanCheckStamp {
|
|
@@ -206,10 +208,14 @@ const loneImplementerBlockReason =
|
|
|
206
208
|
"a lone agent call runs unisolated and produces no worktree diff. " +
|
|
207
209
|
"To disable this gate, set piGauntlet.closureReview.enforce: false.";
|
|
208
210
|
|
|
209
|
-
const fixRoundCapBlockReason = (used: number, cap: number): string =>
|
|
210
|
-
`Conformance fix loop: fix round
|
|
211
|
-
"with the verdict trail instead of re-looping
|
|
212
|
-
"
|
|
211
|
+
const fixRoundCapBlockReason = (used: number, cap: number, settingsPath: string): string =>
|
|
212
|
+
`Conformance fix loop: ${used} fix round(s) used against a cap of ${cap} (granted rounds included); ` +
|
|
213
|
+
"escalate to the human with the verdict trail instead of re-looping.\n" +
|
|
214
|
+
"If the human explicitly approves more rounds, record it and retry as a tasks wave: " +
|
|
215
|
+
'phase_tracker({ action: "grant_fix_rounds", rounds: <N>, reason: "<their words>" }).\n' +
|
|
216
|
+
`Last resort: set piGauntlet.closureReview.enforce: false in ${settingsPath} (disables all closure guards; ` +
|
|
217
|
+
"applies on the next tool call, no restart - a gauntlet_setting read in the same message sees the write; " +
|
|
218
|
+
"a repo closureReview block replaces the preset's whole block, so restate model and maxFixRounds alongside).";
|
|
213
219
|
|
|
214
220
|
const closureModelBlockReason = (model: string, missing: number, total: number): string =>
|
|
215
221
|
`Blocked: ${missing} of ${total} conformance-reviewer ${total === 1 ? "dispatch" : "entries"} ` +
|
|
@@ -255,7 +261,7 @@ const pathInSpecDirs = (rawPath: string, specDirs: string[]): boolean => {
|
|
|
255
261
|
};
|
|
256
262
|
|
|
257
263
|
const PhaseTrackerParams = Type.Object({
|
|
258
|
-
action: StringEnum(["start", "complete", "skip", "status", "reset", "substep"] as const, {
|
|
264
|
+
action: StringEnum(["start", "complete", "skip", "status", "reset", "substep", "grant_fix_rounds"] as const, {
|
|
259
265
|
description: "Action to perform",
|
|
260
266
|
}),
|
|
261
267
|
phase: Type.Optional(
|
|
@@ -265,7 +271,14 @@ const PhaseTrackerParams = Type.Object({
|
|
|
265
271
|
),
|
|
266
272
|
reason: Type.Optional(
|
|
267
273
|
Type.String({
|
|
268
|
-
description: "Reason
|
|
274
|
+
description: "Reason (required for skip and grant_fix_rounds; for grant, the human's approval quoted)",
|
|
275
|
+
}),
|
|
276
|
+
),
|
|
277
|
+
rounds: Type.Optional(
|
|
278
|
+
Type.Integer({
|
|
279
|
+
minimum: 1,
|
|
280
|
+
maximum: Number.MAX_SAFE_INTEGER,
|
|
281
|
+
description: "Extra fix rounds the human explicitly approved (grant_fix_rounds only)",
|
|
269
282
|
}),
|
|
270
283
|
),
|
|
271
284
|
force: Type.Optional(
|
|
@@ -328,6 +341,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
328
341
|
let phases: PhaseMap = emptyPhases();
|
|
329
342
|
let conformanceDispatched = false;
|
|
330
343
|
let fixRounds = 0;
|
|
344
|
+
let fixRoundCredits = 0;
|
|
331
345
|
let gauntletEntered = false;
|
|
332
346
|
// Shared by replay and the live tool_result hook so a resumed session enforces
|
|
333
347
|
// the same budget. Evaluated BEFORE the same result may set the latch, so the
|
|
@@ -341,6 +355,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
341
355
|
isImplementerWave(details, isError) &&
|
|
342
356
|
resolveClosureReview(loadGauntletSettings(ctx.cwd).gauntlet).enforce
|
|
343
357
|
) {
|
|
358
|
+
if (fixRoundCredits > 0) fixRoundCredits -= 1;
|
|
344
359
|
fixRounds += 1;
|
|
345
360
|
}
|
|
346
361
|
};
|
|
@@ -402,7 +417,10 @@ export default function (pi: ExtensionAPI) {
|
|
|
402
417
|
// skills; plan-tracker tracks tasks independently.
|
|
403
418
|
const applyPlanActivity = (tasks?: { status: string }[]) => {
|
|
404
419
|
if (!tasks || tasks.length === 0) return;
|
|
405
|
-
if (
|
|
420
|
+
if (
|
|
421
|
+
phases.implement.status === "in_progress" &&
|
|
422
|
+
tasks.every((t) => t.status === "complete" || t.status === "skipped")
|
|
423
|
+
) {
|
|
406
424
|
phases = { ...phases, implement: { status: "complete" } };
|
|
407
425
|
firedGuards.clear();
|
|
408
426
|
}
|
|
@@ -420,6 +438,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
420
438
|
phases = emptyPhases();
|
|
421
439
|
conformanceDispatched = false;
|
|
422
440
|
fixRounds = 0;
|
|
441
|
+
fixRoundCredits = 0;
|
|
423
442
|
gauntletEntered = false;
|
|
424
443
|
planCheckStamp = undefined;
|
|
425
444
|
attemptedRecoveryEdges.clear();
|
|
@@ -444,14 +463,17 @@ export default function (pi: ExtensionAPI) {
|
|
|
444
463
|
const details = msg.details as PhaseTrackerDetails | undefined;
|
|
445
464
|
if (details && !details.error) {
|
|
446
465
|
phases = details.phases;
|
|
466
|
+
if (details.action === "grant_fix_rounds" && typeof details.rounds === "number") fixRoundCredits = details.rounds;
|
|
447
467
|
gauntletEntered = nextGauntletEntered(gauntletEntered, details.action, details.phases.brainstorm.status);
|
|
448
468
|
if (details.action === "start" && details.phases.implement.status === "in_progress") {
|
|
449
469
|
conformanceDispatched = false;
|
|
450
470
|
fixRounds = 0;
|
|
471
|
+
fixRoundCredits = 0;
|
|
451
472
|
}
|
|
452
473
|
if (details.action === "reset") {
|
|
453
474
|
conformanceDispatched = false;
|
|
454
475
|
fixRounds = 0;
|
|
476
|
+
fixRoundCredits = 0;
|
|
455
477
|
planCheckStamp = undefined;
|
|
456
478
|
}
|
|
457
479
|
}
|
|
@@ -544,7 +566,12 @@ export default function (pi: ExtensionAPI) {
|
|
|
544
566
|
}
|
|
545
567
|
if (fixLoopWindow && collectAgents(event.input, "implementer").length > 0) {
|
|
546
568
|
const cap = resolveClosureReview(g()).maxFixRounds;
|
|
547
|
-
if (fixRounds >= cap)
|
|
569
|
+
if (fixRounds >= cap) {
|
|
570
|
+
const funded = fixRoundCredits > 0 && (event.input as { async?: unknown })?.async !== true;
|
|
571
|
+
if (!funded) {
|
|
572
|
+
return { block: true, reason: fixRoundCapBlockReason(fixRounds, cap, join(ctx.cwd, ".pi", "settings.json")) };
|
|
573
|
+
}
|
|
574
|
+
}
|
|
548
575
|
}
|
|
549
576
|
const model = closureReviewModel();
|
|
550
577
|
if (model && !hasAction) {
|
|
@@ -730,6 +757,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
730
757
|
label: "Gauntlet Setting",
|
|
731
758
|
description: "Resolve merged piGauntlet.* settings (repo over preset); for skill use only.",
|
|
732
759
|
parameters: GauntletSettingParams,
|
|
760
|
+
executionMode: "sequential",
|
|
733
761
|
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
|
734
762
|
const { gauntlet, errors } = loadGauntletSettings(ctx.cwd);
|
|
735
763
|
const payload =
|
|
@@ -914,6 +942,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
914
942
|
if (params.phase === "implement") {
|
|
915
943
|
conformanceDispatched = false;
|
|
916
944
|
fixRounds = 0;
|
|
945
|
+
fixRoundCredits = 0;
|
|
917
946
|
}
|
|
918
947
|
gauntletEntered = nextGauntletEntered(gauntletEntered, "start", phases.brainstorm.status);
|
|
919
948
|
firedGuards.clear();
|
|
@@ -1080,6 +1109,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
1080
1109
|
) as PhaseMap;
|
|
1081
1110
|
conformanceDispatched = false;
|
|
1082
1111
|
fixRounds = 0;
|
|
1112
|
+
fixRoundCredits = 0;
|
|
1083
1113
|
gauntletEntered = nextGauntletEntered(gauntletEntered, "reset", phases.brainstorm.status);
|
|
1084
1114
|
firedGuards.clear();
|
|
1085
1115
|
updateWidget(ctx);
|
|
@@ -1089,6 +1119,39 @@ export default function (pi: ExtensionAPI) {
|
|
|
1089
1119
|
};
|
|
1090
1120
|
}
|
|
1091
1121
|
|
|
1122
|
+
case "grant_fix_rounds": {
|
|
1123
|
+
const reject = (error: string) => ({
|
|
1124
|
+
content: [{ type: "text" as const, text: `Error: ${error}` }],
|
|
1125
|
+
details: { action: "grant_fix_rounds", phases: { ...phases }, error } as PhaseTrackerDetails,
|
|
1126
|
+
});
|
|
1127
|
+
if (params.rounds === undefined) return reject("grant_fix_rounds requires rounds: a positive integer");
|
|
1128
|
+
const reason = params.reason?.trim() ?? "";
|
|
1129
|
+
if (reason.length === 0) return reject("grant_fix_rounds requires reason: the human's approval, quoted");
|
|
1130
|
+
const closure = resolveClosureReview(loadGauntletSettings(ctx.cwd).gauntlet);
|
|
1131
|
+
const capLive =
|
|
1132
|
+
!isSubagentChild &&
|
|
1133
|
+
gauntletEntered &&
|
|
1134
|
+
phases.verify.status === "in_progress" &&
|
|
1135
|
+
conformanceDispatched &&
|
|
1136
|
+
closure.enforce &&
|
|
1137
|
+
fixRounds >= closure.maxFixRounds;
|
|
1138
|
+
if (!capLive) {
|
|
1139
|
+
return reject(
|
|
1140
|
+
`grant_fix_rounds: no fix-round cap block is active (${fixRounds} used, cap ${closure.maxFixRounds}); nothing to overrule`,
|
|
1141
|
+
);
|
|
1142
|
+
}
|
|
1143
|
+
if (fixRoundCredits > 0) {
|
|
1144
|
+
return reject(`grant_fix_rounds: ${fixRoundCredits} granted round(s) still unused; spend them before granting more`);
|
|
1145
|
+
}
|
|
1146
|
+
fixRoundCredits = params.rounds;
|
|
1147
|
+
return {
|
|
1148
|
+
content: [
|
|
1149
|
+
{ type: "text", text: `Granted ${params.rounds} extra fix round(s) - reason: "${reason}"\n${formatStatus(phases)}` },
|
|
1150
|
+
],
|
|
1151
|
+
details: { action: "grant_fix_rounds", rounds: params.rounds, reason, phases: { ...phases } } as PhaseTrackerDetails,
|
|
1152
|
+
};
|
|
1153
|
+
}
|
|
1154
|
+
|
|
1092
1155
|
default:
|
|
1093
1156
|
return {
|
|
1094
1157
|
content: [{ type: "text", text: `Unknown action: ${params.action}` }],
|
|
@@ -1160,6 +1223,14 @@ export default function (pi: ExtensionAPI) {
|
|
|
1160
1223
|
}
|
|
1161
1224
|
case "reset":
|
|
1162
1225
|
return new Text(theme.fg("success", "✓ ") + theme.fg("muted", "Phase tracker reset"), 0, 0);
|
|
1226
|
+
case "grant_fix_rounds":
|
|
1227
|
+
return new Text(
|
|
1228
|
+
theme.fg("success", "+ ") +
|
|
1229
|
+
theme.fg("muted", `${details.rounds} extra fix round(s) granted`) +
|
|
1230
|
+
theme.fg("dim", ` (${details.reason})`),
|
|
1231
|
+
0,
|
|
1232
|
+
0,
|
|
1233
|
+
);
|
|
1163
1234
|
default:
|
|
1164
1235
|
return new Text(theme.fg("dim", "Done"), 0, 0);
|
|
1165
1236
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import assert from "node:assert/strict";
|
|
2
2
|
import { test } from "node:test";
|
|
3
|
-
import registerPlanTracker from "./plan-tracker.ts";
|
|
3
|
+
import registerPlanTracker, { validateSnapshot } from "./plan-tracker.ts";
|
|
4
4
|
|
|
5
5
|
type ToolResult = {
|
|
6
6
|
content: { type: string; text: string }[];
|
|
@@ -10,6 +10,9 @@ type ToolResult = {
|
|
|
10
10
|
function harness(branch: unknown[] = []) {
|
|
11
11
|
const tools: {
|
|
12
12
|
name: string;
|
|
13
|
+
description: string;
|
|
14
|
+
executionMode?: string;
|
|
15
|
+
parameters: any;
|
|
13
16
|
execute: (...args: any[]) => unknown;
|
|
14
17
|
renderResult: (result: unknown, options: unknown, theme: unknown) => { text?: string };
|
|
15
18
|
}[] = [];
|
|
@@ -39,7 +42,16 @@ function harness(branch: unknown[] = []) {
|
|
|
39
42
|
const fire = async (event: string) => {
|
|
40
43
|
for (const h of handlers) if (h.event === event) await h.handler({}, ctx);
|
|
41
44
|
};
|
|
42
|
-
return {
|
|
45
|
+
return {
|
|
46
|
+
call,
|
|
47
|
+
fire,
|
|
48
|
+
tool: () => tools[0],
|
|
49
|
+
theme,
|
|
50
|
+
widget: () => widgetText,
|
|
51
|
+
parameters: () => tools[0].parameters,
|
|
52
|
+
description: () => tools[0].description,
|
|
53
|
+
executionMode: () => tools[0].executionMode,
|
|
54
|
+
};
|
|
43
55
|
}
|
|
44
56
|
|
|
45
57
|
test("add appends pending tasks and preserves existing statuses", async () => {
|
|
@@ -95,7 +107,7 @@ test("update to failed round-trips and is excluded from complete count", async (
|
|
|
95
107
|
res.details.tasks.map((t) => [t.name, t.status]),
|
|
96
108
|
[["a", "complete"], ["b", "failed"], ["c", "pending"]],
|
|
97
109
|
);
|
|
98
|
-
assert.match(res.content[0].text, /1\/3
|
|
110
|
+
assert.match(res.content[0].text, /1\/3 done/);
|
|
99
111
|
assert.match(res.content[0].text, /1 failed/);
|
|
100
112
|
});
|
|
101
113
|
|
|
@@ -140,11 +152,11 @@ test("widget renders failed as \u2717, keeps failed out of complete count and cu
|
|
|
140
152
|
test("renderResult status path shows \u2717 for failed and excludes it from complete", async () => {
|
|
141
153
|
const { call, tool, theme } = harness();
|
|
142
154
|
await call({ action: "init", tasks: ["a", "b"] });
|
|
143
|
-
await call({ action: "update", index:
|
|
155
|
+
await call({ action: "update", index: 0, status: "failed" });
|
|
144
156
|
const res = await call({ action: "status" });
|
|
145
157
|
const rendered = tool().renderResult(res as any, {}, theme as any);
|
|
146
158
|
const text = (rendered as any).text as string;
|
|
147
|
-
assert.match(text, /0\/2
|
|
159
|
+
assert.match(text, /0\/2 done/);
|
|
148
160
|
assert.match(text, /\u2717/);
|
|
149
161
|
});
|
|
150
162
|
|
|
@@ -156,7 +168,7 @@ test("renderResult status header appends failed count when a task has failed", a
|
|
|
156
168
|
const res = await call({ action: "status" });
|
|
157
169
|
const rendered = tool().renderResult(res as any, {}, theme as any);
|
|
158
170
|
const text = (rendered as any).text as string;
|
|
159
|
-
assert.match(text, /1\/3
|
|
171
|
+
assert.match(text, /1\/3 done, 1 failed/);
|
|
160
172
|
});
|
|
161
173
|
|
|
162
174
|
test("renderResult status header omits failed count when no task has failed", async () => {
|
|
@@ -166,7 +178,7 @@ test("renderResult status header omits failed count when no task has failed", as
|
|
|
166
178
|
const res = await call({ action: "status" });
|
|
167
179
|
const rendered = tool().renderResult(res as any, {}, theme as any);
|
|
168
180
|
const text = (rendered as any).text as string;
|
|
169
|
-
assert.match(text, /^1\/2
|
|
181
|
+
assert.match(text, /^1\/2 done\n/);
|
|
170
182
|
assert.doesNotMatch(text, /failed/);
|
|
171
183
|
});
|
|
172
184
|
|
|
@@ -177,7 +189,7 @@ test("renderResult update case appends failed count when a task has failed", asy
|
|
|
177
189
|
const res = await call({ action: "update", index: 1, status: "failed" });
|
|
178
190
|
const rendered = tool().renderResult(res as any, {}, theme as any);
|
|
179
191
|
const text = (rendered as any).text as string;
|
|
180
|
-
assert.match(text, /^\u2713 Updated \(1\/3
|
|
192
|
+
assert.match(text, /^\u2713 Updated \(1\/3 done, 1 failed\)$/);
|
|
181
193
|
});
|
|
182
194
|
|
|
183
195
|
test("renderResult update case omits failed count when no task has failed", async () => {
|
|
@@ -186,6 +198,196 @@ test("renderResult update case omits failed count when no task has failed", asyn
|
|
|
186
198
|
const res = await call({ action: "update", index: 0, status: "complete" });
|
|
187
199
|
const rendered = tool().renderResult(res as any, {}, theme as any);
|
|
188
200
|
const text = (rendered as any).text as string;
|
|
189
|
-
assert.match(text, /^\u2713 Updated \(1\/2
|
|
201
|
+
assert.match(text, /^\u2713 Updated \(1\/2 done\)$/);
|
|
190
202
|
assert.doesNotMatch(text, /failed/);
|
|
191
203
|
});
|
|
204
|
+
|
|
205
|
+
const S = (code: string) =>
|
|
206
|
+
[...code].map((c, i) => ({
|
|
207
|
+
name: `t${i}`,
|
|
208
|
+
status: ({ o: "pending", ">": "in_progress", k: "complete", x: "failed", "/": "skipped" } as const)[c as "o" | ">" | "k" | "x" | "/"],
|
|
209
|
+
}));
|
|
210
|
+
|
|
211
|
+
test("validateSnapshot: pending must trail every non-pending task", () => {
|
|
212
|
+
for (const code of ["kkkoo", "kk>>oo", "kk>kk", "x>o", "ooo", "///", "", "k"]) {
|
|
213
|
+
assert.deepEqual(validateSnapshot(S(code)), [], code);
|
|
214
|
+
}
|
|
215
|
+
assert.deepEqual(validateSnapshot(S("okoo")), [0]);
|
|
216
|
+
assert.deepEqual(validateSnapshot(S("kkk>ookkkk")), [4, 5]);
|
|
217
|
+
assert.deepEqual(validateSnapshot(S("o>")), [0]);
|
|
218
|
+
assert.deepEqual(validateSnapshot(S("ook")), [0, 1]);
|
|
219
|
+
});
|
|
220
|
+
|
|
221
|
+
test("update past a pending index is rejected with every offender, fixes, and the current snapshot", async () => {
|
|
222
|
+
const { call } = harness();
|
|
223
|
+
await call({ action: "init", tasks: ["gather", "resolve evidence", "claim-check", "provision worktree", "review", "verify claims"] });
|
|
224
|
+
for (const i of [0, 1, 2]) await call({ action: "update", index: i, status: "complete" });
|
|
225
|
+
const res = await call({ action: "update", index: 5, status: "complete" });
|
|
226
|
+
assert.equal(res.details.error, "pending-suffix violation: 3,4");
|
|
227
|
+
assert.deepEqual(
|
|
228
|
+
res.details.tasks.map((t) => t.status),
|
|
229
|
+
["complete", "complete", "complete", "pending", "pending", "pending"],
|
|
230
|
+
);
|
|
231
|
+
const text = res.content[0].text;
|
|
232
|
+
assert.equal(
|
|
233
|
+
text,
|
|
234
|
+
[
|
|
235
|
+
'Error: cannot set task 5 "verify claims" to complete: pending tasks precede it: 3 "provision worktree", 4 "review".',
|
|
236
|
+
"Rule: pending tasks must trail every started or finished task.",
|
|
237
|
+
"Fix first, then retry: update each listed task to in_progress (working on it now), complete, failed, or skipped (not applicable in this run); or, if the list order itself is wrong, re-init with {name, status}[] keeping every task and its true status so the untouched ones trail.",
|
|
238
|
+
"Never clear, drop tasks, or record a status the work has not actually reached.",
|
|
239
|
+
"Current: 0 gather=complete, 1 resolve evidence=complete, 2 claim-check=complete, 3 provision worktree=pending, 4 review=pending, 5 verify claims=pending",
|
|
240
|
+
].join("\n"),
|
|
241
|
+
);
|
|
242
|
+
const after = await call({ action: "status" });
|
|
243
|
+
assert.deepEqual(after.details.tasks.map((t) => t.status), ["complete", "complete", "complete", "pending", "pending", "pending"]);
|
|
244
|
+
});
|
|
245
|
+
|
|
246
|
+
test("update -> pending is rejected in execute and names reopen and re-init", async () => {
|
|
247
|
+
const { call } = harness();
|
|
248
|
+
await call({ action: "init", tasks: ["a", "b", "claim-check"] });
|
|
249
|
+
for (const i of [0, 1, 2]) await call({ action: "update", index: i, status: "complete" });
|
|
250
|
+
const res = await call({ action: "update", index: 2, status: "pending" });
|
|
251
|
+
assert.equal(res.details.error, "update cannot set pending");
|
|
252
|
+
assert.deepEqual(res.details.tasks.map((t) => t.status), ["complete", "complete", "complete"]);
|
|
253
|
+
assert.equal(
|
|
254
|
+
res.content[0].text,
|
|
255
|
+
[
|
|
256
|
+
'Error: cannot set task 2 "claim-check" to pending: update never sets pending; a task is pending only from init/add.',
|
|
257
|
+
"To redo it, update it to in_progress. To recreate the whole list, re-init with {name, status}[].",
|
|
258
|
+
"Current: 0 a=complete, 1 b=complete, 2 claim-check=complete",
|
|
259
|
+
].join("\n"),
|
|
260
|
+
);
|
|
261
|
+
});
|
|
262
|
+
|
|
263
|
+
test("update at index 0 / first pending index, direct jumps, and reopens are accepted", async () => {
|
|
264
|
+
const { call } = harness();
|
|
265
|
+
await call({ action: "init", tasks: ["a", "b", "c", "d"] });
|
|
266
|
+
assert.equal((await call({ action: "update", index: 0, status: "complete" })).details.error, undefined);
|
|
267
|
+
assert.equal((await call({ action: "update", index: 1, status: "failed" })).details.error, undefined);
|
|
268
|
+
assert.equal((await call({ action: "update", index: 2, status: "skipped" })).details.error, undefined);
|
|
269
|
+
assert.equal((await call({ action: "update", index: 0, status: "in_progress" })).details.error, undefined);
|
|
270
|
+
assert.equal((await call({ action: "update", index: 1, status: "in_progress" })).details.error, undefined);
|
|
271
|
+
const res = await call({ action: "status" });
|
|
272
|
+
assert.deepEqual(res.details.tasks.map((t) => t.status), ["in_progress", "in_progress", "skipped", "pending"]);
|
|
273
|
+
});
|
|
274
|
+
|
|
275
|
+
test("wave fan-out: in_progress in increasing order succeeds, out of order is rejected naming earlier pending indices", async () => {
|
|
276
|
+
const ok = harness();
|
|
277
|
+
await ok.call({ action: "init", tasks: ["a", "b", "c"] });
|
|
278
|
+
for (const i of [0, 1, 2]) assert.equal((await ok.call({ action: "update", index: i, status: "in_progress" })).details.error, undefined);
|
|
279
|
+
const bad = harness();
|
|
280
|
+
await bad.call({ action: "init", tasks: ["a", "b", "c"] });
|
|
281
|
+
const res = await bad.call({ action: "update", index: 2, status: "in_progress" });
|
|
282
|
+
assert.equal(res.details.error, "pending-suffix violation: 0,1");
|
|
283
|
+
assert.match(res.content[0].text, /pending tasks precede it: 0 "a", 1 "b"\./);
|
|
284
|
+
});
|
|
285
|
+
|
|
286
|
+
test("init: strings, objects, and mixed forms normalize; ordering violation rejects the whole list", async () => {
|
|
287
|
+
const { call } = harness();
|
|
288
|
+
const strings = await call({ action: "init", tasks: ["a", "b"] });
|
|
289
|
+
assert.deepEqual(strings.details.tasks, [{ name: "a", status: "pending" }, { name: "b", status: "pending" }]);
|
|
290
|
+
const objects = await call({ action: "init", tasks: [{ name: "a", status: "complete" }, { name: "b", status: "skipped" }, "c"] });
|
|
291
|
+
assert.equal(objects.details.error, undefined);
|
|
292
|
+
assert.deepEqual(objects.details.tasks, [
|
|
293
|
+
{ name: "a", status: "complete" },
|
|
294
|
+
{ name: "b", status: "skipped" },
|
|
295
|
+
{ name: "c", status: "pending" },
|
|
296
|
+
]);
|
|
297
|
+
const bad = await call({ action: "init", tasks: [{ name: "a", status: "complete" }, "b", { name: "c", status: "complete" }] });
|
|
298
|
+
assert.equal(bad.details.error, "pending-suffix violation: 1");
|
|
299
|
+
assert.equal(
|
|
300
|
+
bad.content[0].text,
|
|
301
|
+
[
|
|
302
|
+
'Error: cannot init: pending tasks precede started or finished ones: 1 "b".',
|
|
303
|
+
"Rule: pending tasks must trail every started or finished task.",
|
|
304
|
+
"Reorder the list or restate those statuses truthfully, then retry.",
|
|
305
|
+
"Proposed: 0 a=complete, 1 b=pending, 2 c=complete",
|
|
306
|
+
].join("\n"),
|
|
307
|
+
);
|
|
308
|
+
assert.doesNotMatch(bad.content[0].text, /Current:/);
|
|
309
|
+
assert.deepEqual(bad.details.tasks.map((t) => t.status), ["complete", "skipped", "pending"]);
|
|
310
|
+
});
|
|
311
|
+
|
|
312
|
+
test("init ordering violation with no prior plan leaves no plan", async () => {
|
|
313
|
+
const { call, widget } = harness();
|
|
314
|
+
const bad = await call({ action: "init", tasks: ["a", { name: "b", status: "complete" }] });
|
|
315
|
+
assert.equal(bad.details.error, "pending-suffix violation: 0");
|
|
316
|
+
assert.deepEqual(bad.details.tasks, []);
|
|
317
|
+
assert.equal(widget(), undefined);
|
|
318
|
+
});
|
|
319
|
+
|
|
320
|
+
test("skipped round-trips, replays, renders \u2298, counts as done, and is never current", async () => {
|
|
321
|
+
const { call, widget, tool, theme } = harness();
|
|
322
|
+
await call({ action: "init", tasks: ["a", "b", "c"] });
|
|
323
|
+
await call({ action: "update", index: 0, status: "complete" });
|
|
324
|
+
const res = await call({ action: "update", index: 1, status: "skipped" });
|
|
325
|
+
assert.equal(res.details.error, undefined);
|
|
326
|
+
assert.match(res.content[0].text, /Plan: 2\/3 done \(1 pending, 1 skipped\)/);
|
|
327
|
+
assert.match(res.content[0].text, /\u2298 \[1\] b/);
|
|
328
|
+
const w = widget()!;
|
|
329
|
+
assert.match(w, /\u2298/);
|
|
330
|
+
assert.match(w, /\(2\/3\)/);
|
|
331
|
+
assert.match(w, /c$/);
|
|
332
|
+
const rendered = (tool().renderResult(res as any, {}, theme as any) as any).text as string;
|
|
333
|
+
assert.match(rendered, /^\u2713 Updated \(2\/3 done, 1 skipped\)$/);
|
|
334
|
+
const status = await call({ action: "status" });
|
|
335
|
+
const statusText = (tool().renderResult(status as any, {}, theme as any) as any).text as string;
|
|
336
|
+
assert.match(statusText, /^2\/3 done, 1 skipped\n/);
|
|
337
|
+
|
|
338
|
+
const replay = harness([{ type: "message", message: { role: "toolResult", toolName: "plan_tracker", details: res.details } }]);
|
|
339
|
+
await replay.fire("session_start");
|
|
340
|
+
const after = await replay.call({ action: "status" });
|
|
341
|
+
assert.deepEqual(after.details.tasks.map((t) => t.status), ["complete", "skipped", "pending"]);
|
|
342
|
+
});
|
|
343
|
+
|
|
344
|
+
test("all tasks skipped renders (M/M)", async () => {
|
|
345
|
+
const { call, widget } = harness();
|
|
346
|
+
await call({ action: "init", tasks: ["a", "b"] });
|
|
347
|
+
await call({ action: "update", index: 0, status: "skipped" });
|
|
348
|
+
await call({ action: "update", index: 1, status: "skipped" });
|
|
349
|
+
assert.match(widget()!, /\(2\/2\)/);
|
|
350
|
+
});
|
|
351
|
+
|
|
352
|
+
test("legacy-invalid snapshot: first update is rejected listing legacy offenders; truthful object-form re-init repairs it", async () => {
|
|
353
|
+
const branch = [
|
|
354
|
+
{
|
|
355
|
+
type: "message",
|
|
356
|
+
message: {
|
|
357
|
+
role: "toolResult",
|
|
358
|
+
toolName: "plan_tracker",
|
|
359
|
+
details: {
|
|
360
|
+
action: "update",
|
|
361
|
+
tasks: [
|
|
362
|
+
{ name: "a", status: "pending" },
|
|
363
|
+
{ name: "b", status: "pending" },
|
|
364
|
+
{ name: "c", status: "complete" },
|
|
365
|
+
],
|
|
366
|
+
},
|
|
367
|
+
},
|
|
368
|
+
},
|
|
369
|
+
];
|
|
370
|
+
const { call, fire } = harness(branch);
|
|
371
|
+
await fire("session_start");
|
|
372
|
+
const rejected = await call({ action: "update", index: 0, status: "complete" });
|
|
373
|
+
assert.equal(rejected.details.error, "pending-suffix violation: 1");
|
|
374
|
+
assert.deepEqual(rejected.details.tasks.map((t) => t.status), ["pending", "pending", "complete"]);
|
|
375
|
+
const repaired = await call({
|
|
376
|
+
action: "init",
|
|
377
|
+
tasks: [{ name: "a", status: "complete" }, { name: "b", status: "skipped" }, { name: "c", status: "complete" }],
|
|
378
|
+
});
|
|
379
|
+
assert.equal(repaired.details.error, undefined);
|
|
380
|
+
assert.deepEqual(repaired.details.tasks.map((t) => t.status), ["complete", "skipped", "complete"]);
|
|
381
|
+
});
|
|
382
|
+
|
|
383
|
+
test("registration: sequential, tasks schema is Array<Union<String, Object>>, description carries the contract", () => {
|
|
384
|
+
const { executionMode, parameters, description } = harness();
|
|
385
|
+
assert.equal(executionMode(), "sequential");
|
|
386
|
+
const tasksSchema = parameters().args[0].tasks.args[0];
|
|
387
|
+
assert.equal(tasksSchema.kind, "Array");
|
|
388
|
+
assert.equal(tasksSchema.args[0].kind, "Union");
|
|
389
|
+
assert.deepEqual(tasksSchema.args[0].args[0].map((s: { kind: string }) => s.kind), ["String", "Object"]);
|
|
390
|
+
assert.match(description(), /skipped/);
|
|
391
|
+
assert.match(description(), /pending tasks must trail every started or finished task/);
|
|
392
|
+
assert.match(description(), /update never sets pending/);
|
|
393
|
+
});
|
|
@@ -11,87 +11,68 @@ import type { ExtensionAPI, ExtensionContext, Theme } from "@earendil-works/pi-c
|
|
|
11
11
|
import { Text } from "@earendil-works/pi-tui";
|
|
12
12
|
import { type Static, Type } from "@sinclair/typebox";
|
|
13
13
|
|
|
14
|
-
|
|
14
|
+
const TASK_STATUSES = ["pending", "in_progress", "complete", "failed", "skipped"] as const;
|
|
15
|
+
type TaskStatus = (typeof TASK_STATUSES)[number];
|
|
15
16
|
|
|
16
|
-
interface Task {
|
|
17
|
-
|
|
18
|
-
status: TaskStatus;
|
|
19
|
-
}
|
|
20
|
-
|
|
21
|
-
interface PlanTrackerDetails {
|
|
22
|
-
action: "init" | "add" | "update" | "status" | "clear";
|
|
23
|
-
tasks: Task[];
|
|
24
|
-
error?: string;
|
|
25
|
-
}
|
|
17
|
+
interface Task { name: string; status: TaskStatus; }
|
|
18
|
+
interface PlanTrackerDetails { action: "init" | "add" | "update" | "status" | "clear"; tasks: Task[]; error?: string; }
|
|
26
19
|
|
|
27
20
|
const PlanTrackerParams = Type.Object({
|
|
28
|
-
action: StringEnum(["init", "add", "update", "status", "clear"] as const, {
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
),
|
|
36
|
-
index: Type.Optional(
|
|
37
|
-
Type.Integer({
|
|
38
|
-
minimum: 0,
|
|
39
|
-
description: "Task index, 0-based (for update)",
|
|
40
|
-
}),
|
|
41
|
-
),
|
|
42
|
-
status: Type.Optional(
|
|
43
|
-
StringEnum(["pending", "in_progress", "complete", "failed"] as const, {
|
|
44
|
-
description: "New status (for update); failed is terminal-negative (ran and did not pass)",
|
|
45
|
-
}),
|
|
46
|
-
),
|
|
21
|
+
action: StringEnum(["init", "add", "update", "status", "clear"] as const, { description: "Action to perform" }),
|
|
22
|
+
tasks: Type.Optional(Type.Array(Type.Union([
|
|
23
|
+
Type.String(),
|
|
24
|
+
Type.Object({ name: Type.String(), status: StringEnum(TASK_STATUSES, { description: "Status to recreate the task with (init only)" }) }),
|
|
25
|
+
]), { description: "Tasks for init and add. A string is a pending task. init also accepts { name, status } to recreate a list with known statuses (fresh session, amendment re-init); the whole list must satisfy the pending-suffix rule." })),
|
|
26
|
+
index: Type.Optional(Type.Integer({ minimum: 0, description: "Task index, 0-based (for update)" })),
|
|
27
|
+
status: Type.Optional(StringEnum(TASK_STATUSES, { description: "New status (for update). failed is terminal-negative (ran and did not pass); skipped is terminal, not applicable in this run, counted as done. pending is set only by init/add; to redo a task, reopen it as in_progress, or re-init with {name, status}[] to recreate a whole list." })),
|
|
47
28
|
});
|
|
48
29
|
|
|
49
30
|
export type PlanTrackerInput = Static<typeof PlanTrackerParams>;
|
|
50
31
|
|
|
51
|
-
|
|
52
|
-
|
|
32
|
+
/** Indices of every pending task that has a non-pending task after it (empty = valid snapshot). */
|
|
33
|
+
export function validateSnapshot(tasks: Task[]): number[] {
|
|
34
|
+
const offenders: number[] = [];
|
|
35
|
+
let seenNonPending = false;
|
|
36
|
+
for (let i = tasks.length - 1; i >= 0; i--) {
|
|
37
|
+
if (tasks[i].status !== "pending") seenNonPending = true;
|
|
38
|
+
else if (seenNonPending) offenders.unshift(i);
|
|
39
|
+
}
|
|
40
|
+
return offenders;
|
|
41
|
+
}
|
|
53
42
|
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
return theme.fg("error", "✗");
|
|
64
|
-
default:
|
|
65
|
-
return theme.fg("dim", "○");
|
|
66
|
-
}
|
|
67
|
-
})
|
|
68
|
-
.join("");
|
|
43
|
+
const RULE_LINE = "Rule: pending tasks must trail every started or finished task.";
|
|
44
|
+
const snapshotLine = (tasks: Task[]): string => tasks.map((t, i) => `${i} ${t.name}=${t.status}`).join(", ");
|
|
45
|
+
const offenderList = (tasks: Task[], indices: number[]): string => indices.map((i) => `${i} "${tasks[i].name}"`).join(", ");
|
|
46
|
+
const normalizeTask = (t: string | { name: string; status: TaskStatus }): Task => typeof t === "string" ? { name: t, status: "pending" } : { name: t.name, status: t.status };
|
|
47
|
+
|
|
48
|
+
const glyph = (status: TaskStatus, theme?: Theme): string => {
|
|
49
|
+
const [color, icon]: [string, string] = status === "complete" ? ["success", "✓"] : status === "in_progress" ? ["warning", "→"] : status === "failed" ? ["error", "✗"] : status === "skipped" ? ["dim", "⊘"] : ["dim", "○"];
|
|
50
|
+
return theme ? theme.fg(color as Parameters<Theme["fg"]>[0], icon) : icon;
|
|
51
|
+
};
|
|
69
52
|
|
|
70
|
-
|
|
53
|
+
const counts = (tasks: Task[]) => {
|
|
54
|
+
const by = (s: TaskStatus) => tasks.filter((t) => t.status === s).length;
|
|
55
|
+
const failed = by("failed");
|
|
56
|
+
const skipped = by("skipped");
|
|
57
|
+
return { done: by("complete") + skipped, inProgress: by("in_progress"), pending: by("pending"), failed, skipped, suffix: `${failed > 0 ? `, ${failed} failed` : ""}${skipped > 0 ? `, ${skipped} skipped` : ""}` };
|
|
58
|
+
};
|
|
59
|
+
|
|
60
|
+
function formatWidget(tasks: Task[], theme: Theme): string {
|
|
61
|
+
if (tasks.length === 0) return "";
|
|
62
|
+
const { done } = counts(tasks);
|
|
63
|
+
const icons = tasks.map((t) => glyph(t.status, theme)).join("");
|
|
71
64
|
const current = tasks.find((t) => t.status === "in_progress") ?? tasks.find((t) => t.status === "pending");
|
|
72
65
|
const currentName = current ? ` ${current.name}` : "";
|
|
73
|
-
|
|
74
|
-
return `${theme.fg("muted", "Tasks:")} ${icons} ${theme.fg("muted", `(${complete}/${tasks.length})`)}${currentName}`;
|
|
66
|
+
return `${theme.fg("muted", "Tasks:")} ${icons} ${theme.fg("muted", `(${done}/${tasks.length})`)}${currentName}`;
|
|
75
67
|
}
|
|
76
68
|
|
|
77
69
|
function formatStatus(tasks: Task[]): string {
|
|
78
70
|
if (tasks.length === 0) return "No plan active.";
|
|
79
|
-
|
|
80
|
-
const
|
|
81
|
-
const
|
|
82
|
-
const
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
const lines: string[] = [];
|
|
86
|
-
lines.push(
|
|
87
|
-
`Plan: ${complete}/${tasks.length} complete (${inProgress} in progress, ${pending} pending, ${failed} failed)`,
|
|
88
|
-
);
|
|
89
|
-
lines.push("");
|
|
90
|
-
for (let i = 0; i < tasks.length; i++) {
|
|
91
|
-
const t = tasks[i];
|
|
92
|
-
const icon = t.status === "complete" ? "✓" : t.status === "in_progress" ? "→" : t.status === "failed" ? "✗" : "○";
|
|
93
|
-
lines.push(` ${icon} [${i}] ${t.name}`);
|
|
94
|
-
}
|
|
71
|
+
const c = counts(tasks);
|
|
72
|
+
const groups = [[c.inProgress, "in progress"], [c.pending, "pending"], [c.failed, "failed"], [c.skipped, "skipped"]] as const;
|
|
73
|
+
const detail = groups.filter(([n]) => n > 0).map(([n, label]) => `${n} ${label}`);
|
|
74
|
+
const lines: string[] = [`Plan: ${c.done}/${tasks.length} done${detail.length > 0 ? ` (${detail.join(", ")})` : ""}`, ""];
|
|
75
|
+
for (let i = 0; i < tasks.length; i++) lines.push(` ${glyph(tasks[i].status)} [${i}] ${tasks[i].name}`);
|
|
95
76
|
return lines.join("\n");
|
|
96
77
|
}
|
|
97
78
|
|
|
@@ -134,8 +115,9 @@ export default function (pi: ExtensionAPI) {
|
|
|
134
115
|
name: "plan_tracker",
|
|
135
116
|
label: "Plan Tracker",
|
|
136
117
|
description:
|
|
137
|
-
"Track progress while EXECUTING an implementation plan (the implement phase), a verify-phase conformance fix wave, or another bounded gate checklist (e.g. pre-merge PR verification). Statuses: pending, in_progress, complete, failed (terminal-negative:
|
|
118
|
+
"Track progress while EXECUTING an implementation plan (the implement phase), a verify-phase conformance fix wave, or another bounded gate checklist (e.g. pre-merge PR verification). Statuses: pending (not yet touched), in_progress (started; several at once is fine), complete, failed (terminal-negative: ran and did not pass; never counted done), skipped (terminal: not applicable in this run; counted done). Rule: pending tasks must trail every started or finished task - an update that would leave a pending task ahead of a non-pending one is rejected with the fix. update never sets pending; init and add do. Actions: init (set task list; elements are task-name strings or { name, status } to recreate a list with known statuses), add (append tasks as pending; existing statuses preserved), update (change task status), status (show current state), clear (remove plan). Do NOT use for brainstorming, research, or planning checklists: those phases are open-ended and a bounded task list misrepresents them as a fixed N-step process.",
|
|
138
119
|
parameters: PlanTrackerParams,
|
|
120
|
+
executionMode: "sequential",
|
|
139
121
|
|
|
140
122
|
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
|
141
123
|
switch (params.action) {
|
|
@@ -150,7 +132,20 @@ export default function (pi: ExtensionAPI) {
|
|
|
150
132
|
} as PlanTrackerDetails,
|
|
151
133
|
};
|
|
152
134
|
}
|
|
153
|
-
|
|
135
|
+
const proposed = params.tasks.map(normalizeTask);
|
|
136
|
+
const offenders = validateSnapshot(proposed);
|
|
137
|
+
if (offenders.length > 0) {
|
|
138
|
+
return {
|
|
139
|
+
content: [{ type: "text", text: [
|
|
140
|
+
`Error: cannot init: pending tasks precede started or finished ones: ${offenderList(proposed, offenders)}.`,
|
|
141
|
+
RULE_LINE,
|
|
142
|
+
"Reorder the list or restate those statuses truthfully, then retry.",
|
|
143
|
+
`Proposed: ${snapshotLine(proposed)}`,
|
|
144
|
+
].join("\n") }],
|
|
145
|
+
details: { action: "init", tasks: tasks.map((t) => ({ ...t })), error: `pending-suffix violation: ${offenders.join(",")}` } as PlanTrackerDetails,
|
|
146
|
+
};
|
|
147
|
+
}
|
|
148
|
+
tasks = proposed;
|
|
154
149
|
updateWidget(ctx);
|
|
155
150
|
return {
|
|
156
151
|
content: [
|
|
@@ -174,7 +169,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
174
169
|
} as PlanTrackerDetails,
|
|
175
170
|
};
|
|
176
171
|
}
|
|
177
|
-
tasks.push(...params.tasks.map((
|
|
172
|
+
tasks.push(...params.tasks.map((t) => ({ name: normalizeTask(t).name, status: "pending" as TaskStatus })));
|
|
178
173
|
updateWidget(ctx);
|
|
179
174
|
return {
|
|
180
175
|
content: [
|
|
@@ -223,7 +218,32 @@ export default function (pi: ExtensionAPI) {
|
|
|
223
218
|
} as PlanTrackerDetails,
|
|
224
219
|
};
|
|
225
220
|
}
|
|
226
|
-
tasks[params.index]
|
|
221
|
+
const target = tasks[params.index];
|
|
222
|
+
if (params.status === "pending") {
|
|
223
|
+
return {
|
|
224
|
+
content: [{ type: "text", text: [
|
|
225
|
+
`Error: cannot set task ${params.index} "${target.name}" to pending: update never sets pending; a task is pending only from init/add.`,
|
|
226
|
+
"To redo it, update it to in_progress. To recreate the whole list, re-init with {name, status}[].",
|
|
227
|
+
`Current: ${snapshotLine(tasks)}`,
|
|
228
|
+
].join("\n") }],
|
|
229
|
+
details: { action: "update", tasks: tasks.map((t) => ({ ...t })), error: "update cannot set pending" } as PlanTrackerDetails,
|
|
230
|
+
};
|
|
231
|
+
}
|
|
232
|
+
const candidate = tasks.map((t, i) => (i === params.index ? { ...t, status: params.status as TaskStatus } : { ...t }));
|
|
233
|
+
const offenders = validateSnapshot(candidate);
|
|
234
|
+
if (offenders.length > 0) {
|
|
235
|
+
return {
|
|
236
|
+
content: [{ type: "text", text: [
|
|
237
|
+
`Error: cannot set task ${params.index} "${target.name}" to ${params.status}: pending tasks precede it: ${offenderList(tasks, offenders)}.`,
|
|
238
|
+
RULE_LINE,
|
|
239
|
+
"Fix first, then retry: update each listed task to in_progress (working on it now), complete, failed, or skipped (not applicable in this run); or, if the list order itself is wrong, re-init with {name, status}[] keeping every task and its true status so the untouched ones trail.",
|
|
240
|
+
"Never clear, drop tasks, or record a status the work has not actually reached.",
|
|
241
|
+
`Current: ${snapshotLine(tasks)}`,
|
|
242
|
+
].join("\n") }],
|
|
243
|
+
details: { action: "update", tasks: tasks.map((t) => ({ ...t })), error: `pending-suffix violation: ${offenders.join(",")}` } as PlanTrackerDetails,
|
|
244
|
+
};
|
|
245
|
+
}
|
|
246
|
+
tasks = candidate;
|
|
227
247
|
updateWidget(ctx);
|
|
228
248
|
return {
|
|
229
249
|
content: [
|
|
@@ -309,12 +329,9 @@ export default function (pi: ExtensionAPI) {
|
|
|
309
329
|
0,
|
|
310
330
|
);
|
|
311
331
|
case "update": {
|
|
312
|
-
const
|
|
313
|
-
const failed = taskList.filter((t) => t.status === "failed").length;
|
|
314
|
-
const suffix = failed > 0 ? `, ${failed} failed` : "";
|
|
332
|
+
const c = counts(taskList);
|
|
315
333
|
return new Text(
|
|
316
|
-
theme.fg("success", "✓ ") +
|
|
317
|
-
theme.fg("muted", `Updated (${complete}/${taskList.length} complete${suffix})`),
|
|
334
|
+
theme.fg("success", "✓ ") + theme.fg("muted", `Updated (${c.done}/${taskList.length} done${c.suffix})`),
|
|
318
335
|
0,
|
|
319
336
|
0,
|
|
320
337
|
);
|
|
@@ -323,19 +340,10 @@ export default function (pi: ExtensionAPI) {
|
|
|
323
340
|
if (taskList.length === 0) {
|
|
324
341
|
return new Text(theme.fg("dim", "No plan active"), 0, 0);
|
|
325
342
|
}
|
|
326
|
-
const
|
|
327
|
-
|
|
328
|
-
const suffix = failed > 0 ? `, ${failed} failed` : "";
|
|
329
|
-
let text = theme.fg("muted", `${complete}/${taskList.length} complete${suffix}`);
|
|
343
|
+
const c = counts(taskList);
|
|
344
|
+
let text = theme.fg("muted", `${c.done}/${taskList.length} done${c.suffix}`);
|
|
330
345
|
for (const t of taskList) {
|
|
331
|
-
const icon =
|
|
332
|
-
t.status === "complete"
|
|
333
|
-
? theme.fg("success", "✓")
|
|
334
|
-
: t.status === "in_progress"
|
|
335
|
-
? theme.fg("warning", "→")
|
|
336
|
-
: t.status === "failed"
|
|
337
|
-
? theme.fg("error", "✗")
|
|
338
|
-
: theme.fg("dim", "○");
|
|
346
|
+
const icon = glyph(t.status, theme);
|
|
339
347
|
text += `\n${icon} ${theme.fg("muted", t.name)}`;
|
|
340
348
|
}
|
|
341
349
|
return new Text(text, 0, 0);
|
|
@@ -22,16 +22,16 @@ const sources = {
|
|
|
22
22
|
}
|
|
23
23
|
`,
|
|
24
24
|
"@sinclair/typebox": `
|
|
25
|
-
const schema = (...args) => ({ args });
|
|
25
|
+
const schema = (kind) => (...args) => ({ kind, args });
|
|
26
26
|
export const Type = {
|
|
27
|
-
Object: schema,
|
|
28
|
-
Optional: schema,
|
|
29
|
-
String: schema,
|
|
30
|
-
Boolean: schema,
|
|
31
|
-
Union: schema,
|
|
32
|
-
Null: schema,
|
|
33
|
-
Array: schema,
|
|
34
|
-
Integer: schema,
|
|
27
|
+
Object: schema("Object"),
|
|
28
|
+
Optional: schema("Optional"),
|
|
29
|
+
String: schema("String"),
|
|
30
|
+
Boolean: schema("Boolean"),
|
|
31
|
+
Union: schema("Union"),
|
|
32
|
+
Null: schema("Null"),
|
|
33
|
+
Array: schema("Array"),
|
|
34
|
+
Integer: schema("Integer"),
|
|
35
35
|
};
|
|
36
36
|
`,
|
|
37
37
|
};
|
package/package.json
CHANGED
|
@@ -354,7 +354,7 @@ Execute this section in place from any later phase. Do not invoke `/skill:brains
|
|
|
354
354
|
|
|
355
355
|
1. Edit the spec. Show `git --no-pager diff -- <spec path>` and one line of impact (affected plan tasks / waves, or "no plan yet").
|
|
356
356
|
2. Wait for approval. Change request -> revise, re-show.
|
|
357
|
-
3. No plan yet -> commit the spec; continue. Plan exists -> update affected anchors and tasks: `plan_tracker` `add` for new tasks
|
|
357
|
+
3. No plan yet -> commit the spec; continue. Plan exists -> update affected anchors and tasks: `plan_tracker` `add` for new tasks; anchor-changed completed tasks are reopened as `in_progress` and re-run the task loop (`update` never sets `pending`). A removed task is deleted from the plan; then re-`init` the tracker with `{ name, status }` elements: preserved tasks keep their order and statuses, reopened tasks are `in_progress` in place, every still-`pending` task (including newly added ones, whatever wave label they carry) trails the non-pending ones, removed tasks are the only deletions (the only permitted `init` after handoff; never `clear`). Re-run `plan_check` until it passes, commit spec + plan together; continue. A task reopened while `verify` or `ship` is in progress: `phase_tracker({ action: "skip", phase: "<current>", reason: "amendment reopened Task N" })`, then `phase_tracker({ action: "start", phase: "implement", force: true })`; later phases re-enter with `force: true` and rerun in full.
|
|
358
358
|
|
|
359
359
|
Redraw test: the diff changes the problem statement, adds or removes a component, or moves a component boundary -> redraw. A change inside one component (a persistence mechanism, a worker's HTTP client, dropping a fallback and its task) -> amend. State the call in the same message as the diff; the user overrides either way.
|
|
360
360
|
|
|
@@ -125,11 +125,12 @@ configured) all emitted for manual execution, none auto-posted.
|
|
|
125
125
|
stage plus one task per AC - status mappings below apply only after that
|
|
126
126
|
init: pass / `satisfied` / `not externally observable` /
|
|
127
127
|
`unverified: no delivery target` / `allowed gap` / `proposed descope` ->
|
|
128
|
-
`complete`; failed stage / `unexplained gap` -> `failed`; skipped stage 2 ->
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
128
|
+
`complete`; failed stage / `unexplained gap` -> `failed`; skipped stage 2 ->
|
|
129
|
+
`skipped` (rendered ⊘, counted as done - never `complete` under a renamed
|
|
130
|
+
title). AC tasks follow the four stage tasks; record each AC verdict while
|
|
131
|
+
stage 3 is `in_progress` - the tracker rejects a verdict recorded behind a
|
|
132
|
+
still-pending stage. Optional-degrading: a native task list, or no tracking
|
|
133
|
+
at all, on harnesses without `plan_tracker`; absence is never a hard stop.
|
|
133
134
|
|
|
134
135
|
**Stage 0 - Pre-flight.** Fetch the ticket and all comments. Check the
|
|
135
136
|
current tracker status first: if it is already in a terminal/done state,
|
|
@@ -83,14 +83,18 @@ configuration.
|
|
|
83
83
|
|
|
84
84
|
## Progress tracking
|
|
85
85
|
|
|
86
|
-
Use `plan_tracker`, never `phase_tracker`. Init with the
|
|
87
|
-
`provision worktree`, `resolve evidence`, `claim-check
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
(shown crossed, error color)
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
86
|
+
Use `plan_tracker`, never `phase_tracker`. Init with the first four stages:
|
|
87
|
+
`gather`, `provision worktree`, `resolve evidence`, `claim-check`. While
|
|
88
|
+
`claim-check` is `in_progress`, `add` one task per material claim as the
|
|
89
|
+
Verifier enumerates them and record each verdict: a matched claim ->
|
|
90
|
+
`complete`, a contradicted claim -> `failed` (shown crossed, error color).
|
|
91
|
+
Once every claim is terminal and `claim-check` is closed, `add` `review` and
|
|
92
|
+
`consent menu` and continue. Stages are never inited ahead of the claims:
|
|
93
|
+
the tracker rejects a verdict recorded behind a still-pending stage. A
|
|
94
|
+
failed stage or claim stays `failed` while the skill stops at the menu -
|
|
95
|
+
never marked complete to move on. On a harness without the `plan_tracker`
|
|
96
|
+
tool: fall back to a plain checklist (or skip if none is available);
|
|
97
|
+
functionality is unchanged either way.
|
|
94
98
|
|
|
95
99
|
## Assessment
|
|
96
100
|
|
|
@@ -171,9 +171,9 @@ Auto-selected at handoff by `writing-plans` (any wave with ≥2 tasks) when the
|
|
|
171
171
|
**Progress tracking (`plan_tracker`).** `plan_tracker` is a flat list with no native group concept, so waves are *encoded*, not modeled:
|
|
172
172
|
|
|
173
173
|
- **Consume, preserve, recover only if absent:** consume the wave-ordered list initialized at writing-plans handoff; indices are positional and stable, so never re-init on continuation or mid-run. Only direct recovery with no tracker initializes the full plan list once before dispatch.
|
|
174
|
-
- **Wave fan-out → `in_progress`:** unconditionally mark every task index in the wave `in_progress` before dispatch. Multiple simultaneous entries are expected (sequential mode has one).
|
|
174
|
+
- **Wave fan-out → `in_progress`:** unconditionally mark every task index in the wave `in_progress` before dispatch, in increasing index order (the tracker validates each call against the previous one and rejects a start while an earlier index is still `pending`). Multiple simultaneous entries are expected (sequential mode has one).
|
|
175
175
|
- **Wave commit → `complete`:** after the wave's gate passes and it commits, unconditionally mark all those same indices `complete`. `complete` = durably committed, so a task in conflict fallback stays `in_progress` until its wave commits.
|
|
176
|
-
- **Lifecycle per task:** `pending → in_progress (wave fan-out) → complete (wave commit)
|
|
176
|
+
- **Lifecycle per task:** `pending → in_progress (wave fan-out) → complete (wave commit)`; `failed` when a task ran and did not pass; `skipped` when the plan drops it as not applicable (terminal, counted done). A rejected `update` names the earlier pending tasks and the legal fixes - record those tasks' true state, never `clear` or re-`init` to move on.
|
|
177
177
|
- **Widget caveat (known, deliberately unfixed).** The persistent `plan_tracker` widget's icon strip (`○ → ✓`) and `(c/total)` count reflect every task, but its trailing *name* shows only the **first** `in_progress` task. In parallel mode the icon strip and the `status` action are the full in-flight view; a richer multi-task widget is a separate extension change, out of scope (YAGNI).
|
|
178
178
|
- **Sequential mode:** consume the same existing full list, one `in_progress` index at a time; wave prefixes are harmless.
|
|
179
179
|
|
|
@@ -126,8 +126,10 @@ partition from an earlier round.
|
|
|
126
126
|
|
|
127
127
|
Mirrors `subagent-driven-development` Parallel-Wave Mode and reuses its
|
|
128
128
|
`plan_tracker` progress surface. Runs entirely inside the gate — it invokes
|
|
129
|
-
**no** `phase_tracker` calls (`
|
|
130
|
-
|
|
129
|
+
**no** phase-transition `phase_tracker` calls (`start`/`complete`/`skip`/`reset`
|
|
130
|
+
- `phase_tracker({ phase: "implement" })` errors while verify is `in_progress`;
|
|
131
|
+
the only `phase_tracker` call inside the loop is the human-approval
|
|
132
|
+
`grant_fix_rounds` in step 6) and does **not** enter SDD's phase machinery.
|
|
131
133
|
Only the fan-out/integrate/review shape and `plan_tracker` are reused. Every execution dispatch is foreground with top-level `async: false`, including retries and prose-described dispatches; an unexpected async handle is a configuration failure: stop and report, never poll or relaunch. `forceTopLevelAsync` is incompatible; see [pi-cohort dispatch configuration](https://github.com/jjuraszek/pi-cohort/blob/main/doc/configuration.md).
|
|
132
134
|
|
|
133
135
|
**Precondition — worktree required.** The loop needs a worktree HEAD to branch
|
|
@@ -141,7 +143,7 @@ prerequisites hold.
|
|
|
141
143
|
|
|
142
144
|
Per round:
|
|
143
145
|
|
|
144
|
-
1. **Synchronize gap tasks** — append only a genuinely new gap that is entering remediation, named `Gn: <gap origin clause verbatim, truncated>`; never `init`. Find existing gaps by their exact `Gn:` prefix and reuse that index even if origin wording changes. Carried-OPEN inventory-only gaps add nothing. Before dispatch, mark every remediated gap's existing index `in_progress
|
|
146
|
+
1. **Synchronize gap tasks** — append only a genuinely new gap that is entering remediation, named `Gn: <gap origin clause verbatim, truncated>`; never `init`. Find existing gaps by their exact `Gn:` prefix and reuse that index even if origin wording changes. Carried-OPEN inventory-only gaps add nothing. Before dispatch, mark every remediated gap's existing index `in_progress`, in increasing index order (the tracker rejects a start while a lower index is still `pending`); a re-audit needing more work reopens that same `Gn` index. The lifecycle traces `[T1,T2]`, then `[T1,T2,G1]`, then `[T1,T2,G1,G2]`; no test-retry or review-round wrapper task.
|
|
145
147
|
2. **Fix wave** — select gaps greedily in `Gn` order: take each `fix` gap unless a gap it `conflicts` with (per the report's `Parallel-safe:` line) is already taken; the certificate's `disjoint` grouping is ignored, and a certificate still malformed after the one re-ask means every gap `conflicts` with every other. Held gaps carry to the next round; the wave is never empty while an eligible `fix` gap exists. Dispatch **one** call - `subagent({ context: "fresh", async: false, tasks: [...] })` - with one `implementer` task per selected gap (`worktree: true`, `cwd` = the conformance worktree, task = the gap block verbatim with `touched-files` as the ownership boundary). A single gap is a one-task `tasks` call; a lone `agent: "implementer"` call never appears in this loop. For an `UNAUTHORIZED` `fix` gap
|
|
146
148
|
whose `evidence` opens with the over-spec provenance (`spec "<section>" - "<clause>" (over-spec)`), the orchestrator adds the spec path to that gap's `touched-files` before dispatch,
|
|
147
149
|
so the implementer deletes the surface **and** the clause/AC line in the same
|
|
@@ -178,8 +180,11 @@ Per round:
|
|
|
178
180
|
default `3`, floors negatives at `0`, coerces non-integers to `3`) reached
|
|
179
181
|
with an open `fix` gap or repair item → **escalate to the human** with the per-gap
|
|
180
182
|
round-by-round verdict trail. Escalation is the sole non-completing
|
|
181
|
-
terminal state — no silent re-loop, no auto-ship.
|
|
182
|
-
|
|
183
|
+
terminal state — no silent re-loop, no auto-ship. If the human explicitly
|
|
184
|
+
approves N more rounds, record `phase_tracker({ action: "grant_fix_rounds",
|
|
185
|
+
rounds: N, reason: "<their words>" })` and re-enter step 2; without that
|
|
186
|
+
approval, escalation stays terminal. Inside a brainstorming-entered flow
|
|
187
|
+
the phase tracker enforces both rules at
|
|
183
188
|
tool-call time: a lone `agent: "implementer"` dispatch is blocked, and the
|
|
184
189
|
wave after the cap is blocked (`closureReview.enforce: false` disables).
|
|
185
190
|
|
|
@@ -187,7 +192,7 @@ Per round:
|
|
|
187
192
|
|
|
188
193
|
a. Run the full plan-header `Verification` set once (ad-hoc: the project's canonical test command). After R0 `CONFORMS` with no round run, the pre-R0 full run counts.
|
|
189
194
|
b. Dispatch `code-reviewer` directly (foreground, `async: false`, `SCOPED_TEST_COMMANDS: none`) over `git diff <r0-head>..HEAD`; never via `/skill:requesting-code-review`. An empty diff is nothing to review - no dispatch.
|
|
190
|
-
c. Repair items = every failing command from a + every Critical/Moderate finding from b (`Behaviour-change: yes` included; the re-audit is its origin check). None → write the closure block; done. Any at the cap → escalate per step 6 with the test/CR trail. Any under the cap → re-enter step 2 as a one-task `tasks` wave: one `implementer` whose task is every repair item verbatim (ownership boundary = the files in `git diff <r0-head>..HEAD`, the CR findings' `touched-files`, and the files each failing command's output names, `SCOPED_TEST_COMMANDS: none`, no `Gn` tracker task, no gap selection), integrate as one `conformance fix CR`, re-audit, then Convergence again. That wave counts against `maxFixRounds`. Later Convergence CRs keep the same `<r0-head>..HEAD` range.
|
|
195
|
+
c. Repair items = every failing command from a + every Critical/Moderate finding from b (`Behaviour-change: yes` included; the re-audit is its origin check). None → write the closure block; done. Any at the cap → escalate per step 6 with the test/CR trail (an explicit human approval re-enters via `grant_fix_rounds`, as step 6 describes). Any under the cap → re-enter step 2 as a one-task `tasks` wave: one `implementer` whose task is every repair item verbatim (ownership boundary = the files in `git diff <r0-head>..HEAD`, the CR findings' `touched-files`, and the files each failing command's output names, `SCOPED_TEST_COMMANDS: none`, no `Gn` tracker task, no gap selection), integrate as one `conformance fix CR`, re-audit, then Convergence again. That wave counts against `maxFixRounds`. Later Convergence CRs keep the same `<r0-head>..HEAD` range.
|
|
191
196
|
|
|
192
197
|
`conformance fix CR` is not a gap fix: it is absent from the `auto-applied fix commits` index and has no `revert conformance fix Gn` action at the finish gate.
|
|
193
198
|
|