pi-gauntlet 5.5.3 → 5.5.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,11 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## v5.5.4 - 2026-09-14
|
|
4
|
+
|
|
5
|
+
- `phase-tracker`: new `phase_tracker` action `grant_fix_rounds` records an explicit human approval of N more conformance fix rounds (reason quotes the human), accepted only while the cap block is live; each qualifying implementer wave spends one granted round, credits replay with the session and reset with the audit latch. The cap-block message now leads with that action and names the exact `.pi/settings.json` for the `enforce: false` last resort (applies without restart, whole-block precedence).
|
|
6
|
+
- `gauntlet_setting` is registered with sequential execution, so a settings write and a verifying read batched in one message run in order.
|
|
7
|
+
- `conformance-check.md`: an explicit human approval re-enters the fix loop via `grant_fix_rounds`; without it escalation stays terminal.
|
|
8
|
+
|
|
3
9
|
## v5.5.3 - 2026-09-13
|
|
4
10
|
|
|
5
11
|
- `code-reviewer`: verdict is a stated function of severity - a Critical or Moderate finding means `FIX_FIRST`, Minor-only and clean reports mean `SHIP`, matching the orchestrating skills. (#30)
|
|
@@ -58,7 +58,7 @@ const resumedBranch = (rest: Partial<Record<Phase, Status>>) => [
|
|
|
58
58
|
|
|
59
59
|
function harness(options: { cwd?: string; branch?: unknown[]; idle?: boolean; beforeSettled?: (setIdle: (idle: boolean) => void) => void; sendThrows?: boolean; model?: { provider: string; id: string }; thinkingLevel?: string } = {}) {
|
|
60
60
|
const handlers = new Map<string, ((event: unknown, ctx: unknown) => unknown)[]>();
|
|
61
|
-
const tools: { name: string; execute: (...args: any[]) => unknown }[] = [];
|
|
61
|
+
const tools: { name: string; executionMode?: string; parameters?: any; execute: (...args: any[]) => unknown }[] = [];
|
|
62
62
|
const sent: { message: any; options: any }[] = [];
|
|
63
63
|
let idle = options.idle ?? true;
|
|
64
64
|
let branch = options.branch ?? [];
|
|
@@ -76,7 +76,7 @@ function harness(options: { cwd?: string; branch?: unknown[]; idle?: boolean; be
|
|
|
76
76
|
registered.push(handler);
|
|
77
77
|
handlers.set(event, registered);
|
|
78
78
|
},
|
|
79
|
-
registerTool(tool: { name: string; executionMode?: string; execute: (...args: any[]) => unknown }) {
|
|
79
|
+
registerTool(tool: { name: string; executionMode?: string; parameters?: any; execute: (...args: any[]) => unknown }) {
|
|
80
80
|
tools.push(tool);
|
|
81
81
|
},
|
|
82
82
|
sendMessage(message: unknown, sendOptions: unknown) {
|
|
@@ -1275,7 +1275,7 @@ test("cap guard: default 3 waves pass, the fourth is blocked with the escalation
|
|
|
1275
1275
|
}
|
|
1276
1276
|
const blocked = await firstCallResult(h, "c4", implementerWave(1));
|
|
1277
1277
|
assert.equal(blocked?.block, true);
|
|
1278
|
-
assert.match(blocked?.reason ?? "", /fix round
|
|
1278
|
+
assert.match(blocked?.reason ?? "", /3 fix round\(s\) used against a cap of 3 \(granted rounds included\); escalate to the human/);
|
|
1279
1279
|
});
|
|
1280
1280
|
|
|
1281
1281
|
test("cap guard: maxFixRounds 0 blocks the first wave; 1 blocks the second; a chain implementer is cap-checked and counted", async () => {
|
|
@@ -1422,3 +1422,298 @@ test("replay: two implementer waves after the audit in verify restore fixRounds
|
|
|
1422
1422
|
await inShip.emit("session_start");
|
|
1423
1423
|
assert.equal(await firstCallResult(inShip, "c1", implementerWave(1)), undefined);
|
|
1424
1424
|
});
|
|
1425
|
+
|
|
1426
|
+
// --- Human overrule of the fix-round cap (spec 2026-09-14-fix-round-human-overrule) ---
|
|
1427
|
+
|
|
1428
|
+
const grantResult = (rounds: number, reason = "human approved") => ({
|
|
1429
|
+
type: "message",
|
|
1430
|
+
message: { role: "toolResult", toolName: "phase_tracker", details: {
|
|
1431
|
+
action: "grant_fix_rounds", rounds, reason,
|
|
1432
|
+
phases: phases({ brainstorm: "complete", plan: "complete", implement: "complete", verify: "in_progress" }),
|
|
1433
|
+
} },
|
|
1434
|
+
});
|
|
1435
|
+
|
|
1436
|
+
const grant = async (h: ReturnType<typeof harness>, id: string, input: Record<string, unknown>) =>
|
|
1437
|
+
(await h.tools.find((t) => t.name === "phase_tracker")!.execute(id, { action: "grant_fix_rounds", ...input }, undefined, undefined, h.ctx)) as {
|
|
1438
|
+
content: { type: string; text: string }[];
|
|
1439
|
+
details: { action: string; rounds?: number; reason?: string; error?: string };
|
|
1440
|
+
};
|
|
1441
|
+
|
|
1442
|
+
const exhaust = async (h: ReturnType<typeof harness>, rounds: number, prefix = "x") => {
|
|
1443
|
+
for (let i = 1; i <= rounds; i++) {
|
|
1444
|
+
assert.equal(await firstCallResult(h, `${prefix}${i}`, implementerWave(1)), undefined);
|
|
1445
|
+
await h.emitEvent("tool_result", okWave(`${prefix}${i}`));
|
|
1446
|
+
}
|
|
1447
|
+
};
|
|
1448
|
+
|
|
1449
|
+
test("grant: funds extra waves and rejects another grant while credits remain", async () => {
|
|
1450
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { enforce: true } } });
|
|
1451
|
+
await h.emit("session_start");
|
|
1452
|
+
await exhaust(h, 3);
|
|
1453
|
+
const res = await grant(h, "g1", { rounds: 2, reason: "I'm approving 2 more rounds" });
|
|
1454
|
+
assert.equal(res.details.error, undefined);
|
|
1455
|
+
assert.deepEqual([res.details.rounds, res.details.reason], [2, "I'm approving 2 more rounds"]);
|
|
1456
|
+
assert.equal((await grant(h, "g2", { rounds: 1, reason: "more" })).details.error, "grant_fix_rounds: 2 granted round(s) still unused; spend them before granting more");
|
|
1457
|
+
await exhaust(h, 2, "c");
|
|
1458
|
+
const blocked = await firstCallResult(h, "c3", implementerWave(1));
|
|
1459
|
+
assert.equal(blocked?.block, true);
|
|
1460
|
+
assert.match(blocked?.reason ?? "", /5 fix round\(s\) used against a cap of 3 \(granted rounds included\)/);
|
|
1461
|
+
});
|
|
1462
|
+
|
|
1463
|
+
test("grant: schema declares rounds as integer 1..MAX_SAFE_INTEGER; missing rounds or empty reason error without state change", async () => {
|
|
1464
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1465
|
+
await h.emit("session_start");
|
|
1466
|
+
const params = h.tools.find((t) => t.name === "phase_tracker")!.parameters;
|
|
1467
|
+
const roundOptions = params.args[0].rounds.args[0].args[0];
|
|
1468
|
+
assert.equal(params.args[0].rounds.args[0].kind, "Integer");
|
|
1469
|
+
assert.deepEqual([roundOptions.minimum, roundOptions.maximum], [1, Number.MAX_SAFE_INTEGER]);
|
|
1470
|
+
assert.ok(params.args[0].action.values.includes("grant_fix_rounds"));
|
|
1471
|
+
|
|
1472
|
+
const noRounds = await grant(h, "g1", { reason: "ok" });
|
|
1473
|
+
assert.equal(noRounds.details.error, "grant_fix_rounds requires rounds: a positive integer");
|
|
1474
|
+
assert.equal(noRounds.details.rounds, undefined);
|
|
1475
|
+
const noReason = await grant(h, "g2", { rounds: 1, reason: " " });
|
|
1476
|
+
assert.equal(noReason.details.error, "grant_fix_rounds requires reason: the human's approval, quoted");
|
|
1477
|
+
assert.equal(noReason.details.rounds, undefined);
|
|
1478
|
+
assert.equal((await firstCallResult(h, "c1", implementerWave(1)))?.block, true, "no credit was granted");
|
|
1479
|
+
});
|
|
1480
|
+
|
|
1481
|
+
test("grant: rejected when no cap block is live - before the audit, in implement, below the cap, enforce off, and in a child", async () => {
|
|
1482
|
+
const expectNotLive = async (h: ReturnType<typeof harness>, used: number, cap: number) => {
|
|
1483
|
+
const res = await grant(h, "g", { rounds: 1, reason: "ok" });
|
|
1484
|
+
assert.equal(res.details.error, `grant_fix_rounds: no fix-round cap block is active (${used} used, cap ${cap}); nothing to overrule`);
|
|
1485
|
+
};
|
|
1486
|
+
|
|
1487
|
+
const preAudit = harness({ cwd: tempCwd({ piGauntlet: { closureReview: { maxFixRounds: 0 } } }), branch: verifyBranch() });
|
|
1488
|
+
await preAudit.emit("session_start");
|
|
1489
|
+
await expectNotLive(preAudit, 0, 0);
|
|
1490
|
+
|
|
1491
|
+
const implement = harness({ cwd: tempCwd({ piGauntlet: { closureReview: { maxFixRounds: 0 } } }), branch: implementBranch([subagentResult(["conformance-reviewer"])]) });
|
|
1492
|
+
await implement.emit("session_start");
|
|
1493
|
+
await expectNotLive(implement, 0, 0);
|
|
1494
|
+
|
|
1495
|
+
const below = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 3 } } });
|
|
1496
|
+
await below.emit("session_start");
|
|
1497
|
+
await exhaust(below, 2);
|
|
1498
|
+
await expectNotLive(below, 2, 3);
|
|
1499
|
+
|
|
1500
|
+
const off = guardedHarness({ piGauntlet: { closureReview: { enforce: false, maxFixRounds: 0 } } });
|
|
1501
|
+
await off.emit("session_start");
|
|
1502
|
+
await expectNotLive(off, 0, 0);
|
|
1503
|
+
|
|
1504
|
+
const priorDepth = process.env.PI_SUBAGENT_DEPTH;
|
|
1505
|
+
process.env.PI_SUBAGENT_DEPTH = "1";
|
|
1506
|
+
let child: ReturnType<typeof harness>;
|
|
1507
|
+
try {
|
|
1508
|
+
child = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1509
|
+
} finally {
|
|
1510
|
+
if (priorDepth === undefined) delete process.env.PI_SUBAGENT_DEPTH;
|
|
1511
|
+
else process.env.PI_SUBAGENT_DEPTH = priorDepth;
|
|
1512
|
+
}
|
|
1513
|
+
await child.emit("session_start");
|
|
1514
|
+
await expectNotLive(child, 0, 0);
|
|
1515
|
+
});
|
|
1516
|
+
|
|
1517
|
+
test("grant: only qualifying synchronous parent task waves spend credits", async () => {
|
|
1518
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1519
|
+
await h.emit("session_start");
|
|
1520
|
+
await grant(h, "g", { rounds: 1, reason: "ok" });
|
|
1521
|
+
await h.emitEvent("tool_result", waveResult("e1", [{ agent: "code-reviewer", exitCode: 0 }]));
|
|
1522
|
+
await h.emitEvent("tool_result", waveResult("e2", [{ agent: "implementer", exitCode: 0 }], true));
|
|
1523
|
+
await h.emitEvent("tool_result", waveResult("e3", []));
|
|
1524
|
+
assert.equal((await firstCallResult(h, "a", { ...implementerWave(1), async: true }))?.block, true);
|
|
1525
|
+
assert.equal((await firstCallResult(h, "l", loneImplementer()))?.block, true);
|
|
1526
|
+
assert.equal(await firstCallResult(h, "c", implementerWave(1)), undefined);
|
|
1527
|
+
await h.emitEvent("tool_result", okWave("c"));
|
|
1528
|
+
assert.equal((await firstCallResult(h, "d", implementerWave(1)))?.block, true);
|
|
1529
|
+
|
|
1530
|
+
// In a child (PI_SUBAGENT_DEPTH >= 1, fixed at registration) both the cap gate and observeFixWave
|
|
1531
|
+
// are dormant, so credit consumption has no public effect; the observable contract is dormancy.
|
|
1532
|
+
const priorDepth = process.env.PI_SUBAGENT_DEPTH;
|
|
1533
|
+
process.env.PI_SUBAGENT_DEPTH = "1";
|
|
1534
|
+
let child: ReturnType<typeof harness>;
|
|
1535
|
+
try {
|
|
1536
|
+
child = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } }, [grantResult(1), subagentResult(["implementer"])]);
|
|
1537
|
+
} finally {
|
|
1538
|
+
if (priorDepth === undefined) delete process.env.PI_SUBAGENT_DEPTH;
|
|
1539
|
+
else process.env.PI_SUBAGENT_DEPTH = priorDepth;
|
|
1540
|
+
}
|
|
1541
|
+
await child.emit("session_start");
|
|
1542
|
+
assert.equal(await firstCallResult(child, "child-wave", implementerWave(1)), undefined, "child session: cap gate and observer are dormant, so the replayed grant and implementer result have no observable effect");
|
|
1543
|
+
});
|
|
1544
|
+
|
|
1545
|
+
test("grant replay restores and spends the recorded pool independent of current cap; rejected grants restore no credit", async () => {
|
|
1546
|
+
const trail = [subagentResult(["implementer"]), subagentResult(["implementer"]), subagentResult(["implementer"]), grantResult(2), subagentResult(["implementer"])];
|
|
1547
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 3 } } }, trail);
|
|
1548
|
+
await h.emit("session_start");
|
|
1549
|
+
assert.equal(
|
|
1550
|
+
(await grant(h, "g", { rounds: 1, reason: "more" })).details.error,
|
|
1551
|
+
"grant_fix_rounds: 1 granted round(s) still unused; spend them before granting more",
|
|
1552
|
+
);
|
|
1553
|
+
assert.equal(await firstCallResult(h, "c", implementerWave(1)), undefined);
|
|
1554
|
+
await h.emitEvent("tool_result", okWave("c"));
|
|
1555
|
+
const blocked = await firstCallResult(h, "d", implementerWave(1));
|
|
1556
|
+
assert.equal(blocked?.block, true);
|
|
1557
|
+
assert.match(blocked?.reason ?? "", /5 fix round\(s\) used against a cap of 3 \(granted rounds included\)/);
|
|
1558
|
+
|
|
1559
|
+
// Same trail at cap 5: replay yields fixRounds = 4 (cap not live yet); lowering the live cap
|
|
1560
|
+
// exposes the one replayed credit, then wave 5 passes on the counter and wave 6 blocks.
|
|
1561
|
+
const capChanged = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 5 } } }, trail);
|
|
1562
|
+
await capChanged.emit("session_start");
|
|
1563
|
+
assert.equal(
|
|
1564
|
+
(await grant(capChanged, "g", { rounds: 1, reason: "more" })).details.error,
|
|
1565
|
+
"grant_fix_rounds: no fix-round cap block is active (4 used, cap 5); nothing to overrule",
|
|
1566
|
+
"replayed fixRounds = 4 regardless of cap on disk",
|
|
1567
|
+
);
|
|
1568
|
+
// Lower the cap on disk without spending a wave: the block goes live at 4 >= 3 and the grant
|
|
1569
|
+
// probe now reads the replayed pool - exactly one credit, independent of the cap at replay time.
|
|
1570
|
+
writeFileSync(join(capChanged.ctx.cwd, ".pi", "settings.json"), JSON.stringify({ piGauntlet: { closureReview: { maxFixRounds: 3 } } }));
|
|
1571
|
+
assert.equal(
|
|
1572
|
+
(await grant(capChanged, "g2", { rounds: 1, reason: "more" })).details.error,
|
|
1573
|
+
"grant_fix_rounds: 1 granted round(s) still unused; spend them before granting more",
|
|
1574
|
+
"same credits regardless of cap on disk",
|
|
1575
|
+
);
|
|
1576
|
+
writeFileSync(join(capChanged.ctx.cwd, ".pi", "settings.json"), JSON.stringify({ piGauntlet: { closureReview: { maxFixRounds: 5 } } }));
|
|
1577
|
+
assert.equal(await firstCallResult(capChanged, "c", implementerWave(1)), undefined);
|
|
1578
|
+
await capChanged.emitEvent("tool_result", okWave("c"));
|
|
1579
|
+
const blockedAt5 = await firstCallResult(capChanged, "d", implementerWave(1));
|
|
1580
|
+
assert.equal(blockedAt5?.block, true);
|
|
1581
|
+
assert.match(blockedAt5?.reason ?? "", /5 fix round\(s\) used against a cap of 5 \(granted rounds included\)/);
|
|
1582
|
+
|
|
1583
|
+
const rejected = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } }, [{
|
|
1584
|
+
type: "message",
|
|
1585
|
+
message: {
|
|
1586
|
+
role: "toolResult",
|
|
1587
|
+
toolName: "phase_tracker",
|
|
1588
|
+
details: { action: "grant_fix_rounds", error: "grant_fix_rounds requires rounds: a positive integer", phases: phases({ verify: "in_progress" }) },
|
|
1589
|
+
},
|
|
1590
|
+
}]);
|
|
1591
|
+
await rejected.emit("session_start");
|
|
1592
|
+
assert.equal((await firstCallResult(rejected, "c1", implementerWave(1)))?.block, true, "no credits from a rejected grant");
|
|
1593
|
+
});
|
|
1594
|
+
|
|
1595
|
+
test("grant reset: start implement and reset zero credits (live and replay); start verify --force keeps them", async () => {
|
|
1596
|
+
const settings = { piGauntlet: { closureReview: { maxFixRounds: 0 }, flowGuards: { enforce: false } } };
|
|
1597
|
+
|
|
1598
|
+
const force = guardedHarness(settings, [grantResult(1)]);
|
|
1599
|
+
await force.emit("session_start");
|
|
1600
|
+
await force.tools.find((t) => t.name === "phase_tracker")!.execute("p", { action: "start", phase: "verify", force: true }, undefined, undefined, force.ctx);
|
|
1601
|
+
assert.equal(await firstCallResult(force, "c", implementerWave(1)), undefined);
|
|
1602
|
+
|
|
1603
|
+
const liveImpl = guardedHarness(settings);
|
|
1604
|
+
await liveImpl.emit("session_start");
|
|
1605
|
+
await grant(liveImpl, "g", { rounds: 1, reason: "ok" });
|
|
1606
|
+
const tool = liveImpl.tools.find((t) => t.name === "phase_tracker")!;
|
|
1607
|
+
for (const [id, input] of [
|
|
1608
|
+
["p1", { action: "skip", phase: "verify", reason: "amendment" }],
|
|
1609
|
+
["p2", { action: "start", phase: "implement", force: true }],
|
|
1610
|
+
["p3", { action: "complete", phase: "implement" }],
|
|
1611
|
+
["p4", { action: "start", phase: "verify", force: true }],
|
|
1612
|
+
] as const) {
|
|
1613
|
+
assert.equal(((await tool.execute(id, input, undefined, undefined, liveImpl.ctx)) as { details: { error?: string } }).details.error, undefined);
|
|
1614
|
+
}
|
|
1615
|
+
await liveImpl.emitEvent("tool_result", waveResult("audit", [{ agent: "conformance-reviewer", exitCode: 0 }]));
|
|
1616
|
+
assert.equal((await firstCallResult(liveImpl, "c1", implementerWave(1)))?.block, true, "credits zeroed by start implement (cap 0 blocks)");
|
|
1617
|
+
|
|
1618
|
+
const reset = guardedHarness(settings);
|
|
1619
|
+
await reset.emit("session_start");
|
|
1620
|
+
await grant(reset, "g", { rounds: 1, reason: "ok" });
|
|
1621
|
+
const resetTool = reset.tools.find((t) => t.name === "phase_tracker")!;
|
|
1622
|
+
await resetTool.execute("p", { action: "reset" }, undefined, undefined, reset.ctx);
|
|
1623
|
+
assert.match((await grant(reset, "g2", { rounds: 1, reason: "ok" })).details.error ?? "", /no fix-round cap block is active/);
|
|
1624
|
+
for (const [id, input] of [
|
|
1625
|
+
["p1", { action: "start", phase: "brainstorm" }],
|
|
1626
|
+
["p2", { action: "complete", phase: "brainstorm" }],
|
|
1627
|
+
["p3", { action: "start", phase: "plan" }],
|
|
1628
|
+
["p4", { action: "complete", phase: "plan" }],
|
|
1629
|
+
["p5", { action: "start", phase: "implement" }],
|
|
1630
|
+
["p6", { action: "complete", phase: "implement" }],
|
|
1631
|
+
["p7", { action: "start", phase: "verify" }],
|
|
1632
|
+
] as const) {
|
|
1633
|
+
assert.equal(((await resetTool.execute(id, input, undefined, undefined, reset.ctx)) as { details: { error?: string } }).details.error, undefined);
|
|
1634
|
+
}
|
|
1635
|
+
await reset.emitEvent("tool_result", waveResult("audit", [{ agent: "conformance-reviewer", exitCode: 0 }]));
|
|
1636
|
+
assert.equal((await firstCallResult(reset, "c1", implementerWave(1)))?.block, true, "credits zeroed by reset (cap 0 blocks)");
|
|
1637
|
+
|
|
1638
|
+
const replayImpl = harness({
|
|
1639
|
+
cwd: tempCwd(settings),
|
|
1640
|
+
branch: verifyBranch([
|
|
1641
|
+
subagentResult(["conformance-reviewer"]),
|
|
1642
|
+
grantResult(1),
|
|
1643
|
+
phaseResult("skip", phases({ brainstorm: "complete", plan: "complete", implement: "complete", verify: "skipped" })),
|
|
1644
|
+
phaseResult("start", phases({ brainstorm: "complete", plan: "complete", implement: "in_progress", verify: "skipped" })),
|
|
1645
|
+
phaseResult("complete", phases({ brainstorm: "complete", plan: "complete", implement: "complete", verify: "skipped" })),
|
|
1646
|
+
phaseResult("start", phases({ brainstorm: "complete", plan: "complete", implement: "complete", verify: "in_progress" })),
|
|
1647
|
+
subagentResult(["conformance-reviewer"]),
|
|
1648
|
+
]),
|
|
1649
|
+
});
|
|
1650
|
+
await replayImpl.emit("session_start");
|
|
1651
|
+
assert.equal((await firstCallResult(replayImpl, "c1", implementerWave(1)))?.block, true, "replayed start implement zeroed credits");
|
|
1652
|
+
|
|
1653
|
+
const replayReset = harness({
|
|
1654
|
+
cwd: tempCwd(settings),
|
|
1655
|
+
branch: verifyBranch([
|
|
1656
|
+
subagentResult(["conformance-reviewer"]),
|
|
1657
|
+
grantResult(1),
|
|
1658
|
+
phaseResult("reset", phases({})),
|
|
1659
|
+
phaseResult("start", phases({ brainstorm: "in_progress" })),
|
|
1660
|
+
phaseResult("complete", phases({ brainstorm: "complete" })),
|
|
1661
|
+
phaseResult("start", phases({ brainstorm: "complete", plan: "in_progress" })),
|
|
1662
|
+
phaseResult("complete", phases({ brainstorm: "complete", plan: "complete" })),
|
|
1663
|
+
phaseResult("start", phases({ brainstorm: "complete", plan: "complete", implement: "in_progress" })),
|
|
1664
|
+
phaseResult("complete", phases({ brainstorm: "complete", plan: "complete", implement: "complete" })),
|
|
1665
|
+
phaseResult("start", phases({ brainstorm: "complete", plan: "complete", implement: "complete", verify: "in_progress" })),
|
|
1666
|
+
subagentResult(["conformance-reviewer"]),
|
|
1667
|
+
]),
|
|
1668
|
+
});
|
|
1669
|
+
await replayReset.emit("session_start");
|
|
1670
|
+
assert.equal((await firstCallResult(replayReset, "c1", implementerWave(1)))?.block, true, "replayed reset zeroed credits");
|
|
1671
|
+
});
|
|
1672
|
+
|
|
1673
|
+
test("grant: block reason gives action, settings path, and live-read guidance", async () => {
|
|
1674
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1675
|
+
await h.emit("session_start");
|
|
1676
|
+
const reason = (await firstCallResult(h, "c", implementerWave(1)))?.reason ?? "";
|
|
1677
|
+
assert.match(reason, /phase_tracker\(\{ action: "grant_fix_rounds"/);
|
|
1678
|
+
assert.ok(reason.includes(join(h.ctx.cwd, ".pi", "settings.json")));
|
|
1679
|
+
assert.match(reason, /no restart.*restate model and maxFixRounds/s);
|
|
1680
|
+
});
|
|
1681
|
+
|
|
1682
|
+
test("gauntlet_setting registration requests sequential execution", () => {
|
|
1683
|
+
assert.equal(harness().tools.find((t) => t.name === "gauntlet_setting")!.executionMode, "sequential");
|
|
1684
|
+
});
|
|
1685
|
+
|
|
1686
|
+
test("grant renderResult shows count and reason", async () => {
|
|
1687
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1688
|
+
await h.emit("session_start");
|
|
1689
|
+
const res = await grant(h, "g", { rounds: 2, reason: "go on" });
|
|
1690
|
+
const tool = h.tools.find((t) => t.name === "phase_tracker")! as any;
|
|
1691
|
+
const rendered = tool.renderResult(res, {}, { fg: (_c: string, s: string) => s, bold: (s: string) => s });
|
|
1692
|
+
assert.match(JSON.stringify(rendered), /2.*go on/);
|
|
1693
|
+
});
|
|
1694
|
+
|
|
1695
|
+
test("grant: child sessions cannot overrule even a zero cap", async () => {
|
|
1696
|
+
const priorDepth = process.env.PI_SUBAGENT_DEPTH;
|
|
1697
|
+
process.env.PI_SUBAGENT_DEPTH = "1";
|
|
1698
|
+
let child: ReturnType<typeof harness>;
|
|
1699
|
+
try {
|
|
1700
|
+
child = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1701
|
+
} finally {
|
|
1702
|
+
if (priorDepth === undefined) delete process.env.PI_SUBAGENT_DEPTH;
|
|
1703
|
+
else process.env.PI_SUBAGENT_DEPTH = priorDepth;
|
|
1704
|
+
}
|
|
1705
|
+
await child.emit("session_start");
|
|
1706
|
+
assert.match((await grant(child, "g", { rounds: 1, reason: "ok" })).details.error ?? "", /no fix-round cap block is active/);
|
|
1707
|
+
});
|
|
1708
|
+
|
|
1709
|
+
test("grant: accepts MAX_SAFE_INTEGER and current on-disk cap controls whether the block is live", async () => {
|
|
1710
|
+
const h = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1711
|
+
await h.emit("session_start");
|
|
1712
|
+
const result = await grant(h, "g", { rounds: Number.MAX_SAFE_INTEGER, reason: "approved" });
|
|
1713
|
+
assert.equal(result.details.rounds, Number.MAX_SAFE_INTEGER);
|
|
1714
|
+
|
|
1715
|
+
const changed = guardedHarness({ piGauntlet: { closureReview: { maxFixRounds: 0 } } });
|
|
1716
|
+
await changed.emit("session_start");
|
|
1717
|
+
writeFileSync(join(changed.ctx.cwd, ".pi", "settings.json"), JSON.stringify({ piGauntlet: { closureReview: { maxFixRounds: 1 } } }));
|
|
1718
|
+
assert.match((await grant(changed, "g", { rounds: 1, reason: "ok" })).details.error ?? "", /0 used, cap 1/);
|
|
1719
|
+
});
|
|
@@ -56,9 +56,11 @@ interface PhaseState {
|
|
|
56
56
|
type PhaseMap = Record<Phase, PhaseState>;
|
|
57
57
|
|
|
58
58
|
interface PhaseTrackerDetails {
|
|
59
|
-
action: "start" | "complete" | "skip" | "status" | "reset" | "substep";
|
|
59
|
+
action: "start" | "complete" | "skip" | "status" | "reset" | "substep" | "grant_fix_rounds";
|
|
60
60
|
phases: PhaseMap;
|
|
61
61
|
error?: string;
|
|
62
|
+
rounds?: number;
|
|
63
|
+
reason?: string;
|
|
62
64
|
}
|
|
63
65
|
|
|
64
66
|
interface PlanCheckStamp {
|
|
@@ -206,10 +208,14 @@ const loneImplementerBlockReason =
|
|
|
206
208
|
"a lone agent call runs unisolated and produces no worktree diff. " +
|
|
207
209
|
"To disable this gate, set piGauntlet.closureReview.enforce: false.";
|
|
208
210
|
|
|
209
|
-
const fixRoundCapBlockReason = (used: number, cap: number): string =>
|
|
210
|
-
`Conformance fix loop: fix round
|
|
211
|
-
"with the verdict trail instead of re-looping
|
|
212
|
-
"
|
|
211
|
+
const fixRoundCapBlockReason = (used: number, cap: number, settingsPath: string): string =>
|
|
212
|
+
`Conformance fix loop: ${used} fix round(s) used against a cap of ${cap} (granted rounds included); ` +
|
|
213
|
+
"escalate to the human with the verdict trail instead of re-looping.\n" +
|
|
214
|
+
"If the human explicitly approves more rounds, record it and retry as a tasks wave: " +
|
|
215
|
+
'phase_tracker({ action: "grant_fix_rounds", rounds: <N>, reason: "<their words>" }).\n' +
|
|
216
|
+
`Last resort: set piGauntlet.closureReview.enforce: false in ${settingsPath} (disables all closure guards; ` +
|
|
217
|
+
"applies on the next tool call, no restart - a gauntlet_setting read in the same message sees the write; " +
|
|
218
|
+
"a repo closureReview block replaces the preset's whole block, so restate model and maxFixRounds alongside).";
|
|
213
219
|
|
|
214
220
|
const closureModelBlockReason = (model: string, missing: number, total: number): string =>
|
|
215
221
|
`Blocked: ${missing} of ${total} conformance-reviewer ${total === 1 ? "dispatch" : "entries"} ` +
|
|
@@ -255,7 +261,7 @@ const pathInSpecDirs = (rawPath: string, specDirs: string[]): boolean => {
|
|
|
255
261
|
};
|
|
256
262
|
|
|
257
263
|
const PhaseTrackerParams = Type.Object({
|
|
258
|
-
action: StringEnum(["start", "complete", "skip", "status", "reset", "substep"] as const, {
|
|
264
|
+
action: StringEnum(["start", "complete", "skip", "status", "reset", "substep", "grant_fix_rounds"] as const, {
|
|
259
265
|
description: "Action to perform",
|
|
260
266
|
}),
|
|
261
267
|
phase: Type.Optional(
|
|
@@ -265,7 +271,14 @@ const PhaseTrackerParams = Type.Object({
|
|
|
265
271
|
),
|
|
266
272
|
reason: Type.Optional(
|
|
267
273
|
Type.String({
|
|
268
|
-
description: "Reason
|
|
274
|
+
description: "Reason (required for skip and grant_fix_rounds; for grant, the human's approval quoted)",
|
|
275
|
+
}),
|
|
276
|
+
),
|
|
277
|
+
rounds: Type.Optional(
|
|
278
|
+
Type.Integer({
|
|
279
|
+
minimum: 1,
|
|
280
|
+
maximum: Number.MAX_SAFE_INTEGER,
|
|
281
|
+
description: "Extra fix rounds the human explicitly approved (grant_fix_rounds only)",
|
|
269
282
|
}),
|
|
270
283
|
),
|
|
271
284
|
force: Type.Optional(
|
|
@@ -328,6 +341,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
328
341
|
let phases: PhaseMap = emptyPhases();
|
|
329
342
|
let conformanceDispatched = false;
|
|
330
343
|
let fixRounds = 0;
|
|
344
|
+
let fixRoundCredits = 0;
|
|
331
345
|
let gauntletEntered = false;
|
|
332
346
|
// Shared by replay and the live tool_result hook so a resumed session enforces
|
|
333
347
|
// the same budget. Evaluated BEFORE the same result may set the latch, so the
|
|
@@ -341,6 +355,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
341
355
|
isImplementerWave(details, isError) &&
|
|
342
356
|
resolveClosureReview(loadGauntletSettings(ctx.cwd).gauntlet).enforce
|
|
343
357
|
) {
|
|
358
|
+
if (fixRoundCredits > 0) fixRoundCredits -= 1;
|
|
344
359
|
fixRounds += 1;
|
|
345
360
|
}
|
|
346
361
|
};
|
|
@@ -420,6 +435,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
420
435
|
phases = emptyPhases();
|
|
421
436
|
conformanceDispatched = false;
|
|
422
437
|
fixRounds = 0;
|
|
438
|
+
fixRoundCredits = 0;
|
|
423
439
|
gauntletEntered = false;
|
|
424
440
|
planCheckStamp = undefined;
|
|
425
441
|
attemptedRecoveryEdges.clear();
|
|
@@ -444,14 +460,17 @@ export default function (pi: ExtensionAPI) {
|
|
|
444
460
|
const details = msg.details as PhaseTrackerDetails | undefined;
|
|
445
461
|
if (details && !details.error) {
|
|
446
462
|
phases = details.phases;
|
|
463
|
+
if (details.action === "grant_fix_rounds" && typeof details.rounds === "number") fixRoundCredits = details.rounds;
|
|
447
464
|
gauntletEntered = nextGauntletEntered(gauntletEntered, details.action, details.phases.brainstorm.status);
|
|
448
465
|
if (details.action === "start" && details.phases.implement.status === "in_progress") {
|
|
449
466
|
conformanceDispatched = false;
|
|
450
467
|
fixRounds = 0;
|
|
468
|
+
fixRoundCredits = 0;
|
|
451
469
|
}
|
|
452
470
|
if (details.action === "reset") {
|
|
453
471
|
conformanceDispatched = false;
|
|
454
472
|
fixRounds = 0;
|
|
473
|
+
fixRoundCredits = 0;
|
|
455
474
|
planCheckStamp = undefined;
|
|
456
475
|
}
|
|
457
476
|
}
|
|
@@ -544,7 +563,12 @@ export default function (pi: ExtensionAPI) {
|
|
|
544
563
|
}
|
|
545
564
|
if (fixLoopWindow && collectAgents(event.input, "implementer").length > 0) {
|
|
546
565
|
const cap = resolveClosureReview(g()).maxFixRounds;
|
|
547
|
-
if (fixRounds >= cap)
|
|
566
|
+
if (fixRounds >= cap) {
|
|
567
|
+
const funded = fixRoundCredits > 0 && (event.input as { async?: unknown })?.async !== true;
|
|
568
|
+
if (!funded) {
|
|
569
|
+
return { block: true, reason: fixRoundCapBlockReason(fixRounds, cap, join(ctx.cwd, ".pi", "settings.json")) };
|
|
570
|
+
}
|
|
571
|
+
}
|
|
548
572
|
}
|
|
549
573
|
const model = closureReviewModel();
|
|
550
574
|
if (model && !hasAction) {
|
|
@@ -730,6 +754,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
730
754
|
label: "Gauntlet Setting",
|
|
731
755
|
description: "Resolve merged piGauntlet.* settings (repo over preset); for skill use only.",
|
|
732
756
|
parameters: GauntletSettingParams,
|
|
757
|
+
executionMode: "sequential",
|
|
733
758
|
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
|
734
759
|
const { gauntlet, errors } = loadGauntletSettings(ctx.cwd);
|
|
735
760
|
const payload =
|
|
@@ -914,6 +939,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
914
939
|
if (params.phase === "implement") {
|
|
915
940
|
conformanceDispatched = false;
|
|
916
941
|
fixRounds = 0;
|
|
942
|
+
fixRoundCredits = 0;
|
|
917
943
|
}
|
|
918
944
|
gauntletEntered = nextGauntletEntered(gauntletEntered, "start", phases.brainstorm.status);
|
|
919
945
|
firedGuards.clear();
|
|
@@ -1080,6 +1106,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
1080
1106
|
) as PhaseMap;
|
|
1081
1107
|
conformanceDispatched = false;
|
|
1082
1108
|
fixRounds = 0;
|
|
1109
|
+
fixRoundCredits = 0;
|
|
1083
1110
|
gauntletEntered = nextGauntletEntered(gauntletEntered, "reset", phases.brainstorm.status);
|
|
1084
1111
|
firedGuards.clear();
|
|
1085
1112
|
updateWidget(ctx);
|
|
@@ -1089,6 +1116,39 @@ export default function (pi: ExtensionAPI) {
|
|
|
1089
1116
|
};
|
|
1090
1117
|
}
|
|
1091
1118
|
|
|
1119
|
+
case "grant_fix_rounds": {
|
|
1120
|
+
const reject = (error: string) => ({
|
|
1121
|
+
content: [{ type: "text" as const, text: `Error: ${error}` }],
|
|
1122
|
+
details: { action: "grant_fix_rounds", phases: { ...phases }, error } as PhaseTrackerDetails,
|
|
1123
|
+
});
|
|
1124
|
+
if (params.rounds === undefined) return reject("grant_fix_rounds requires rounds: a positive integer");
|
|
1125
|
+
const reason = params.reason?.trim() ?? "";
|
|
1126
|
+
if (reason.length === 0) return reject("grant_fix_rounds requires reason: the human's approval, quoted");
|
|
1127
|
+
const closure = resolveClosureReview(loadGauntletSettings(ctx.cwd).gauntlet);
|
|
1128
|
+
const capLive =
|
|
1129
|
+
!isSubagentChild &&
|
|
1130
|
+
gauntletEntered &&
|
|
1131
|
+
phases.verify.status === "in_progress" &&
|
|
1132
|
+
conformanceDispatched &&
|
|
1133
|
+
closure.enforce &&
|
|
1134
|
+
fixRounds >= closure.maxFixRounds;
|
|
1135
|
+
if (!capLive) {
|
|
1136
|
+
return reject(
|
|
1137
|
+
`grant_fix_rounds: no fix-round cap block is active (${fixRounds} used, cap ${closure.maxFixRounds}); nothing to overrule`,
|
|
1138
|
+
);
|
|
1139
|
+
}
|
|
1140
|
+
if (fixRoundCredits > 0) {
|
|
1141
|
+
return reject(`grant_fix_rounds: ${fixRoundCredits} granted round(s) still unused; spend them before granting more`);
|
|
1142
|
+
}
|
|
1143
|
+
fixRoundCredits = params.rounds;
|
|
1144
|
+
return {
|
|
1145
|
+
content: [
|
|
1146
|
+
{ type: "text", text: `Granted ${params.rounds} extra fix round(s) - reason: "${reason}"\n${formatStatus(phases)}` },
|
|
1147
|
+
],
|
|
1148
|
+
details: { action: "grant_fix_rounds", rounds: params.rounds, reason, phases: { ...phases } } as PhaseTrackerDetails,
|
|
1149
|
+
};
|
|
1150
|
+
}
|
|
1151
|
+
|
|
1092
1152
|
default:
|
|
1093
1153
|
return {
|
|
1094
1154
|
content: [{ type: "text", text: `Unknown action: ${params.action}` }],
|
|
@@ -1160,6 +1220,14 @@ export default function (pi: ExtensionAPI) {
|
|
|
1160
1220
|
}
|
|
1161
1221
|
case "reset":
|
|
1162
1222
|
return new Text(theme.fg("success", "✓ ") + theme.fg("muted", "Phase tracker reset"), 0, 0);
|
|
1223
|
+
case "grant_fix_rounds":
|
|
1224
|
+
return new Text(
|
|
1225
|
+
theme.fg("success", "+ ") +
|
|
1226
|
+
theme.fg("muted", `${details.rounds} extra fix round(s) granted`) +
|
|
1227
|
+
theme.fg("dim", ` (${details.reason})`),
|
|
1228
|
+
0,
|
|
1229
|
+
0,
|
|
1230
|
+
);
|
|
1163
1231
|
default:
|
|
1164
1232
|
return new Text(theme.fg("dim", "Done"), 0, 0);
|
|
1165
1233
|
}
|
|
@@ -22,16 +22,16 @@ const sources = {
|
|
|
22
22
|
}
|
|
23
23
|
`,
|
|
24
24
|
"@sinclair/typebox": `
|
|
25
|
-
const schema = (...args) => ({ args });
|
|
25
|
+
const schema = (kind) => (...args) => ({ kind, args });
|
|
26
26
|
export const Type = {
|
|
27
|
-
Object: schema,
|
|
28
|
-
Optional: schema,
|
|
29
|
-
String: schema,
|
|
30
|
-
Boolean: schema,
|
|
31
|
-
Union: schema,
|
|
32
|
-
Null: schema,
|
|
33
|
-
Array: schema,
|
|
34
|
-
Integer: schema,
|
|
27
|
+
Object: schema("Object"),
|
|
28
|
+
Optional: schema("Optional"),
|
|
29
|
+
String: schema("String"),
|
|
30
|
+
Boolean: schema("Boolean"),
|
|
31
|
+
Union: schema("Union"),
|
|
32
|
+
Null: schema("Null"),
|
|
33
|
+
Array: schema("Array"),
|
|
34
|
+
Integer: schema("Integer"),
|
|
35
35
|
};
|
|
36
36
|
`,
|
|
37
37
|
};
|
package/package.json
CHANGED
|
@@ -126,8 +126,10 @@ partition from an earlier round.
|
|
|
126
126
|
|
|
127
127
|
Mirrors `subagent-driven-development` Parallel-Wave Mode and reuses its
|
|
128
128
|
`plan_tracker` progress surface. Runs entirely inside the gate — it invokes
|
|
129
|
-
**no** `phase_tracker` calls (`
|
|
130
|
-
|
|
129
|
+
**no** phase-transition `phase_tracker` calls (`start`/`complete`/`skip`/`reset`
|
|
130
|
+
- `phase_tracker({ phase: "implement" })` errors while verify is `in_progress`;
|
|
131
|
+
the only `phase_tracker` call inside the loop is the human-approval
|
|
132
|
+
`grant_fix_rounds` in step 6) and does **not** enter SDD's phase machinery.
|
|
131
133
|
Only the fan-out/integrate/review shape and `plan_tracker` are reused. Every execution dispatch is foreground with top-level `async: false`, including retries and prose-described dispatches; an unexpected async handle is a configuration failure: stop and report, never poll or relaunch. `forceTopLevelAsync` is incompatible; see [pi-cohort dispatch configuration](https://github.com/jjuraszek/pi-cohort/blob/main/doc/configuration.md).
|
|
132
134
|
|
|
133
135
|
**Precondition — worktree required.** The loop needs a worktree HEAD to branch
|
|
@@ -178,8 +180,11 @@ Per round:
|
|
|
178
180
|
default `3`, floors negatives at `0`, coerces non-integers to `3`) reached
|
|
179
181
|
with an open `fix` gap or repair item → **escalate to the human** with the per-gap
|
|
180
182
|
round-by-round verdict trail. Escalation is the sole non-completing
|
|
181
|
-
terminal state — no silent re-loop, no auto-ship.
|
|
182
|
-
|
|
183
|
+
terminal state — no silent re-loop, no auto-ship. If the human explicitly
|
|
184
|
+
approves N more rounds, record `phase_tracker({ action: "grant_fix_rounds",
|
|
185
|
+
rounds: N, reason: "<their words>" })` and re-enter step 2; without that
|
|
186
|
+
approval, escalation stays terminal. Inside a brainstorming-entered flow
|
|
187
|
+
the phase tracker enforces both rules at
|
|
183
188
|
tool-call time: a lone `agent: "implementer"` dispatch is blocked, and the
|
|
184
189
|
wave after the cap is blocked (`closureReview.enforce: false` disables).
|
|
185
190
|
|
|
@@ -187,7 +192,7 @@ Per round:
|
|
|
187
192
|
|
|
188
193
|
a. Run the full plan-header `Verification` set once (ad-hoc: the project's canonical test command). After R0 `CONFORMS` with no round run, the pre-R0 full run counts.
|
|
189
194
|
b. Dispatch `code-reviewer` directly (foreground, `async: false`, `SCOPED_TEST_COMMANDS: none`) over `git diff <r0-head>..HEAD`; never via `/skill:requesting-code-review`. An empty diff is nothing to review - no dispatch.
|
|
190
|
-
c. Repair items = every failing command from a + every Critical/Moderate finding from b (`Behaviour-change: yes` included; the re-audit is its origin check). None → write the closure block; done. Any at the cap → escalate per step 6 with the test/CR trail. Any under the cap → re-enter step 2 as a one-task `tasks` wave: one `implementer` whose task is every repair item verbatim (ownership boundary = the files in `git diff <r0-head>..HEAD`, the CR findings' `touched-files`, and the files each failing command's output names, `SCOPED_TEST_COMMANDS: none`, no `Gn` tracker task, no gap selection), integrate as one `conformance fix CR`, re-audit, then Convergence again. That wave counts against `maxFixRounds`. Later Convergence CRs keep the same `<r0-head>..HEAD` range.
|
|
195
|
+
c. Repair items = every failing command from a + every Critical/Moderate finding from b (`Behaviour-change: yes` included; the re-audit is its origin check). None → write the closure block; done. Any at the cap → escalate per step 6 with the test/CR trail (an explicit human approval re-enters via `grant_fix_rounds`, as step 6 describes). Any under the cap → re-enter step 2 as a one-task `tasks` wave: one `implementer` whose task is every repair item verbatim (ownership boundary = the files in `git diff <r0-head>..HEAD`, the CR findings' `touched-files`, and the files each failing command's output names, `SCOPED_TEST_COMMANDS: none`, no `Gn` tracker task, no gap selection), integrate as one `conformance fix CR`, re-audit, then Convergence again. That wave counts against `maxFixRounds`. Later Convergence CRs keep the same `<r0-head>..HEAD` range.
|
|
191
196
|
|
|
192
197
|
`conformance fix CR` is not a gap fix: it is absent from the `auto-applied fix commits` index and has no `revert conformance fix Gn` action at the finish gate.
|
|
193
198
|
|