pi-gauntlet 5.3.7 → 5.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +4 -0
- package/extensions/lib/gauntlet-settings.test.ts +37 -0
- package/extensions/lib/gauntlet-settings.ts +24 -0
- package/extensions/phase-tracker.test.ts +14 -1
- package/extensions/phase-tracker.ts +10 -2
- package/package.json +1 -1
- package/skills/subagent-driven-development/SKILL.md +6 -3
- package/skills/subagent-driven-development/stop-note.md +48 -0
- package/skills/verification-before-completion/reference/settings-precedence.md +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,9 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## v5.4.0 - 2026-09-10
|
|
4
|
+
|
|
5
|
+
- `subagent-driven-development`: a stalled review fix loop runs one escalated fix round (`implementer`, `context: fresh`, model from new `piGauntlet.escalationLoop.implModel`, default main-loop model + thinking) before stopping; the stop is a one-screen problem note (`stop-note.md`) with concrete fix options, replacing the trajectory-log escalation report. `gauntlet_setting` gains the `escalationLoop` key. (#29)
|
|
6
|
+
|
|
3
7
|
## v5.3.7 - 2026-09-10
|
|
4
8
|
|
|
5
9
|
- `brainstorming`: the standalone "does this replace a prior spec" question is gone - the gather scout names candidate predecessor specs and round 1 states them; the design is presented in two rounds (architecture/components/data flow, then errors/testing/docs) with one approval each. (#27)
|
|
@@ -4,6 +4,8 @@ import {
|
|
|
4
4
|
mergeGauntlet,
|
|
5
5
|
resolveSpecCouncil,
|
|
6
6
|
resolveClosureReview,
|
|
7
|
+
resolveEscalationLoop,
|
|
8
|
+
mainLoopModel,
|
|
7
9
|
resolveFlowGuards,
|
|
8
10
|
resolveVerifyBeforeShip,
|
|
9
11
|
settingsErrorWarning,
|
|
@@ -24,6 +26,41 @@ test("mergeGauntlet: undefined layers -> {}", () => {
|
|
|
24
26
|
assert.deepEqual(mergeGauntlet(undefined, undefined), {});
|
|
25
27
|
});
|
|
26
28
|
|
|
29
|
+
test("mergeGauntlet: repo escalationLoop replaces preset whole-object", () => {
|
|
30
|
+
const preset = { escalationLoop: { implModel: "p/preset:high" }, closureReview: { model: "m" } };
|
|
31
|
+
const repo = { escalationLoop: {} };
|
|
32
|
+
const merged = mergeGauntlet(preset, repo);
|
|
33
|
+
assert.deepEqual(merged.escalationLoop, {});
|
|
34
|
+
assert.deepEqual(merged.closureReview, { model: "m" });
|
|
35
|
+
});
|
|
36
|
+
|
|
37
|
+
test("escalationLoop: absent/empty/null/non-string -> mainLoop", () => {
|
|
38
|
+
const main = "p/main:medium";
|
|
39
|
+
assert.equal(resolveEscalationLoop({}, main).implModel, main);
|
|
40
|
+
assert.equal(resolveEscalationLoop({ escalationLoop: {} }, main).implModel, main);
|
|
41
|
+
assert.equal(resolveEscalationLoop({ escalationLoop: { implModel: "" } }, main).implModel, main);
|
|
42
|
+
assert.equal(resolveEscalationLoop({ escalationLoop: { implModel: null } }, main).implModel, main);
|
|
43
|
+
assert.equal(resolveEscalationLoop({ escalationLoop: { implModel: 42 } }, main).implModel, main);
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
test("escalationLoop: non-empty string wins, trimmed; undefined mainLoop passes through", () => {
|
|
47
|
+
assert.equal(
|
|
48
|
+
resolveEscalationLoop({ escalationLoop: { implModel: " p/x:high " } }, "p/main:medium").implModel,
|
|
49
|
+
"p/x:high",
|
|
50
|
+
);
|
|
51
|
+
assert.equal(resolveEscalationLoop({}, undefined).implModel, undefined);
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
test("mainLoopModel: always suffixed; unset -> off, max -> xhigh, recognised pass through", () => {
|
|
55
|
+
const m = { provider: "p", id: "id" };
|
|
56
|
+
assert.equal(mainLoopModel(m, undefined), "p/id:off");
|
|
57
|
+
assert.equal(mainLoopModel(m, "off"), "p/id:off");
|
|
58
|
+
assert.equal(mainLoopModel(m, "medium"), "p/id:medium");
|
|
59
|
+
assert.equal(mainLoopModel(m, "xhigh"), "p/id:xhigh");
|
|
60
|
+
assert.equal(mainLoopModel(m, "max"), "p/id:xhigh");
|
|
61
|
+
assert.equal(mainLoopModel(undefined, "medium"), undefined);
|
|
62
|
+
});
|
|
63
|
+
|
|
27
64
|
test("specCouncil: non-empty string array -> council", () => {
|
|
28
65
|
const r = resolveSpecCouncil({ specCouncil: { members: ["p/m1", " p/m2 "], chair: "p/c" } });
|
|
29
66
|
assert.equal(r.verdict, "council");
|
|
@@ -8,6 +8,7 @@ export interface PiGauntlet {
|
|
|
8
8
|
closureReview?: { enforce?: unknown; model?: unknown; maxFixRounds?: unknown };
|
|
9
9
|
flowGuards?: { enforce?: unknown; specDirs?: unknown };
|
|
10
10
|
verifyBeforeShip?: { testCommands?: unknown; warningReference?: unknown };
|
|
11
|
+
escalationLoop?: { implModel?: unknown };
|
|
11
12
|
}
|
|
12
13
|
|
|
13
14
|
// Whole-object second-level merge: each piGauntlet key present in the repo layer
|
|
@@ -87,6 +88,29 @@ export function resolveClosureReview(g: PiGauntlet): ClosureReviewResolved {
|
|
|
87
88
|
return { model, enforce, maxFixRounds };
|
|
88
89
|
}
|
|
89
90
|
|
|
91
|
+
export interface EscalationLoopResolved {
|
|
92
|
+
implModel: string | undefined;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
export function resolveEscalationLoop(g: PiGauntlet, mainLoop: string | undefined): EscalationLoopResolved {
|
|
96
|
+
const raw = g.escalationLoop?.implModel;
|
|
97
|
+
return { implModel: nonEmptyString(raw) ? raw.trim() : mainLoop };
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
const THINKING_SUFFIXES = new Set(["off", "minimal", "low", "medium", "high", "xhigh"]);
|
|
101
|
+
|
|
102
|
+
// Always emit a suffix so pi-cohort's applyThinkingSuffix never falls back to the
|
|
103
|
+
// implementer's configured thinking; pi's "max" has no pi-cohort equivalent -> xhigh.
|
|
104
|
+
export function mainLoopModel(
|
|
105
|
+
model: { provider: string; id: string } | undefined,
|
|
106
|
+
thinkingLevel: string | undefined,
|
|
107
|
+
): string | undefined {
|
|
108
|
+
if (!model) return undefined;
|
|
109
|
+
const level =
|
|
110
|
+
thinkingLevel === "max" ? "xhigh" : THINKING_SUFFIXES.has(thinkingLevel ?? "") ? thinkingLevel : "off";
|
|
111
|
+
return `${model.provider}/${model.id}:${level}`;
|
|
112
|
+
}
|
|
113
|
+
|
|
90
114
|
export interface FlowGuardsResolved {
|
|
91
115
|
enforce: boolean;
|
|
92
116
|
specDirs: string[];
|
|
@@ -51,7 +51,7 @@ const resumedBranch = (rest: Partial<Record<Phase, Status>>) => [
|
|
|
51
51
|
phaseResult("complete", phases({ brainstorm: "skipped", ...rest })),
|
|
52
52
|
];
|
|
53
53
|
|
|
54
|
-
function harness(options: { cwd?: string; branch?: unknown[]; idle?: boolean; beforeSettled?: (setIdle: (idle: boolean) => void) => void; sendThrows?: boolean } = {}) {
|
|
54
|
+
function harness(options: { cwd?: string; branch?: unknown[]; idle?: boolean; beforeSettled?: (setIdle: (idle: boolean) => void) => void; sendThrows?: boolean; model?: { provider: string; id: string }; thinkingLevel?: string } = {}) {
|
|
55
55
|
const handlers = new Map<string, ((event: unknown, ctx: unknown) => unknown)[]>();
|
|
56
56
|
const tools: { name: string; execute: (...args: any[]) => unknown }[] = [];
|
|
57
57
|
const sent: { message: any; options: any }[] = [];
|
|
@@ -62,6 +62,8 @@ function harness(options: { cwd?: string; branch?: unknown[]; idle?: boolean; be
|
|
|
62
62
|
hasUI: false,
|
|
63
63
|
isIdle: () => idle,
|
|
64
64
|
sessionManager: { getBranch: () => branch },
|
|
65
|
+
model: options.model,
|
|
66
|
+
thinkingLevel: options.thinkingLevel,
|
|
65
67
|
};
|
|
66
68
|
const pi = {
|
|
67
69
|
on(event: string, handler: (event: unknown, context: unknown) => unknown) {
|
|
@@ -1141,3 +1143,14 @@ test("replay via session_switch: pass then fail clears the stamp, rebuilt from r
|
|
|
1141
1143
|
};
|
|
1142
1144
|
assert.match(res.details.error ?? "", /plan_check/);
|
|
1143
1145
|
});
|
|
1146
|
+
|
|
1147
|
+
test("gauntlet_setting escalationLoop: setting absent -> ctx-derived main-loop model; setting wins when set", async () => {
|
|
1148
|
+
const h = harness({ cwd: tempCwd({ piGauntlet: {} }), model: { provider: "p", id: "main" }, thinkingLevel: "medium" });
|
|
1149
|
+
const tool = h.tools.find((t) => t.name === "gauntlet_setting")!;
|
|
1150
|
+
const absent = (await tool.execute("g1", { key: "escalationLoop" }, undefined, undefined, h.ctx)) as { details: { key: string; implModel?: string; errors: string[] } };
|
|
1151
|
+
assert.deepEqual(absent.details, { key: "escalationLoop", implModel: "p/main:medium", errors: [] });
|
|
1152
|
+
|
|
1153
|
+
const set = harness({ cwd: tempCwd({ piGauntlet: { escalationLoop: { implModel: "p/strong:high" } } }), model: { provider: "p", id: "main" }, thinkingLevel: "medium" });
|
|
1154
|
+
const res = (await set.tools.find((t) => t.name === "gauntlet_setting")!.execute("g2", { key: "escalationLoop" }, undefined, undefined, set.ctx)) as { details: { implModel?: string } };
|
|
1155
|
+
assert.equal(res.details.implModel, "p/strong:high");
|
|
1156
|
+
});
|
|
@@ -17,7 +17,9 @@ import type { ExtensionAPI, ExtensionContext, Theme } from "@earendil-works/pi-c
|
|
|
17
17
|
import { Text } from "@earendil-works/pi-tui";
|
|
18
18
|
import { type Static, Type } from "@sinclair/typebox";
|
|
19
19
|
import {
|
|
20
|
+
mainLoopModel,
|
|
20
21
|
resolveClosureReview,
|
|
22
|
+
resolveEscalationLoop,
|
|
21
23
|
resolveFlowGuards,
|
|
22
24
|
resolveSpecCouncil,
|
|
23
25
|
settingsErrorWarning,
|
|
@@ -666,7 +668,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
666
668
|
});
|
|
667
669
|
|
|
668
670
|
const GauntletSettingParams = Type.Object({
|
|
669
|
-
key: StringEnum(["specCouncil", "closureReview"] as const, {
|
|
671
|
+
key: StringEnum(["specCouncil", "closureReview", "escalationLoop"] as const, {
|
|
670
672
|
description: "Which gauntlet setting to resolve (merged repo-over-preset).",
|
|
671
673
|
}),
|
|
672
674
|
});
|
|
@@ -681,7 +683,13 @@ export default function (pi: ExtensionAPI) {
|
|
|
681
683
|
const payload =
|
|
682
684
|
params.key === "specCouncil"
|
|
683
685
|
? { key: "specCouncil" as const, ...resolveSpecCouncil(gauntlet), errors }
|
|
684
|
-
:
|
|
686
|
+
: params.key === "closureReview"
|
|
687
|
+
? { key: "closureReview" as const, ...resolveClosureReview(gauntlet), errors }
|
|
688
|
+
: {
|
|
689
|
+
key: "escalationLoop" as const,
|
|
690
|
+
...resolveEscalationLoop(gauntlet, mainLoopModel(ctx.model, ctx.thinkingLevel)),
|
|
691
|
+
errors,
|
|
692
|
+
};
|
|
685
693
|
return {
|
|
686
694
|
content: [{ type: "text", text: "```json\n" + JSON.stringify(payload, null, 2) + "\n```" }],
|
|
687
695
|
details: payload,
|
package/package.json
CHANGED
|
@@ -30,7 +30,7 @@ You are the **orchestrator**. You read the plan, dispatch, review the review, de
|
|
|
30
30
|
**Do not pause to check in with the user between tasks.** The plan is already approved. Pause only when:
|
|
31
31
|
|
|
32
32
|
- A subagent returns `NEEDS_CONTEXT` or `BLOCKED` (see [Implementer Status](#implementer-status))
|
|
33
|
-
-
|
|
33
|
+
- An escalated round fails (stop note per [Fix-Loop Rounds](#fix-loop-rounds))
|
|
34
34
|
- A ⚠️ workflow warning fires
|
|
35
35
|
|
|
36
36
|
Reaching the end of the plan is not a pause: continue through verification and invoke `/skill:finishing-a-development-branch` as defined in [After All Tasks](#after-all-tasks-complete).
|
|
@@ -88,12 +88,12 @@ Every fix re-dispatch (implementer) and code-review re-review carries the consum
|
|
|
88
88
|
|
|
89
89
|
Every dispatched fix is verified by a re-review before escalation or task progression - the loop only ever exits on a clean review or an escalation.
|
|
90
90
|
|
|
91
|
-
**
|
|
91
|
+
**Escalate** = one escalated round; stop only on failure. Call `gauntlet_setting({ key: "escalationLoop" })` (unavailable -> stop and report). `implModel` undefined -> stop note. Otherwise re-dispatch that fix - same payload and isolation knobs (`cwd`, `worktree`, `SCOPED_TEST_COMMANDS`, status protocol, prior patch, spec anchors) plus the prior report verbatim - overriding only `model: <implModel>`, `context: "fresh"`, `async: false`; one implementer, no fan-out. Run the normal fix-round review gate (SR then CR on `Behaviour-change: yes`, else the triggering reviewer with the re-review marker). All clean -> proceed; in wave mode the escalated patch supersedes the prior one at integrate. Any review with issues, a non-`DONE` status, or a dispatch error -> stop note per `stop-note.md`; no second dispatch. Once per loop; independent of the convergence exception. No `plan_tracker` write during escalation - the task stays `in_progress` until the human decides.
|
|
92
92
|
|
|
93
93
|
**Worked examples:**
|
|
94
94
|
|
|
95
95
|
- Review-2 verdict `TRAJECTORY: STAGNANT (repeat of: unchecked error path in parser)` -> escalate now, before fix 2 - earlier than the ordinary budget.
|
|
96
|
-
- Review-3 verdict `TRAJECTORY: CONVERGING (3 -> 1, max severity Moderate)` -> dispatch fix 3; if review 4 still finds issues, escalate
|
|
96
|
+
- Review-3 verdict `TRAJECTORY: CONVERGING (3 -> 1, max severity Moderate)` -> dispatch fix 3; if review 4 still finds issues, escalate.
|
|
97
97
|
- Review-3 verdict `TRAJECTORY: CONVERGING (3 -> 2, max severity Critical)` or `TRAJECTORY: DIVERGING` or no `TRAJECTORY:` line -> escalate.
|
|
98
98
|
|
|
99
99
|
## Implementer Status
|
|
@@ -139,6 +139,9 @@ When in doubt, default. Don't downgrade reviewers — false negatives are expens
|
|
|
139
139
|
// implementer
|
|
140
140
|
subagent({ agent: "implementer", async: false, task: "<task text + context + SCOPED_TEST_COMMANDS + status protocol>" })
|
|
141
141
|
|
|
142
|
+
// escalated fix round (Fix-Loop Rounds): same fix payload, model from gauntlet_setting({ key: "escalationLoop" }).implModel
|
|
143
|
+
subagent({ agent: "implementer", model: "<implModel>", context: "fresh", async: false, task: "<the just-dispatched fix payload + prior review report verbatim>" })
|
|
144
|
+
|
|
142
145
|
// spec compliance
|
|
143
146
|
subagent({ agent: "spec-reviewer", async: false, task: "<task text + patch diff + absolute spec path + task's Spec: anchors>" })
|
|
144
147
|
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# Stop note (subagent-driven-development companion)
|
|
2
|
+
|
|
3
|
+
Emitted when no escalation model resolves, the escalated dispatch errors, or its one escalated round fails (see `SKILL.md` "Fix-Loop Rounds"); an unavailable `gauntlet_setting` tool is a configuration error - stop and report, no stop note. Inline in the reply; the turn ends; phase stays `implement`, the task stays `in_progress`; no further tasks start. Wave mode: one note per stalled task, after the current batch returns.
|
|
4
|
+
|
|
5
|
+
## Template
|
|
6
|
+
|
|
7
|
+
```
|
|
8
|
+
Stopped on task <n> (<title>)[; escalated round on <implModel> did not resolve it].
|
|
9
|
+
|
|
10
|
+
Problem: <one sentence: what is wrong and why the fixes could not resolve it>
|
|
11
|
+
<file:line> - <quoted finding from the final review>
|
|
12
|
+
<failing test/command + 1-3 line output snippet, when present>
|
|
13
|
+
|
|
14
|
+
Fix options:
|
|
15
|
+
a) <concrete change>
|
|
16
|
+
b) <concrete change - amending spec section X / plan task n is a normal option>
|
|
17
|
+
c) <optional third>
|
|
18
|
+
|
|
19
|
+
Pick one, or give another fix.
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
## Rules
|
|
23
|
+
|
|
24
|
+
- Bracketed header clause only when an escalated round actually ran; omit it when no model was resolvable or the dispatch errored.
|
|
25
|
+
- Residual issues from the final review report only; quote, do not summarise history.
|
|
26
|
+
- When no escalated round ran, the Problem is the reason: "no escalation model resolvable", or the implementer's non-DONE status text / the dispatch error, quoted; skip the file:line and test lines.
|
|
27
|
+
- Options are actionable edits. Spec/plan amendment is first-class - stalls are usually a slightly contradictory spec, not a capability gap.
|
|
28
|
+
- Never offer "skip the task". If the task is genuinely droppable, say so and name the plan tasks that depend on it.
|
|
29
|
+
- No trajectory verdicts, round history, review counts, or paths to spec/plan/review reports.
|
|
30
|
+
- Plain words, ASCII, no headings. One screen.
|
|
31
|
+
|
|
32
|
+
## Example
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
Stopped on task 4 (retry policy for the outbound client); escalated round on <provider>/<model>:high did not resolve it.
|
|
36
|
+
|
|
37
|
+
Problem: `RetryPolicy.next()` returns 0 ms for the first retry, but the client treats 0 as "no
|
|
38
|
+
retry", so the first failure is never retried.
|
|
39
|
+
src/net/retry.ts:41 - `return attempt * this.baseMs;`
|
|
40
|
+
npm test -- retry > "retries once after a transient failure":
|
|
41
|
+
expected 1 call after failure, got 0
|
|
42
|
+
|
|
43
|
+
Fix options:
|
|
44
|
+
a) start the backoff at `baseMs` (`(attempt + 1) * this.baseMs`)
|
|
45
|
+
b) amend plan task 4 so the client retries on any non-negative delay, and keep the policy as is
|
|
46
|
+
|
|
47
|
+
Pick one, or give another fix.
|
|
48
|
+
```
|
|
@@ -15,7 +15,7 @@ If the repo file defines a `piGauntlet.<key>` at all, that definition **replaces
|
|
|
15
15
|
the preset's for that key entirely - the two are never merged leaf-by-leaf. If the
|
|
16
16
|
repo file does not define the key, the preset's value is used unchanged. This is
|
|
17
17
|
exactly pi's own `deepMergeSettings` behaviour: it spreads the second-level keys
|
|
18
|
-
(`specCouncil`, `closureReview`, `flowGuards`, `verifyBeforeShip`) wholesale, and
|
|
18
|
+
(`specCouncil`, `closureReview`, `flowGuards`, `verifyBeforeShip`, `escalationLoop`) wholesale, and
|
|
19
19
|
does not recurse into their leaves.
|
|
20
20
|
|
|
21
21
|
**Caveat - partial definitions drop siblings.** Because the replace is
|