pi-gauntlet 5.3.7 → 5.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -36,6 +36,12 @@ interface FileEntry {
36
36
  path: string;
37
37
  }
38
38
 
39
+ interface TestsBullet {
40
+ line: number;
41
+ text: string;
42
+ value: string;
43
+ }
44
+
39
45
  interface Task {
40
46
  number: number;
41
47
  line: number;
@@ -48,6 +54,12 @@ interface Task {
48
54
  specAnchorLine: number | undefined;
49
55
  anchors: Anchor[];
50
56
  anchorParseError: boolean;
57
+ testsHeadingLine: number | undefined;
58
+ testsMisplacedLines: number[];
59
+ tests: TestsBullet[];
60
+ testsVia: TestsBullet[];
61
+ testsNone: TestsBullet[];
62
+ testsMalformed: TestsBullet[];
51
63
  }
52
64
 
53
65
  interface Wave {
@@ -235,6 +247,50 @@ function parsePlan(planText: string): ParsedPlan {
235
247
  }
236
248
  }
237
249
 
250
+ let testsHeadingLine: number | undefined;
251
+ const testsMisplacedLines: number[] = [];
252
+ const tests: TestsBullet[] = [];
253
+ const testsVia: TestsBullet[] = [];
254
+ const testsNone: TestsBullet[] = [];
255
+ const testsMalformed: TestsBullet[] = [];
256
+
257
+ let headingIdx = -1;
258
+ if (filesLine !== undefined) {
259
+ let lastEntry = filesLine - 1;
260
+ let p = filesLine;
261
+ while (p <= bodyEndIdx) {
262
+ const l = lines[p];
263
+ if (l.trim() === "") { p++; continue; }
264
+ if (/^- \w+: /.test(l)) { lastEntry = p; p++; continue; }
265
+ break;
266
+ }
267
+ let q = lastEntry + 1;
268
+ while (q <= bodyEndIdx && lines[q].trim() === "") q++;
269
+ if (q <= bodyEndIdx && !mask[q] && lines[q] === "**Tests:**") {
270
+ headingIdx = q;
271
+ testsHeadingLine = q + 1;
272
+ }
273
+ }
274
+ for (let k = i; k <= bodyEndIdx; k++) {
275
+ if (mask[k] || k === headingIdx) continue;
276
+ if (/^\*\*Tests:\*\*/.test(lines[k])) testsMisplacedLines.push(k + 1);
277
+ }
278
+ if (headingIdx !== -1) {
279
+ for (let p = headingIdx + 1; p <= bodyEndIdx; p++) {
280
+ const l = lines[p];
281
+ if (l.trim() === "") continue;
282
+ if (mask[p] || !l.startsWith("- ") || l.startsWith("- [ ]")) break;
283
+ const cmd = /^- `([^`]+)`$/.exec(l);
284
+ const via = /^- via: (\S.*)$/.exec(l);
285
+ const none = /^- none: (\S.*)$/.exec(l);
286
+ const bullet = { line: p + 1, text: l, value: (cmd ?? via ?? none)?.[1] ?? "" };
287
+ if (cmd) tests.push(bullet);
288
+ else if (via) testsVia.push(bullet);
289
+ else if (none) testsNone.push(bullet);
290
+ else testsMalformed.push(bullet);
291
+ }
292
+ }
293
+
238
294
  tasks.push({
239
295
  number: Number(m[1]),
240
296
  line: i + 1,
@@ -247,6 +303,12 @@ function parsePlan(planText: string): ParsedPlan {
247
303
  specAnchorLine,
248
304
  anchors,
249
305
  anchorParseError,
306
+ testsHeadingLine,
307
+ testsMisplacedLines,
308
+ tests,
309
+ testsVia,
310
+ testsNone,
311
+ testsMalformed,
250
312
  });
251
313
  }
252
314
 
@@ -343,6 +405,99 @@ function isGlob(p: string): boolean {
343
405
  return /[*?{[\]]/.test(p);
344
406
  }
345
407
 
408
+ function norm(s: string): string {
409
+ return s.replaceAll("`", "").replace(/\s+/g, " ").trim();
410
+ }
411
+
412
+ function backtickSpans(s: string): string[] {
413
+ const out: string[] = [];
414
+ const re = /`([^`]+)`/g;
415
+ let m: RegExpExecArray | null;
416
+ while ((m = re.exec(s))) out.push(m[1]);
417
+ return out;
418
+ }
419
+
420
+ function commandSegments(cmd: string): string[] {
421
+ return norm(cmd).split(/\s*(?:&&|\|\||;|\|)\s*/).map((s) => s.trim()).filter(Boolean);
422
+ }
423
+
424
+ function headerSegments(parsed: ParsedPlan): string[] {
425
+ const value = parsed.header.verificationText ?? "";
426
+ const spans = backtickSpans(value);
427
+ const parts = spans.length > 0 ? spans : [value];
428
+ return parts.flatMap((p) => norm(p).split(/\s*(?:&&|\|\||;|,)\s*/)).map((s) => s.trim()).filter(Boolean);
429
+ }
430
+
431
+ const RUN_RE = /^\s*(- \[ \] )?Run:\s*(.*)$/;
432
+
433
+ function runPayloadSegments(line: string): string[] | undefined {
434
+ const m = RUN_RE.exec(line);
435
+ if (!m) return undefined;
436
+ const spans = backtickSpans(m[2]);
437
+ return (spans.length > 0 ? spans : [m[2]]).flatMap(commandSegments);
438
+ }
439
+
440
+ const UNSUPPORTED_SHELL = ["cd ", "sh -c", "bash -c", "eval ", "$("];
441
+
442
+ function isBroadening(token: string): boolean {
443
+ return /[*?[]/.test(token) || token.endsWith("/");
444
+ }
445
+
446
+ function checkTestsBlock(parsed: ParsedPlan, fs: FsPort): PlanCheckFinding[] {
447
+ const findings: PlanCheckFinding[] = [];
448
+ const push = (task: Task, line: number, text: string, reason: string) =>
449
+ findings.push({ check: "tests-block", line, text, reason: `Task ${task.number}: ${reason}` });
450
+ const createPaths = new Set(
451
+ parsed.tasks.flatMap((t) => t.files.filter((f) => f.kind === "create").map((f) => stripLineSuffix(f.path))),
452
+ );
453
+ const header = headerSegments(parsed);
454
+
455
+ for (const task of parsed.tasks) {
456
+ const testEntries = task.files.filter((f) => f.kind === "test");
457
+ const testPaths = testEntries.map((f) => stripLineSuffix(f.path));
458
+
459
+ for (const f of testEntries) {
460
+ const p = stripLineSuffix(f.path);
461
+ if (!createPaths.has(p) && !fs.exists(p)) push(task, f.line, f.text, `unknown \`Test:\` path "${p}" (neither exists nor is a Create: path of any task)`);
462
+ }
463
+
464
+ if (task.testsMisplacedLines.length > 0) {
465
+ for (const ln of task.testsMisplacedLines) push(task, ln, parsed.lines[ln - 1], "misplaced block: `**Tests:**` must be the bare line directly after the Files: entries");
466
+ } else if (task.testsHeadingLine === undefined) {
467
+ push(task, task.line, task.text, "block missing: no `**Tests:**` directly after the Files: entries");
468
+ }
469
+ if (task.testsHeadingLine === undefined) continue;
470
+
471
+ const headingText = parsed.lines[task.testsHeadingLine - 1];
472
+ if (task.tests.length === 0 && task.testsNone.length === 0) push(task, task.testsHeadingLine, headingText, "block empty: no command bullet and no `none:`");
473
+ for (const b of task.testsMalformed) push(task, b.line, b.text, "malformed bullet: expected `- \\`command\\``, `- via: <seam>`, or `- none: <category>`");
474
+ if (task.testsNone.length > 1 || (task.testsNone.length > 0 && (task.tests.length > 0 || task.testsVia.length > 0))) {
475
+ push(task, task.testsNone[0].line, task.testsNone[0].text, "contradictory block: `none:` with a command or `via:`, or more than one `none:`");
476
+ }
477
+ if (task.testsNone.length > 0) {
478
+ for (const f of testEntries) push(task, f.line, f.text, "unused `Test:` path: task declares `none:`");
479
+ }
480
+
481
+ const anchors = testPaths.filter((p) => !isBroadening(p));
482
+ for (const b of task.tests) {
483
+ if (UNSUPPORTED_SHELL.some((s) => b.value.includes(s))) {
484
+ push(task, b.line, b.text, "unsupported shell: `cd `, `sh -c`, `bash -c`, `eval `, `$(` are not allowed");
485
+ continue;
486
+ }
487
+ for (const seg of commandSegments(b.value)) {
488
+ const tokens = seg.split(" ");
489
+ const isAnchor = (t: string) => anchors.some((p) => t === p || t.startsWith(p + "::") || t.startsWith(p + "#") || t.startsWith(p + ":"));
490
+ if (!tokens.some(isAnchor)) push(task, b.line, b.text, `segment not anchored: "${seg}" names no Test: path of this task`);
491
+ for (const t of tokens) {
492
+ if (!isAnchor(t) && isBroadening(t)) push(task, b.line, b.text, `broadening selector "${t}" in "${seg}"`);
493
+ }
494
+ if (header.includes(seg)) push(task, b.line, b.text, `full-suite command in task: "${seg}" equals a header **Verification:** segment`);
495
+ }
496
+ }
497
+ }
498
+ return findings;
499
+ }
500
+
346
501
  function taskBodyText(task: Task, lines: string[]): string {
347
502
  return lines.slice(task.bodyStartLine - 1, task.bodyEndLine).join("\n");
348
503
  }
@@ -687,10 +842,14 @@ function checkPlaceholderScan(parsed: ParsedPlan, requiredLiterals: Map<number,
687
842
  return findings;
688
843
  }
689
844
 
690
- function fileEntries(task: Task): { path: string; kind: "literal" | "glob" }[] {
845
+ function fileEntries(task: Task): { path: string; kind: "literal" | "glob"; role: "test" | "write" }[] {
691
846
  return task.files.map((f) => {
692
847
  const p = stripLineSuffix(f.path);
693
- return { path: p, kind: (isGlob(p) ? "glob" : "literal") as "literal" | "glob" };
848
+ return {
849
+ path: p,
850
+ kind: (isGlob(p) ? "glob" : "literal") as "literal" | "glob",
851
+ role: f.kind === "test" ? "test" : "write",
852
+ };
694
853
  });
695
854
  }
696
855
 
@@ -719,6 +878,7 @@ function checkWaveFileDisjointness(parsed: ParsedPlan, fs: FsPort): PlanCheckFin
719
878
  const b = fileEntries(tasks[j]);
720
879
  for (const ea of a) {
721
880
  for (const eb of b) {
881
+ if (ea.role === "test" && eb.role === "test") continue;
722
882
  let overlap = false;
723
883
  let errFinding: PlanCheckFinding | undefined;
724
884
  if (ea.kind === "literal" && eb.kind === "literal") {
@@ -798,6 +958,15 @@ function checkHeaderEntrypoint(parsed: ParsedPlan): PlanCheckFinding[] {
798
958
  const entrypoint = parsed.header.verificationText.trim();
799
959
  if (!entrypoint) return findings;
800
960
 
961
+ const header = headerSegments(parsed);
962
+ const executable = new Set<number>();
963
+ for (const task of parsed.tasks) {
964
+ for (const bullet of [...task.tests, ...task.testsVia, ...task.testsNone, ...task.testsMalformed]) {
965
+ executable.add(bullet.line);
966
+ }
967
+ }
968
+ const mask = fenceMask(parsed.lines);
969
+
801
970
  const waveBoundaryRe = /^##\s/;
802
971
  const inScope = new Set<number>();
803
972
  for (const wave of parsed.waves) {
@@ -814,6 +983,21 @@ function checkHeaderEntrypoint(parsed: ParsedPlan): PlanCheckFinding[] {
814
983
 
815
984
  for (const ln of [...inScope].sort((a, b) => a - b)) {
816
985
  const line = parsed.lines[ln - 1];
986
+ if (executable.has(ln)) continue;
987
+ const segments = runPayloadSegments(line);
988
+ if (segments) {
989
+ if (mask[ln - 1]) continue;
990
+ const hit = segments.find((segment) => header.includes(segment));
991
+ if (hit !== undefined) {
992
+ findings.push({
993
+ check: "header-entrypoint",
994
+ line: ln,
995
+ text: line,
996
+ reason: `Run: segment "${hit}" equals a header **Verification:** segment (full suite belongs to the verify phase)`,
997
+ });
998
+ }
999
+ continue;
1000
+ }
817
1001
  if (line.includes(entrypoint)) {
818
1002
  findings.push({
819
1003
  check: "header-entrypoint",
@@ -870,6 +1054,7 @@ export function checkPlan(planText: string, specText: string, fs: FsPort): PlanC
870
1054
  findings.push(...checkQuoteIntegrity(parsed, specLines));
871
1055
  findings.push(...checkAnchorResolution(parsed, specLines));
872
1056
  findings.push(...checkPathsExist(parsed, fs));
1057
+ findings.push(...checkTestsBlock(parsed, fs));
873
1058
  const requiredLiterals = computeRequiredLiteralsPerTask(parsed, specLines);
874
1059
  findings.push(...checkPlaceholderScan(parsed, requiredLiterals));
875
1060
  findings.push(...checkWaveFileDisjointness(parsed, fs));
@@ -51,7 +51,7 @@ const resumedBranch = (rest: Partial<Record<Phase, Status>>) => [
51
51
  phaseResult("complete", phases({ brainstorm: "skipped", ...rest })),
52
52
  ];
53
53
 
54
- function harness(options: { cwd?: string; branch?: unknown[]; idle?: boolean; beforeSettled?: (setIdle: (idle: boolean) => void) => void; sendThrows?: boolean } = {}) {
54
+ function harness(options: { cwd?: string; branch?: unknown[]; idle?: boolean; beforeSettled?: (setIdle: (idle: boolean) => void) => void; sendThrows?: boolean; model?: { provider: string; id: string }; thinkingLevel?: string } = {}) {
55
55
  const handlers = new Map<string, ((event: unknown, ctx: unknown) => unknown)[]>();
56
56
  const tools: { name: string; execute: (...args: any[]) => unknown }[] = [];
57
57
  const sent: { message: any; options: any }[] = [];
@@ -62,6 +62,8 @@ function harness(options: { cwd?: string; branch?: unknown[]; idle?: boolean; be
62
62
  hasUI: false,
63
63
  isIdle: () => idle,
64
64
  sessionManager: { getBranch: () => branch },
65
+ model: options.model,
66
+ thinkingLevel: options.thinkingLevel,
65
67
  };
66
68
  const pi = {
67
69
  on(event: string, handler: (event: unknown, context: unknown) => unknown) {
@@ -705,7 +707,12 @@ const FIXTURE_PLAN = `# Fixture Plan
705
707
 
706
708
  **Files:**
707
709
  - Create: lib/task1.ts
710
+ - Create: lib/task1.test.ts
708
711
  - Modify: file-a.ts
712
+ - Test: lib/task1.test.ts
713
+
714
+ **Tests:**
715
+ - \`node --test lib/task1.test.ts\`
709
716
 
710
717
  This task implements helperFn() for parsing.
711
718
 
@@ -715,7 +722,12 @@ This task implements helperFn() for parsing.
715
722
 
716
723
  **Files:**
717
724
  - Create: lib/task2.ts
725
+ - Create: lib/task2.test.ts
718
726
  - Modify: file-b.ts
727
+ - Test: lib/task2.test.ts
728
+
729
+ **Tests:**
730
+ - \`node --test lib/task2.test.ts\`
719
731
 
720
732
  This task handles naming details.
721
733
 
@@ -1141,3 +1153,14 @@ test("replay via session_switch: pass then fail clears the stamp, rebuilt from r
1141
1153
  };
1142
1154
  assert.match(res.details.error ?? "", /plan_check/);
1143
1155
  });
1156
+
1157
+ test("gauntlet_setting escalationLoop: setting absent -> ctx-derived main-loop model; setting wins when set", async () => {
1158
+ const h = harness({ cwd: tempCwd({ piGauntlet: {} }), model: { provider: "p", id: "main" }, thinkingLevel: "medium" });
1159
+ const tool = h.tools.find((t) => t.name === "gauntlet_setting")!;
1160
+ const absent = (await tool.execute("g1", { key: "escalationLoop" }, undefined, undefined, h.ctx)) as { details: { key: string; implModel?: string; errors: string[] } };
1161
+ assert.deepEqual(absent.details, { key: "escalationLoop", implModel: "p/main:medium", errors: [] });
1162
+
1163
+ const set = harness({ cwd: tempCwd({ piGauntlet: { escalationLoop: { implModel: "p/strong:high" } } }), model: { provider: "p", id: "main" }, thinkingLevel: "medium" });
1164
+ const res = (await set.tools.find((t) => t.name === "gauntlet_setting")!.execute("g2", { key: "escalationLoop" }, undefined, undefined, set.ctx)) as { details: { implModel?: string } };
1165
+ assert.equal(res.details.implModel, "p/strong:high");
1166
+ });
@@ -17,7 +17,9 @@ import type { ExtensionAPI, ExtensionContext, Theme } from "@earendil-works/pi-c
17
17
  import { Text } from "@earendil-works/pi-tui";
18
18
  import { type Static, Type } from "@sinclair/typebox";
19
19
  import {
20
+ mainLoopModel,
20
21
  resolveClosureReview,
22
+ resolveEscalationLoop,
21
23
  resolveFlowGuards,
22
24
  resolveSpecCouncil,
23
25
  settingsErrorWarning,
@@ -666,7 +668,7 @@ export default function (pi: ExtensionAPI) {
666
668
  });
667
669
 
668
670
  const GauntletSettingParams = Type.Object({
669
- key: StringEnum(["specCouncil", "closureReview"] as const, {
671
+ key: StringEnum(["specCouncil", "closureReview", "escalationLoop"] as const, {
670
672
  description: "Which gauntlet setting to resolve (merged repo-over-preset).",
671
673
  }),
672
674
  });
@@ -681,7 +683,13 @@ export default function (pi: ExtensionAPI) {
681
683
  const payload =
682
684
  params.key === "specCouncil"
683
685
  ? { key: "specCouncil" as const, ...resolveSpecCouncil(gauntlet), errors }
684
- : { key: "closureReview" as const, ...resolveClosureReview(gauntlet), errors };
686
+ : params.key === "closureReview"
687
+ ? { key: "closureReview" as const, ...resolveClosureReview(gauntlet), errors }
688
+ : {
689
+ key: "escalationLoop" as const,
690
+ ...resolveEscalationLoop(gauntlet, mainLoopModel(ctx.model, ctx.thinkingLevel)),
691
+ errors,
692
+ };
685
693
  return {
686
694
  content: [{ type: "text", text: "```json\n" + JSON.stringify(payload, null, 2) + "\n```" }],
687
695
  details: payload,
@@ -697,7 +705,7 @@ export default function (pi: ExtensionAPI) {
697
705
  name: "plan_check",
698
706
  label: "Plan Check",
699
707
  description:
700
- "Deterministically verify an implementation plan against its spec (9 mechanical checks); " +
708
+ "Deterministically verify an implementation plan against its spec (mechanical checks); " +
701
709
  "a pass stamps the plan for implement-start.",
702
710
  parameters: PlanCheckParams,
703
711
  async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-gauntlet",
3
- "version": "5.3.7",
3
+ "version": "5.5.0",
4
4
  "description": "Opinionated, gated workflow skills, subagent personas, and runtime extensions for the pi coding agent.",
5
5
  "author": "Jacek Juraszek",
6
6
  "type": "module",
@@ -30,7 +30,7 @@ You are the **orchestrator**. You read the plan, dispatch, review the review, de
30
30
  **Do not pause to check in with the user between tasks.** The plan is already approved. Pause only when:
31
31
 
32
32
  - A subagent returns `NEEDS_CONTEXT` or `BLOCKED` (see [Implementer Status](#implementer-status))
33
- - A fix loop escalates per [Fix-Loop Rounds](#fix-loop-rounds) (stagnation, or budget exhausted without convergence)
33
+ - An escalated round fails (stop note per [Fix-Loop Rounds](#fix-loop-rounds))
34
34
  - A ⚠️ workflow warning fires
35
35
 
36
36
  Reaching the end of the plan is not a pause: continue through verification and invoke `/skill:finishing-a-development-branch` as defined in [After All Tasks](#after-all-tasks-complete).
@@ -55,11 +55,13 @@ Before the first task, enter the implement phase: `phase_tracker({ action: "star
55
55
 
56
56
  For each task in `plan_tracker`:
57
57
 
58
- 1. **Start, then dispatch implementer.** Mark the task's existing `plan_tracker` index `in_progress` before dispatch. Pass the full task text + scene-setting context + the task's plan-declared test commands as `SCOPED_TEST_COMMANDS` (or `none`). Don't make the subagent re-read the plan.
58
+ **`SCOPED_TEST_COMMANDS`** = the task's `**Tests:**` command bullets, backticks stripped, one per line; `- none:` -> `none`; a wave = the union of its tasks' commands. `via:` bullets and `- Test:` paths are contract, not commands: they ride with the task text, never in `SCOPED_TEST_COMMANDS`.
59
+
60
+ 1. **Start, then dispatch implementer.** Mark the task's existing `plan_tracker` index `in_progress` before dispatch. Pass the full task text + scene-setting context + its `SCOPED_TEST_COMMANDS` (definition above) and its `**Tests:**` block verbatim. Don't make the subagent re-read the plan.
59
61
  2. **Handle implementer status** (see below).
60
- 3. **Dispatch spec reviewer.** Pass the task text, the patch diff, the absolute spec path, and the task's `**Spec:**` anchors — SR reads the anchored ranges from the spec file itself; never inline spec excerpts. The spec wins every dispute; the authority hierarchy and finding labels live in `./spec-reviewer-prompt.md`. Anchor-less tasks (no `**Spec:**` line): task text alone is the contract. Verify the change satisfies the anchored spec — nothing missing, nothing extra.
62
+ 3. **Dispatch spec reviewer.** Pass the task text, the patch diff, the absolute spec path, and the task's `**Spec:**` anchors, plus the task's `**Tests:**` block and `- Test:` paths as contract (never as commands - SR never executes) — SR reads the anchored ranges from the spec file itself; never inline spec excerpts. The spec wins every dispute; the authority hierarchy and finding labels live in `./spec-reviewer-prompt.md`. Anchor-less tasks (no `**Spec:**` line): task text alone is the contract. Verify the change satisfies the anchored spec — nothing missing, nothing extra.
61
63
  4. If spec reviewer finds gaps → re-dispatch implementer to fix → re-review. Loop until ✅, within [Fix-Loop Rounds](#fix-loop-rounds).
62
- 5. **Dispatch code-quality reviewer.** Only after spec is ✅. Skip for doc-only tasks (every file in the task's `Files:` block documentation-only) — SR-only, same exemption as doc-only waves. Pass `SCOPED_TEST_COMMANDS` = the task's plan-declared commands.
64
+ 5. **Dispatch code-quality reviewer.** Only after spec is ✅. Skip for doc-only tasks (every file in the task's `Files:` block documentation-only) — SR-only, same exemption as doc-only waves. Pass the task's `SCOPED_TEST_COMMANDS`.
63
65
  6. If quality reviewer finds issues → re-dispatch implementer → re-review. Loop until ✅, within [Fix-Loop Rounds](#fix-loop-rounds).
64
66
  7. After its existing reviews accept the work (and its required commit point), mark that same task index `complete` in `plan_tracker`. A subagent exit or green tests alone are not acceptance.
65
67
 
@@ -73,11 +75,11 @@ One rule governs both review loops - spec-compliance and code-quality - in seque
73
75
 
74
76
  **Re-review dispatch rule:** every re-review task includes the complete prior review report verbatim under the marker `## Previous review report (re-review trigger)`, plus the trajectory block from the reviewer's prompt template. The marker's presence is what obligates the reviewer to emit the `TRAJECTORY:` line. You never select, summarize, or diff findings yourself - pattern-match the sentinel line only.
75
77
 
76
- **Fix fan-out.** When the triggering review's `Parallel-safe:` line certifies a `disjoint` group of ≥ 2 findings, dispatch that fix round per `dispatching-parallel-agents` "Fix fan-out"; the fan-out counts as **one** fix against this budget, its scoped test gate is the consuming task/wave's plan-declared commands, and one re-review of the integrated delta follows.
78
+ **Fix fan-out.** When the triggering review's `Parallel-safe:` line certifies a `disjoint` group of ≥ 2 findings, dispatch that fix round per `dispatching-parallel-agents` "Fix fan-out"; the fan-out counts as **one** fix against this budget, its scoped test gate is the consuming task/wave's `SCOPED_TEST_COMMANDS`, and one re-review of the integrated delta follows.
77
79
 
78
- **Behaviour-change reroute.** When the triggering CR report carries `Behaviour-change: yes`, the fix round's re-review is SR first, then CR. The SR dispatch reviews the fix diff against the task's spec anchors as a first review (no `## Previous review report` marker, so no `TRAJECTORY:` line; SR carries no test commands), but its ordinal continues the task's SR-loop count - a task whose SR loop ended at review 2 gets review 3 here, and an issue-bearing rerouted SR after review 4 escalates. Issues follow the normal sequence: fix, then SR re-review pasting this SR's report. CR round numbering is unchanged. `Behaviour-change: no` re-reviews with CR only. A missing or malformed `Behaviour-change:` line is re-asked once like `Parallel-safe:` (see `dispatching-parallel-agents` "Structural probe"); still missing -> route through SR, never default to `no`. In wave mode the SR re-review targets the task(s) whose files the fix touched.
80
+ **Behaviour-change reroute.** When the triggering CR report carries `Behaviour-change: yes`, the fix round's re-review is SR first, then CR. The SR dispatch reviews the fix diff against the task's spec anchors as a first review (no `## Previous review report` marker, so no `TRAJECTORY:` line; SR carries no test commands but does carry the task's `**Tests:**` block and `- Test:` paths), but its ordinal continues the task's SR-loop count - a task whose SR loop ended at review 2 gets review 3 here, and an issue-bearing rerouted SR after review 4 escalates. Issues follow the normal sequence: fix, then SR re-review pasting this SR's report. CR round numbering is unchanged. `Behaviour-change: no` re-reviews with CR only. A missing or malformed `Behaviour-change:` line is re-asked once like `Parallel-safe:` (see `dispatching-parallel-agents` "Structural probe"); still missing -> route through SR, never default to `no`. In wave mode the SR re-review targets the task(s) whose files the fix touched.
79
81
 
80
- Every fix re-dispatch (implementer) and code-review re-review carries the consuming task/wave's `SCOPED_TEST_COMMANDS`; spec-reviewer re-reviews carry none - SR never executes.
82
+ Every fix re-dispatch (implementer) and code-review re-review carries the consuming task/wave's `SCOPED_TEST_COMMANDS`; spec-reviewer re-reviews carry the `**Tests:**` block and `- Test:` paths, no commands - SR never executes.
81
83
 
82
84
  **The sequence.** Each review that finds issues is a decision point: read the `TRAJECTORY:` line before dispatching anything (review 1 has no line - on issues, dispatch fix 1). Any clean review ends the loop.
83
85
 
@@ -88,12 +90,12 @@ Every fix re-dispatch (implementer) and code-review re-review carries the consum
88
90
 
89
91
  Every dispatched fix is verified by a re-review before escalation or task progression - the loop only ever exits on a clean review or an escalation.
90
92
 
91
- **Escalation report:** report reviews run per loop and the final `TRAJECTORY` verdict - report `TRAJECTORY: MISSING` if the line was absent, or quote the raw line if malformed. If the exception ran, say so explicitly: `fix 3 was the convergence exception (CONVERGING, no Critical)`.
93
+ **Escalate** = one escalated round; stop only on failure. Call `gauntlet_setting({ key: "escalationLoop" })` (unavailable -> stop and report). `implModel` undefined -> stop note. Otherwise re-dispatch that fix - same payload and isolation knobs (`cwd`, `worktree`, `SCOPED_TEST_COMMANDS`, status protocol, prior patch, spec anchors) plus the prior report verbatim - overriding only `model: <implModel>`, `context: "fresh"`, `async: false`; one implementer, no fan-out. Run the normal fix-round review gate (SR then CR on `Behaviour-change: yes`, else the triggering reviewer with the re-review marker). All clean -> proceed; in wave mode the escalated patch supersedes the prior one at integrate. Any review with issues, a non-`DONE` status, or a dispatch error -> stop note per `stop-note.md`; no second dispatch. Once per loop; independent of the convergence exception. No `plan_tracker` write during escalation - the task stays `in_progress` until the human decides.
92
94
 
93
95
  **Worked examples:**
94
96
 
95
97
  - Review-2 verdict `TRAJECTORY: STAGNANT (repeat of: unchecked error path in parser)` -> escalate now, before fix 2 - earlier than the ordinary budget.
96
- - Review-3 verdict `TRAJECTORY: CONVERGING (3 -> 1, max severity Moderate)` -> dispatch fix 3; if review 4 still finds issues, escalate with the exception named in the report.
98
+ - Review-3 verdict `TRAJECTORY: CONVERGING (3 -> 1, max severity Moderate)` -> dispatch fix 3; if review 4 still finds issues, escalate.
97
99
  - Review-3 verdict `TRAJECTORY: CONVERGING (3 -> 2, max severity Critical)` or `TRAJECTORY: DIVERGING` or no `TRAJECTORY:` line -> escalate.
98
100
 
99
101
  ## Implementer Status
@@ -139,8 +141,11 @@ When in doubt, default. Don't downgrade reviewers — false negatives are expens
139
141
  // implementer
140
142
  subagent({ agent: "implementer", async: false, task: "<task text + context + SCOPED_TEST_COMMANDS + status protocol>" })
141
143
 
144
+ // escalated fix round (Fix-Loop Rounds): same fix payload, model from gauntlet_setting({ key: "escalationLoop" }).implModel
145
+ subagent({ agent: "implementer", model: "<implModel>", context: "fresh", async: false, task: "<the just-dispatched fix payload + prior review report verbatim>" })
146
+
142
147
  // spec compliance
143
- subagent({ agent: "spec-reviewer", async: false, task: "<task text + patch diff + absolute spec path + task's Spec: anchors>" })
148
+ subagent({ agent: "spec-reviewer", async: false, task: "<task text + patch diff + absolute spec path + task's Spec: anchors + Tests: block + Test: paths>" })
144
149
 
145
150
  // code quality
146
151
  subagent({ agent: "code-reviewer", async: false, task: "<diff range + SCOPED_TEST_COMMANDS (task commands; wave: union; whole-diff: none) + ask: production-ready?>" })
@@ -174,12 +179,12 @@ Auto-selected at handoff by `writing-plans` (any wave with ≥2 tasks) when the
174
179
 
175
180
  **Per-wave loop:**
176
181
 
177
- 1. **Independence check.** Parse the wave's tasks' `Files:` blocks; assert pairwise-disjoint (mechanical). Runtime-resource disjointness (DB/schema, port, fixture, external service, shared temp path) is not machine-checkable here — trust the plan's wave grouping, which `writing-plans`' D5 contract guarantees. Either kind of overlap → the wave is mis-grouped; run those tasks as sequential single-task waves and note it.
182
+ 1. **Independence check.** Parse the wave's tasks' `Files:` blocks; assert pairwise-disjoint (mechanical, mirroring `wave-file-disjointness`: `Test`/`Test` on one path is not overlap, `Test` vs another task's `Create`/`Modify` is, `Modify`/`Modify` is). Runtime-resource disjointness (DB/schema, port, fixture, external service, shared temp path) is not machine-checkable here — trust the plan's wave grouping, which `writing-plans`' D5 contract guarantees. Either kind of overlap → the wave is mis-grouped; run those tasks as sequential single-task waves and note it.
178
183
  2. **Start, then fan out.** Mark every wave index `in_progress` before one parallel foreground dispatch (shape below): `implementer` per task, `context: "fresh"`, `worktree: true`. Each returns a status + a patch.
179
- 3. **Status + spec review per task.** Parse each `DONE`/`BLOCKED`/etc. (see [Implementer Status](#implementer-status)) **first**. Then **dispatch a `spec-reviewer` per accepted patch** (`DONE`, or a `DONE_WITH_CONCERNS` you proceeded with) in one parallel fan-out — `context: "fresh"`, `cwd: <this worktree>`, **no `worktree` flag** (read-only) — each passed its task text, the returned **patch diff**, the absolute spec path, and the task's `**Spec:**` anchors — SR reads the anchored ranges itself (never inline excerpts; authority hierarchy in `./spec-reviewer-prompt.md`; anchor-less tasks are task-text-only). Review starts from the diff (its hunks carry `file:line`) and reads each touched file in full, and test execution is never the reviewer's job - in either mode (persona rule; the wave test gate in step 5 runs the wave's declared test commands). Inline verdicts are fine at normal wave sizes; large waves use `output:` + `outputMode: "file-only"` to keep verdicts out of your context. **Re-dispatch by cause:** `BLOCKED`/`NEEDS_CONTEXT` per the [Implementer Status](#implementer-status) matrix; a **spec gap** re-dispatches the implementer (fresh, `worktree: true`) carrying the prior patch + the reviewer's findings, the new patch superseding the old at step 4. Loop until accepted + spec ✅, within [Fix-Loop Rounds](#fix-loop-rounds), same as sequential.
184
+ 3. **Status + spec review per task.** Parse each `DONE`/`BLOCKED`/etc. (see [Implementer Status](#implementer-status)) **first**. Then **dispatch a `spec-reviewer` per accepted patch** (`DONE`, or a `DONE_WITH_CONCERNS` you proceeded with) in one parallel fan-out — `context: "fresh"`, `cwd: <this worktree>`, **no `worktree` flag** (read-only) — each passed its task text, the returned **patch diff**, the absolute spec path, and the task's `**Spec:**` anchors, plus its `**Tests:**` block and `- Test:` paths as contract — SR reads the anchored ranges itself (never inline excerpts; authority hierarchy in `./spec-reviewer-prompt.md`; anchor-less tasks are task-text-only). Review starts from the diff (its hunks carry `file:line`) and reads each touched file in full, and test execution is never the reviewer's job - in either mode (persona rule; the wave test gate in step 5 runs the wave's `SCOPED_TEST_COMMANDS`). Inline verdicts are fine at normal wave sizes; large waves use `output:` + `outputMode: "file-only"` to keep verdicts out of your context. **Re-dispatch by cause:** `BLOCKED`/`NEEDS_CONTEXT` per the [Implementer Status](#implementer-status) matrix; a **spec gap** re-dispatches the implementer (fresh, `worktree: true`) carrying the prior patch + the reviewer's findings, the new patch superseding the old at step 4. Loop until accepted + spec ✅, within [Fix-Loop Rounds](#fix-loop-rounds), same as sequential.
180
185
  4. **Integrate.** `git apply` each task's patch sequentially onto HEAD. Apply fails = textual conflict → drop that task, finish the rest, re-run the dropped task sequentially on the updated HEAD.
181
- 5. **Test gate.** Run the union of the wave's tasks' declared test commands on the integrated tree — the full verification set is the verify phase's job, run once. Failure = semantic conflict or bug → re-run the offending task sequentially, else fix per [When a Subagent Fails](#when-a-subagent-fails).
182
- 6. **Quality review.** CR binds to the wave: exactly one **initial** code-review dispatch per code-touching wave, over the integrated wave diff - never per task within a wave, never batched across waves. Subsequent dispatches within the wave are re-reviews triggered only by findings, per Fix-Loop Rounds. Pass `SCOPED_TEST_COMMANDS` = the union of the wave's tasks' declared commands. Code-quality review on the integrated wave diff; loop fixes to ✅ within [Fix-Loop Rounds](#fix-loop-rounds), same as sequential. Skip for doc-only waves (SR-only per the commit precondition below).
186
+ 5. **Test gate.** Run the wave's `SCOPED_TEST_COMMANDS` (union of its tasks' `Tests:` commands) on the integrated tree — the full verification set is the verify phase's job, run once. Failure = semantic conflict or bug → re-run the offending task sequentially, else fix per [When a Subagent Fails](#when-a-subagent-fails).
187
+ 6. **Quality review.** CR binds to the wave: exactly one **initial** code-review dispatch per code-touching wave, over the integrated wave diff - never per task within a wave, never batched across waves. Subsequent dispatches within the wave are re-reviews triggered only by findings, per Fix-Loop Rounds. Pass the wave's `SCOPED_TEST_COMMANDS`. Code-quality review on the integrated wave diff; loop fixes to ✅ within [Fix-Loop Rounds](#fix-loop-rounds), same as sequential. Skip for doc-only waves (SR-only per the commit precondition below).
183
188
  7. **Commit and complete the wave.** After the gate passes and the wave commits, mark all of its existing indices `complete`. Leaves a clean tree; the next wave's children branch from this commit and so see the integrated work.
184
189
 
185
190
  **Two-stage review is preserved:** spec review per task (pre-integration, dispatched `spec-reviewer` — not inline), quality review per wave (post-integration). A wave commit requires one spec-review verdict per accepted task, plus one code-review verdict on the integrated diff for waves that touch code. A doc-only wave (every task's `Files:` block documentation-only, per `writing-plans`' Wave Grouping) is SR-only — the CR gate does not apply.
@@ -240,11 +245,11 @@ For the fan-out + worktree + patch-integration + conflict mechanics, see `dispat
240
245
  - Writing code yourself instead of dispatching.
241
246
  - Pausing between tasks for anything other than `NEEDS_CONTEXT`, `BLOCKED`, a fix-loop escalation, a workflow warning, or a spec amendment.
242
247
  - Dispatching parallel implementers on overlapping files, on a shared mutable runtime resource, or without `worktree: true`.
243
- - Making a subagent read the plan, inlining spec excerpts to the spec reviewer, or dispatching without a `SCOPED_TEST_COMMANDS` value.
248
+ - Making a subagent read the plan, inlining spec excerpts to the spec reviewer, or dispatching with a `SCOPED_TEST_COMMANDS` value missing or not copied from the plan's `Tests:` bullets.
244
249
  - Dispatching `code-reviewer` before every in-scope spec-review verdict is ✅, or per task inside a wave.
245
250
  - Moving to the next task with either review still showing issues, or skipping the `Implementer Status` parse.
246
251
  - Dispatching fix 3 without a reviewer-emitted `CONVERGING` verdict, or continuing past `STAGNANT` instead of escalating.
247
- - Running the full verification entrypoint during the implement phase.
252
+ - Running the full verification entrypoint, or any test command outside `SCOPED_TEST_COMMANDS`, during the implement phase.
248
253
  - Dispatching a repair before reopening the plan-task indices that own its files, whole-diff CR before parent verification passes, or conformance before the CR result is accepted.
249
254
  - Polling, joining, or relaunching an unexpectedly asynchronous dispatch, or starting on main without explicit user consent.
250
255
 
@@ -14,7 +14,7 @@ Dispatch a subagent with the code-reviewer template:
14
14
  PLAN_OR_REQUIREMENTS: Task N from [plan-file]
15
15
  BASE_SHA: [commit before task]
16
16
  HEAD_SHA: [current commit]
17
- SCOPED_TEST_COMMANDS: [the consuming task's plan-declared commands; wave reviews: the union of the wave's tasks' declared commands; `none` for the whole-diff verify-phase review]
17
+ SCOPED_TEST_COMMANDS: [the consuming task's Tests: commands; wave reviews: the union of the wave's tasks' Tests: commands; `none` for the whole-diff verify-phase review]
18
18
  ```
19
19
 
20
20
  **In addition to standard code quality concerns, the reviewer should check:**
@@ -38,12 +38,17 @@ Dispatch a subagent with this prompt:
38
38
 
39
39
  Work from: [directory]
40
40
 
41
- SCOPED_TEST_COMMANDS: [the task's plan-declared test commands, verbatim | none]
41
+ SCOPED_TEST_COMMANDS: [the task's Tests: bullets, backticks stripped, verbatim | none]
42
42
 
43
43
  Run ONLY these commands for verification. Never run a repo-wide suite,
44
44
  linter, or type-checker. If the value is `none`, run nothing and say so
45
45
  in your report.
46
46
 
47
+ TEST_CONTRACT: [the task's **Tests:** block verbatim (commands, via:, none:) and its Files: Create:/Test: paths]
48
+
49
+ Tests call the via: seam directly - not a wrapper, not the internals behind it.
50
+ Create every Create: path at exactly that path.
51
+
47
52
  **While you work:** If you encounter something unexpected or unclear, **ask questions**.
48
53
  It's always OK to pause and clarify. Don't guess or make assumptions.
49
54
 
@@ -109,6 +114,9 @@ Dispatch a subagent with this prompt:
109
114
  - **Status:** `DONE` | `DONE_WITH_CONCERNS` | `BLOCKED` | `NEEDS_CONTEXT`
110
115
  - What you implemented (or what you attempted, if blocked)
111
116
  - What you tested and test results
117
+ - Test contract: one line per SCOPED_TEST_COMMANDS command - `met` (exit 0; quote the last output line) or `unmet <reason>`;
118
+ one line per via: - `met <test file:line calling it>` or `unmet <reason>`; `none` when the block is `none:`.
119
+ Any `unmet` -> `DONE_WITH_CONCERNS`.
112
120
  - Files changed
113
121
  - Self-review findings (if any)
114
122
  - Any issues or concerns
@@ -24,6 +24,7 @@ Dispatch a subagent with this prompt:
24
24
 
25
25
  Spec: [absolute spec path]
26
26
  Anchors: [the task's **Spec:** anchor list, e.g. § "Design" L34-L37 — or "omitted: anchor-less mechanical task"]
27
+ Task contract: [the task's **Tests:** block verbatim + its `Files:` paths]
27
28
 
28
29
  The spec is the sole authority — human-approved; the task never wins a dispute. Read the anchored ranges from the spec file yourself. Requirements in scope are ONLY the cited anchor ranges; do not extract, review, or flag the rest of the spec file.
29
30
 
@@ -35,6 +36,7 @@ Dispatch a subagent with this prompt:
35
36
  - **Finding grammar:** divergence findings use the existing F1..Fn finding grammar - a finding kind by prose label, not a new schema; the `Parallel-safe:` and `TRAJECTORY:` grammars are untouched.
36
37
  - **Condition match:** for every anchored clause that fixes a value, threshold, comparison, or trigger ("only when", "unless", "if", a literal), the clause row carries two indented sub-lines, before `touched-files:` where present: `spec-condition: <clause fragment quoted from the spec>` and `code-condition: <what the code checks, file:line>`. If the two differ, the clause is `PARTIAL` at most - regardless of passing tests. A plausible condition is not the specified condition.
37
38
  - **Plan/task code snippets:** implementation guidance, not review authority; a diff matching a snippet never proves compliance. For anchor-less tasks the task text's prose requirements remain your contract.
39
+ - **Task contract (Tests:/via:/Files:):** The task's `**Tests:**` block, `via:`, and its `Files:` paths supplement the anchored spec where it is silent; the anchored spec wins a conflict - report the divergence once, against the plan, never against code corrected to the spec. Findings: a `Create:` path absent from the diff or created elsewhere; a test that does not call the `via:` entry point; a `Tests:` block the diff contradicts. Existing files need no diff touch. You never run the commands.
38
40
 
39
41
  ## CRITICAL: Do Not Trust the Report
40
42
 
@@ -64,7 +66,7 @@ Dispatch a subagent with this prompt:
64
66
 
65
67
  ## Your Job
66
68
 
67
- <!-- clause decomposition / snippet non-authority / whole-file reads: keep in lockstep with agents/spec-reviewer.md — change them together or not at all -->
69
+ <!-- clause decomposition / snippet non-authority / whole-file reads / task-contract supplement: keep in lockstep with agents/spec-reviewer.md — change them together or not at all -->
68
70
 
69
71
  Decompose the binding contract - the anchored spec lines, or the task text when anchors are omitted - into atomic clauses, covering every requirement, acceptance criterion, and explicit non-goal. Each independently checkable statement is one clause; a sentence listing three requirements yields three clauses.
70
72
 
@@ -0,0 +1,48 @@
1
+ # Stop note (subagent-driven-development companion)
2
+
3
+ Emitted when no escalation model resolves, the escalated dispatch errors, or its one escalated round fails (see `SKILL.md` "Fix-Loop Rounds"); an unavailable `gauntlet_setting` tool is a configuration error - stop and report, no stop note. Inline in the reply; the turn ends; phase stays `implement`, the task stays `in_progress`; no further tasks start. Wave mode: one note per stalled task, after the current batch returns.
4
+
5
+ ## Template
6
+
7
+ ```
8
+ Stopped on task <n> (<title>)[; escalated round on <implModel> did not resolve it].
9
+
10
+ Problem: <one sentence: what is wrong and why the fixes could not resolve it>
11
+ <file:line> - <quoted finding from the final review>
12
+ <failing test/command + 1-3 line output snippet, when present>
13
+
14
+ Fix options:
15
+ a) <concrete change>
16
+ b) <concrete change - amending spec section X / plan task n is a normal option>
17
+ c) <optional third>
18
+
19
+ Pick one, or give another fix.
20
+ ```
21
+
22
+ ## Rules
23
+
24
+ - Bracketed header clause only when an escalated round actually ran; omit it when no model was resolvable or the dispatch errored.
25
+ - Residual issues from the final review report only; quote, do not summarise history.
26
+ - When no escalated round ran, the Problem is the reason: "no escalation model resolvable", or the implementer's non-DONE status text / the dispatch error, quoted; skip the file:line and test lines.
27
+ - Options are actionable edits. Spec/plan amendment is first-class - stalls are usually a slightly contradictory spec, not a capability gap.
28
+ - Never offer "skip the task". If the task is genuinely droppable, say so and name the plan tasks that depend on it.
29
+ - No trajectory verdicts, round history, review counts, or paths to spec/plan/review reports.
30
+ - Plain words, ASCII, no headings. One screen.
31
+
32
+ ## Example
33
+
34
+ ```
35
+ Stopped on task 4 (retry policy for the outbound client); escalated round on <provider>/<model>:high did not resolve it.
36
+
37
+ Problem: `RetryPolicy.next()` returns 0 ms for the first retry, but the client treats 0 as "no
38
+ retry", so the first failure is never retried.
39
+ src/net/retry.ts:41 - `return attempt * this.baseMs;`
40
+ npm test -- retry > "retries once after a transient failure":
41
+ expected 1 call after failure, got 0
42
+
43
+ Fix options:
44
+ a) start the backoff at `baseMs` (`(attempt + 1) * this.baseMs`)
45
+ b) amend plan task 4 so the client retries on any non-negative delay, and keep the policy as is
46
+
47
+ Pick one, or give another fix.
48
+ ```
@@ -117,11 +117,11 @@ Don't add features, refactor other code, or "improve" beyond what the test requi
117
117
 
118
118
  Run the test. Confirm:
119
119
  - New test passes
120
- - The task's scoped commands pass (full-suite verification belongs to the verify phase)
120
+ - The task's `Tests:` commands pass (full-suite verification belongs to the verify phase)
121
121
  - Output is pristine (no errors, no warnings)
122
122
 
123
123
  **Test fails?** Fix code, not test.
124
- **Scoped commands fail?** Fix now — don't move on with broken tests.
124
+ **`Tests:` commands fail?** Fix now — don't move on with broken tests.
125
125
 
126
126
  ### REFACTOR — Clean Up
127
127
 
@@ -176,7 +176,7 @@ Before marking work complete:
176
176
  - [ ] Watched each test fail before implementing
177
177
  - [ ] Each test failed for expected reason (feature missing, not typo)
178
178
  - [ ] Wrote minimal code to pass each test
179
- - [ ] The task's scoped commands pass (full suite belongs to the verify phase)
179
+ - [ ] The task's `Tests:` commands pass (full suite belongs to the verify phase)
180
180
  - [ ] Output pristine (no errors, warnings)
181
181
  - [ ] Tests use real code (mocks only if unavoidable)
182
182
  - [ ] Edge cases and errors covered
@@ -15,7 +15,7 @@ If the repo file defines a `piGauntlet.<key>` at all, that definition **replaces
15
15
  the preset's for that key entirely - the two are never merged leaf-by-leaf. If the
16
16
  repo file does not define the key, the preset's value is used unchanged. This is
17
17
  exactly pi's own `deepMergeSettings` behaviour: it spreads the second-level keys
18
- (`specCouncil`, `closureReview`, `flowGuards`, `verifyBeforeShip`) wholesale, and
18
+ (`specCouncil`, `closureReview`, `flowGuards`, `verifyBeforeShip`, `escalationLoop`) wholesale, and
19
19
  does not recurse into their leaves.
20
20
 
21
21
  **Caveat - partial definitions drop siblings.** Because the replace is