devflow-kit 3.1.0 → 3.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (138) hide show
  1. package/CHANGELOG.md +52 -0
  2. package/README.md +2 -2
  3. package/dist/cli/agents-view/render.js +69 -15
  4. package/dist/cli/agents-view/state.js +40 -14
  5. package/dist/cli/commands/agents.js +135 -45
  6. package/dist/cli/commands/init.js +128 -53
  7. package/dist/cli/commands/learning.js +61 -13
  8. package/dist/cli/commands/memory.js +35 -14
  9. package/dist/cli/commands/uninstall.js +163 -39
  10. package/dist/commands/code-review.md +1 -3
  11. package/dist/commands/debug.md +15 -12
  12. package/dist/commands/dynamic-build.md +172 -135
  13. package/dist/commands/dynamic-plan.md +9 -3
  14. package/dist/commands/explore.md +10 -4
  15. package/dist/commands/implement.md +149 -145
  16. package/dist/commands/plan.md +13 -9
  17. package/dist/commands/release.md +8 -2
  18. package/dist/commands/research.md +8 -2
  19. package/dist/commands/resolve.md +28 -19
  20. package/dist/commands/self-review.md +16 -13
  21. package/dist/core/agent-frontmatter.js +25 -0
  22. package/dist/core/agent-models.js +201 -36
  23. package/dist/core/agent-state.js +27 -5
  24. package/dist/core/assets.js +1 -1
  25. package/dist/core/feature-config.js +68 -10
  26. package/dist/core/flags.js +24 -0
  27. package/dist/core/learning-queue-cleanup.js +10 -11
  28. package/dist/core/learning-tuning-config.js +8 -0
  29. package/dist/core/linked-path.js +46 -0
  30. package/dist/core/plugins.js +16 -5
  31. package/dist/core/queue-drain.js +31 -0
  32. package/dist/hud/components/learning-counts.js +54 -8
  33. package/dist/skills/git/references/tracker/github/create-release.md +2 -2
  34. package/dist/skills/git/references/tracker/github/gather-release-evidence.md +1 -1
  35. package/dist/skills/git/references/tracker/jira/create-release.md +2 -2
  36. package/dist/skills/git/references/tracker/jira/gather-release-evidence.md +1 -1
  37. package/dist/skills/git/references/tracker/linear/create-release.md +2 -2
  38. package/dist/skills/git/references/tracker/linear/gather-release-evidence.md +1 -1
  39. package/dist/targets/claude-code/installer.js +36 -9
  40. package/dist/targets/claude-code/post-install.js +128 -38
  41. package/package.json +1 -1
  42. package/src/assets/agents/code.md +85 -35
  43. package/src/assets/agents/design.md +12 -0
  44. package/src/assets/agents/diagnose.md +18 -11
  45. package/src/assets/agents/evaluate.md +17 -24
  46. package/src/assets/agents/knowledge.md +7 -3
  47. package/src/assets/agents/learning.md +4 -6
  48. package/src/assets/agents/research.md +21 -0
  49. package/src/assets/agents/review.md +12 -0
  50. package/src/assets/agents/scrutinize.md +37 -9
  51. package/src/assets/agents/simplify.md +24 -0
  52. package/src/assets/agents/skim.md +6 -2
  53. package/src/assets/agents/synthesize.md +18 -0
  54. package/src/assets/agents/test.md +19 -11
  55. package/src/assets/agents/triage.md +8 -0
  56. package/src/assets/agents/validate.md +20 -11
  57. package/src/assets/commands/_partials/_engine.mds +36 -55
  58. package/src/assets/commands/_partials/_knowledge.mds +1 -3
  59. package/src/assets/commands/_partials/_plan_contract.mds +1 -1
  60. package/src/assets/commands/_partials/_tracker.mds +1 -1
  61. package/src/assets/commands/_partials/_wave.mds +8 -6
  62. package/src/assets/commands/code-review.mds +1 -3
  63. package/src/assets/commands/debug.mds +13 -8
  64. package/src/assets/commands/dynamic-build.mds +126 -72
  65. package/src/assets/commands/dynamic-plan.mds +7 -1
  66. package/src/assets/commands/explore.mds +9 -1
  67. package/src/assets/commands/implement.mds +147 -141
  68. package/src/assets/commands/plan.mds +12 -8
  69. package/src/assets/commands/release.md +8 -2
  70. package/src/assets/commands/research.mds +8 -2
  71. package/src/assets/commands/resolve.mds +27 -16
  72. package/src/assets/commands/self-review.mds +15 -10
  73. package/src/assets/mds/tracker/_common.mds +1 -1
  74. package/src/assets/mds/tracker/_github.mds +2 -2
  75. package/src/assets/mds/tracker/_jira.mds +2 -2
  76. package/src/assets/mds/tracker/_linear.mds +2 -2
  77. package/src/assets/scripts/ci-wait.cjs +636 -0
  78. package/src/assets/scripts/hooks/assets/orchestrator-charter.md +4 -3
  79. package/src/assets/scripts/hooks/background-memory-update +356 -17
  80. package/src/assets/scripts/hooks/capture-prompt +4 -3
  81. package/src/assets/scripts/hooks/capture-question +4 -3
  82. package/src/assets/scripts/hooks/capture-turn +4 -3
  83. package/src/assets/scripts/hooks/ensure-devflow-init +13 -1
  84. package/src/assets/scripts/hooks/ensure-root-gitignore +122 -10
  85. package/src/assets/scripts/hooks/git-marker +71 -0
  86. package/src/assets/scripts/hooks/json-helper.cjs +12 -145
  87. package/src/assets/scripts/hooks/json-parse +24 -129
  88. package/src/assets/scripts/hooks/lib/learning-store.cjs +169 -64
  89. package/src/assets/scripts/hooks/lib/render-decisions.cjs +1 -1
  90. package/src/assets/scripts/hooks/memory-worker +10 -0
  91. package/src/assets/scripts/hooks/pre-compact-memory +66 -14
  92. package/src/assets/scripts/hooks/preamble +9 -1
  93. package/src/assets/scripts/hooks/queue-append +53 -21
  94. package/src/assets/scripts/hooks/session-start-context +108 -29
  95. package/src/assets/scripts/hooks/session-start-memory +33 -11
  96. package/src/assets/scripts/release-trace.cjs +27 -10
  97. package/src/assets/skills/accessibility/SKILL.md +1 -1
  98. package/src/assets/skills/apply-decisions/SKILL.md +12 -82
  99. package/src/assets/skills/apply-feature-knowledge/SKILL.md +8 -42
  100. package/src/assets/skills/architecture/SKILL.md +1 -1
  101. package/src/assets/skills/boundary-validation/SKILL.md +1 -1
  102. package/src/assets/skills/complexity/SKILL.md +1 -1
  103. package/src/assets/skills/compliance/SKILL.md +1 -1
  104. package/src/assets/skills/consistency/SKILL.md +1 -1
  105. package/src/assets/skills/database/SKILL.md +1 -1
  106. package/src/assets/skills/dependencies/SKILL.md +1 -1
  107. package/src/assets/skills/dependency-research/SKILL.md +3 -6
  108. package/src/assets/skills/design-review/SKILL.md +1 -1
  109. package/src/assets/skills/docs-framework/SKILL.md +1 -1
  110. package/src/assets/skills/documentation/SKILL.md +1 -1
  111. package/src/assets/skills/gap-analysis/SKILL.md +1 -1
  112. package/src/assets/skills/git/SKILL.md +1 -1
  113. package/src/assets/skills/go/SKILL.md +1 -1
  114. package/src/assets/skills/java/SKILL.md +1 -1
  115. package/src/assets/skills/patterns/SKILL.md +1 -1
  116. package/src/assets/skills/performance/SKILL.md +1 -1
  117. package/src/assets/skills/python/SKILL.md +1 -1
  118. package/src/assets/skills/qa/SKILL.md +1 -3
  119. package/src/assets/skills/quality-gates/SKILL.md +9 -12
  120. package/src/assets/skills/quality-gates/references/report-template.md +20 -20
  121. package/src/assets/skills/react/SKILL.md +1 -1
  122. package/src/assets/skills/regression/SKILL.md +1 -1
  123. package/src/assets/skills/reliability/SKILL.md +1 -1
  124. package/src/assets/skills/research-codebase/SKILL.md +1 -1
  125. package/src/assets/skills/research-competitor/SKILL.md +1 -1
  126. package/src/assets/skills/research-external/SKILL.md +1 -1
  127. package/src/assets/skills/research-technology/SKILL.md +1 -1
  128. package/src/assets/skills/review-methodology/SKILL.md +1 -1
  129. package/src/assets/skills/rust/SKILL.md +1 -1
  130. package/src/assets/skills/security/SKILL.md +1 -1
  131. package/src/assets/skills/software-design/SKILL.md +1 -1
  132. package/src/assets/skills/test-driven-development/SKILL.md +15 -33
  133. package/src/assets/skills/testing/SKILL.md +1 -1
  134. package/src/assets/skills/typescript/SKILL.md +1 -1
  135. package/src/assets/skills/ui-design/SKILL.md +1 -1
  136. package/src/assets/skills/worktree-support/SKILL.md +3 -55
  137. package/src/assets/skills/worktree-support/references/discovery.md +48 -0
  138. package/src/assets/skills/worktree-support/references/roots.md +2 -2
@@ -206,7 +206,13 @@ Note the `budget` value from the Workflow tool context (or default to "medium" i
206
206
 
207
207
  **3. Detect mode: SINGLE or WAVE**
208
208
 
209
- - **SINGLE mode:** input is one ticket, one issue, one task description, or one plan document
209
+ What follows the command is bound once, here. Every later step names it `COMMAND_INPUT` and never restates it:
210
+
211
+ <command-input>
212
+ $ARGUMENTS
213
+ </command-input>
214
+
215
+ - **SINGLE mode:** `COMMAND_INPUT` is one ticket, one issue, one task description, or one plan document
210
216
  - **WAVE mode:** input is a set of tracker issues (wave labels, milestone, issue list), or the user says "wave" / "all tickets in wave N"
211
217
  - **A `/devflow:dynamic-tickets` ticket directory** (`{worktree}/.devflow/docs/tickets/{slug}/{ts}/`) is WAVE input: in each ticket file (every `.md` there but `tracking-issue.md`), the `**Issue:**` line directly after `**Depends on:**` is one raw `ISSUE_REFS` token, forwarded to the wave's pre-fetch verbatim — never rendered, normalised or re-derived. A ticket file with no `**Issue:**` line, or more than one, contributes no token; name it in the run summary as `not filed`.
212
218
 
@@ -244,11 +250,11 @@ If none found: build proceeds Gate-1-only (Gate 2 skipped with a note). Never re
244
250
  **5. Resolve tracking-issue number (optional)**
245
251
 
246
252
  Check, in priority order:
247
- - An explicit candidate issue reference or issue URL in the user's input (e.g. `#42`, `42`, or `https://github.com/…/issues/42`)
253
+ - An explicit candidate issue reference or issue URL in `COMMAND_INPUT` (e.g. `#42`, `42`, or `https://github.com/…/issues/42`)
248
254
  - The `**Issue:**` line directly after the H1 of the ticket set's `tracking-issue.md` (written by `/devflow:dynamic-tickets`' filing step; the file is at `{worktree}/.devflow/docs/tickets/{slug}/{ts}/tracking-issue.md`), as a raw token — only when the file holds exactly one `**Issue:**` line
249
255
  - Otherwise: none
250
256
 
251
- **Issue-reference grammar (L1 — command layer, permissive and provider-blind):** scan `$ARGUMENTS` for candidate issue references — a `#`-prefixed token and a bare digit run are both candidates — and collect them in source order as the raw token list `ISSUE_REFS`. Forward that list to the Git agent **verbatim**: the command never renders, normalises, pads, strips or coerces a token, and never rules a candidate out. Under `github` a token matching `^#?[1-9][0-9]{0,8}$` **is** a reference and the Git agent renders it as `#{n}`.
257
+ **Issue-reference grammar (L1 — command layer, permissive and provider-blind):** scan `COMMAND_INPUT` for candidate issue references — a `#`-prefixed token and a bare digit run are both candidates — and collect them in source order as the raw token list `ISSUE_REFS`. Forward that list to the Git agent **verbatim**: the command never renders, normalises, pads, strips or coerces a token, and never rules a candidate out. Under `github` a token matching `^#?[1-9][0-9]{0,8}$` **is** a reference and the Git agent renders it as `#{n}`.
252
258
 
253
259
  **A token of any other shape is neither coerced nor dropped silently — and no producer-side grammar check rejects it before the fetch.** Adjudication belongs to the operation that runs, and each one answers in its own Output block: `fetch-issue` strips a leading `#` and takes the text branch, so a non-numeric token is used as a **search term** and the operation returns the first open match or nothing; `fetch-issues-batch` resolves each token to an issue number, drops the ones it cannot resolve, and names them in `NOT_FOUND ({refs})` beside the issues it did fetch. Read the outcome from the operation that ran — a token's shape is a verdict nowhere, and there is nothing upstream holding it back.
254
260
 
@@ -316,7 +322,8 @@ if (BRANCH === "(none)") {
316
322
 
317
323
  // Phase 2: Implement
318
324
  await phase("implement", () =>
319
- agent(`Implement the following ticket on branch ${BRANCH}:
325
+ agent(`OPERATION: implement
326
+ Implement the following ticket on branch ${BRANCH}:
320
327
 
321
328
  ${TICKET}
322
329
 
@@ -329,7 +336,7 @@ ISSUE_NUMBER: ${ISSUE_NUMBER}
329
336
  ISSUE_PR_LINK: ${ISSUE_PR_LINK}
330
337
  COMPLIANCE_FRAMEWORKS: ${COMPLIANCE_FRAMEWORKS}
331
338
 
332
- When you build or run tests to verify your work, use your "Long-running commands" discipline (background-Bash + Monitor poll) for anything that may run silent >120s, and prefer package-scoped commands.
339
+ When you build or run tests to verify your work, follow your Running commands block.
333
340
 
334
341
  After implementing, commit your changes with a conventional-commit message and report:
335
342
  - Files changed
@@ -337,39 +344,51 @@ After implementing, commit your changes with a conventional-commit message and r
337
344
  - Any open questions or blockers`, { agentType: "Code" })
338
345
  );
339
346
 
340
- // Phase 3: Gate 1 — post-code pipeline
347
+ // Phase 3: Gate 1 #1 — post-code pipeline: Simplify agent → Scrutinize agent → Validate agent.
348
+ // Validate runs last and unconditionally, so its one full run covers every commit the pass made (see gate1_postcode()).
341
349
  const gate1 = await phase("gate1", async () => {
342
- // Gate 1 #1 — runs once, immediately after the initial implementation.
350
+ await agent(`Simplify and reduce complexity of recent changes on branch ${BRANCH}. Commit any improvements.`, { agentType: "Simplify" });
351
+
352
+ const scrutiny = await agent(`9-pillar self-review of recent changes on branch ${BRANCH}. Fix P0/P1 issues and commit the fixes.
353
+ Return: {"status": "PASS" | "FIXED" | "BLOCKED"}`, { agentType: "Scrutinize" });
354
+ // Anything but an explicit PASS or FIXED is a stop: a Scrutinize that returned no status, or one this skeleton does not know, has not accepted the code.
355
+ if (!["PASS", "FIXED"].includes(scrutiny?.status)) {
356
+ return { verdict: "ESCALATED", type: "scrutiny-blocked", reason: scrutiny?.status === "BLOCKED" ? "Scrutinize agent returned BLOCKED: a P0 issue cannot be fixed in scope" : "Scrutinize agent returned no recognised status: counted as BLOCKED" };
357
+ }
358
+
343
359
  const validation = await agent(`Run build, typecheck, lint, and tests on branch ${BRANCH}.
344
- For any build/test that may run silent >120s, use the background-Bash + Monitor poll procedure (your "Long-running commands" discipline) so you never trip the 180s watchdog; prefer package-scoped commands.
345
- Report: PASS or FAIL with details.`, { agentType: "Validate" });
360
+ Follow your Running commands block for every build and test command.
361
+ Return: {"verdict": "PASS" | "FAIL", "details": "..."}`, { agentType: "Validate" });
346
362
 
347
- if (validation.verdict === "FAIL") {
348
- // Up to 2 fix attempts
363
+ // Anything but an explicit PASS is a failure: a Validate that returned no verdict has not validated the branch.
364
+ if (validation?.verdict !== "PASS") {
365
+ let failureDetails = validation?.details || "Validate agent returned no PASS verdict";
366
+ // Up to 2 fix attempts, each followed by a Validate re-run
349
367
  for (let attempt = 1; attempt <= 2; attempt++) {
350
- await agent(`Fix the validation failures on branch ${BRANCH}:
351
- ${validation.details}
368
+ await agent(`OPERATION: validation-fix
369
+ Fix the validation failures on branch ${BRANCH}:
370
+ ${failureDetails}
352
371
  ISSUE_NUMBER: ${ISSUE_NUMBER}
353
372
  ISSUE_PR_LINK: ${ISSUE_PR_LINK}
354
373
  COMPLIANCE_FRAMEWORKS: ${COMPLIANCE_FRAMEWORKS}
355
374
  Commit fixes with conventional-commit message.`, { agentType: "Code" });
356
- const recheck = await agent(`Re-run build, typecheck, lint, tests on branch ${BRANCH}. Report: PASS or FAIL.`, { agentType: "Validate" });
357
- if (recheck.verdict === "PASS") break;
358
- if (attempt === 2) return { verdict: "ESCALATED", reason: "Validation exhausted after 2 Code agent fix attempts" };
375
+ const recheck = await agent(`Re-run build, typecheck, lint, tests on branch ${BRANCH} after fix attempt ${attempt} (follow your Running commands block).
376
+ Return: {"verdict": "PASS" | "FAIL", "details": "..."}`, { agentType: "Validate" });
377
+ if (recheck?.verdict === "PASS") break;
378
+ failureDetails = recheck?.details || failureDetails;
379
+ if (attempt === 2) return { verdict: "ESCALATED", type: "validation-exhausted", reason: "Validation exhausted after 2 Code agent fix attempts" };
359
380
  }
360
381
  }
361
382
 
362
- await agent(`Simplify and reduce complexity of recent changes on branch ${BRANCH}. Commit any improvements.`, { agentType: "Simplify" });
363
-
364
- const scrutiny = await agent(`9-pillar self-review of recent changes on branch ${BRANCH}. Report any code you changed and your findings.`, { agentType: "Scrutinize" });
365
-
366
- if (scrutiny.codeChanged) {
367
- await agent(`Re-run build, typecheck, lint, tests on branch ${BRANCH} (Scrutinize agent made changes). Report: PASS or FAIL.`, { agentType: "Validate" });
368
- }
369
-
370
383
  return { verdict: "PASS" };
371
384
  });
372
385
 
386
+ // Gate 1 #1 stop, before Gate 2: an exhausted Validate or a BLOCKED Scrutinize leaves a broken or unfinished build,
387
+ // so Gate 2 and the review pass never run on it — a correctness stop, like the ticket-link and branch stops.
388
+ if (gate1.verdict === "ESCALATED") {
389
+ return { ticket: TICKET, branch: BRANCH, verdict: "ESCALATED", issueId: ISSUE_NUMBER, issuePrLink: ISSUE_PR_LINK, escalations: [{ type: gate1.type, description: gate1.reason }] };
390
+ }
391
+
373
392
  // Phase 4: Gate 2 — acceptance gate (once, before review pass)
374
393
  const gate2 = await phase("gate2", async () => {
375
394
  if (!PLAN && !CRITERIA && !TEST_PLAN) {
@@ -378,25 +397,23 @@ const gate2 = await phase("gate2", async () => {
378
397
 
379
398
  let evalVerdict = "SKIPPED";
380
399
  if (PLAN) {
381
- const panel = await parallel([
382
- () => agent(`Acceptance-criteria evaluator: does the implementation on branch ${BRANCH} satisfy each numbered criterion, INCLUDING negative criteria?
400
+ const evaluation = await agent(`Acceptance and scope check on branch ${BRANCH}. Answer both questions:
401
+ 1. Does the implementation satisfy each numbered criterion, INCLUDING negative criteria (what it must NOT do)?
402
+ 2. Did the Code agent introduce any unplanned changes, smuggled anti-features, or drift from the plan's intent?
383
403
  Plan: ${PLAN}
384
404
  Criteria: ${CRITERIA || "see plan"}
385
- Report: PASS or FAIL per criterion with rationale.`, { agentType: "Evaluate" }),
386
- () => agent(`Scope/intent-drift evaluator: did the Code agent on branch ${BRANCH} introduce any unplanned changes, smuggled anti-features, or drift from the plan's intent?
387
- Plan: ${PLAN}
388
- Report: PASS or FAIL with rationale.`, { agentType: "Evaluate" }),
389
- ]);
390
- evalVerdict = panel.every(p => p.verdict === "PASS") ? "PASS" : "FAIL";
405
+ Return: {"verdict": "PASS" | "FAIL", "rationale": "..."} — FAIL if either answer is no, naming the failing criterion or the drift.`, { agentType: "Evaluate" });
406
+ evalVerdict = evaluation?.verdict === "PASS" ? "PASS" : "FAIL";
391
407
  if (evalVerdict === "FAIL") {
392
408
  // fix-and-continue — no re-evaluate, no inline Gate 1. The final Gate 1 (#2)
393
409
  // after the review pass is the build gate before the branch is handed back.
394
- await agent(`Fix the alignment issues identified by the Evaluate agent panel on branch ${BRANCH}:
395
- ${panel.filter(p => p.verdict === "FAIL").map(p => p.rationale).join("\n")}
410
+ await agent(`OPERATION: alignment-fix
411
+ Fix the alignment issues identified by the Evaluate agent on branch ${BRANCH}:
412
+ ${evaluation?.rationale || "see the Evaluate agent's report"}
396
413
  ISSUE_NUMBER: ${ISSUE_NUMBER}
397
414
  ISSUE_PR_LINK: ${ISSUE_PR_LINK}
398
415
  COMPLIANCE_FRAMEWORKS: ${COMPLIANCE_FRAMEWORKS}
399
- Self-verify your fix compiles (background-Bash + Monitor for any build >120s — see your "Long-running commands" discipline). Commit fixes.`, { agentType: "Code" });
416
+ Self-verify your fix compiles, following your Running commands block. Commit fixes.`, { agentType: "Code" });
400
417
  evalVerdict = "FAIL-FIXED"; // issues found, fixes applied, not re-evaluated by design
401
418
  }
402
419
  }
@@ -406,17 +423,18 @@ Self-verify your fix compiles (background-Bash + Monitor for any build >120s —
406
423
  const testResult = await agent(`Run scenario-based acceptance tests on branch ${BRANCH} against these criteria:
407
424
  ${CRITERIA || "(none)"}
408
425
  TEST_PLAN: ${TEST_PLAN || "(none)"}
409
- For any test/build command that may run silent >120s, use the background-Bash + Monitor poll procedure (your "Long-running commands" discipline) so you never trip the 180s watchdog.
426
+ Follow your Running commands block for every scenario command.
410
427
  Cover: functionality, API contracts, performance, and cover every TEST_PLAN scenario. Report: PASS or FAIL per scenario.`, { agentType: "Test" });
411
428
  testVerdict = testResult.verdict;
412
429
  if (testVerdict === "FAIL") {
413
430
  // fix-and-continue — no re-test, no inline Gate 1.
414
- await agent(`Fix the failing acceptance test scenarios on branch ${BRANCH}:
431
+ await agent(`OPERATION: qa-fix
432
+ Fix the failing acceptance test scenarios on branch ${BRANCH}:
415
433
  ${testResult.failures}
416
434
  ISSUE_NUMBER: ${ISSUE_NUMBER}
417
435
  ISSUE_PR_LINK: ${ISSUE_PR_LINK}
418
436
  COMPLIANCE_FRAMEWORKS: ${COMPLIANCE_FRAMEWORKS}
419
- Self-verify your fix compiles and the scenarios pass (background-Bash + Monitor for any build/test >120s). Commit fixes.`, { agentType: "Code" });
437
+ Self-verify your fix compiles and the scenarios pass, following your Running commands block. Commit fixes.`, { agentType: "Code" });
420
438
  testVerdict = "FAIL-FIXED"; // issues found, fixes applied, not re-evaluated by design
421
439
  }
422
440
  }
@@ -438,7 +456,7 @@ const reviewResult = await phase("review", async () => {
438
456
  const chunkSize = 5;
439
457
  const reviewThunks = reviewFocuses.map(focus =>
440
458
  () => agent(`${focus.charAt(0).toUpperCase() + focus.slice(1)} review. ${reviewScope}
441
- Focus discipline: ${focus}. Return: {"focus": "${focus}", "reviewed": true, "filesExamined": ["<path>"], "findings": [{"file": "<path-or-null>", "description": "<one-sentence issue>", "severity": "critical|high|medium|low"}]}`, { agentType: "Review" })
459
+ Focus discipline: ${focus}. Return: {"focus": "${focus}", "reviewed": true, "filesExamined": ["<path>"], "findings": [{"file": "<path-or-null>", "line": <number-or-null>, "description": "<one-sentence issue>", "severity": "critical|high|medium|low"}]}`, { agentType: "Review" })
442
460
  );
443
461
 
444
462
  let rawReviews = [];
@@ -455,7 +473,7 @@ Focus discipline: ${focus}. Return: {"focus": "${focus}", "reviewed": true, "fil
455
473
  const focus = reviewFocuses[i];
456
474
  if (!r || r.reviewed !== true) {
457
475
  // Dead Review agent: retry once sequentially
458
- const retry = await agent(`Retry ${focus} review. ${reviewScope} Focus: ${focus} discipline. Return: {"focus": "${focus}", "reviewed": true, "filesExamined": ["<path>"], "findings": [{"file": "<path-or-null>", "description": "<one-sentence issue>", "severity": "critical|high|medium|low"}]}`, { agentType: "Review" });
476
+ const retry = await agent(`Retry ${focus} review. ${reviewScope} Focus: ${focus} discipline. Return: {"focus": "${focus}", "reviewed": true, "filesExamined": ["<path>"], "findings": [{"file": "<path-or-null>", "line": <number-or-null>, "description": "<one-sentence issue>", "severity": "critical|high|medium|low"}]}`, { agentType: "Review" });
459
477
  if (!retry || retry.reviewed !== true) {
460
478
  coverageGaps.push(focus);
461
479
  log(`review coverage incomplete: ${focus}`);
@@ -520,17 +538,22 @@ ${JSON.stringify(allFindings.map((f, i) => ({ index: i, description: f.descripti
520
538
  // Distinct-file groups are paced in chunks of FIX_CHUNK — same pacing bar as the Review spawn
521
539
  // path — to avoid launching all Code agents at once on a large branch (reliability: explicit bound).
522
540
  const FIX_CHUNK = 5;
523
- const fixThunks = Object.entries(fileGroups).map(([fileKey, findings]) => async () => {
541
+ // issue-fix SCOPE per finding: CRITICAL or HIGH takes the Careful protocol (failing regression test first); anything else is Standard.
542
+ const CAREFUL_SEVERITIES = new Set(["critical", "high"]);
543
+ const fixThunks = Object.values(fileGroups).map(findings => async () => {
524
544
  const chunkResults = [];
525
545
  for (let i = 0; i < findings.length; i += MAX_BATCH) {
526
546
  const chunk = findings.slice(i, i + MAX_BATCH);
527
- const r = await agent(`Fix the following confirmed review findings on branch ${BRANCH}${fileKey.startsWith("__nofile") ? "" : ` in ${fileKey}`}:
528
- ${chunk.map(f => `- ${f.description} (${f.severity})`).join("\n")}
529
-
547
+ const r = await agent(`OPERATION: issue-fix
548
+ Fix the following confirmed review findings on branch ${BRANCH}:
549
+ ISSUES:
550
+ ${chunk.map((f, n) => `${n + 1}. ${f.file ? `${f.file}${f.line ? `:${f.line}` : ""} — ` : ""}${f.description} (${f.severity})`).join("\n")}
551
+ SCOPE: ${chunk.map((f, n) => `${n + 1}=${CAREFUL_SEVERITIES.has(String(f.severity).toLowerCase()) ? "Careful" : "Standard"}`).join(", ")}
552
+ PUSH: false
530
553
  ISSUE_NUMBER: ${ISSUE_NUMBER}
531
554
  ISSUE_PR_LINK: ${ISSUE_PR_LINK}
532
555
  COMPLIANCE_FRAMEWORKS: ${COMPLIANCE_FRAMEWORKS}
533
- Fix all findings in this batch. Self-verify your fix compiles (background-Bash + Monitor for any build >120s — see your "Long-running commands" discipline). Commit with conventional-commit message.
556
+ Fix all findings in this batch. Self-verify your fix compiles, following your Running commands block. Commit with conventional-commit message.
534
557
  Return: {"status": "fixed"|"blocked", "commitShas": ["<sha>"], "unresolved": ["<description of any finding that could not be fixed>"]}`, { agentType: "Code" });
535
558
  chunkResults.push({ chunk, result: r });
536
559
  }
@@ -565,38 +588,40 @@ Return: {"status": "fixed"|"blocked", "commitShas": ["<sha>"], "unresolved": ["<
565
588
  return { survivingFindings: notAddressed, fixedFindings, coverageGaps };
566
589
  });
567
590
 
568
- // Phase 5.5: Gate 1 #2 — FINAL post-fix gate (full Validate agent → Simplify agent → Scrutinize agent).
591
+ // Phase 5.5: Gate 1 #2 — FINAL post-fix gate (Simplify agent → Scrutinize agent → full Validate agent, Validate last and unconditional).
569
592
  // Runs ONCE, after the review pass. Inside the pass the Code agent self-verified its own fixes;
570
593
  // this is the build gate before the branch is handed back. See gate1_postcode() cadence.
571
594
  const gate1Final = await phase("gate1-final", async () => {
595
+ await agent(`Simplify and reduce complexity of recent changes on branch ${BRANCH}. Commit any improvements.`, { agentType: "Simplify" });
596
+
597
+ const scrutiny = await agent(`9-pillar self-review of recent changes on branch ${BRANCH}. Fix P0/P1 issues and commit the fixes.
598
+ Return: {"status": "PASS" | "FIXED" | "BLOCKED"}`, { agentType: "Scrutinize" });
599
+ if (!["PASS", "FIXED"].includes(scrutiny?.status)) {
600
+ return { verdict: "ESCALATED", type: "scrutiny-blocked", reason: scrutiny?.status === "BLOCKED" ? "Final Gate 1 Scrutinize agent returned BLOCKED: a P0 issue cannot be fixed in scope" : "Final Gate 1 Scrutinize agent returned no recognised status: counted as BLOCKED" };
601
+ }
602
+
572
603
  const validation = await agent(`Run build, typecheck, lint, and tests on branch ${BRANCH} (final gate after all fixing).
573
- For any build/test that may run silent >120s, use the background-Bash + Monitor poll procedure (your "Long-running commands" discipline) so you never trip the 180s watchdog; prefer package-scoped commands.
574
- Report: PASS or FAIL with details.`, { agentType: "Validate" });
604
+ Follow your Running commands block for every build and test command.
605
+ Return: {"verdict": "PASS" | "FAIL", "details": "..."}`, { agentType: "Validate" });
575
606
 
576
- if (validation.verdict === "FAIL") {
577
- let failureDetails = validation.details;
607
+ if (validation?.verdict !== "PASS") {
608
+ let failureDetails = validation?.details || "Validate agent returned no PASS verdict";
578
609
  for (let attempt = 1; attempt <= 2; attempt++) {
579
- await agent(`Fix the final validation failures on branch ${BRANCH}:
610
+ await agent(`OPERATION: validation-fix
611
+ Fix the final validation failures on branch ${BRANCH}:
580
612
  ${failureDetails}
581
613
  ISSUE_NUMBER: ${ISSUE_NUMBER}
582
614
  ISSUE_PR_LINK: ${ISSUE_PR_LINK}
583
615
  COMPLIANCE_FRAMEWORKS: ${COMPLIANCE_FRAMEWORKS}
584
616
  Self-verify your fix compiles. Commit fixes with conventional-commit message.`, { agentType: "Code" });
585
- const recheck = await agent(`Re-run build, typecheck, lint, tests on branch ${BRANCH} (background+Monitor for long commands). Report: PASS or FAIL.`, { agentType: "Validate" });
586
- if (recheck.verdict === "PASS") break;
587
- failureDetails = recheck.details || failureDetails;
588
- if (attempt === 2) return { verdict: "ESCALATED", reason: "Final Gate 1 validation exhausted after 2 Code agent fix attempts" };
617
+ const recheck = await agent(`Re-run build, typecheck, lint, tests on branch ${BRANCH} (final gate, after fix attempt ${attempt}; follow your Running commands block).
618
+ Return: {"verdict": "PASS" | "FAIL", "details": "..."}`, { agentType: "Validate" });
619
+ if (recheck?.verdict === "PASS") break;
620
+ failureDetails = recheck?.details || failureDetails;
621
+ if (attempt === 2) return { verdict: "ESCALATED", type: "validation-exhausted", reason: "Final Gate 1 validation exhausted after 2 Code agent fix attempts" };
589
622
  }
590
623
  }
591
624
 
592
- await agent(`Simplify and reduce complexity of recent changes on branch ${BRANCH}. Commit any improvements.`, { agentType: "Simplify" });
593
-
594
- const scrutiny = await agent(`9-pillar self-review of recent changes on branch ${BRANCH}. Report any code you changed and your findings.`, { agentType: "Scrutinize" });
595
-
596
- if (scrutiny.codeChanged) {
597
- await agent(`Re-run build, typecheck, lint, tests on branch ${BRANCH} (Scrutinize agent made changes; background+Monitor for long commands). Report: PASS or FAIL.`, { agentType: "Validate" });
598
- }
599
-
600
625
  return { verdict: "PASS" };
601
626
  });
602
627
 
@@ -625,7 +650,7 @@ Write a concise report. Present only surviving findings as outstanding — never
625
650
  survivingFindings: reviewResult.survivingFindings || [], fixedFindings: reviewResult.fixedFindings || [],
626
651
  reviewCoverage: { failedFocuses: coverageGaps, complete: coverageGaps.length === 0 },
627
652
  escalations: [
628
- ...(gate1Final.verdict === "ESCALATED" ? [{ type: "validation-exhausted", description: gate1Final.reason || "final Gate 1 escalated" }] : []),
653
+ ...(gate1Final.verdict === "ESCALATED" ? [{ type: gate1Final.type || "validation-exhausted", description: gate1Final.reason || "final Gate 1 escalated" }] : []),
629
654
  ...coverageGaps.map(focus => ({ type: "review-coverage-incomplete", description: `review coverage incomplete: ${focus}` })),
630
655
  ],
631
656
  gate2, report,
@@ -647,54 +672,38 @@ Write a concise report. Present only surviving findings as outstanding — never
647
672
 
648
673
  This applies to both: multiple Code agents working on a single ticket AND multi-ticket scheduling in a wave.
649
674
 
650
- ### Build execution doctrine — long-running commands (LOAD-BEARING)
675
+ ### Build execution doctrine — the Running commands block (LOAD-BEARING)
651
676
 
652
- The Workflow runtime KILLS any sub-agent that emits no output for 180 seconds. A cold `cargo build`, `cargo test`, a large `tsc`, `gradle build`, `go build ./...`, etc. routinely runs silent far longer and trips this watchdog (the failure reads `agent stalled on all N attempts`). Plain foreground `Bash` also defaults to a 120s timeout.
677
+ The Validate, Code and Test agents each carry this `## Running commands` block in their bodies, so each prompt in this engine names the block ("your Running commands block") instead of copying it:
653
678
 
654
- **RULE: any agent (Validate agent, Code agent, Test agent) running a build / test / compile / install that may run silent for more than ~120s MUST run it in the BACKGROUND and POLL — never as a single silent foreground command.**
679
+ Run builds, typechecks, lints and tests in the foreground, each with an explicit Bash `timeout` above its expected run time. The ceiling is 600000 ms, or `BASH_MAX_TIMEOUT_MS` when set (`echo ${BASH_MAX_TIMEOUT_MS:-600000}`).
655
680
 
656
- Build commands are NEVER wrapped in `sh -c`, `bash -c`, or inline interpreters (`python3 -c`, `node -e`). Invoke commands directly with the step-1 redirect form — permission systems deny wrapper-invoked commands that would be allowed directly.
657
-
658
- Mechanical procedure (spike-verified — a workflow sub-agent survived a 253s job this way):
659
-
660
- 0. **Pre-load Monitor:** before launching any background task, load the `Monitor` tool via ToolSearch (`select:Monitor`).
661
- 1. Choose ONE unique base path for this run and reuse it verbatim in steps 1–3, e.g. `BASE=/tmp/df-build-<ticket-slug>`. Launch the command with the Bash tool using `run_in_background: true`:
662
- ```
663
- <build/test command> > <BASE>.log 2>&1; echo "EXIT=$?" > <BASE>.done
664
- ```
665
- This returns immediately with a background task id — do NOT block on it.
666
- 2. Arm ONE Monitor that emits a heartbeat well under 180s AND exits when the job finishes:
667
- - description: short, e.g. `await <build cmd>`
668
- - persistent: false
669
- - timeout_ms: comfortably ABOVE the expected job time (e.g. 600000)
670
- - command: `until [ -f <BASE>.done ]; do echo building; sleep 25; done; echo BUILD_DONE; cat <BASE>.done`
671
-
672
- The `building` heartbeat every 25s (≪ 180s) is delivered as a notification that re-invokes you, so the watchdog never sees a >180s gap. `BUILD_DONE` + the `EXIT=` line signal completion.
673
-
674
- **Exit-code honesty:** the background task's own exit status is meaningless (the trailing `echo` always exits 0). ALWAYS read the `EXIT=` value written inside `<BASE>.done` — that is the authoritative result.
675
-
676
- **Bounded polling:** arm ONE Monitor then stop acting. On Monitor timeout, re-arm at most 2× (never more than 3 total Monitor calls per build). If the build has not finished after 3 Monitor calls: record the state and escalate — never babysit. A Code agent burned 241k tokens polling one build.
677
- 3. When the monitor reports `BUILD_DONE`: the job PASSES iff `<BASE>.done` contains `EXIT=0`. Read `<BASE>.log` for output/failure detail.
681
+ - Capture, then tail, in one Bash call (shell state does not persist): `LOG=$(mktemp); echo "LOG=$LOG"; <command> >"$LOG" 2>&1; rc=$?; tail -n 40 "$LOG"; echo "EXIT=$rc"`. The printed `EXIT=` value is the result; never decide one from a grep count.
682
+ - Never background a command and wait on it, and never poll across turns: no `sleep` or `true` turns, no sentinel-file checks, no Monitor.
683
+ - Prefer the scoped command for the change (a package, a path or a test file); for the whole set, one workspace-level command over a per-package loop.
684
+ - A run that exceeds its timeout is BLOCKED: report its duration and log path. Do not wait on it, poll it or re-run it.
685
+ - A run expected to exceed the ceiling is split into parts, each under about 90% of it, run in sequence. If it cannot be split, report BLOCKED with the remedy `devflow flags --set bash-max-timeout-ms=<ms>`.
686
+ - Never re-run a command when nothing it reads has changed.
687
+ - Never wrap a build or test command in `sh -c`, `bash -c`, `python3 -c` or `node -e`: permission rules deny wrapped commands they would allow directly.
688
+ - The same rules hold inside a dynamic Workflow sub-agent.
678
689
 
679
690
  **Cheapest-sufficient validation:** iterate with the fastest check that proves the change — `tsc --noEmit`, `cargo check`, `go vet`, a single test file — the ecosystem's cheapest sufficient signal. Reserve expensive full/optimized builds and whole-suite runs for the final gates only.
680
691
 
681
692
  **One build gate per phase:** batch related fixes, validate once. A fix-pass Code agent runs ONE light check over its whole batch — never several invocations per small fix. Do NOT validate after every individual mutation; validate once after the batch is complete.
682
693
 
683
- **Scope commands to stay short.** During the engine, PREFER crate/package-scoped builds and tests — `cargo build -p <crate>`, `cargo test -p <crate>`, `npm test -- <path>`, `go test ./pkg/...` — over the whole workspace. The full-workspace regression is the human's job after the wave (the wave already hands the integrated branch back to the user). Scoping keeps most commands under the watchdog window and under budget.
684
-
685
- **Invariants:** heartbeat interval MUST stay well under 180s (25–30s is the tested value); Monitor `timeout_ms` MUST exceed the expected job duration (a too-short timeout kills the poll, not the build). Never substitute a single silent long command for this procedure.
694
+ **Scope commands to stay short.** During the engine, prefer the scoped command for the change over the whole workspace — `cargo build -p <crate>`, `cargo test -p <crate>`, `npm test -- <path>`, `go test ./pkg/...` — and, when the whole set is needed, one workspace-level command over a per-package loop. After the wave, the full-workspace regression is the human's job (the wave already hands the integrated branch back to the user).
686
695
 
687
696
  ### Implement bundle
688
697
 
689
698
  The standard implementation unit for one ticket. Run in order:
690
699
 
691
700
  ```
692
- Code(agentType:"Code", prompt: full task + plan + DECISIONS_CONTEXT + handoff if sequential)
701
+ Code(agentType:"Code", prompt: "OPERATION: implement" first line, then full task + plan + DECISIONS_CONTEXT + handoff if sequential)
693
702
  → gate1_postcode()
694
703
  → gate2_acceptance() ← Gate 2 runs HERE — before the review pass, not after
695
704
  ```
696
705
 
697
- The Code agent prompt must include: task description, implementation plan (if one exists), relevant DECISIONS_CONTEXT (the index you loaded before authoring), the compliance lens (`COMPLIANCE_FRAMEWORKS` — every Code prompt carries it, fix prompts included), and any PRIOR_PHASE_SUMMARY / HANDOFF_FILE for sequential multi-phase tickets.
706
+ Every Code agent prompt opens with `OPERATION: <mode>` as its first line (`implement`, `issue-fix`, `validation-fix`, `alignment-fix` or `qa-fix`). It must include: task description, implementation plan (if one exists), relevant DECISIONS_CONTEXT (the index you loaded before authoring), the compliance lens (`COMPLIANCE_FRAMEWORKS` — every Code prompt carries it, fix prompts included), and any PRIOR_PHASE_SUMMARY / HANDOFF_FILE for sequential multi-phase tickets.
698
707
 
699
708
  Gate 2 runs at implementation acceptance — this matches devflow's deliberate placement: "evaluation is part of implementation acceptance, not post-review" (§6.1).
700
709
 
@@ -702,23 +711,23 @@ Gate 2 runs at implementation acceptance — this matches devflow's deliberate p
702
711
 
703
712
  ORDER IS LOAD-BEARING. Run exactly in this sequence:
704
713
 
705
- 1. **Validate agent** — build / typecheck / lint / test
706
- - Build/test commands that may run silent for >~120s MUST follow `build_execution_doctrine()` (background Bash + Monitor poll), or they trip the 180s workflow watchdog.
707
- - FAIL → Code agent fix (max 2 retries) → re-run Validate agent
708
- - If still FAIL after 2 retries → escalate (do not loop endlessly)
709
- 2. **Simplify agent** — reduce complexity, remove duplication
710
- 3. **Scrutinize agent** — 9-pillar self-review (deep structural analysis)
711
- - If Scrutinize agent changed code → re-run Validate agent (verify the Scrutinize agent's edits compile/pass)
714
+ 1. **Simplify agent** — reduce complexity, remove duplication
715
+ 2. **Scrutinize agent** — 9-pillar self-review (deep structural analysis); returns `{"status": "PASS" | "FIXED" | "BLOCKED"}`
716
+ - BLOCKED (a P0 issue cannot be fixed in scope), or a missing or unrecognised status, which counts as BLOCKED → stop the pass and escalate as `scrutiny-blocked`; Validate does not run on code Scrutinize could not accept
717
+ 3. **Validate agent** — build / typecheck / lint / test; returns `{"verdict": "PASS" | "FAIL", "details": "..."}`. It runs LAST and UNCONDITIONALLY: one full run over the HEAD that Simplify and Scrutinize left, so it covers their commits whether or not Scrutinize changed code
718
+ - Build/test commands follow `build_execution_doctrine()`: the Validate agent's Running commands block, foreground under an explicit Bash timeout.
719
+ - FAIL → Code agent fix (max 2 retries), each followed by a Validate agent re-run
720
+ - If still FAIL after 2 retries → escalate as `validation-exhausted` (do not loop endlessly)
712
721
 
713
722
  **Gate 1 contains NO Evaluate agent and NO Test agent.** Those are Gate 2 only.
714
723
 
715
724
  Depth scales to change size + budget: a trivial one-line fix warrants a lighter pass; a multi-file refactor warrants the full depth.
716
725
 
717
726
  **Cadence — Gate 1 runs at exactly TWO points per ticket:**
718
- 1. **Gate 1 #1** — immediately after the initial Code agent implementation (inside `implement_bundle()`).
719
- 2. **Gate 1 #2** — the FINAL gate, after ALL Gate-2 fixes AND the entire review pass have completed.
727
+ 1. **Gate 1 #1** — immediately after the initial Code agent implementation (inside `implement_bundle()`). An escalated result — Validate exhausted, or Scrutinize BLOCKED — returns early: the ticket's engine returns `ESCALATED` with one escalation, and Gate 2 and the review pass never run on a broken or unfinished build.
728
+ 2. **Gate 1 #2** — the FINAL gate, after ALL Gate-2 fixes AND the entire review pass have completed. Scrutinize BLOCKED here is `ESCALATED` as well.
720
729
 
721
- It does NOT run inside the review pass, nor after each individual Gate-2 / review / QA fix. At those points the fixing Code agent self-verifies its OWN build compiles (see `review_pass()` and the Code agent's "Long-running commands" discipline). The final Gate 1 #2 is the invariant that all written code passes before merge.
730
+ It does NOT run inside the review pass, nor after each individual Gate-2 / review / QA fix. At those points the fixing Code agent self-verifies its OWN build compiles (see `review_pass()` and the Code agent's Running commands block). The final Gate 1 #2 is the invariant that all written code passes before merge.
722
731
 
723
732
  ### GATE 2 — Acceptance gate (per ticket, plan-scoped — fires ONCE at implementation acceptance)
724
733
 
@@ -726,30 +735,27 @@ Gate 2 fires ONCE: after the implement-bundle and BEFORE the review pass. It doe
726
735
 
727
736
  Gate 2 inputs are produced by `/devflow:dynamic-plan`'s plan-challenge step — the acceptance criteria and test plan written for the Evaluate agent and Test agent.
728
737
 
729
- **Evaluate agent panel** (only if a plan exists):
730
- - Run `evaluator_panel()` — see that block for the panel composition
731
- - If any critical lens returns MISALIGNED: fix-and-continue — the demanded fixes are applied by a Code agent that self-verifies its own build (batched per the review-pass batching doctrine if numerous). The recorded verdict becomes `FAIL-FIXED` (issues found, fixes applied, not re-evaluated by design); Gate 2 then proceeds. In SINGLE mode the run reports it as `UNVERIFIED`, never PASS.
738
+ **Evaluate agent** (only if a plan exists):
739
+ - Run `evaluator_panel()` — see that block for the one spawn and the two lenses it names
740
+ - If the Evaluate agent returns FAIL (either lens failed): fix-and-continue — the demanded fixes are applied by a Code agent that self-verifies its own build (batched per the review-pass batching doctrine if numerous). The recorded verdict becomes `FAIL-FIXED` (issues found, fixes applied, not re-evaluated by design); Gate 2 then proceeds. In SINGLE mode the run reports it as `UNVERIFIED`, never PASS.
732
741
 
733
742
  **Test agent** (only if acceptance criteria or a test plan exist):
734
743
  - Scenario-based acceptance tests covering functionality, API contracts, performance
735
744
  - FAIL → fix-and-continue — a Code agent applies the demanded fixes and self-verifies its own build. The recorded verdict becomes `FAIL-FIXED`; Gate 2 then proceeds. In SINGLE mode the run reports it as `UNVERIFIED`, never PASS.
736
745
 
737
746
  **When Gate 2 inputs are absent:**
738
- - No plan → skip Evaluate agent panel silently (note in output: "Gate 2 Evaluate agent skipped — no plan available")
747
+ - No plan → skip the Evaluate agent silently (note in output: "Gate 2 Evaluate agent skipped — no plan available")
739
748
  - No acceptance criteria and no test plan → skip Test agent silently (note in output: "Gate 2 Test agent skipped — no criteria available")
740
749
  - Build proceeds Gate-1-only. Never refuse to build; never force-generate fake criteria. Trust the user.
741
750
 
742
- ### Evaluate agent panel (§12 — diverse-lens verification)
751
+ ### Evaluate agent spawn (one agent, two lenses)
743
752
 
744
- Run 2–3 Evaluate agents with DISTINCT lenses. Diversity not redundancy — each agent asks a different question:
753
+ Run ONE Evaluate agent whose prompt names both lenses, so a ticket pays for one spawn and one read of the plan:
745
754
 
746
- 1. **Acceptance-criteria evaluator** — "Does the implementation satisfy each numbered acceptance criterion, INCLUDING the negative criteria (what it must NOT do)?"
747
- 2. **Scope / intent-drift evaluator** — "Did the Code agent smuggle in unplanned changes, deviate from the plan's intent, or introduce anti-features?"
748
- 3. **Cross-ticket-consistency evaluator** — (wave-only; skip for single-ticket runs) "Does this ticket honor the API contracts and invariants that other wave tickets depend on?"
755
+ 1. **Acceptance criteria** — "Does the implementation satisfy each numbered acceptance criterion, INCLUDING the negative criteria (what it must NOT do)?"
756
+ 2. **Scope / intent drift** — "Did the Code agent smuggle in unplanned changes, deviate from the plan's intent, or introduce anti-features?"
749
757
 
750
- Gate = **all-critical-must-pass**: a single critical FAIL from any lens blocks acceptance.
751
-
752
- Keep the panel to 2–3 agents. Do not spawn redundant agents asking the same question — that is cost with no diversity benefit.
758
+ It returns `{"verdict": "PASS" | "FAIL", "rationale": "..."}`: FAIL when either lens fails, the rationale naming the failing criterion or the drift. A FAIL on either lens blocks acceptance. A wave ticket runs this same skeleton through `runSingleTicketEngine`, so it gets the same single spawn.
753
759
 
754
760
  ### Acceptance criteria + test plan contract
755
761
 
@@ -794,7 +800,7 @@ Every line of a `## Test Plan` section, or of a PR's test-plan block, follows th
794
800
 
795
801
  #### Consumption by Gate 2
796
802
 
797
- The Evaluate agent panel receives: the per-ticket plan + the numbered acceptance criteria (positive and negative).
803
+ The Evaluate agent receives: the per-ticket plan + the numbered acceptance criteria (positive and negative).
798
804
 
799
805
  The Test agent receives: the test plan's TP lines, once `check tp` has admitted them.
800
806
 
@@ -850,13 +856,13 @@ Majority-survives: a finding needs >50% of verification lenses to confirm it. St
850
856
 
851
857
  If no surviving findings: return early (no fixes needed). Any coverageGaps are carried in the return — they block a PASS verdict downstream, not the early exit.
852
858
 
853
- If survivors remain: batch the confirmed findings for fixing: group findings by file — one file per set of sub-batches, chunked at max 5 findings per sub-batch; never mix two files in one batch. A finding with no `file` field is its own singleton batch. Sub-batches for the SAME file run sequentially (never two Code agents editing the same file concurrently — same-file edits in `parallel()` cause index contention and lost fixes); sub-batches for DISTINCT files run via `parallel()` in staggered chunks of ~5, same pacing bar as the Review spawn path (different code areas — safe per concurrency doctrine). Each Code agent's prompt pins a return contract: `{"status": "fixed"|"blocked", "commitShas": [...], "unresolved": [...]}` — a chunk is FIXED only when `result.status === "fixed"` AND `commitShas` is non-empty AND `result.unresolved` is empty; never decide disposition from status alone. A non-empty `unresolved` list means the agent named work it could not complete — carry the whole chunk into `survivingFindings` rather than guessing which findings the strings map to. `survivingFindings` = findings NOT addressed: fix Code agent dead/failed/blocked/deferred OR committed but left work named in `unresolved`. The fixing Code agent **self-verifies its own fix builds** (build/typecheck per the Code agent's "Long-running commands" discipline). Do **NOT** run Gate 1 or Gate 2 inside the pass (no Validate agent, no Simplify agent, no Scrutinize agent, no Evaluate agent, no Test agent). The engine runs ONE final Gate 1 after the pass exits — see the `gate1_postcode()` cadence (Gate 1 #2).
859
+ If survivors remain: batch the confirmed findings for fixing: group findings by file — one file per set of sub-batches, chunked at max 5 findings per sub-batch; never mix two files in one batch. A finding with no `file` field is its own singleton batch. Sub-batches for the SAME file run sequentially (never two Code agents editing the same file concurrently — same-file edits in `parallel()` cause index contention and lost fixes); sub-batches for DISTINCT files run via `parallel()` in staggered chunks of ~5, same pacing bar as the Review spawn path (different code areas — safe per concurrency doctrine). Each Code agent's prompt pins a return contract: `{"status": "fixed"|"blocked", "commitShas": [...], "unresolved": [...]}` — a chunk is FIXED only when `result.status === "fixed"` AND `commitShas` is non-empty AND `result.unresolved` is empty; never decide disposition from status alone. A non-empty `unresolved` list means the agent named work it could not complete — carry the whole chunk into `survivingFindings` rather than guessing which findings the strings map to. `survivingFindings` = findings NOT addressed: fix Code agent dead/failed/blocked/deferred OR committed but left work named in `unresolved`. The fixing Code agent **self-verifies its own fix builds** (build/typecheck per the Code agent's Running commands block, once over its whole batch). Do **NOT** run Gate 1 or Gate 2 inside the pass (no Validate agent, no Simplify agent, no Scrutinize agent, no Evaluate agent, no Test agent). The engine runs ONE final Gate 1 after the pass exits — see the `gate1_postcode()` cadence (Gate 1 #2).
854
860
 
855
861
  ### Engine invariants (non-negotiable)
856
862
 
857
863
  1. **Code is written ONLY by Code agents.** No other agent type writes code — not Review agent, not Evaluate agent.
858
864
  2. **Findings are verified before any fix is written.** The adversarial verification step is not optional; unverified findings are not passed to the Code agent.
859
- 3. **All written code passes Gate 1.** No code merge, commit, or handoff before Validate agent + Simplify agent + Scrutinize agent (in that order).
865
+ 3. **All written code passes Gate 1.** No code merge, commit, or handoff before Simplify agent + Scrutinize agent + Validate agent (in that order — Validate last, so its one run covers the commits of the two before it).
860
866
  4. **Gate 2 runs once, at implementation acceptance.** It does not re-run after review-fixes.
861
867
  5. **NEVER auto-merge to main or master.** All merges target the integration branch. The user merges to main themselves.
862
868
  6. **No unauthorized tracker or remote side-effects.** Sub-agents NEVER create issues/PRs on the tracker, comment on them, or push beyond the ticket-authorized branch unless the ticket, plan, or user explicitly authorizes that exact action. This applies to whatever tracker is resolved, not to one vendor. Proposed follow-ups go in the run report.
@@ -898,7 +904,7 @@ Each ticket engine run returns a structured result. The Synthesize agent or the
898
904
  },
899
905
  "escalations": [
900
906
  {
901
- "type": "merge-conflict | gate2-fail | validation-exhausted | ambiguous-resolution | review-coverage-incomplete | dependency-blocked | engine-crash | ticket-link-missing | branch-missing",
907
+ "type": "merge-conflict | gate2-fail | validation-exhausted | scrutiny-blocked | ambiguous-resolution | review-coverage-incomplete | dependency-blocked | engine-crash | ticket-link-missing | branch-missing",
902
908
  "description": "string"
903
909
  }
904
910
  ],
@@ -957,8 +963,8 @@ For each ready ticket (sequentially by default; parallel only past the §7.1 bar
957
963
  - Branch setup: the engine's setup-task creates the ticket's branch off integration HEAD at ready-time (so it already contains merged deps); every later phase, and the merge, uses the branch setup-task created, and a setup-task that reports none stops the ticket before implementing
958
964
  - Run the single-ticket engine inside a try/catch — one ticket's crash/stall never kills the wave; catch the exception, quarantine that ticket, and continue with the remaining ready set
959
965
  - The engine gets the ticket's own reference, the one the pre-fetch printed, as its setup-task input — never the wave's tracking issue
960
- - On engine PASS or UNVERIFIED: merge to integration branch, run Validate agent (build + test)
961
- - Merge FAIL (build red after merge): quarantine ticket, mark as escalated, continue
966
+ - On engine PASS or UNVERIFIED: merge to the integration branch locally through Git (no push, no build or test), then the workflow spawns the Validate agent (build + test) over the merge. The merge counts as kept only when that Validate returns PASS
967
+ - Validate FAIL or no PASS (build red after merge): the workflow spawns Git to undo the merge, quarantines the ticket, marks it as escalated, and continues. If the undo is refused, the wave stops taking merges and rounds
962
968
  - On any other verdict (PARTIAL, FAIL, ESCALATED) or none: quarantine ticket, do not block independent siblings
963
969
 
964
970
  **Cascade quarantine:** when a ticket is quarantined for any reason (Gate-1 exhausted, engine crash/stall, build-red after merge, review coverage incomplete after retry), the quarantine cascades to its direct and transitive dependents — each is marked blocked with the named reason, naming the blocker by its `{ISSUE_REF}` (e.g. "blocked: depends on {ISSUE_REF} which failed Gate-1"). Independent siblings are never affected. The quarantined list is injected into every subsequent Design agent reader prompt so the reader never schedules dependents of failed tickets.
@@ -985,7 +991,9 @@ MAX_ROUNDS = LLM judgment based on ticket count (heuristic: ticket_count * 2 + 5
985
991
 
986
992
  **Parallel independent tickets:** each gets its own `git worktree add` + durable branch managed by the Git agent. Use explicit `git worktree add` — NOT the Workflow tool's ephemeral `isolation:'worktree'`. The branch must persist across implement → review → resolve → merge stages; ephemeral worktrees are gone when the agent call ends.
987
993
 
988
- **Post-merge validation:** after EVERY merge into the integration branch, run Validate agent (build + test). A red build immediately after merge is the cheapest possible conflict detector. Red → quarantine the merged result + escalate.
994
+ **Post-merge validation:** after EVERY merge into the integration branch, the workflow itself spawns the Validate agent (build + test) over the integration branch; Git runs no build or test. A red build immediately after merge is the cheapest possible conflict detector. It always runs — the merge reports `treeEqual` (whether the merge commit's tree equals the ticket branch head's tree) and the wave row records it, but a true value never skips the Validate.
995
+
996
+ **Undo of a red merge:** on FAIL or no PASS verdict the workflow spawns Git to undo the merge. The undo runs only when no remote branch contains the merge commit and the integration HEAD still equals it, as `git reset --keep {mergeSha}^1` on the integration branch — never `--hard`, never a push. It returns `{"undone": true}` or `{"undone": false, "reason": "..."}`. Undone → the row reads `merged: false` and the ticket is quarantined ("post-merge build red; merge undone"). Not undone → the wave stops taking merges and rounds, and its report names the red integration HEAD.
989
997
 
990
998
  **Commit discipline:** Git agent creates atomic commits per logical change, conventional-commit format, on the ticket branch before merge.
991
999
 
@@ -996,13 +1004,13 @@ Two parallel sibling tickets can produce real git conflicts. The resolution is *
996
1004
  **Resolution procedure:**
997
1005
 
998
1006
  1. Git agent detects the conflict and reports the conflicting files + sections
999
- 2. Spawn a Code agent with FULL intent context:
1007
+ 2. Spawn a Code agent whose prompt opens with `OPERATION: implement` and carries `COMPLIANCE_FRAMEWORKS`, with FULL intent context:
1000
1008
  - Both ticket descriptions and plans
1001
1009
  - The conflicting diff sections (both sides)
1002
1010
  - Relevant ADRs from DECISIONS_CONTEXT (loaded by main model before authoring)
1003
1011
  3. Code agent resolves to PRESERVE BOTH INTENTS — the resolution must honor what both tickets were trying to achieve
1004
1012
  4. If the correct resolution is NOT UNAMBIGUOUS from the intent context: **do NOT guess** → quarantine + escalate
1005
- 5. After any resolution: Validate agent (build + test) immediately
1013
+ 5. After any resolution: the post-merge Validate covers it — the workflow spawns it after the merge, and no Validate is requested from Git
1006
1014
 
1007
1015
  **Conservative-or-escalate is absolute.** An LLM silently guessing a wrong merge is the highest-danger failure mode in the whole design. When in doubt: quarantine + surface in report. The user re-runs (resume) with the escalated context.
1008
1016
 
@@ -1017,7 +1025,7 @@ A workflow cannot pause mid-run (F4). "Escalate" means: quarantine-and-continue
1017
1025
  - Ticket engine FAIL after max retries (Gate 1 exhausted)
1018
1026
  - Gate 2 FAIL after max retries (Evaluate agent/Test agent not satisfied)
1019
1027
  - Circular dependency detected (all remaining tickets blocked on each other)
1020
- - Build red after merge (Validate agent fails post-merge)
1028
+ - Build red after merge (the post-merge Validate agent fails: the merge is undone and the ticket quarantined; a refused undo halts the wave)
1021
1029
  - Review coverage incomplete after retry (a focus area failed to produce a live Review agent result after the retry)
1022
1030
  - Ticket engine crash/stall (unrecoverable exception or watchdog kill — quarantine cascades to dependents)
1023
1031
  - No ticket link while issues are required (the engine stops before implementing)
@@ -1035,7 +1043,7 @@ A workflow cannot pause mid-run (F4). "Escalate" means: quarantine-and-continue
1035
1043
 
1036
1044
  **Wave workflow structure (author after the SINGLE engine blocks above):**
1037
1045
 
1038
- The wave workflow uses the same phases as SINGLE but wraps them in a wave loop. The integration branch is `wave/<initiative>` — the initiative slug, referenced below as `{slug}` — (or the user's current branch). Each ticket works on the branch setup-task created. The Git agent manages worktrees for parallel-eligible tickets. After every merge: Validate agent (build + test). Escalations accumulate in a list; the final report lists all of them.
1046
+ The wave workflow uses the same phases as SINGLE but wraps them in a wave loop. The integration branch is `wave/<initiative>` — the initiative slug, referenced below as `{slug}` — (or the user's current branch). Each ticket works on the branch setup-task created. The Git agent manages worktrees for parallel-eligible tickets. After every merge the workflow spawns the Validate agent (build + test) and undoes a red merge. Escalations accumulate in a list; the final report lists all of them.
1039
1047
 
1040
1048
  Wave skeleton — compact reference (see `wave_loop()` doctrine for full semantics):
1041
1049
 
@@ -1046,11 +1054,12 @@ Wave skeleton — compact reference (see `wave_loop()` doctrine for full semanti
1046
1054
  const WAVE_TICKETS = [...remainingTickets]; // the pre-fetch's refs, in input order — the wave block's row order
1047
1055
  const results = {}; // ticketId → its row of the workflow's return; a ticket with no entry never ran
1048
1056
  // A row from an engine result: the fields step 3 after the workflow renders
1049
- const rowOf = (ticketId, r, merged) => ({ ticket: ticketId, ran: true, verdict: r?.verdict || r?.overallVerdict || null, merged, issuePrLink: r?.issuePrLink || "(none)", evaluateVerdict: r?.gate2?.evaluateVerdict, testVerdict: r?.gate2?.testVerdict, surviving: r?.survivingFindings?.length, coverageComplete: r?.reviewCoverage?.complete });
1057
+ const rowOf = (ticketId, r, merged, treeEqual = null) => ({ ticket: ticketId, ran: true, verdict: r?.verdict || r?.overallVerdict || null, merged, treeEqual, issuePrLink: r?.issuePrLink || "(none)", evaluateVerdict: r?.gate2?.evaluateVerdict, testVerdict: r?.gate2?.testVerdict, surviving: r?.survivingFindings?.length, coverageComplete: r?.reviewCoverage?.complete });
1050
1058
  const MAX_ROUNDS = Math.max(10, remainingTickets.length * 2 + 5); // heuristic; always finite
1051
- const waveState = { quarantined: [], round: 0 };
1059
+ const MERGE_SHA = /^[0-9a-f]{40}$/; // a merge commit as git prints it: the only Git-reported value that reaches a later prompt
1060
+ const waveState = { quarantined: [], round: 0, halted: null };
1052
1061
  let reAskedThisDeadlock = false; // re-ask guard: re-ask once on empty ready-set, then escalate
1053
- while (remainingTickets.length > 0 && waveState.round < MAX_ROUNDS) {
1062
+ while (remainingTickets.length > 0 && waveState.round < MAX_ROUNDS && waveState.halted === null) {
1054
1063
  waveState.round++;
1055
1064
 
1056
1065
  // Design agent reader (opus) applies vacuous-truth rule: a ticket with no unmet dependencies is ALWAYS ready.
@@ -1083,10 +1092,36 @@ Return: {"ready": [...ticket-ids], "blocked": [{"ticket": "id", "namedBlocker":
1083
1092
  // Check both verdict (engine_output_schema) and overallVerdict (SINGLE skeleton alias).
1084
1093
  // PASS and UNVERIFIED merge; PARTIAL, FAIL, ESCALATED (the ticket-link and branch stops included) or no verdict quarantine.
1085
1094
  if (["PASS", "UNVERIFIED"].includes(engineResult.verdict || engineResult.overallVerdict)) {
1086
- const merge = await agent(`Merge ${engineResult.branch} to ${INTEGRATION_BRANCH}. Include ticket ID ${ticketId} in the merge commit message. Run Validate agent (build + test) after merge.
1087
- Return: {"merged": true} — or {"merged": false, "reason": "<why>"} when the merge or the post-merge build failed and the merge was not kept.`, { agentType: "Git" });
1088
- results[ticketId] = rowOf(ticketId, engineResult, merge?.merged === true);
1089
- if (merge?.merged !== true) waveState.quarantined.push({ ticket: ticketId, reason: merge?.reason || "merge or post-merge build failed" });
1095
+ // Git merges locally and validates nothing: the workflow owns the post-merge Validate (D-WAVE-MERGE-VALIDATE).
1096
+ const merge = await agent(`Merge ${engineResult.branch} to ${INTEGRATION_BRANCH} locally. Include ticket ID ${ticketId} in the merge commit message. Do not push, and run no build or test.
1097
+ Return: {"merged": true, "mergeSha": "<40-hex merge commit>", "treeEqual": <true when mergeSha^{tree} equals the tree of ${engineResult.branch}'s head>} — or {"merged": false, "reason": "<why>"} when no merge is kept.`, { agentType: "Git" });
1098
+ let kept = false;
1099
+ let reason = merge?.reason || "merge failed";
1100
+ if (merge?.merged === true && !MERGE_SHA.test(String(merge.mergeSha))) {
1101
+ // A merge may sit on the integration branch with no SHA to validate or undo it by: stop taking merges.
1102
+ waveState.halted = { ticket: ticketId, mergeSha: null, reason: "merge reported without a 40-hex mergeSha; the integration branch may hold an unvalidated merge" };
1103
+ reason = waveState.halted.reason;
1104
+ } else if (merge?.merged === true) {
1105
+ // A merge counts as kept only when this Validate returns PASS. It always runs: treeEqual is recorded, never a reason to skip.
1106
+ const post = await agent(`Run build, typecheck, lint, and tests on ${INTEGRATION_BRANCH} after merging ${engineResult.branch} (merge commit ${merge.mergeSha}).
1107
+ Follow your Running commands block for every build and test command.
1108
+ Return: {"verdict": "PASS" | "FAIL", "details": "..."}`, { agentType: "Validate" });
1109
+ if (post?.verdict === "PASS") {
1110
+ kept = true;
1111
+ } else {
1112
+ const undo = await agent(`Undo the merge ${merge.mergeSha} on ${INTEGRATION_BRANCH}, only when no remote branch contains ${merge.mergeSha} and the HEAD of ${INTEGRATION_BRANCH} still equals ${merge.mergeSha}. On the integration branch run exactly this one command and no other that changes a branch or a remote: git reset --keep ${merge.mergeSha}^1
1113
+ Return: {"undone": true} — or {"undone": false, "reason": "<why>"} when you did not undo it.`, { agentType: "Git" });
1114
+ if (undo?.undone === true) {
1115
+ reason = "post-merge build red; merge undone";
1116
+ } else {
1117
+ // A merge that stays red would poison every later ticket: stop taking merges and rounds, and name the red HEAD.
1118
+ reason = `post-merge build red; merge not undone (${undo?.reason || "no reason given"})`;
1119
+ waveState.halted = { ticket: ticketId, mergeSha: merge.mergeSha, reason };
1120
+ }
1121
+ }
1122
+ }
1123
+ results[ticketId] = rowOf(ticketId, engineResult, kept, merge?.merged === true && typeof merge.treeEqual === "boolean" ? merge.treeEqual : null);
1124
+ if (!kept) waveState.quarantined.push({ ticket: ticketId, reason });
1090
1125
  } else {
1091
1126
  results[ticketId] = rowOf(ticketId, engineResult, false);
1092
1127
  waveState.quarantined.push({ ticket: ticketId, reason: engineResult.escalations?.[0]?.description || "engine fail/escalated" });
@@ -1097,11 +1132,12 @@ Return: {"merged": true} — or {"merged": false, "reason": "<why>"} when the me
1097
1132
  waveState.quarantined.push({ ticket: ticketId, reason: `engine crash: ${String(err)}` });
1098
1133
  }
1099
1134
  remainingTickets = remainingTickets.filter(t => t !== ticketId);
1135
+ if (waveState.halted !== null) break; // a red integration HEAD takes no further merge in this round
1100
1136
  }
1101
1137
  }
1102
1138
 
1103
1139
  // After the wave report is written: one row per wave ticket, in input order. No entry ⇒ never ran (cascade, deadlock, MAX_ROUNDS).
1104
- return { tickets: WAVE_TICKETS.map(t => results[t] || { ticket: t, ran: false, verdict: null, merged: false, issuePrLink: "(none)" }), quarantined: waveState.quarantined };
1140
+ return { tickets: WAVE_TICKETS.map(t => results[t] || { ticket: t, ran: false, verdict: null, merged: false, treeEqual: null, issuePrLink: "(none)" }), quarantined: waveState.quarantined, halted: waveState.halted };
1105
1141
  ```
1106
1142
 
1107
1143
  ---
@@ -1125,6 +1161,7 @@ A workflow cannot pause mid-run. After the build/wave workflow returns, you (the
1125
1161
  In WAVE mode, if no tracking-issue number was resolved in Pre-authoring step 5: state `TRACEABILITY: DEGRADED (no tracking issue for this run)` in the run summary and skip — never skip silently.
1126
1162
  3. **Compose the wave PR inputs** (WAVE mode only — skip this step entirely in SINGLE mode). The workflow returned `tickets`: one entry per wave ticket, in its input order, each `{ticket, ran, verdict, merged, issuePrLink, evaluateVerdict, testVerdict, surviving, coverageComplete}`. Every value in it is agent-reported and treated as untrusted: nothing below is repaired, and no reference is ever composed from a number.
1127
1163
  - **Branch check.** Run `git -C "{integration worktree root}" branch --show-current`. Only a name matching `^wave/[a-z0-9][a-z0-9-]{0,59}$` opens a wave PR, and its part after `wave/` is the `{slug}` below. Any other name: record `TRACEABILITY: DEGRADED (not a wave branch)` and go to step 4 with no wave PR.
1164
+ - **Halted** (the workflow returned a non-null `halted`: a post-merge build it could not undo, or a merge reported with no SHA): the integration HEAD may be red. Record `Wave PR: skipped (integration HEAD red, wave halted at <halted.ticket>)`, name `halted.mergeSha` (or `none reported`) and `halted.reason` as the red integration HEAD in the run summary, and go to step 4 with no wave PR.
1128
1165
  - **Nothing merged** (no entry has `merged: true`): record `Wave PR: skipped (nothing merged)` and go to step 4.
1129
1166
  - Otherwise compose (a) and then (b).
1130
1167