@mmerterden/multi-agent-pipeline 16.18.0 → 16.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (238) hide show
  1. package/CHANGELOG.md +68 -0
  2. package/README.md +5 -5
  3. package/README.tr.md +2 -2
  4. package/docs/FIGMA_PIPELINE.md +1 -1
  5. package/docs/adr/0001-three-model-triage.md +4 -2
  6. package/docs/architecture.md +2 -2
  7. package/docs/ecosystem.md +13 -11
  8. package/docs/features.md +2 -2
  9. package/index.js +1 -1
  10. package/install/_codex-agents.mjs +2 -2
  11. package/install/_common.mjs +25 -1
  12. package/install/_dev-only-files.mjs +3 -2
  13. package/install/_mcp-register.mjs +4 -3
  14. package/install/_plugin-skills.mjs +1 -3
  15. package/install/copilot.mjs +18 -9
  16. package/install/index.mjs +2 -4
  17. package/install/templates/copilot-instructions.md +7 -7
  18. package/package.json +4 -3
  19. package/pipeline/agents/android-architect.md +1 -0
  20. package/pipeline/agents/backend-architect.md +1 -0
  21. package/pipeline/agents/code-reviewer.md +1 -0
  22. package/pipeline/agents/dev-critic.md +2 -1
  23. package/pipeline/agents/explorer.md +1 -0
  24. package/pipeline/agents/ios-architect.md +1 -0
  25. package/pipeline/agents/security-auditor.md +1 -0
  26. package/pipeline/agents/task-clarifier.md +1 -0
  27. package/pipeline/claude-md-template.md +2 -2
  28. package/pipeline/commands/multi-agent/SKILL.md +5 -5
  29. package/pipeline/commands/multi-agent/complaint-analysis/SKILL.md +1 -1
  30. package/pipeline/commands/multi-agent/design-check/SKILL.md +1 -1
  31. package/pipeline/commands/multi-agent/help/SKILL.md +2 -2
  32. package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +1 -1
  33. package/pipeline/commands/multi-agent/manual-test/SKILL.md +5 -1
  34. package/pipeline/commands/multi-agent/resume/SKILL.md +1 -0
  35. package/pipeline/commands/multi-agent/review/SKILL.md +27 -14
  36. package/pipeline/commands/multi-agent/review-analysis/SKILL.md +1 -1
  37. package/pipeline/commands/multi-agent/store-ready/SKILL.md +24 -2
  38. package/pipeline/commands/multi-agent/sync/SKILL.md +2 -2
  39. package/pipeline/commands/sim-test.md +64 -20
  40. package/pipeline/lib/credential-inventory.sh +15 -2
  41. package/pipeline/lib/credential-store-resolver.sh +14 -4
  42. package/pipeline/lib/credential-store.sh +8 -2
  43. package/pipeline/lib/extract-conventions.sh +1 -14
  44. package/pipeline/lib/fetch-confluence.sh +12 -4
  45. package/pipeline/lib/fetch-crashlytics.sh +11 -8
  46. package/pipeline/lib/fetch-document.sh +1 -1
  47. package/pipeline/lib/fetch-figma-annotations.sh +7 -5
  48. package/pipeline/lib/fetch-fortify.sh +5 -3
  49. package/pipeline/lib/fetch-graylog.sh +5 -3
  50. package/pipeline/lib/figma-mcp-refresh.sh +1 -1
  51. package/pipeline/lib/figma-screenshot.sh +27 -24
  52. package/pipeline/lib/figma-token.sh +8 -4
  53. package/pipeline/lib/issue-fetcher.sh +0 -1
  54. package/pipeline/lib/jira-publish.sh +7 -5
  55. package/pipeline/lib/md2confluence-v3.py +13 -7
  56. package/pipeline/lib/multi-repo-pipeline.sh +18 -8
  57. package/pipeline/lib/plan-todos.sh +11 -0
  58. package/pipeline/lib/post-pr-review.sh +9 -2
  59. package/pipeline/lib/repo-cache.sh +18 -10
  60. package/pipeline/lib/review-watch.sh +60 -14
  61. package/pipeline/lib/shadow-git.sh +8 -4
  62. package/pipeline/lib/vercel-deploy.sh +2 -2
  63. package/pipeline/multi-agent-refs/_dev-context.md +5 -2
  64. package/pipeline/multi-agent-refs/analysis/locked.md +4 -4
  65. package/pipeline/multi-agent-refs/analysis/render.md +1 -1
  66. package/pipeline/multi-agent-refs/channels/pr.md +22 -4
  67. package/pipeline/multi-agent-refs/cross-cli-contract.md +1 -1
  68. package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +15 -2
  69. package/pipeline/multi-agent-refs/features/review-delta.md +89 -0
  70. package/pipeline/multi-agent-refs/features/scope-check.md +41 -0
  71. package/pipeline/multi-agent-refs/features/verify-by-test.md +6 -5
  72. package/pipeline/multi-agent-refs/features/worktree-finalize.md +1 -1
  73. package/pipeline/multi-agent-refs/outside-the-pipeline.md +6 -6
  74. package/pipeline/multi-agent-refs/payload-contracts.md +1 -1
  75. package/pipeline/multi-agent-refs/phases/log-format.md +1 -1
  76. package/pipeline/multi-agent-refs/phases/modes.md +1 -1
  77. package/pipeline/multi-agent-refs/phases/phase-0-init.md +4 -2
  78. package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +3 -3
  79. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +5 -5
  80. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +17 -2
  81. package/pipeline/multi-agent-refs/phases/phase-4-review.md +67 -24
  82. package/pipeline/multi-agent-refs/phases/phase-5-test.md +10 -0
  83. package/pipeline/multi-agent-refs/phases/phase-6-commit.md +2 -0
  84. package/pipeline/multi-agent-refs/phases/phase-7-report.md +4 -2
  85. package/pipeline/multi-agent-refs/rules.md +2 -2
  86. package/pipeline/multi-agent-refs/tracker-contract.md +1 -1
  87. package/pipeline/rules/figma-pipeline.md +18 -72
  88. package/pipeline/rules/outside-the-pipeline.md +4 -3
  89. package/pipeline/schemas/agent-state.schema.json +130 -1
  90. package/pipeline/schemas/dev-critic-output.schema.json +5 -0
  91. package/pipeline/schemas/prefs.schema.json +47 -0
  92. package/pipeline/schemas/reviewer-output.schema.json +7 -2
  93. package/pipeline/schemas/scope-check.schema.json +55 -0
  94. package/pipeline/schemas/token-budget.json +3 -3
  95. package/pipeline/schemas/triage-output.schema.json +12 -2
  96. package/pipeline/scripts/README.md +3 -2
  97. package/pipeline/scripts/_fingerprint.mjs +173 -0
  98. package/pipeline/scripts/_stack-routing.mjs +1 -1
  99. package/pipeline/scripts/agent-guard.py +102 -21
  100. package/pipeline/scripts/anonymize-findings.mjs +7 -6
  101. package/pipeline/scripts/build-skills-index.mjs +14 -3
  102. package/pipeline/scripts/cost-budget-check.mjs +5 -3
  103. package/pipeline/scripts/cost-lib.sh +0 -15
  104. package/pipeline/scripts/diff-explain.mjs +22 -12
  105. package/pipeline/scripts/evidence-gate.mjs +73 -5
  106. package/pipeline/scripts/finding-fingerprint.mjs +101 -0
  107. package/pipeline/scripts/gc-refs.sh +6 -2
  108. package/pipeline/scripts/gc-tmp.sh +1 -1
  109. package/pipeline/scripts/gc-worktrees.sh +1 -1
  110. package/pipeline/scripts/gen-mode-dispatch.mjs +3 -3
  111. package/pipeline/scripts/github-ssh-setup.sh +7 -2
  112. package/pipeline/scripts/graph-build.mjs +2 -2
  113. package/pipeline/scripts/jira-wiki-escape.mjs +2 -1
  114. package/pipeline/scripts/keychain.py +12 -11
  115. package/pipeline/scripts/learning-curve.mjs +1 -1
  116. package/pipeline/scripts/migrate-prefs.mjs +1 -1
  117. package/pipeline/scripts/output-quality-check.sh +3 -1
  118. package/pipeline/scripts/phase-tracker.sh +1 -1
  119. package/pipeline/scripts/phase0-exit-gate.mjs +2 -1
  120. package/pipeline/scripts/plan-coverage-gate.mjs +2 -1
  121. package/pipeline/scripts/pre-commit-check.sh +23 -13
  122. package/pipeline/scripts/prune-logs.sh +1 -1
  123. package/pipeline/scripts/render-agent-log-cost.sh +3 -1
  124. package/pipeline/scripts/render-cost-summary.sh +4 -2
  125. package/pipeline/scripts/render-work-summary.sh +5 -3
  126. package/pipeline/scripts/repo-map.mjs +3 -2
  127. package/pipeline/scripts/review-delta.mjs +217 -0
  128. package/pipeline/scripts/run-metrics.mjs +20 -0
  129. package/pipeline/scripts/scan-skills.sh +6 -2
  130. package/pipeline/scripts/scope-check-gate.mjs +90 -0
  131. package/pipeline/scripts/search-logs.sh +8 -6
  132. package/pipeline/scripts/sign-skills.sh +3 -1
  133. package/pipeline/scripts/smoke-cross-cli-behavior.sh +12 -5
  134. package/pipeline/scripts/triage-memory.mjs +25 -4
  135. package/pipeline/scripts/uninstall.mjs +20 -12
  136. package/pipeline/scripts/update-check.sh +2 -2
  137. package/pipeline/scripts/update-issue-progress.sh +6 -5
  138. package/pipeline/scripts/validate-analysis-doc.mjs +6 -6
  139. package/pipeline/scripts/validate-reviewer.mjs +6 -0
  140. package/pipeline/scripts/validate-triage.mjs +20 -0
  141. package/pipeline/scripts/verify-skills.sh +3 -1
  142. package/pipeline/scripts/worktree-finalize.sh +22 -10
  143. package/pipeline/skills/.skill-manifest.json +81 -57
  144. package/pipeline/skills/.skills-index.json +19 -19
  145. package/pipeline/skills/shared/README.md +8 -8
  146. package/pipeline/skills/shared/core/multi-agent/SKILL.md +7 -7
  147. package/pipeline/skills/shared/core/multi-agent-design-check/SKILL.md +1 -1
  148. package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +2 -2
  149. package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +1 -1
  150. package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +2 -2
  151. package/pipeline/skills/shared/core/multi-agent-review/SKILL.md +23 -14
  152. package/pipeline/skills/shared/core/multi-agent-store-ready/SKILL.md +5 -0
  153. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +2 -2
  154. package/pipeline/skills/shared/external/NOTICE-dimillian-skills.md +56 -0
  155. package/pipeline/skills/shared/external/accessibility-compliance-accessibility-audit/SKILL.md +0 -4
  156. package/pipeline/skills/shared/external/api-patterns/SKILL.md +12 -24
  157. package/pipeline/skills/shared/external/app-store-changelog/references/release-notes-guidelines.md +34 -0
  158. package/pipeline/skills/shared/external/app-store-changelog/scripts/collect_release_changes.sh +33 -0
  159. package/pipeline/skills/shared/external/architecture/SKILL.md +7 -9
  160. package/pipeline/skills/shared/external/debugging-strategies/SKILL.md +0 -4
  161. package/pipeline/skills/shared/external/fastapi-pro/SKILL.md +0 -1
  162. package/pipeline/skills/shared/external/github-actions-templates/SKILL.md +0 -14
  163. package/pipeline/skills/shared/external/hig-components-content/SKILL.md +13 -13
  164. package/pipeline/skills/shared/external/hig-components-layout/SKILL.md +16 -16
  165. package/pipeline/skills/shared/external/hig-components-status/SKILL.md +6 -6
  166. package/pipeline/skills/shared/external/hig-components-system/SKILL.md +13 -13
  167. package/pipeline/skills/shared/external/hig-foundations/SKILL.md +23 -23
  168. package/pipeline/skills/shared/external/hig-inputs/SKILL.md +18 -18
  169. package/pipeline/skills/shared/external/hig-patterns/SKILL.md +30 -30
  170. package/pipeline/skills/shared/external/hig-platforms/SKILL.md +11 -11
  171. package/pipeline/skills/shared/external/hig-technologies/SKILL.md +33 -33
  172. package/pipeline/skills/shared/external/ios-coding-standard/references/STANDARD.md +52 -52
  173. package/pipeline/skills/shared/external/ios-coding-standard/references/lint-local.sh +1 -1
  174. package/pipeline/skills/shared/external/ios-coding-standard/references/rules.yml +11 -11
  175. package/pipeline/skills/shared/external/ios-developer/SKILL.md +0 -1
  176. package/pipeline/skills/shared/external/ios-module-structure/SKILL.md +7 -3
  177. package/pipeline/skills/shared/external/localization-reuse-map/SKILL.md +9 -15
  178. package/pipeline/skills/shared/external/macos-spm-app-packaging/SKILL.md +0 -5
  179. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/Package.swift +17 -0
  180. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/Sources/MyApp/Resources/.keep +0 -0
  181. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/Sources/MyApp/main.swift +11 -0
  182. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/version.env +2 -0
  183. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/build_icon.sh +49 -0
  184. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/compile_and_run.sh +63 -0
  185. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/launch.sh +28 -0
  186. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/make_appcast.sh +82 -0
  187. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/package_app.sh +206 -0
  188. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/setup_dev_signing.sh +52 -0
  189. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/sign-and-notarize.sh +52 -0
  190. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/version.env +2 -0
  191. package/pipeline/skills/shared/external/macos-spm-app-packaging/references/packaging.md +17 -0
  192. package/pipeline/skills/shared/external/macos-spm-app-packaging/references/release.md +32 -0
  193. package/pipeline/skills/shared/external/macos-spm-app-packaging/references/scaffold.md +79 -0
  194. package/pipeline/skills/shared/external/monorepo-architect/SKILL.md +0 -1
  195. package/pipeline/skills/shared/external/nodejs-backend-patterns/SKILL.md +0 -4
  196. package/pipeline/skills/shared/external/swift-concurrency-expert/references/approachable-concurrency.md +63 -0
  197. package/pipeline/skills/shared/external/swift-concurrency-expert/references/swift-6-2-concurrency.md +272 -0
  198. package/pipeline/skills/shared/external/swift-concurrency-expert/references/swiftui-concurrency-tour-wwdc.md +33 -0
  199. package/pipeline/skills/shared/external/swiftui-performance-audit/references/code-smells.md +150 -0
  200. package/pipeline/skills/shared/external/swiftui-performance-audit/references/demystify-swiftui-performance-wwdc23.md +46 -0
  201. package/pipeline/skills/shared/external/swiftui-performance-audit/references/optimizing-swiftui-performance-instruments.md +29 -0
  202. package/pipeline/skills/shared/external/swiftui-performance-audit/references/profiling-intake.md +44 -0
  203. package/pipeline/skills/shared/external/swiftui-performance-audit/references/report-template.md +47 -0
  204. package/pipeline/skills/shared/external/swiftui-performance-audit/references/understanding-hangs-in-your-app.md +33 -0
  205. package/pipeline/skills/shared/external/swiftui-performance-audit/references/understanding-improving-swiftui-performance.md +52 -0
  206. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/app-wiring.md +201 -0
  207. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/async-state.md +96 -0
  208. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/components-index.md +46 -0
  209. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/controls.md +57 -0
  210. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/deeplinks.md +66 -0
  211. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/focus.md +90 -0
  212. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/form.md +97 -0
  213. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/grids.md +71 -0
  214. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/haptics.md +71 -0
  215. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/input-toolbar.md +51 -0
  216. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/lightweight-clients.md +93 -0
  217. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/list.md +86 -0
  218. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/loading-placeholders.md +38 -0
  219. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/macos-settings.md +71 -0
  220. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/matched-transitions.md +59 -0
  221. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/media.md +73 -0
  222. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/menu-bar.md +101 -0
  223. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/navigationstack.md +159 -0
  224. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/overlay.md +45 -0
  225. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/performance.md +62 -0
  226. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/previews.md +48 -0
  227. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/scroll-reveal.md +133 -0
  228. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/scrollview.md +87 -0
  229. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/searchable.md +71 -0
  230. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/sheets.md +155 -0
  231. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/split-views.md +72 -0
  232. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/tabview.md +114 -0
  233. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/theming.md +71 -0
  234. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/title-menus.md +93 -0
  235. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/top-bar.md +49 -0
  236. package/pipeline/skills/shared/external/swiftui-view-refactor/references/mv-patterns.md +161 -0
  237. package/pipeline/skills/skills-index.md +8 -8
  238. package/pipeline/skills/shared/external/help-skills/SKILL.md +0 -166
@@ -104,7 +104,7 @@ If changes include UI files (iOS: `*View.swift`, `*Screen.swift`, `*Cell.swift`;
104
104
  - Missing safe area / keyboard avoidance → **important**
105
105
  - Hardcoded colors instead of system/semantic colors → **suggestion**
106
106
 
107
- **iOS - SwiftUI interaction & accessibility conventions.** Gated to changed SwiftUI files. These are rule-registry territory as of v14.0.0, not a list transcribed here: Step 1.78 resolves them from whichever registry declares SwiftUI scope, so the criteria and their severities live in one place instead of drifting between this doc and the skill. Reviewers receive the resolved rule IDs. Native-SwiftUI-first unless the project's `figma-config` `ui.*` declares a custom system, in which case check against that system. Reference skills, when no registry covers the change: `figma-navigation`, `figma-overlays`, `figma-bottom-sheets`, `figma-to-swiftui`.
107
+ **iOS - SwiftUI interaction & accessibility conventions.** Gated to changed SwiftUI files. These are rule-registry territory as of v14.0.0, not a list transcribed here: Step 1.78 resolves them from whichever registry declares SwiftUI scope, so the criteria and their severities live in one place instead of drifting between this doc and the skill. Reviewers receive the resolved rule IDs. Native-SwiftUI-first unless the project's `figma-config` `ui.*` declares a custom system, in which case check against that system. Reference skills, when no registry covers the change: the enabled stack plugin's `navigation`, `overlays` and `bottom-sheets` skills (`ai-ios-toolkit:*` / `ai-android-toolkit:*`), plus `figma-to-swiftui`.
108
108
 
109
109
  **Android - Material Design compliance** (skills: `ai-android-toolkit:compose-components`, `ai-android-toolkit:android-architecture`):
110
110
  - Non-Material3 component when M3 equivalent exists → **suggestion**
@@ -152,6 +152,12 @@ echo "$RISK_FULL" | node $HOME/.claude/scripts/validate-diff-risk.mjs - >/dev/nu
152
152
  RISK_JSON=$([ -n "$RISK_FULL" ] && jq -c '.files |= (sort_by(-.score) | .[:5])' <<< "$RISK_FULL" || echo "")
153
153
  ```
154
154
 
155
+ Persist the totals as `state.diffRisk` (Phase 6 `risk` section, Phase 7, `run-metrics.mjs` read them):
156
+
157
+ ```bash
158
+ [ -n "$RISK_FULL" ] && jq -c '{diffRisk: (.totals + {signals: ([.files[].signals[]?.name] | unique)})}' <<< "$RISK_FULL" | node $HOME/.claude/scripts/write-state.mjs "$STATE_FILE"
159
+ ```
160
+
155
161
  **Signals & weights** (see `$HOME/.claude/schemas/diff-risk.schema.json`):
156
162
 
157
163
  | Signal | Weight | Triggers when |
@@ -179,7 +185,9 @@ On success, emit a single summary metric:
179
185
  $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 review.diff_risk \
180
186
  top_files=$(jq '.files | length' <<< "$RISK_JSON") \
181
187
  max_score=$(jq '.totals.max_score' <<< "$RISK_JSON") \
182
- loc_added=$(jq '.totals.loc_added' <<< "$RISK_JSON")
188
+ loc_added=$(jq '.totals.loc_added' <<< "$RISK_JSON") \
189
+ loc_removed=$(jq '.totals.loc_removed' <<< "$RISK_JSON") \
190
+ files=$(jq '.totals.files' <<< "$RISK_JSON")
183
191
  ```
184
192
 
185
193
  **Opt-out**: `prefs.global.diffRiskAdvisory = false` skips this step entirely (no script invocation, no priority block injection). Default `true` because the cost is bounded and the signal-to-noise has been measured against the golden-task fixture set.
@@ -269,7 +277,7 @@ When `state.figmaAccess.tier === 3` (user-attached screenshot, no Code Connect s
269
277
 
270
278
  Phase 4 sends the same diff to every reviewer and then to triage, so the diff is the dominant token cost. Two measures keep it bounded:
271
279
 
272
- **Shared cache prefix.** Build the reviewer and triage prompts so the large invariant context - the full diff, the `${CRITERIA}` block from Step 1.78, the Phase 1 analysis summary, the Phase 2 plan - is a byte-identical leading block across all dispatches in this iteration. Only the per-reviewer focus + skill line varies, and it goes AFTER the shared block. `${CRITERIA}` goes in the prefix, identical for every reviewer: subsetting it per reviewer would invalidate the prefix for the whole panel and re-bill the largest block in the phase. Per-reviewer emphasis stays a one-line pointer in the suffix. When the host supports prompt caching, the 2nd/3rd reviewer and the triage call then read that prefix at the discounted cache-read rate instead of re-billing it as fresh input. Forward the host-reported cache-read count as `tokens_cached` per the Token telemetry contract so the saving lands in the cost ledger.
280
+ **Shared cache prefix.** Build the reviewer and triage prompts so the large invariant context - the full diff, the `${CRITERIA}` block from Step 1.78, the Phase 1 analysis summary, the Phase 2 plan - is a byte-identical leading block across all dispatches in this iteration. Only the per-reviewer focus + skill line varies, and it goes AFTER the shared block. `${CRITERIA}` goes in the prefix, identical for every reviewer: subsetting it per reviewer would invalidate the prefix for the whole panel and re-bill the largest block in the phase. Per-reviewer emphasis stays a one-line pointer in the suffix. When the host supports prompt caching, the 2nd/3rd reviewer and the triage call then read that prefix at the discounted cache-read rate instead of re-billing it as fresh input. Forward the host-reported cache-read count as `tokens_cached` per the Token telemetry contract so the saving lands in the cost ledger. The `<scope-self-check>` block and, from iteration 2, the `<previous-round-findings>` block (Step 2.1) close the shared block, after the plan and before the per-reviewer suffix.
273
281
 
274
282
  **Single-repo diff cap.** If the diff exceeds the Phase 4 token allowance (`token-budget.json`), truncate the largest files and append a footer `[truncated - full diff in file://$WORKTREE/.review-diff.txt]`, writing the full diff to that path. Reviewers and triage receive the same capped view + the marker so they can flag "review the full diff manually." Log `review.diff_truncated bytes_dropped=<N>`. (Multi-repo already caps the combined diff at 80% of budget; this is the single-repo equivalent.)
275
283
 
@@ -303,17 +311,19 @@ looks like it ran. The full statement lives in the managed block at `~/.codex/AG
303
311
  Sub-agent delegation itself is authorized by that same managed block; without it Phase 4
304
312
  degrades to a single in-thread review.
305
313
 
306
- **Single-vendor caveat.** Every Codex reviewer is an OpenAI model, so the
307
- cross-vendor disagreement that Claude Code and Copilot CLI get for free is absent.
308
- The diversity budget shifts to reasoning effort and persona focus: Reviewer 1 runs
309
- `xhigh` on security and architecture, Reviewer 2 runs a different model family
310
- member on edge cases, Reviewer 3 runs `medium` on quality. Treat consensus among
311
- them as weaker evidence than the same consensus on a two-vendor host, and say so
312
- in the triage note when all three agree on a borderline finding.
314
+ **Single-vendor caveat.** Every Codex reviewer is an OpenAI model, and every Claude
315
+ Code reviewer is an Anthropic model, so the cross-vendor disagreement that Copilot CLI
316
+ gets for free (GPT-5.4 beside two Claude models) is absent on both. The diversity budget
317
+ shifts to model generation, reasoning effort and persona focus: on Codex Reviewer 1 runs
318
+ `xhigh` on security and architecture, Reviewer 2 runs a different model family member
319
+ on edge cases, Reviewer 3 runs `medium` on quality; on Claude Code the three slots are
320
+ three different Claude tiers. Treat consensus among a single-vendor panel as weaker
321
+ evidence than the same consensus on Copilot CLI, and say so in the triage note when all
322
+ three agree on a borderline finding.
313
323
 
314
324
  Each reviewer inherits the `code-reviewer` agent's focus areas (Security, Architecture, Quality, Performance) and output contract. The orchestrator overrides only the model and the stack-specific skill per-reviewer - no prompt duplication.
315
325
 
316
- **Model override wiring:** `code-reviewer.md` declares `preferredModel: fable`, so Reviewer 1 uses the persona default (Fable 5). Reviewer 2 (`claude-opus-5` on Claude Code, `gpt-5.4` elsewhere) and Reviewer 3 (`claude-sonnet-5`) set `PHASE_MODEL_OVERRIDE=<model>` before dispatch - the orchestrator exports `CLAUDE_CODE_SUBAGENT_MODEL` on Claude Code, or passes `--model` on Copilot CLI. Full precedence rule: `skills/shared/core/multi-agent/SKILL.md#agent-dispatch--per-persona-model-routing-v610`. Fable dispatches are subject to the fallback contract (`$HOME/.claude/multi-agent-refs/features/model-fallback.md`): dispatch-error retry walks `fable -> opus -> sonnet` and budget-ceiling downgrade.
326
+ **Model override wiring:** `code-reviewer.md` declares `preferredModel: fable`, so Reviewer 1 uses the persona default (Fable 5). Reviewer 2 (`claude-opus-5` on Claude Code, `gpt-5.4` elsewhere) and Reviewer 3 (`claude-sonnet-5`) set `PHASE_MODEL_OVERRIDE=<model>` before dispatch - the orchestrator exports `CLAUDE_CODE_SUBAGENT_MODEL` on Claude Code, or passes `--model` on Copilot CLI. Full precedence rule: `skills/shared/core/multi-agent/SKILL.md#agent-dispatch--per-persona-model-routing`. Fable dispatches are subject to the fallback contract (`$HOME/.claude/multi-agent-refs/features/model-fallback.md`): dispatch-error retry walks `fable -> opus -> sonnet` and budget-ceiling downgrade.
317
327
 
318
328
  **Stack-specific skills loaded per reviewer** (from Phase 1 `detectedStack`). All three columns are used on every host; Reviewer 2 reads them as Opus on Claude Code and as GPT-5.4 elsewhere.
319
329
 
@@ -326,7 +336,9 @@ Each reviewer inherits the `code-reviewer` agent's focus areas (Security, Archit
326
336
  | Docker | `ai-backend-toolkit:docker-expert` | `ai-backend-toolkit:docker-expert` | `ai-backend-toolkit:ci-cd-pipelines` |
327
337
  | Generic | `security-review` | `ai-backend-toolkit:clean-code` | `ai-backend-toolkit:clean-code` |
328
338
 
329
- Skills are injected into reviewer prompt context - the reviewer uses them as reference, not as commands.
339
+ ##### 2.1 Previous-round findings (iteration >= 2) and 2.2 scope self-check (every iteration)
340
+
341
+ A reviewer has no memory of the round before, so it rediscovers last round's findings in new words. From iteration 2, render the previous round's accepted blocking/important findings (`.pipeline/triage-round-$((ITERATION-1)).json`, max 40) into a `<previous-round-findings>` block at the end of the shared prefix: a still-present issue is reported with the SAME fingerprint and the current line, a fixed one is omitted, anything new leaves `fingerprint` unset. Every iteration also renders `.pipeline/scope-check.json` (Phase 3 Step 3.7) plus `scope-check-gate.mjs --advisory` output as `<scope-self-check>`: file reasons, unjustified files, and `notDone[]` (never re-raised as findings); a missing record logs `review.scope_check=missing`. Block text and recipes: `$HOME/.claude/multi-agent-refs/features/review-delta.md`.
330
342
 
331
343
  #### Step 2.8 - Visual conformance gate (component / screen work only)
332
344
 
@@ -380,11 +392,14 @@ Step 2 produces N reviewer-output objects (one per dispatched reviewer), each co
380
392
  REVIEWER_FILE="$WORKTREE/.pipeline/reviewer-$N.json"
381
393
  printf '%s' "$REVIEWER_JSON" > "$REVIEWER_FILE"
382
394
  node $HOME/.claude/scripts/validate-reviewer.mjs "$REVIEWER_FILE" \
383
- --criteria "$WORKTREE/.pipeline/criteria-manifest.json"
395
+ --criteria "$WORKTREE/.pipeline/criteria-manifest.json" \
396
+ && node $HOME/.claude/scripts/finding-fingerprint.mjs annotate --in-place "$REVIEWER_FILE"
384
397
  ```
385
398
 
386
399
  Progress line: ` → checking validator validate-reviewer ({reviewer})`
387
400
 
401
+ `finding-fingerprint.mjs` stamps each finding with its cross-round id once the validator passes; an echoed one is kept, and anonymization leaves it intact.
402
+
388
403
  Exit 0 = valid. Exit 2 = contradiction (approved=true with blocking findings) - flip `approved` to `false`, continue. With `--criteria`, exit 1 also covers the conformance checklist: a selected rule ID with no verdict, a verdict for an ID that was never selected, a `conformant` row with no file evidence, or a `violated` row with no matching finding. Those are the four ways a review can look complete without being complete, and the validator is what makes the checklist more than decoration - it is hand-written and does not apply `additionalProperties`, so an unchecked array would otherwise pass. Exit 1 = malformed; gate protocol (fails CLOSED, same handling as the evidence gate): emit the validator stderr + `errors[]` verbatim, attempt ONE self-correction rework (re-invoke that reviewer with the errors quoted, overwrite the file), re-run the validator. If it fails again -> HALT the phase (no merge, no triage). Recovery hint: `ERR: reviewer output failed validate-reviewer.mjs twice. Inspect $REVIEWER_FILE against $HOME/.claude/schemas/reviewer-output.schema.json, then resume with /multi-agent:resume #N.`
389
404
 
390
405
  #### Step 2.5 - Disagreement-round loop (opt-in)
@@ -401,7 +416,7 @@ Exit 0 = valid. Exit 2 = contradiction (approved=true with blocking findings) -
401
416
  - Max one round. Results replace the original outputs.
402
417
  4. Proceed to Step 3 triage with the round-2 outputs.
403
418
 
404
- **Parity contract:** both CLI sides (Claude 2-model, Copilot 3-model) run the round identically. Telemetry emits `review_round_count={1|2}` per reviewer for Phase 7 rollup.
419
+ **Parity contract:** every host (three reviewers each: Claude Code, Copilot CLI, Codex CLI) runs the round identically. Telemetry emits `review_round_count={1|2}` per reviewer for Phase 7 rollup.
405
420
 
406
421
  **Cost ceiling:** rebuttal round consumes ~1× the original Step 2 token budget. Smoke + budget tests treat this as opt-in so the default cost stays the same.
407
422
 
@@ -495,8 +510,9 @@ bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 memory.hit rows=$CITED_COU
495
510
  **Triage prompt skeleton:**
496
511
 
497
512
  ```
498
- You are the Review Triage agent. Two reviewers returned findings on this diff.
513
+ You are the Review Triage agent. Three reviewers (fewer on a single-scope run) returned findings on this diff.
499
514
  Your job: separate signal from noise. Do NOT add new findings. Do NOT re-review code.
515
+ Preserve `fingerprint` verbatim on every finding you carry through; it is the finding's identity across rounds.
500
516
 
501
517
  For each finding, decide:
502
518
  - ACCEPTED: real issue, in scope, must be fixed now
@@ -527,14 +543,19 @@ Return ONLY valid JSON conforming to $HOME/.claude/schemas/triage-output.schema.
527
543
  Run on the persisted file immediately after the triage agent returns, before acting on the verdict; the validator's exit code decides, not the LLM turn:
528
544
 
529
545
  ```bash
530
- TRIAGE_FILE="$WORKTREE/.pipeline/triage.json"
546
+ ITERATION=$(jq '.reviewIterations | length' "$STATE_FILE")
547
+ TRIAGE_FILE="$WORKTREE/.pipeline/triage-round-$ITERATION.json"
531
548
  mkdir -p "$(dirname "$TRIAGE_FILE")"
532
549
  printf '%s' "$TRIAGE_JSON" > "$TRIAGE_FILE"
533
- node $HOME/.claude/scripts/validate-triage.mjs "$TRIAGE_FILE"
550
+ node $HOME/.claude/scripts/validate-triage.mjs "$TRIAGE_FILE" \
551
+ && node $HOME/.claude/scripts/finding-fingerprint.mjs annotate --in-place "$TRIAGE_FILE" \
552
+ && cp "$TRIAGE_FILE" "$WORKTREE/triage-output.json"
534
553
  ```
535
554
 
536
555
  Progress line: ` → checking validator validate-triage`
537
556
 
557
+ One file per round; `triage-output.json` is the latest copy every downstream reader expects (Phase 7, finalize, work-summary, diff-explain). Step 3.7 rewrites `$TRIAGE_FILE`; repeat the `cp` after it.
558
+
538
559
  | Exit | Meaning | Action |
539
560
  | ----- | ---------------------------- | ------------------------------------------------- |
540
561
  | **0** | Valid and clean | Act on triage output as-is |
@@ -567,10 +588,13 @@ emit() { # $1=event $2=model $3=duration $4=tokens_in $5=tokens_out
567
588
  model="$2" duration_ms="$3" tokens_in="$4" tokens_out="$5"
568
589
  }
569
590
  emit review.reviewer_call fable "$R1_DURATION" "$R1_IN" "$R1_OUT" # opus on Copilot CLI
570
- emit review.reviewer_call sonnet "$SONNET_DURATION" "$SONNET_IN" "$SONNET_OUT"
571
591
  # Reviewer 2 is Opus on Claude Code and GPT-5.4 elsewhere:
572
- [ "${CLI_HOST:-claude}" = "copilot" ] && \
573
- emit review.reviewer_call gpt-5.4 "$GPT_DURATION" "$GPT_IN" "$GPT_OUT"
592
+ if [ "${CLI_HOST:-claude}" = "claude" ]; then
593
+ emit review.reviewer_call opus "$R2_DURATION" "$R2_IN" "$R2_OUT"
594
+ else
595
+ emit review.reviewer_call gpt-5.4 "$R2_DURATION" "$R2_IN" "$R2_OUT"
596
+ fi
597
+ emit review.reviewer_call sonnet "$SONNET_DURATION" "$SONNET_IN" "$SONNET_OUT"
574
598
  emit review.triage_call fable "$TRIAGE_DURATION" "$TRIAGE_IN" "$TRIAGE_OUT"
575
599
  bash "$M" "$TASK_ID" 4 review.completed raw_count=$RAW accepted=$ACC \
576
600
  deferred=$DEF rejected=$REJ approved=$APPROVED duration_ms=$DURATION
@@ -584,7 +608,7 @@ Opt-in via `prefs.global.triageCrossCheck.enabled` (default `false`). Sampled ru
584
608
 
585
609
  ##### 3.6 Consensus surfacing (anti-correlation)
586
610
 
587
- **Rationale:** Reviewer 1 (Fable) and Reviewer 3 (Sonnet) are both Anthropic Claude models, so unanimous agreement on a *judgment call* is not independent confirmation - same-family models drift the same way on ambiguous prompts. Treating "both approved" as proof produces false-consensus passes. Triage therefore records a `consensus` block (schema v3.1.0) and surfaces disagreement and unverified agreement to the user rather than burying it.
611
+ **Rationale:** On Claude Code all three reviewers are Anthropic models, on Codex CLI all three are OpenAI, and on Copilot CLI two of three are Anthropic, so unanimous agreement on a *judgment call* is not independent confirmation: same-family models drift the same way on ambiguous prompts, and "all approved" taken as proof produces false-consensus passes. Triage therefore records a `consensus` block (schema v3.1.0) and surfaces disagreement and unverified agreement instead of burying it.
588
612
 
589
613
  After the triage verdict is computed, populate `triage.consensus`:
590
614
 
@@ -602,7 +626,26 @@ After the triage verdict is computed, populate `triage.consensus`:
602
626
 
603
627
  A triage verdict is judgment; a failing repro test is proof. Runs only when `prefs.global.verifyByTest.enabled` is `true` AND `accepted` contains a `blocking` finding; otherwise skip silently. **Full contract (verdict table, cleanup invariant, prompts): `$HOME/.claude/multi-agent-refs/features/verify-by-test.md` - read it before executing this step.**
604
628
 
605
- Compressed flow: dispatch ONE verifier agent (model `verifyByTest.model`, default `sonnet`) for up to `maxFindings` (default 3) accepted blocking findings. Per finding it writes ONE minimal repro test and runs ONLY that test (Phase 3 single-test invocation, build lock, log tee'd to `$WORKTREE/.pipeline/verify-<i>.test.log`). Outcomes: test FAILS as predicted -> `confirmed`, finding stays blocking and the test is KEPT in `redTests[]` as the Phase 3 rework RED test; test PASSES -> `not-reproduced` ONLY if `evidence-gate.mjs --claim test --status passed` exits 0 on the log, finding moves to `deferred[]`, test deleted; compile error / timeout / not unit-testable -> `inconclusive`, judgment verdict stands. Stamp findings with `verification` (schema v3.2.0), persist `state.reviewIterations[-1].verifyByTest = {attempted, confirmed, downgraded, inconclusive, redTests[]}`, recompute `approved`, re-run `validate-triage.mjs` under the 3.2.1 gate. Whole step bounded by `stepTimeoutSec` (default 600); on breach or crash remaining findings keep judgment verdicts - never blocks. Telemetry per 3.4: `review.verify_by_test attempted= confirmed= downgraded= inconclusive= duration_ms=`.
629
+ Compressed flow: dispatch ONE verifier agent (model `verifyByTest.model`, default `sonnet`) for up to `maxFindings` (default 3) accepted blocking findings. Per finding it writes ONE minimal repro test and runs ONLY that test (Phase 3 single-test invocation, build lock, log tee'd to `$WORKTREE/.pipeline/verify-<i>.test.log`). Outcomes: test FAILS as predicted -> `confirmed`, finding stays blocking and the test is KEPT in `redTests[]` as the Phase 3 rework RED test; test PASSES on every one of `verifyByTest.repeatCount` runs (default 3) -> `not-reproduced` ONLY if `evidence-gate.mjs --claim test --status passed` exits 0 on each log; a run that disagrees with the others -> `inconclusive` with `flaky: passed k/N`, finding moves to `deferred[]`, test deleted; compile error / timeout / not unit-testable -> `inconclusive`, judgment verdict stands. Stamp findings with `verification` (schema v3.2.0), persist `state.reviewIterations[-1].verifyByTest = {attempted, confirmed, downgraded, inconclusive, redTests[]}`, recompute `approved`, re-run `validate-triage.mjs` under the 3.2.1 gate. Whole step bounded by `stepTimeoutSec` (default 600); on breach or crash remaining findings keep judgment verdicts - never blocks. Telemetry per 3.4: `review.verify_by_test attempted= confirmed= downgraded= inconclusive= duration_ms=`.
630
+
631
+ ##### 3.8 Cross-round delta + circuit-breaker trigger 2 (iteration >= 2)
632
+
633
+ **Full contract (state merge, telemetry line, picker wording): `$HOME/.claude/multi-agent-refs/features/review-delta.md`.**
634
+
635
+ ```bash
636
+ TRIP=$(jq -r '.global.autopilotCircuitBreaker.identicalFindingCycles // 2' "$PREFS_FILE")
637
+ DELTA_JSON=$(node $HOME/.claude/scripts/review-delta.mjs --rounds-dir "$WORKTREE/.pipeline" --iteration "$ITERATION" --trip-cycles "$TRIP"); DELTA_RC=$?
638
+ ```
639
+
640
+ Merge the JSON into `state.reviewIterations[-1].delta` via `write-state.mjs`; emit `review.delta iteration= new= still_present= resolved= plateau= tripped=`. Progress line: ` → comparing round {N} vs {N-1} (new={n} still={n} resolved={n})`.
641
+
642
+ | Exit | Action |
643
+ |---|---|
644
+ | **0** | Continue to Step 4 (also iteration 1 and a missing previous round). |
645
+ | **3** | Autopilot with `prefs.global.autopilotCircuitBreaker.enabled` (default true): write `state.circuitBreaker = {tripped: true, trigger: 2, detail, checkpoint: {phase: 4, step: "3.8", iteration: N}, trippedAt, counters}`, then the `operations.md` halt protocol with `haltReason="4:circuit-breaker:identical-finding"`. Interactive: show `delta.stillPresent`, ask `Continue rework` / `Escalate to me` / `Accept as deferred`. |
646
+ | **1** | Log `review.delta_skipped reason=invalid`, continue; the delta never blocks on its own failure. |
647
+
648
+ `plateau` is logged, not acted on; trigger 3 (the rework cap) is recorded by the Phase 3 re-entry.
606
649
 
607
650
  #### Step 4 - Consensus + Action (triage-driven)
608
651
 
@@ -610,7 +653,7 @@ If `triage.consensus.verdict` is `split` or `unverified`, surface `consensus.dis
610
653
 
611
654
  Act **only on triage.accepted**:
612
655
 
613
- - **accepted.blocking** → back to Phase 3 (max 3 iterations, with reflection prompt citing only accepted items). When Step 3.7 ran and `state.reviewIterations[-1].verifyByTest.redTests[]` is non-empty, the reflection prompt cites each red test: "a failing repro test already exists at <testRef>; make it green; do not delete or weaken it."
656
+ - **accepted.blocking** → back to Phase 3 (max 3 iterations, with reflection prompt citing only accepted items). The reflection prompt names findings by `fingerprint` and quotes `delta.stillPresent` first, marked `STILL PRESENT after round N-1's fix`. When Step 3.7 ran and `state.reviewIterations[-1].verifyByTest.redTests[]` is non-empty, the reflection prompt cites each red test: "a failing repro test already exists at <testRef>; make it green; do not delete or weaken it."
614
657
  - **accepted.important** → fix and re-review
615
658
  - **accepted.suggestion** → apply if reasonable
616
659
  - **deferred** → append to Phase 7 report as "follow-up items" (do not block)
@@ -631,7 +674,7 @@ Statement shape: the durable rule/root cause, not the symptom - "force-unwrapp
631
674
 
632
675
  Progress line: ` → writing lesson to learnings ledger ({N} entries)`
633
676
 
634
- Log: "Phase 4: Review - raw={N1+N2} accepted={Na} deferred={Nd} rejected={Nr} approved={bool} consensus={verdict}"
677
+ Log: "Phase 4: Review - raw={N1+N2+N3} accepted={Na} deferred={Nd} rejected={Nr} approved={bool} consensus={verdict}"
635
678
 
636
679
  ---
637
680
 
@@ -103,6 +103,16 @@ Tier 1 / Tier 2 records print `screenshotUrl` from the captured evidence (Tier 2
103
103
  bash $HOME/.claude/scripts/phase-tracker.sh render
104
104
  ```
105
105
  The waiting state persists in `tracker-state.json` across the handoff; `/multi-agent:resume-local` and `/multi-agent:manual-test` CONTINUE this state file and never re-init it (`$HOME/.claude/multi-agent-refs/tracker-contract.md` "Continuation runs").
106
+
107
+ **"ok" is a structured result, not a word.** Before "ok" is accepted, the run writes `$WORKTREE/.pipeline/manual-test.json`: one entry per acceptance criterion, the criteria taken from the analysis doc test plan (Section 15 / 20), the plan tasks, and the user's own words in the reply. Every criterion records what was seen; a criterion that was not tried says so with a reason.
108
+ ```json
109
+ {"criteria":[{"spec":"<quote>","source":"analysis 15.2 | plan task 3 | user","observed":"<what was seen>","verdict":"pass|fail|not-tested","reason":"<required when not-tested>","screenshot":"<path or null>"}],"verdict":"passed|failed"}
110
+ ```
111
+ Then gate it:
112
+ ```bash
113
+ node $HOME/.claude/scripts/evidence-gate.mjs --claim manual --status passed --evidence "$WORKTREE/.pipeline/manual-test.json"
114
+ ```
115
+ Exit 1 means the "ok" is not accepted: tell the user which criterion is missing evidence (a `fail` verdict, or `not-tested` without a reason) and wait for the next reply. Exit 0 marks Phase 5 completed with `Result "local test passed (user)"`. The "fix: ..." path below is unchanged.
106
116
  6. If fix needed:
107
117
  - Branch already has WIP commit (from step 2) - changes are safe
108
118
  - **Heal stale admin state first** (same contract as Phase 0 - step 3's
@@ -136,6 +136,8 @@ Branch **deterministically**, no implicit fallback. Read `agent-state.json` and
136
136
 
137
137
  Generate a structured PR description based on task type. The PR body targets **code reviewers** - it should be technical: what changed, why, architecture decisions, how to verify.
138
138
 
139
+ Two inputs are read from state and the worktree before writing, not recalled from the conversation: `$WORKTREE/.pipeline/scope-check.json` (Phase 3 Step 3.7) supplies the `## Changes` bullets from `files[].reason` and the "Follow-ups not done in this PR" list under `## Related` from `notDone[]`; `state.diffRisk.signals` (Phase 4 Step 1.75) decides whether the conditional `## Risk and Security` section is required. When a high-stakes signal is present and the section is missing, this step blocks until it is written; a placeholder answer ("TBD") counts as missing.
140
+
139
141
  **required**: Run all generated text (PR body, commit message) through the `humanizer` skill before posting. This removes AI-generated patterns (inflated language, filler phrases, repetitive structure) and makes the output sound like a developer wrote it.
140
142
 
141
143
  **IMPORTANT - Issue auto-close prevention**: Never use these keywords in PR title, body, or commit messages - they auto-close the issue on merge:
@@ -163,8 +163,10 @@ Skipped sections: when `planTodos.enabled` is false or no `plan.todos[]` was emi
163
163
 
164
164
  ## Review Iterations
165
165
 
166
- | Iteration | Blocking | Important | Suggestion | Decision |
167
- | --------- | -------- | --------- | ---------- | -------- |
166
+ | Iteration | Blocking | Important | Suggestion | Decision | Still present | Resolved | New |
167
+ | --------- | -------- | --------- | ---------- | -------- | ------------- | -------- | --- |
168
+
169
+ (the last three columns come from `reviewIterations[i].delta.counts` (Phase 4 Step 3.8) and read `-` on iteration 1 and on runs that predate the delta; a tripped circuit-breaker is one extra line under the table: `Circuit-breaker tripped: trigger {n}, {detail}`)
168
170
 
169
171
  ## Files Changed
170
172
 
@@ -52,10 +52,10 @@ This is the single source of truth. When a contributor or model is unsure where
52
52
  - **NEVER** commit without passing build (all gates in Phase 4 Step 1 must be green).
53
53
  - **NEVER** commit without passing review (at least one AI reviewer must return `approved: true` with no blocking findings).
54
54
  - **NEVER** skip tests. Every public method, every error path, every edge case.
55
- - **NEVER** delete, rename, or weaken an existing test to get a green run. Existing tests are immutable during a task. A test may change only when the task itself changes the spec that test encodes, and the commit body must name the changed test and the spec change. Deterministic backstop: the `test_lines_removed` diff-risk signal (Phase 4 Step 1.75) flags test files that shrink.
55
+ - **NEVER** delete, rename, or weaken an existing test to get a green run. Existing tests are immutable during a task; one may change only when the task changes the spec it encodes, and the commit body names both. Deterministic backstop: the `test_lines_removed` diff-risk signal (Phase 4 Step 1.75) flags test files that shrink. A pass only on retry is a flake signal, not a pass: Phase 3 repeats new and changed tests (`testStability.repeatCount`, default 3) and logs `test.flake_signal` when runs disagree.
56
56
  - **Follow existing code style and conventions.** Read neighbor files before writing new ones - match naming, structure, import order.
57
57
  - **Use design tokens, no magic numbers.** `16` → `.Spacing.spacing16`. `#E31837` → `Color.Primary.primary`. `.font(.system(size: 14))` → `.typographyStyle(.body1)`.
58
- - **Design system primitives before custom views.** Before writing a new SwiftUI / Compose / React / View / Configuration triplet inside a domain or feature module, grep the shared component library (project-specific path, e.g. `Common/UIComponents/`, `core-ui/`, `packages/ui/`) for an existing primitive that solves the same problem. New domain-level wrappers, custom modals, custom buttons, or hand-rolled toasts are forbidden when the design system already has an equivalent. If the primitive exists but lacks a modifier (placeholder, size, error binding), **add the modifier to the primitive** in its `+Modifiers` extension - do not fork the primitive into the consumer domain. The Figma `CodeConnectSnippet` is the authoritative pointer to which primitive to use.
58
+ - **Design system primitives before custom views.** Before writing a new View / Configuration triplet in a feature module, grep the shared component library (e.g. `Common/UIComponents/`, `core-ui/`, `packages/ui/`) for a primitive that already solves it. Domain-level wrappers, custom modals, buttons or toasts are forbidden when the design system has an equivalent. If the primitive exists but lacks a modifier (placeholder, size, error binding), **add the modifier to the primitive** in its `+Modifiers` extension; never fork it into the consumer domain. The Figma `CodeConnectSnippet` is the authoritative pointer to which primitive to use.
59
59
 
60
60
  ## Swift-Specific Rules
61
61
 
@@ -246,7 +246,7 @@ bash $HOME/.claude/scripts/phase-tracker.sh model <N> <model_name>
246
246
 
247
247
  Token counts are additive - multiple calls accumulate. `input_count` is FRESH input (cache-exclusive); the optional 4th arg is the prompt-cache-read count, priced at the discounted rate. `model` tags the phase so the card and the cost helper can price it. Stored per phase in the state file; surfaced in `:log` reports, on the bash card tile (`<elapsed> · <tok> tok · ~$<usd>`), and in the card footer (total USD + cached tokens).
248
248
 
249
- **MANDATORY: per-phase token narration on completion (v9.10.2).** The native
249
+ **Required: per-phase token narration on completion (v9.10.2).** The native
250
250
  TaskList widget cannot display per-phase tokens - it shows name, status, and
251
251
  duration only. Without this rule the user sees durations and nothing else
252
252
  until the Phase 7 Cost Breakdown. So whenever a phase transitions to
@@ -96,7 +96,7 @@ Re-fetching Figma in dev phases:
96
96
 
97
97
  ### Verification
98
98
 
99
- The pipeline ships a smoke gate `pipeline/scripts/smoke-no-mcp-in-dev-phases.sh` that reads `state.telemetry.mcpCalls[]`. If any entry has `phase >= 2`, the gate fails. CI runs this gate after every run as a regression check.
99
+ The pipeline ships a smoke gate `pipeline/scripts/smoke-no-mcp-in-dev-phases.sh` that reads `state.telemetry.mcpCalls[]`. If any entry has `phase >= 2`, the gate fails. It is a maintainer-side regression check: it runs locally as part of `npm test` and the pre-push hook, not in CI (the CI workflow is dormant) and not after every pipeline run.
100
100
 
101
101
  Memory: [[mcp-only-in-analysis]]
102
102
 
@@ -174,12 +174,14 @@ When Figma pipeline is active (`figmaConfigPath` set in project preferences),
174
174
  multi-agent Phase 3 (DEV) dispatches Figma sub-phases instead of standard TDD:
175
175
 
176
176
  ```
177
- Phase 3.0: figma-to-swift-ui-start → Branch, assign, registry
178
- Phase 3.1: figma-to-swiftui → Full 8-phase implementation
179
- Phase 3.2: figma-commit → 14-item review + commit + PR
180
- Phase 3.3: figma-iteration-commit → Batch iteration commit (if iterating)
177
+ Phase 3.0: figma-validate → registry, Code Connect, token compliance (halts Phase 3 on failure)
178
+ Phase 3.1: create-component / create-screen → full implementation by the enabled stack plugin (evolve-component for an existing one)
179
+ Phase 3.2: figma-commit → 14-item review + commit + PR
180
+ Phase 3.3: figma-iteration-commit → Batch iteration commit (if iterating)
181
181
  ```
182
182
 
183
+ Skill names are the enabled stack plugin's (`ai-ios-toolkit:*` / `ai-android-toolkit:*`); the dispatch contract is `multi-agent-refs/component-dispatch.md`.
184
+
183
185
  ### Abstraction Layers
184
186
 
185
187
  | Layer | Provider Options | Config |
@@ -219,70 +221,14 @@ Provider interfaces are defined inline in the plugin skill sets - see the `ai-
219
221
  13. Build verification (xcodebuild)
220
222
  14. Test verification (ViewInspector + Snapshot)
221
223
 
222
- ### Command Catalog (31 commands)
223
-
224
- **Core Pipeline:**
225
-
226
- | Command | What It Does |
227
- |---------|-------------|
228
- | `/figma-to-swiftui <url>` | Full 8-phase pipeline |
229
- | `/figma-to-swift-ui-start #N` | Start from issue: branch, assign |
230
- | `/figma-to-swift-ui-implement <url>` | Phases 0-4 only (implementation) |
231
- | `/figma-to-swift-ui-test <url>` | Phase 5 only (tests) |
232
- | `/figma-to-swift-ui-wiki <Name>` | Phase 7 only (wiki docs) |
233
- | `/figma-to-swift-ui-code-connect <url>` | Phase 6 only (Code Connect) |
234
- | `/figma-to-swift-ui-confluence-sync` | Sync wiki → Confluence |
235
- | `/figma-to-swift-ui-status-update` | Refresh Confluence dashboard |
236
-
237
- **Issue & Board:**
238
-
239
- | Command | What It Does |
240
- |---------|-------------|
241
- | `/figma-issue open <url>` | Create GitHub Issue + optional Jira |
242
- | `/figma-review <Name>` | Interactive review (approve/bug) |
243
- | `/figma-validate <url>` | Pre-implementation validation |
244
-
245
- **Commit & PR:**
246
-
247
- | Command | What It Does |
248
- |---------|-------------|
249
- | `/figma-commit <Name>` | 14-item review + commit + PR |
250
- | `/figma-iteration-commit <Name>` | Commit to iteration/develop + PR |
251
-
252
- **Iteration & Batch:**
253
-
254
- | Command | What It Does |
255
- |---------|-------------|
256
- | `/figma-iterate` | Auto loop: pick → implement → commit |
257
- | `/figma-cli-iterate` | Full CLI iterate (all phases) |
258
- | `/figma-cli-lean-iterate` | Lean iterate (skip test+wiki) |
259
- | `/figma-cli-iterate-mend` | Re-implement discarded components |
260
- | `/figma-cli-skip` | Mark component as skipped |
261
- | `/figma-skip` | Skip + update Confluence |
262
-
263
- **Bugfix:**
264
-
265
- | Command | What It Does |
266
- |---------|-------------|
267
- | `/figma-fix` | Apply targeted bug fixes |
268
- | `/figma-mend` | Re-implement from scratch |
269
-
270
- **Setup & Utility:**
271
-
272
- | Command | What It Does |
273
- |---------|-------------|
274
- | `/figma-setup` | Environment setup wizard |
275
- | `/figma-utility` | Figma data fetch (screenshots, metadata) |
276
- | `/figma-remote-mcp-auth` | Figma MCP OAuth flow |
277
- | `/figma-ui-patterns` | UI pattern library index |
278
- | `/figma-price-integration` | Price protocol adoption guide |
279
-
280
- **Performance (batch production):**
281
-
282
- | Command | What It Does |
283
- |---------|-------------|
284
- | `/performance-start #N` | Start component with perf tracking |
285
- | `/performance-swiftui` | Perf-optimized pipeline |
286
- | `/performance-tour` | Batch produce multiple components |
287
- | `/performance-review-next` | Interactive batch review |
288
- | `/performance-iteration-commit-all` | Batch validate + push all |
224
+ ### Figma Skill Surfaces (current)
225
+
226
+ The old 31-command personal catalog is retired; none of those `/figma-*` commands ships with the pipeline. Figma capabilities live on three surfaces. Screens are ALWAYS drawn 1:1 from the Figma design context resolved in analysis, and a Code Connect-mapped component is ALWAYS used when one exists (see the fallback chain + Code Connect rules above).
227
+
228
+ | Surface | What it carries | When to use |
229
+ |---------|-----------------|-------------|
230
+ | `ai-ios-toolkit` / `ai-android-toolkit` (marketplace stack plugins, enabled per repo by `/multi-agent:stack`; skill names below are the iOS plugin's, the Android plugin carries the Compose equivalents) | Component skills: `create-component`, `create-screen`, `evolve-component`, `figma-validate`, `figma-review`, `figma-commit`, `figma-iteration-commit`, `figma-component-start`, `figma-utility`, `figma-setup`, `code-connect`, `component-docs`, `component-wiki`, plus the `navigation` / `overlays` / `bottom-sheets` / `ui-patterns` reference skills | Direct component work in a repo where the stack plugin is enabled; Phase 3 dispatches here for `taskType === component` |
231
+ | Pipeline commands | `/multi-agent:analysis` (the only phase allowed to fetch Figma), `/multi-agent:design-check` (mock-mode vs Figma conformance, Phase 4 Step 2.8), `/multi-agent:review` (cites the analysis doc, never Figma) | Inside multi-agent pipeline runs |
232
+ | Legacy `/figma-to-swiftui` (single standalone command) | The old full 8-phase flow | Kept for the old flow only; prefer the plugin's `create-component` for new work |
233
+
234
+ The pipeline keeps no Figma command catalog of its own: the routing table for component skills is maintained inside each stack plugin's `index` skill, and `/multi-agent:stack` decides which plugin is active for a repo.
@@ -21,6 +21,7 @@ available. Read the effective `enabledPlugins` and load each enabled toolkit's
21
21
  `ai-common-toolkit` and `ai-analyst-toolkit` are on everywhere. Nothing enabled
22
22
  is a normal state.
23
23
 
24
- **multi-agent-toolkit MCP.** 80+ tools for a running app: `ui-inspect`,
25
- `crash-logs`, `design-check`, `ios-app-store-audit`, `ios-testflight`. Use them
26
- instead of guessing about on-screen state. Not registered is a silent no-op.
24
+ **multi-agent-toolkit MCP.** 80+ tools for a running app: `ios_get_ui_tree` /
25
+ `android_get_ui_tree`, `ios_list_crashes` / `android_list_crashes`, `design_*`,
26
+ `ios_app_store_audit`, `ios_testflight_validate`. Use them instead of guessing
27
+ about on-screen state. Not registered is a silent no-op.
@@ -107,7 +107,7 @@
107
107
  "siblings": {
108
108
  "type": "array",
109
109
  "maxItems": 10,
110
- "description": "Repos the dev-context picker offered that this run does not modify: read-only siblings, plus any extra the user selected. Persisted at Phase 0 because the phases that consume them run much later - Phase 4's platform-parity cross-check reads this and nothing else, so a picker result that is not written here is a step that can never fire.",
110
+ "description": "Repos the dev-context picker offered that this run does not modify: read-only siblings, plus any extra the user selected. Persisted at Phase 0 because the phases that consume them run much later - Phase 4's platform-parity cross-check reads this as the fourth of its four counterpart sources (after --with, prefs.projects[<slug>].counterpartRoots[] and the primary checkout's sibling directories; see multi-agent-refs/platform-parity.md), so a picker result that is not written here is a candidate the check can never see.",
111
111
  "items": {
112
112
  "type": "object",
113
113
  "additionalProperties": false,
@@ -650,6 +650,90 @@
650
650
  "type": "string",
651
651
  "enum": ["fix", "accept", "escalate"]
652
652
  },
653
+ "delta": {
654
+ "type": "object",
655
+ "additionalProperties": true,
656
+ "description": "Written by Phase 4 Step 3.8 from review-delta.mjs: how this round's accepted findings relate to the previous round's rework mandate. Absent on iteration 1 and on runs from before the delta existed.",
657
+ "properties": {
658
+ "previousIteration": { "type": "integer", "minimum": 1 },
659
+ "new": {
660
+ "type": "array",
661
+ "items": {
662
+ "type": "object",
663
+ "additionalProperties": true,
664
+ "properties": {
665
+ "fingerprint": { "type": "string" },
666
+ "severity": { "type": ["string", "null"] },
667
+ "file": { "type": ["string", "null"] },
668
+ "issue": { "type": ["string", "null"] }
669
+ }
670
+ }
671
+ },
672
+ "stillPresent": {
673
+ "type": "array",
674
+ "items": {
675
+ "type": "object",
676
+ "additionalProperties": true,
677
+ "properties": {
678
+ "fingerprint": { "type": "string" },
679
+ "severity": { "type": ["string", "null"] },
680
+ "file": { "type": ["string", "null"] },
681
+ "issue": { "type": ["string", "null"] }
682
+ }
683
+ }
684
+ },
685
+ "resolved": {
686
+ "type": "array",
687
+ "items": {
688
+ "type": "object",
689
+ "additionalProperties": true,
690
+ "properties": {
691
+ "fingerprint": { "type": "string" },
692
+ "severity": { "type": ["string", "null"] },
693
+ "file": { "type": ["string", "null"] },
694
+ "issue": { "type": ["string", "null"] }
695
+ }
696
+ }
697
+ },
698
+ "downgraded": {
699
+ "type": "array",
700
+ "items": {
701
+ "type": "object",
702
+ "additionalProperties": true,
703
+ "properties": {
704
+ "fingerprint": { "type": "string" },
705
+ "severity": { "type": ["string", "null"] },
706
+ "file": { "type": ["string", "null"] },
707
+ "issue": { "type": ["string", "null"] }
708
+ }
709
+ }
710
+ },
711
+ "stillPresentBlocking": {
712
+ "type": "array",
713
+ "items": {
714
+ "type": "object",
715
+ "additionalProperties": true,
716
+ "properties": {
717
+ "fingerprint": { "type": "string" },
718
+ "severity": { "type": ["string", "null"] },
719
+ "file": { "type": ["string", "null"] },
720
+ "issue": { "type": ["string", "null"] }
721
+ }
722
+ }
723
+ },
724
+ "recurrence": {
725
+ "type": "object",
726
+ "additionalProperties": { "type": "integer", "minimum": 1 },
727
+ "description": "fingerprint -> consecutive rework cycles the finding has survived."
728
+ },
729
+ "plateau": {
730
+ "type": "boolean",
731
+ "description": "The still-present set is unchanged from the previous delta and non-empty."
732
+ },
733
+ "tripped": { "type": "boolean" },
734
+ "computedAt": { "type": "string", "format": "date-time" }
735
+ }
736
+ },
653
737
  "reviewers": {
654
738
  "type": "array",
655
739
  "description": "One entry per reviewer dispatch that RETURNED. Typed because two consumers depend on the shape: anonymize-findings.mjs needs model+findings to build the label map, and run-metrics.mjs reports acceptedRatio per reviewer. Extra keys are allowed; nothing is required, so a run written before this shape existed still validates and surfaces as model \"unknown\" rather than failing.",
@@ -699,6 +783,51 @@
699
783
  }
700
784
  }
701
785
  },
786
+ "circuitBreaker": {
787
+ "type": "object",
788
+ "additionalProperties": false,
789
+ "description": "Autopilot circuit-breaker record (refs/features/autopilot-circuit-breaker.md). Written only when a trigger trips: trigger 2 by Phase 4 Step 3.8 (a mandate finding survived identicalFindingCycles rework cycles), trigger 3 by the Phase 3 re-entry hard-kill. /multi-agent:resume clears tripped and keeps counters.",
790
+ "required": ["tripped"],
791
+ "properties": {
792
+ "tripped": { "type": "boolean" },
793
+ "trigger": { "type": ["integer", "null"], "minimum": 1, "maximum": 5 },
794
+ "detail": { "type": "string" },
795
+ "checkpoint": {
796
+ "type": "object",
797
+ "additionalProperties": false,
798
+ "properties": {
799
+ "phase": { "type": "integer", "minimum": 0, "maximum": 7 },
800
+ "step": { "type": "string" },
801
+ "iteration": { "type": "integer", "minimum": 1 }
802
+ }
803
+ },
804
+ "trippedAt": { "type": "string", "format": "date-time" },
805
+ "counters": {
806
+ "type": "object",
807
+ "additionalProperties": false,
808
+ "properties": {
809
+ "identicalFindingCycles": { "type": "integer", "minimum": 0 },
810
+ "reworkCycles": { "type": "integer", "minimum": 0 }
811
+ }
812
+ }
813
+ }
814
+ },
815
+ "diffRisk": {
816
+ "type": "object",
817
+ "additionalProperties": true,
818
+ "description": "Totals from diff-risk-score.mjs, persisted by Phase 4 Step 1.75 so Phase 6 (PR risk section), Phase 7 and run-metrics.mjs read the same numbers the review scope decision used.",
819
+ "properties": {
820
+ "files": { "type": "integer", "minimum": 0 },
821
+ "loc_added": { "type": "integer", "minimum": 0 },
822
+ "loc_removed": { "type": "integer", "minimum": 0 },
823
+ "max_score": { "type": "number" },
824
+ "signals": {
825
+ "type": "array",
826
+ "items": { "type": "string" },
827
+ "description": "Distinct signal names seen across files (security_path, migration, public_api, no_test_change, test_lines_removed, ...)."
828
+ }
829
+ }
830
+ },
702
831
  "confluenceSpace": {
703
832
  "type": ["string", "null"],
704
833
  "description": "Cached Confluence space key for the project - avoids re-asking on every run."
@@ -100,6 +100,11 @@
100
100
  "description": "Citation. Format: 'rules/<file>.md#<anchor>' or 'gate/<name>'."
101
101
  },
102
102
  "issue": { "type": "string", "description": "Short, one-sentence description." },
103
+ "fingerprint": {
104
+ "type": "string",
105
+ "pattern": "^F:[0-9a-f]{8}$",
106
+ "description": "Stable id computed by finding-fingerprint.mjs --kind dev-critic. Round-2 findings carry the round-1 fingerprint; a round-2 finding with none is the scope creep the loop contract forbids."
107
+ },
103
108
  "fix": {
104
109
  "type": "string",
105
110
  "description": "Concrete suggestion the generator can act on without re-reading the rule."