@mmerterden/multi-agent-pipeline 16.18.0 → 16.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +68 -0
- package/README.md +5 -5
- package/README.tr.md +2 -2
- package/docs/FIGMA_PIPELINE.md +1 -1
- package/docs/adr/0001-three-model-triage.md +4 -2
- package/docs/architecture.md +2 -2
- package/docs/ecosystem.md +13 -11
- package/docs/features.md +2 -2
- package/index.js +1 -1
- package/install/_codex-agents.mjs +2 -2
- package/install/_common.mjs +25 -1
- package/install/_dev-only-files.mjs +3 -2
- package/install/_mcp-register.mjs +4 -3
- package/install/_plugin-skills.mjs +1 -3
- package/install/copilot.mjs +18 -9
- package/install/index.mjs +2 -4
- package/install/templates/copilot-instructions.md +7 -7
- package/package.json +4 -3
- package/pipeline/agents/android-architect.md +1 -0
- package/pipeline/agents/backend-architect.md +1 -0
- package/pipeline/agents/code-reviewer.md +1 -0
- package/pipeline/agents/dev-critic.md +2 -1
- package/pipeline/agents/explorer.md +1 -0
- package/pipeline/agents/ios-architect.md +1 -0
- package/pipeline/agents/security-auditor.md +1 -0
- package/pipeline/agents/task-clarifier.md +1 -0
- package/pipeline/claude-md-template.md +2 -2
- package/pipeline/commands/multi-agent/SKILL.md +5 -5
- package/pipeline/commands/multi-agent/complaint-analysis/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/design-check/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/help/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/manual-test/SKILL.md +5 -1
- package/pipeline/commands/multi-agent/resume/SKILL.md +1 -0
- package/pipeline/commands/multi-agent/review/SKILL.md +27 -14
- package/pipeline/commands/multi-agent/review-analysis/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/store-ready/SKILL.md +24 -2
- package/pipeline/commands/multi-agent/sync/SKILL.md +2 -2
- package/pipeline/commands/sim-test.md +64 -20
- package/pipeline/lib/credential-inventory.sh +15 -2
- package/pipeline/lib/credential-store-resolver.sh +14 -4
- package/pipeline/lib/credential-store.sh +8 -2
- package/pipeline/lib/extract-conventions.sh +1 -14
- package/pipeline/lib/fetch-confluence.sh +12 -4
- package/pipeline/lib/fetch-crashlytics.sh +11 -8
- package/pipeline/lib/fetch-document.sh +1 -1
- package/pipeline/lib/fetch-figma-annotations.sh +7 -5
- package/pipeline/lib/fetch-fortify.sh +5 -3
- package/pipeline/lib/fetch-graylog.sh +5 -3
- package/pipeline/lib/figma-mcp-refresh.sh +1 -1
- package/pipeline/lib/figma-screenshot.sh +27 -24
- package/pipeline/lib/figma-token.sh +8 -4
- package/pipeline/lib/issue-fetcher.sh +0 -1
- package/pipeline/lib/jira-publish.sh +7 -5
- package/pipeline/lib/md2confluence-v3.py +13 -7
- package/pipeline/lib/multi-repo-pipeline.sh +18 -8
- package/pipeline/lib/plan-todos.sh +11 -0
- package/pipeline/lib/post-pr-review.sh +9 -2
- package/pipeline/lib/repo-cache.sh +18 -10
- package/pipeline/lib/review-watch.sh +60 -14
- package/pipeline/lib/shadow-git.sh +8 -4
- package/pipeline/lib/vercel-deploy.sh +2 -2
- package/pipeline/multi-agent-refs/_dev-context.md +5 -2
- package/pipeline/multi-agent-refs/analysis/locked.md +4 -4
- package/pipeline/multi-agent-refs/analysis/render.md +1 -1
- package/pipeline/multi-agent-refs/channels/pr.md +22 -4
- package/pipeline/multi-agent-refs/cross-cli-contract.md +1 -1
- package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +15 -2
- package/pipeline/multi-agent-refs/features/review-delta.md +89 -0
- package/pipeline/multi-agent-refs/features/scope-check.md +41 -0
- package/pipeline/multi-agent-refs/features/verify-by-test.md +6 -5
- package/pipeline/multi-agent-refs/features/worktree-finalize.md +1 -1
- package/pipeline/multi-agent-refs/outside-the-pipeline.md +6 -6
- package/pipeline/multi-agent-refs/payload-contracts.md +1 -1
- package/pipeline/multi-agent-refs/phases/log-format.md +1 -1
- package/pipeline/multi-agent-refs/phases/modes.md +1 -1
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +4 -2
- package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +3 -3
- package/pipeline/multi-agent-refs/phases/phase-2-planning.md +5 -5
- package/pipeline/multi-agent-refs/phases/phase-3-dev.md +17 -2
- package/pipeline/multi-agent-refs/phases/phase-4-review.md +67 -24
- package/pipeline/multi-agent-refs/phases/phase-5-test.md +10 -0
- package/pipeline/multi-agent-refs/phases/phase-6-commit.md +2 -0
- package/pipeline/multi-agent-refs/phases/phase-7-report.md +4 -2
- package/pipeline/multi-agent-refs/rules.md +2 -2
- package/pipeline/multi-agent-refs/tracker-contract.md +1 -1
- package/pipeline/rules/figma-pipeline.md +18 -72
- package/pipeline/rules/outside-the-pipeline.md +4 -3
- package/pipeline/schemas/agent-state.schema.json +130 -1
- package/pipeline/schemas/dev-critic-output.schema.json +5 -0
- package/pipeline/schemas/prefs.schema.json +47 -0
- package/pipeline/schemas/reviewer-output.schema.json +7 -2
- package/pipeline/schemas/scope-check.schema.json +55 -0
- package/pipeline/schemas/token-budget.json +3 -3
- package/pipeline/schemas/triage-output.schema.json +12 -2
- package/pipeline/scripts/README.md +3 -2
- package/pipeline/scripts/_fingerprint.mjs +173 -0
- package/pipeline/scripts/_stack-routing.mjs +1 -1
- package/pipeline/scripts/agent-guard.py +102 -21
- package/pipeline/scripts/anonymize-findings.mjs +7 -6
- package/pipeline/scripts/build-skills-index.mjs +14 -3
- package/pipeline/scripts/cost-budget-check.mjs +5 -3
- package/pipeline/scripts/cost-lib.sh +0 -15
- package/pipeline/scripts/diff-explain.mjs +22 -12
- package/pipeline/scripts/evidence-gate.mjs +73 -5
- package/pipeline/scripts/finding-fingerprint.mjs +101 -0
- package/pipeline/scripts/gc-refs.sh +6 -2
- package/pipeline/scripts/gc-tmp.sh +1 -1
- package/pipeline/scripts/gc-worktrees.sh +1 -1
- package/pipeline/scripts/gen-mode-dispatch.mjs +3 -3
- package/pipeline/scripts/github-ssh-setup.sh +7 -2
- package/pipeline/scripts/graph-build.mjs +2 -2
- package/pipeline/scripts/jira-wiki-escape.mjs +2 -1
- package/pipeline/scripts/keychain.py +12 -11
- package/pipeline/scripts/learning-curve.mjs +1 -1
- package/pipeline/scripts/migrate-prefs.mjs +1 -1
- package/pipeline/scripts/output-quality-check.sh +3 -1
- package/pipeline/scripts/phase-tracker.sh +1 -1
- package/pipeline/scripts/phase0-exit-gate.mjs +2 -1
- package/pipeline/scripts/plan-coverage-gate.mjs +2 -1
- package/pipeline/scripts/pre-commit-check.sh +23 -13
- package/pipeline/scripts/prune-logs.sh +1 -1
- package/pipeline/scripts/render-agent-log-cost.sh +3 -1
- package/pipeline/scripts/render-cost-summary.sh +4 -2
- package/pipeline/scripts/render-work-summary.sh +5 -3
- package/pipeline/scripts/repo-map.mjs +3 -2
- package/pipeline/scripts/review-delta.mjs +217 -0
- package/pipeline/scripts/run-metrics.mjs +20 -0
- package/pipeline/scripts/scan-skills.sh +6 -2
- package/pipeline/scripts/scope-check-gate.mjs +90 -0
- package/pipeline/scripts/search-logs.sh +8 -6
- package/pipeline/scripts/sign-skills.sh +3 -1
- package/pipeline/scripts/smoke-cross-cli-behavior.sh +12 -5
- package/pipeline/scripts/triage-memory.mjs +25 -4
- package/pipeline/scripts/uninstall.mjs +20 -12
- package/pipeline/scripts/update-check.sh +2 -2
- package/pipeline/scripts/update-issue-progress.sh +6 -5
- package/pipeline/scripts/validate-analysis-doc.mjs +6 -6
- package/pipeline/scripts/validate-reviewer.mjs +6 -0
- package/pipeline/scripts/validate-triage.mjs +20 -0
- package/pipeline/scripts/verify-skills.sh +3 -1
- package/pipeline/scripts/worktree-finalize.sh +22 -10
- package/pipeline/skills/.skill-manifest.json +81 -57
- package/pipeline/skills/.skills-index.json +19 -19
- package/pipeline/skills/shared/README.md +8 -8
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +7 -7
- package/pipeline/skills/shared/core/multi-agent-design-check/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-review/SKILL.md +23 -14
- package/pipeline/skills/shared/core/multi-agent-store-ready/SKILL.md +5 -0
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +2 -2
- package/pipeline/skills/shared/external/NOTICE-dimillian-skills.md +56 -0
- package/pipeline/skills/shared/external/accessibility-compliance-accessibility-audit/SKILL.md +0 -4
- package/pipeline/skills/shared/external/api-patterns/SKILL.md +12 -24
- package/pipeline/skills/shared/external/app-store-changelog/references/release-notes-guidelines.md +34 -0
- package/pipeline/skills/shared/external/app-store-changelog/scripts/collect_release_changes.sh +33 -0
- package/pipeline/skills/shared/external/architecture/SKILL.md +7 -9
- package/pipeline/skills/shared/external/debugging-strategies/SKILL.md +0 -4
- package/pipeline/skills/shared/external/fastapi-pro/SKILL.md +0 -1
- package/pipeline/skills/shared/external/github-actions-templates/SKILL.md +0 -14
- package/pipeline/skills/shared/external/hig-components-content/SKILL.md +13 -13
- package/pipeline/skills/shared/external/hig-components-layout/SKILL.md +16 -16
- package/pipeline/skills/shared/external/hig-components-status/SKILL.md +6 -6
- package/pipeline/skills/shared/external/hig-components-system/SKILL.md +13 -13
- package/pipeline/skills/shared/external/hig-foundations/SKILL.md +23 -23
- package/pipeline/skills/shared/external/hig-inputs/SKILL.md +18 -18
- package/pipeline/skills/shared/external/hig-patterns/SKILL.md +30 -30
- package/pipeline/skills/shared/external/hig-platforms/SKILL.md +11 -11
- package/pipeline/skills/shared/external/hig-technologies/SKILL.md +33 -33
- package/pipeline/skills/shared/external/ios-coding-standard/references/STANDARD.md +52 -52
- package/pipeline/skills/shared/external/ios-coding-standard/references/lint-local.sh +1 -1
- package/pipeline/skills/shared/external/ios-coding-standard/references/rules.yml +11 -11
- package/pipeline/skills/shared/external/ios-developer/SKILL.md +0 -1
- package/pipeline/skills/shared/external/ios-module-structure/SKILL.md +7 -3
- package/pipeline/skills/shared/external/localization-reuse-map/SKILL.md +9 -15
- package/pipeline/skills/shared/external/macos-spm-app-packaging/SKILL.md +0 -5
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/Package.swift +17 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/Sources/MyApp/Resources/.keep +0 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/Sources/MyApp/main.swift +11 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/version.env +2 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/build_icon.sh +49 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/compile_and_run.sh +63 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/launch.sh +28 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/make_appcast.sh +82 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/package_app.sh +206 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/setup_dev_signing.sh +52 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/sign-and-notarize.sh +52 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/version.env +2 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/references/packaging.md +17 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/references/release.md +32 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/references/scaffold.md +79 -0
- package/pipeline/skills/shared/external/monorepo-architect/SKILL.md +0 -1
- package/pipeline/skills/shared/external/nodejs-backend-patterns/SKILL.md +0 -4
- package/pipeline/skills/shared/external/swift-concurrency-expert/references/approachable-concurrency.md +63 -0
- package/pipeline/skills/shared/external/swift-concurrency-expert/references/swift-6-2-concurrency.md +272 -0
- package/pipeline/skills/shared/external/swift-concurrency-expert/references/swiftui-concurrency-tour-wwdc.md +33 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/code-smells.md +150 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/demystify-swiftui-performance-wwdc23.md +46 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/optimizing-swiftui-performance-instruments.md +29 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/profiling-intake.md +44 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/report-template.md +47 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/understanding-hangs-in-your-app.md +33 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/understanding-improving-swiftui-performance.md +52 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/app-wiring.md +201 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/async-state.md +96 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/components-index.md +46 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/controls.md +57 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/deeplinks.md +66 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/focus.md +90 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/form.md +97 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/grids.md +71 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/haptics.md +71 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/input-toolbar.md +51 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/lightweight-clients.md +93 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/list.md +86 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/loading-placeholders.md +38 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/macos-settings.md +71 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/matched-transitions.md +59 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/media.md +73 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/menu-bar.md +101 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/navigationstack.md +159 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/overlay.md +45 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/performance.md +62 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/previews.md +48 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/scroll-reveal.md +133 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/scrollview.md +87 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/searchable.md +71 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/sheets.md +155 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/split-views.md +72 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/tabview.md +114 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/theming.md +71 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/title-menus.md +93 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/top-bar.md +49 -0
- package/pipeline/skills/shared/external/swiftui-view-refactor/references/mv-patterns.md +161 -0
- package/pipeline/skills/skills-index.md +8 -8
- package/pipeline/skills/shared/external/help-skills/SKILL.md +0 -166
|
@@ -104,7 +104,7 @@ If changes include UI files (iOS: `*View.swift`, `*Screen.swift`, `*Cell.swift`;
|
|
|
104
104
|
- Missing safe area / keyboard avoidance → **important**
|
|
105
105
|
- Hardcoded colors instead of system/semantic colors → **suggestion**
|
|
106
106
|
|
|
107
|
-
**iOS - SwiftUI interaction & accessibility conventions.** Gated to changed SwiftUI files. These are rule-registry territory as of v14.0.0, not a list transcribed here: Step 1.78 resolves them from whichever registry declares SwiftUI scope, so the criteria and their severities live in one place instead of drifting between this doc and the skill. Reviewers receive the resolved rule IDs. Native-SwiftUI-first unless the project's `figma-config` `ui.*` declares a custom system, in which case check against that system. Reference skills, when no registry covers the change: `
|
|
107
|
+
**iOS - SwiftUI interaction & accessibility conventions.** Gated to changed SwiftUI files. These are rule-registry territory as of v14.0.0, not a list transcribed here: Step 1.78 resolves them from whichever registry declares SwiftUI scope, so the criteria and their severities live in one place instead of drifting between this doc and the skill. Reviewers receive the resolved rule IDs. Native-SwiftUI-first unless the project's `figma-config` `ui.*` declares a custom system, in which case check against that system. Reference skills, when no registry covers the change: the enabled stack plugin's `navigation`, `overlays` and `bottom-sheets` skills (`ai-ios-toolkit:*` / `ai-android-toolkit:*`), plus `figma-to-swiftui`.
|
|
108
108
|
|
|
109
109
|
**Android - Material Design compliance** (skills: `ai-android-toolkit:compose-components`, `ai-android-toolkit:android-architecture`):
|
|
110
110
|
- Non-Material3 component when M3 equivalent exists → **suggestion**
|
|
@@ -152,6 +152,12 @@ echo "$RISK_FULL" | node $HOME/.claude/scripts/validate-diff-risk.mjs - >/dev/nu
|
|
|
152
152
|
RISK_JSON=$([ -n "$RISK_FULL" ] && jq -c '.files |= (sort_by(-.score) | .[:5])' <<< "$RISK_FULL" || echo "")
|
|
153
153
|
```
|
|
154
154
|
|
|
155
|
+
Persist the totals as `state.diffRisk` (Phase 6 `risk` section, Phase 7, `run-metrics.mjs` read them):
|
|
156
|
+
|
|
157
|
+
```bash
|
|
158
|
+
[ -n "$RISK_FULL" ] && jq -c '{diffRisk: (.totals + {signals: ([.files[].signals[]?.name] | unique)})}' <<< "$RISK_FULL" | node $HOME/.claude/scripts/write-state.mjs "$STATE_FILE"
|
|
159
|
+
```
|
|
160
|
+
|
|
155
161
|
**Signals & weights** (see `$HOME/.claude/schemas/diff-risk.schema.json`):
|
|
156
162
|
|
|
157
163
|
| Signal | Weight | Triggers when |
|
|
@@ -179,7 +185,9 @@ On success, emit a single summary metric:
|
|
|
179
185
|
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 review.diff_risk \
|
|
180
186
|
top_files=$(jq '.files | length' <<< "$RISK_JSON") \
|
|
181
187
|
max_score=$(jq '.totals.max_score' <<< "$RISK_JSON") \
|
|
182
|
-
loc_added=$(jq '.totals.loc_added' <<< "$RISK_JSON")
|
|
188
|
+
loc_added=$(jq '.totals.loc_added' <<< "$RISK_JSON") \
|
|
189
|
+
loc_removed=$(jq '.totals.loc_removed' <<< "$RISK_JSON") \
|
|
190
|
+
files=$(jq '.totals.files' <<< "$RISK_JSON")
|
|
183
191
|
```
|
|
184
192
|
|
|
185
193
|
**Opt-out**: `prefs.global.diffRiskAdvisory = false` skips this step entirely (no script invocation, no priority block injection). Default `true` because the cost is bounded and the signal-to-noise has been measured against the golden-task fixture set.
|
|
@@ -269,7 +277,7 @@ When `state.figmaAccess.tier === 3` (user-attached screenshot, no Code Connect s
|
|
|
269
277
|
|
|
270
278
|
Phase 4 sends the same diff to every reviewer and then to triage, so the diff is the dominant token cost. Two measures keep it bounded:
|
|
271
279
|
|
|
272
|
-
**Shared cache prefix.** Build the reviewer and triage prompts so the large invariant context - the full diff, the `${CRITERIA}` block from Step 1.78, the Phase 1 analysis summary, the Phase 2 plan - is a byte-identical leading block across all dispatches in this iteration. Only the per-reviewer focus + skill line varies, and it goes AFTER the shared block. `${CRITERIA}` goes in the prefix, identical for every reviewer: subsetting it per reviewer would invalidate the prefix for the whole panel and re-bill the largest block in the phase. Per-reviewer emphasis stays a one-line pointer in the suffix. When the host supports prompt caching, the 2nd/3rd reviewer and the triage call then read that prefix at the discounted cache-read rate instead of re-billing it as fresh input. Forward the host-reported cache-read count as `tokens_cached` per the Token telemetry contract so the saving lands in the cost ledger.
|
|
280
|
+
**Shared cache prefix.** Build the reviewer and triage prompts so the large invariant context - the full diff, the `${CRITERIA}` block from Step 1.78, the Phase 1 analysis summary, the Phase 2 plan - is a byte-identical leading block across all dispatches in this iteration. Only the per-reviewer focus + skill line varies, and it goes AFTER the shared block. `${CRITERIA}` goes in the prefix, identical for every reviewer: subsetting it per reviewer would invalidate the prefix for the whole panel and re-bill the largest block in the phase. Per-reviewer emphasis stays a one-line pointer in the suffix. When the host supports prompt caching, the 2nd/3rd reviewer and the triage call then read that prefix at the discounted cache-read rate instead of re-billing it as fresh input. Forward the host-reported cache-read count as `tokens_cached` per the Token telemetry contract so the saving lands in the cost ledger. The `<scope-self-check>` block and, from iteration 2, the `<previous-round-findings>` block (Step 2.1) close the shared block, after the plan and before the per-reviewer suffix.
|
|
273
281
|
|
|
274
282
|
**Single-repo diff cap.** If the diff exceeds the Phase 4 token allowance (`token-budget.json`), truncate the largest files and append a footer `[truncated - full diff in file://$WORKTREE/.review-diff.txt]`, writing the full diff to that path. Reviewers and triage receive the same capped view + the marker so they can flag "review the full diff manually." Log `review.diff_truncated bytes_dropped=<N>`. (Multi-repo already caps the combined diff at 80% of budget; this is the single-repo equivalent.)
|
|
275
283
|
|
|
@@ -303,17 +311,19 @@ looks like it ran. The full statement lives in the managed block at `~/.codex/AG
|
|
|
303
311
|
Sub-agent delegation itself is authorized by that same managed block; without it Phase 4
|
|
304
312
|
degrades to a single in-thread review.
|
|
305
313
|
|
|
306
|
-
**Single-vendor caveat.** Every Codex reviewer is an OpenAI model,
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
314
|
+
**Single-vendor caveat.** Every Codex reviewer is an OpenAI model, and every Claude
|
|
315
|
+
Code reviewer is an Anthropic model, so the cross-vendor disagreement that Copilot CLI
|
|
316
|
+
gets for free (GPT-5.4 beside two Claude models) is absent on both. The diversity budget
|
|
317
|
+
shifts to model generation, reasoning effort and persona focus: on Codex Reviewer 1 runs
|
|
318
|
+
`xhigh` on security and architecture, Reviewer 2 runs a different model family member
|
|
319
|
+
on edge cases, Reviewer 3 runs `medium` on quality; on Claude Code the three slots are
|
|
320
|
+
three different Claude tiers. Treat consensus among a single-vendor panel as weaker
|
|
321
|
+
evidence than the same consensus on Copilot CLI, and say so in the triage note when all
|
|
322
|
+
three agree on a borderline finding.
|
|
313
323
|
|
|
314
324
|
Each reviewer inherits the `code-reviewer` agent's focus areas (Security, Architecture, Quality, Performance) and output contract. The orchestrator overrides only the model and the stack-specific skill per-reviewer - no prompt duplication.
|
|
315
325
|
|
|
316
|
-
**Model override wiring:** `code-reviewer.md` declares `preferredModel: fable`, so Reviewer 1 uses the persona default (Fable 5). Reviewer 2 (`claude-opus-5` on Claude Code, `gpt-5.4` elsewhere) and Reviewer 3 (`claude-sonnet-5`) set `PHASE_MODEL_OVERRIDE=<model>` before dispatch - the orchestrator exports `CLAUDE_CODE_SUBAGENT_MODEL` on Claude Code, or passes `--model` on Copilot CLI. Full precedence rule: `skills/shared/core/multi-agent/SKILL.md#agent-dispatch--per-persona-model-routing
|
|
326
|
+
**Model override wiring:** `code-reviewer.md` declares `preferredModel: fable`, so Reviewer 1 uses the persona default (Fable 5). Reviewer 2 (`claude-opus-5` on Claude Code, `gpt-5.4` elsewhere) and Reviewer 3 (`claude-sonnet-5`) set `PHASE_MODEL_OVERRIDE=<model>` before dispatch - the orchestrator exports `CLAUDE_CODE_SUBAGENT_MODEL` on Claude Code, or passes `--model` on Copilot CLI. Full precedence rule: `skills/shared/core/multi-agent/SKILL.md#agent-dispatch--per-persona-model-routing`. Fable dispatches are subject to the fallback contract (`$HOME/.claude/multi-agent-refs/features/model-fallback.md`): dispatch-error retry walks `fable -> opus -> sonnet` and budget-ceiling downgrade.
|
|
317
327
|
|
|
318
328
|
**Stack-specific skills loaded per reviewer** (from Phase 1 `detectedStack`). All three columns are used on every host; Reviewer 2 reads them as Opus on Claude Code and as GPT-5.4 elsewhere.
|
|
319
329
|
|
|
@@ -326,7 +336,9 @@ Each reviewer inherits the `code-reviewer` agent's focus areas (Security, Archit
|
|
|
326
336
|
| Docker | `ai-backend-toolkit:docker-expert` | `ai-backend-toolkit:docker-expert` | `ai-backend-toolkit:ci-cd-pipelines` |
|
|
327
337
|
| Generic | `security-review` | `ai-backend-toolkit:clean-code` | `ai-backend-toolkit:clean-code` |
|
|
328
338
|
|
|
329
|
-
|
|
339
|
+
##### 2.1 Previous-round findings (iteration >= 2) and 2.2 scope self-check (every iteration)
|
|
340
|
+
|
|
341
|
+
A reviewer has no memory of the round before, so it rediscovers last round's findings in new words. From iteration 2, render the previous round's accepted blocking/important findings (`.pipeline/triage-round-$((ITERATION-1)).json`, max 40) into a `<previous-round-findings>` block at the end of the shared prefix: a still-present issue is reported with the SAME fingerprint and the current line, a fixed one is omitted, anything new leaves `fingerprint` unset. Every iteration also renders `.pipeline/scope-check.json` (Phase 3 Step 3.7) plus `scope-check-gate.mjs --advisory` output as `<scope-self-check>`: file reasons, unjustified files, and `notDone[]` (never re-raised as findings); a missing record logs `review.scope_check=missing`. Block text and recipes: `$HOME/.claude/multi-agent-refs/features/review-delta.md`.
|
|
330
342
|
|
|
331
343
|
#### Step 2.8 - Visual conformance gate (component / screen work only)
|
|
332
344
|
|
|
@@ -380,11 +392,14 @@ Step 2 produces N reviewer-output objects (one per dispatched reviewer), each co
|
|
|
380
392
|
REVIEWER_FILE="$WORKTREE/.pipeline/reviewer-$N.json"
|
|
381
393
|
printf '%s' "$REVIEWER_JSON" > "$REVIEWER_FILE"
|
|
382
394
|
node $HOME/.claude/scripts/validate-reviewer.mjs "$REVIEWER_FILE" \
|
|
383
|
-
--criteria "$WORKTREE/.pipeline/criteria-manifest.json"
|
|
395
|
+
--criteria "$WORKTREE/.pipeline/criteria-manifest.json" \
|
|
396
|
+
&& node $HOME/.claude/scripts/finding-fingerprint.mjs annotate --in-place "$REVIEWER_FILE"
|
|
384
397
|
```
|
|
385
398
|
|
|
386
399
|
Progress line: ` → checking validator validate-reviewer ({reviewer})`
|
|
387
400
|
|
|
401
|
+
`finding-fingerprint.mjs` stamps each finding with its cross-round id once the validator passes; an echoed one is kept, and anonymization leaves it intact.
|
|
402
|
+
|
|
388
403
|
Exit 0 = valid. Exit 2 = contradiction (approved=true with blocking findings) - flip `approved` to `false`, continue. With `--criteria`, exit 1 also covers the conformance checklist: a selected rule ID with no verdict, a verdict for an ID that was never selected, a `conformant` row with no file evidence, or a `violated` row with no matching finding. Those are the four ways a review can look complete without being complete, and the validator is what makes the checklist more than decoration - it is hand-written and does not apply `additionalProperties`, so an unchecked array would otherwise pass. Exit 1 = malformed; gate protocol (fails CLOSED, same handling as the evidence gate): emit the validator stderr + `errors[]` verbatim, attempt ONE self-correction rework (re-invoke that reviewer with the errors quoted, overwrite the file), re-run the validator. If it fails again -> HALT the phase (no merge, no triage). Recovery hint: `ERR: reviewer output failed validate-reviewer.mjs twice. Inspect $REVIEWER_FILE against $HOME/.claude/schemas/reviewer-output.schema.json, then resume with /multi-agent:resume #N.`
|
|
389
404
|
|
|
390
405
|
#### Step 2.5 - Disagreement-round loop (opt-in)
|
|
@@ -401,7 +416,7 @@ Exit 0 = valid. Exit 2 = contradiction (approved=true with blocking findings) -
|
|
|
401
416
|
- Max one round. Results replace the original outputs.
|
|
402
417
|
4. Proceed to Step 3 triage with the round-2 outputs.
|
|
403
418
|
|
|
404
|
-
**Parity contract:**
|
|
419
|
+
**Parity contract:** every host (three reviewers each: Claude Code, Copilot CLI, Codex CLI) runs the round identically. Telemetry emits `review_round_count={1|2}` per reviewer for Phase 7 rollup.
|
|
405
420
|
|
|
406
421
|
**Cost ceiling:** rebuttal round consumes ~1× the original Step 2 token budget. Smoke + budget tests treat this as opt-in so the default cost stays the same.
|
|
407
422
|
|
|
@@ -495,8 +510,9 @@ bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 memory.hit rows=$CITED_COU
|
|
|
495
510
|
**Triage prompt skeleton:**
|
|
496
511
|
|
|
497
512
|
```
|
|
498
|
-
You are the Review Triage agent.
|
|
513
|
+
You are the Review Triage agent. Three reviewers (fewer on a single-scope run) returned findings on this diff.
|
|
499
514
|
Your job: separate signal from noise. Do NOT add new findings. Do NOT re-review code.
|
|
515
|
+
Preserve `fingerprint` verbatim on every finding you carry through; it is the finding's identity across rounds.
|
|
500
516
|
|
|
501
517
|
For each finding, decide:
|
|
502
518
|
- ACCEPTED: real issue, in scope, must be fixed now
|
|
@@ -527,14 +543,19 @@ Return ONLY valid JSON conforming to $HOME/.claude/schemas/triage-output.schema.
|
|
|
527
543
|
Run on the persisted file immediately after the triage agent returns, before acting on the verdict; the validator's exit code decides, not the LLM turn:
|
|
528
544
|
|
|
529
545
|
```bash
|
|
530
|
-
|
|
546
|
+
ITERATION=$(jq '.reviewIterations | length' "$STATE_FILE")
|
|
547
|
+
TRIAGE_FILE="$WORKTREE/.pipeline/triage-round-$ITERATION.json"
|
|
531
548
|
mkdir -p "$(dirname "$TRIAGE_FILE")"
|
|
532
549
|
printf '%s' "$TRIAGE_JSON" > "$TRIAGE_FILE"
|
|
533
|
-
node $HOME/.claude/scripts/validate-triage.mjs "$TRIAGE_FILE"
|
|
550
|
+
node $HOME/.claude/scripts/validate-triage.mjs "$TRIAGE_FILE" \
|
|
551
|
+
&& node $HOME/.claude/scripts/finding-fingerprint.mjs annotate --in-place "$TRIAGE_FILE" \
|
|
552
|
+
&& cp "$TRIAGE_FILE" "$WORKTREE/triage-output.json"
|
|
534
553
|
```
|
|
535
554
|
|
|
536
555
|
Progress line: ` → checking validator validate-triage`
|
|
537
556
|
|
|
557
|
+
One file per round; `triage-output.json` is the latest copy every downstream reader expects (Phase 7, finalize, work-summary, diff-explain). Step 3.7 rewrites `$TRIAGE_FILE`; repeat the `cp` after it.
|
|
558
|
+
|
|
538
559
|
| Exit | Meaning | Action |
|
|
539
560
|
| ----- | ---------------------------- | ------------------------------------------------- |
|
|
540
561
|
| **0** | Valid and clean | Act on triage output as-is |
|
|
@@ -567,10 +588,13 @@ emit() { # $1=event $2=model $3=duration $4=tokens_in $5=tokens_out
|
|
|
567
588
|
model="$2" duration_ms="$3" tokens_in="$4" tokens_out="$5"
|
|
568
589
|
}
|
|
569
590
|
emit review.reviewer_call fable "$R1_DURATION" "$R1_IN" "$R1_OUT" # opus on Copilot CLI
|
|
570
|
-
emit review.reviewer_call sonnet "$SONNET_DURATION" "$SONNET_IN" "$SONNET_OUT"
|
|
571
591
|
# Reviewer 2 is Opus on Claude Code and GPT-5.4 elsewhere:
|
|
572
|
-
[ "${CLI_HOST:-claude}" = "
|
|
573
|
-
emit review.reviewer_call
|
|
592
|
+
if [ "${CLI_HOST:-claude}" = "claude" ]; then
|
|
593
|
+
emit review.reviewer_call opus "$R2_DURATION" "$R2_IN" "$R2_OUT"
|
|
594
|
+
else
|
|
595
|
+
emit review.reviewer_call gpt-5.4 "$R2_DURATION" "$R2_IN" "$R2_OUT"
|
|
596
|
+
fi
|
|
597
|
+
emit review.reviewer_call sonnet "$SONNET_DURATION" "$SONNET_IN" "$SONNET_OUT"
|
|
574
598
|
emit review.triage_call fable "$TRIAGE_DURATION" "$TRIAGE_IN" "$TRIAGE_OUT"
|
|
575
599
|
bash "$M" "$TASK_ID" 4 review.completed raw_count=$RAW accepted=$ACC \
|
|
576
600
|
deferred=$DEF rejected=$REJ approved=$APPROVED duration_ms=$DURATION
|
|
@@ -584,7 +608,7 @@ Opt-in via `prefs.global.triageCrossCheck.enabled` (default `false`). Sampled ru
|
|
|
584
608
|
|
|
585
609
|
##### 3.6 Consensus surfacing (anti-correlation)
|
|
586
610
|
|
|
587
|
-
**Rationale:**
|
|
611
|
+
**Rationale:** On Claude Code all three reviewers are Anthropic models, on Codex CLI all three are OpenAI, and on Copilot CLI two of three are Anthropic, so unanimous agreement on a *judgment call* is not independent confirmation: same-family models drift the same way on ambiguous prompts, and "all approved" taken as proof produces false-consensus passes. Triage therefore records a `consensus` block (schema v3.1.0) and surfaces disagreement and unverified agreement instead of burying it.
|
|
588
612
|
|
|
589
613
|
After the triage verdict is computed, populate `triage.consensus`:
|
|
590
614
|
|
|
@@ -602,7 +626,26 @@ After the triage verdict is computed, populate `triage.consensus`:
|
|
|
602
626
|
|
|
603
627
|
A triage verdict is judgment; a failing repro test is proof. Runs only when `prefs.global.verifyByTest.enabled` is `true` AND `accepted` contains a `blocking` finding; otherwise skip silently. **Full contract (verdict table, cleanup invariant, prompts): `$HOME/.claude/multi-agent-refs/features/verify-by-test.md` - read it before executing this step.**
|
|
604
628
|
|
|
605
|
-
Compressed flow: dispatch ONE verifier agent (model `verifyByTest.model`, default `sonnet`) for up to `maxFindings` (default 3) accepted blocking findings. Per finding it writes ONE minimal repro test and runs ONLY that test (Phase 3 single-test invocation, build lock, log tee'd to `$WORKTREE/.pipeline/verify-<i>.test.log`). Outcomes: test FAILS as predicted -> `confirmed`, finding stays blocking and the test is KEPT in `redTests[]` as the Phase 3 rework RED test; test PASSES -> `not-reproduced` ONLY if `evidence-gate.mjs --claim test --status passed` exits 0 on
|
|
629
|
+
Compressed flow: dispatch ONE verifier agent (model `verifyByTest.model`, default `sonnet`) for up to `maxFindings` (default 3) accepted blocking findings. Per finding it writes ONE minimal repro test and runs ONLY that test (Phase 3 single-test invocation, build lock, log tee'd to `$WORKTREE/.pipeline/verify-<i>.test.log`). Outcomes: test FAILS as predicted -> `confirmed`, finding stays blocking and the test is KEPT in `redTests[]` as the Phase 3 rework RED test; test PASSES on every one of `verifyByTest.repeatCount` runs (default 3) -> `not-reproduced` ONLY if `evidence-gate.mjs --claim test --status passed` exits 0 on each log; a run that disagrees with the others -> `inconclusive` with `flaky: passed k/N`, finding moves to `deferred[]`, test deleted; compile error / timeout / not unit-testable -> `inconclusive`, judgment verdict stands. Stamp findings with `verification` (schema v3.2.0), persist `state.reviewIterations[-1].verifyByTest = {attempted, confirmed, downgraded, inconclusive, redTests[]}`, recompute `approved`, re-run `validate-triage.mjs` under the 3.2.1 gate. Whole step bounded by `stepTimeoutSec` (default 600); on breach or crash remaining findings keep judgment verdicts - never blocks. Telemetry per 3.4: `review.verify_by_test attempted= confirmed= downgraded= inconclusive= duration_ms=`.
|
|
630
|
+
|
|
631
|
+
##### 3.8 Cross-round delta + circuit-breaker trigger 2 (iteration >= 2)
|
|
632
|
+
|
|
633
|
+
**Full contract (state merge, telemetry line, picker wording): `$HOME/.claude/multi-agent-refs/features/review-delta.md`.**
|
|
634
|
+
|
|
635
|
+
```bash
|
|
636
|
+
TRIP=$(jq -r '.global.autopilotCircuitBreaker.identicalFindingCycles // 2' "$PREFS_FILE")
|
|
637
|
+
DELTA_JSON=$(node $HOME/.claude/scripts/review-delta.mjs --rounds-dir "$WORKTREE/.pipeline" --iteration "$ITERATION" --trip-cycles "$TRIP"); DELTA_RC=$?
|
|
638
|
+
```
|
|
639
|
+
|
|
640
|
+
Merge the JSON into `state.reviewIterations[-1].delta` via `write-state.mjs`; emit `review.delta iteration= new= still_present= resolved= plateau= tripped=`. Progress line: ` → comparing round {N} vs {N-1} (new={n} still={n} resolved={n})`.
|
|
641
|
+
|
|
642
|
+
| Exit | Action |
|
|
643
|
+
|---|---|
|
|
644
|
+
| **0** | Continue to Step 4 (also iteration 1 and a missing previous round). |
|
|
645
|
+
| **3** | Autopilot with `prefs.global.autopilotCircuitBreaker.enabled` (default true): write `state.circuitBreaker = {tripped: true, trigger: 2, detail, checkpoint: {phase: 4, step: "3.8", iteration: N}, trippedAt, counters}`, then the `operations.md` halt protocol with `haltReason="4:circuit-breaker:identical-finding"`. Interactive: show `delta.stillPresent`, ask `Continue rework` / `Escalate to me` / `Accept as deferred`. |
|
|
646
|
+
| **1** | Log `review.delta_skipped reason=invalid`, continue; the delta never blocks on its own failure. |
|
|
647
|
+
|
|
648
|
+
`plateau` is logged, not acted on; trigger 3 (the rework cap) is recorded by the Phase 3 re-entry.
|
|
606
649
|
|
|
607
650
|
#### Step 4 - Consensus + Action (triage-driven)
|
|
608
651
|
|
|
@@ -610,7 +653,7 @@ If `triage.consensus.verdict` is `split` or `unverified`, surface `consensus.dis
|
|
|
610
653
|
|
|
611
654
|
Act **only on triage.accepted**:
|
|
612
655
|
|
|
613
|
-
- **accepted.blocking** → back to Phase 3 (max 3 iterations, with reflection prompt citing only accepted items). When Step 3.7 ran and `state.reviewIterations[-1].verifyByTest.redTests[]` is non-empty, the reflection prompt cites each red test: "a failing repro test already exists at <testRef>; make it green; do not delete or weaken it."
|
|
656
|
+
- **accepted.blocking** → back to Phase 3 (max 3 iterations, with reflection prompt citing only accepted items). The reflection prompt names findings by `fingerprint` and quotes `delta.stillPresent` first, marked `STILL PRESENT after round N-1's fix`. When Step 3.7 ran and `state.reviewIterations[-1].verifyByTest.redTests[]` is non-empty, the reflection prompt cites each red test: "a failing repro test already exists at <testRef>; make it green; do not delete or weaken it."
|
|
614
657
|
- **accepted.important** → fix and re-review
|
|
615
658
|
- **accepted.suggestion** → apply if reasonable
|
|
616
659
|
- **deferred** → append to Phase 7 report as "follow-up items" (do not block)
|
|
@@ -631,7 +674,7 @@ Statement shape: the durable rule/root cause, not the symptom - "force-unwrapp
|
|
|
631
674
|
|
|
632
675
|
Progress line: ` → writing lesson to learnings ledger ({N} entries)`
|
|
633
676
|
|
|
634
|
-
Log: "Phase 4: Review - raw={N1+N2} accepted={Na} deferred={Nd} rejected={Nr} approved={bool} consensus={verdict}"
|
|
677
|
+
Log: "Phase 4: Review - raw={N1+N2+N3} accepted={Na} deferred={Nd} rejected={Nr} approved={bool} consensus={verdict}"
|
|
635
678
|
|
|
636
679
|
---
|
|
637
680
|
|
|
@@ -103,6 +103,16 @@ Tier 1 / Tier 2 records print `screenshotUrl` from the captured evidence (Tier 2
|
|
|
103
103
|
bash $HOME/.claude/scripts/phase-tracker.sh render
|
|
104
104
|
```
|
|
105
105
|
The waiting state persists in `tracker-state.json` across the handoff; `/multi-agent:resume-local` and `/multi-agent:manual-test` CONTINUE this state file and never re-init it (`$HOME/.claude/multi-agent-refs/tracker-contract.md` "Continuation runs").
|
|
106
|
+
|
|
107
|
+
**"ok" is a structured result, not a word.** Before "ok" is accepted, the run writes `$WORKTREE/.pipeline/manual-test.json`: one entry per acceptance criterion, the criteria taken from the analysis doc test plan (Section 15 / 20), the plan tasks, and the user's own words in the reply. Every criterion records what was seen; a criterion that was not tried says so with a reason.
|
|
108
|
+
```json
|
|
109
|
+
{"criteria":[{"spec":"<quote>","source":"analysis 15.2 | plan task 3 | user","observed":"<what was seen>","verdict":"pass|fail|not-tested","reason":"<required when not-tested>","screenshot":"<path or null>"}],"verdict":"passed|failed"}
|
|
110
|
+
```
|
|
111
|
+
Then gate it:
|
|
112
|
+
```bash
|
|
113
|
+
node $HOME/.claude/scripts/evidence-gate.mjs --claim manual --status passed --evidence "$WORKTREE/.pipeline/manual-test.json"
|
|
114
|
+
```
|
|
115
|
+
Exit 1 means the "ok" is not accepted: tell the user which criterion is missing evidence (a `fail` verdict, or `not-tested` without a reason) and wait for the next reply. Exit 0 marks Phase 5 completed with `Result "local test passed (user)"`. The "fix: ..." path below is unchanged.
|
|
106
116
|
6. If fix needed:
|
|
107
117
|
- Branch already has WIP commit (from step 2) - changes are safe
|
|
108
118
|
- **Heal stale admin state first** (same contract as Phase 0 - step 3's
|
|
@@ -136,6 +136,8 @@ Branch **deterministically**, no implicit fallback. Read `agent-state.json` and
|
|
|
136
136
|
|
|
137
137
|
Generate a structured PR description based on task type. The PR body targets **code reviewers** - it should be technical: what changed, why, architecture decisions, how to verify.
|
|
138
138
|
|
|
139
|
+
Two inputs are read from state and the worktree before writing, not recalled from the conversation: `$WORKTREE/.pipeline/scope-check.json` (Phase 3 Step 3.7) supplies the `## Changes` bullets from `files[].reason` and the "Follow-ups not done in this PR" list under `## Related` from `notDone[]`; `state.diffRisk.signals` (Phase 4 Step 1.75) decides whether the conditional `## Risk and Security` section is required. When a high-stakes signal is present and the section is missing, this step blocks until it is written; a placeholder answer ("TBD") counts as missing.
|
|
140
|
+
|
|
139
141
|
**required**: Run all generated text (PR body, commit message) through the `humanizer` skill before posting. This removes AI-generated patterns (inflated language, filler phrases, repetitive structure) and makes the output sound like a developer wrote it.
|
|
140
142
|
|
|
141
143
|
**IMPORTANT - Issue auto-close prevention**: Never use these keywords in PR title, body, or commit messages - they auto-close the issue on merge:
|
|
@@ -163,8 +163,10 @@ Skipped sections: when `planTodos.enabled` is false or no `plan.todos[]` was emi
|
|
|
163
163
|
|
|
164
164
|
## Review Iterations
|
|
165
165
|
|
|
166
|
-
| Iteration | Blocking | Important | Suggestion | Decision |
|
|
167
|
-
| --------- | -------- | --------- | ---------- | -------- |
|
|
166
|
+
| Iteration | Blocking | Important | Suggestion | Decision | Still present | Resolved | New |
|
|
167
|
+
| --------- | -------- | --------- | ---------- | -------- | ------------- | -------- | --- |
|
|
168
|
+
|
|
169
|
+
(the last three columns come from `reviewIterations[i].delta.counts` (Phase 4 Step 3.8) and read `-` on iteration 1 and on runs that predate the delta; a tripped circuit-breaker is one extra line under the table: `Circuit-breaker tripped: trigger {n}, {detail}`)
|
|
168
170
|
|
|
169
171
|
## Files Changed
|
|
170
172
|
|
|
@@ -52,10 +52,10 @@ This is the single source of truth. When a contributor or model is unsure where
|
|
|
52
52
|
- **NEVER** commit without passing build (all gates in Phase 4 Step 1 must be green).
|
|
53
53
|
- **NEVER** commit without passing review (at least one AI reviewer must return `approved: true` with no blocking findings).
|
|
54
54
|
- **NEVER** skip tests. Every public method, every error path, every edge case.
|
|
55
|
-
- **NEVER** delete, rename, or weaken an existing test to get a green run. Existing tests are immutable during a task
|
|
55
|
+
- **NEVER** delete, rename, or weaken an existing test to get a green run. Existing tests are immutable during a task; one may change only when the task changes the spec it encodes, and the commit body names both. Deterministic backstop: the `test_lines_removed` diff-risk signal (Phase 4 Step 1.75) flags test files that shrink. A pass only on retry is a flake signal, not a pass: Phase 3 repeats new and changed tests (`testStability.repeatCount`, default 3) and logs `test.flake_signal` when runs disagree.
|
|
56
56
|
- **Follow existing code style and conventions.** Read neighbor files before writing new ones - match naming, structure, import order.
|
|
57
57
|
- **Use design tokens, no magic numbers.** `16` → `.Spacing.spacing16`. `#E31837` → `Color.Primary.primary`. `.font(.system(size: 14))` → `.typographyStyle(.body1)`.
|
|
58
|
-
- **Design system primitives before custom views.** Before writing a new
|
|
58
|
+
- **Design system primitives before custom views.** Before writing a new View / Configuration triplet in a feature module, grep the shared component library (e.g. `Common/UIComponents/`, `core-ui/`, `packages/ui/`) for a primitive that already solves it. Domain-level wrappers, custom modals, buttons or toasts are forbidden when the design system has an equivalent. If the primitive exists but lacks a modifier (placeholder, size, error binding), **add the modifier to the primitive** in its `+Modifiers` extension; never fork it into the consumer domain. The Figma `CodeConnectSnippet` is the authoritative pointer to which primitive to use.
|
|
59
59
|
|
|
60
60
|
## Swift-Specific Rules
|
|
61
61
|
|
|
@@ -246,7 +246,7 @@ bash $HOME/.claude/scripts/phase-tracker.sh model <N> <model_name>
|
|
|
246
246
|
|
|
247
247
|
Token counts are additive - multiple calls accumulate. `input_count` is FRESH input (cache-exclusive); the optional 4th arg is the prompt-cache-read count, priced at the discounted rate. `model` tags the phase so the card and the cost helper can price it. Stored per phase in the state file; surfaced in `:log` reports, on the bash card tile (`<elapsed> · <tok> tok · ~$<usd>`), and in the card footer (total USD + cached tokens).
|
|
248
248
|
|
|
249
|
-
**
|
|
249
|
+
**Required: per-phase token narration on completion (v9.10.2).** The native
|
|
250
250
|
TaskList widget cannot display per-phase tokens - it shows name, status, and
|
|
251
251
|
duration only. Without this rule the user sees durations and nothing else
|
|
252
252
|
until the Phase 7 Cost Breakdown. So whenever a phase transitions to
|
|
@@ -96,7 +96,7 @@ Re-fetching Figma in dev phases:
|
|
|
96
96
|
|
|
97
97
|
### Verification
|
|
98
98
|
|
|
99
|
-
The pipeline ships a smoke gate `pipeline/scripts/smoke-no-mcp-in-dev-phases.sh` that reads `state.telemetry.mcpCalls[]`. If any entry has `phase >= 2`, the gate fails.
|
|
99
|
+
The pipeline ships a smoke gate `pipeline/scripts/smoke-no-mcp-in-dev-phases.sh` that reads `state.telemetry.mcpCalls[]`. If any entry has `phase >= 2`, the gate fails. It is a maintainer-side regression check: it runs locally as part of `npm test` and the pre-push hook, not in CI (the CI workflow is dormant) and not after every pipeline run.
|
|
100
100
|
|
|
101
101
|
Memory: [[mcp-only-in-analysis]]
|
|
102
102
|
|
|
@@ -174,12 +174,14 @@ When Figma pipeline is active (`figmaConfigPath` set in project preferences),
|
|
|
174
174
|
multi-agent Phase 3 (DEV) dispatches Figma sub-phases instead of standard TDD:
|
|
175
175
|
|
|
176
176
|
```
|
|
177
|
-
Phase 3.0: figma-
|
|
178
|
-
Phase 3.1:
|
|
179
|
-
Phase 3.2: figma-commit
|
|
180
|
-
Phase 3.3: figma-iteration-commit
|
|
177
|
+
Phase 3.0: figma-validate → registry, Code Connect, token compliance (halts Phase 3 on failure)
|
|
178
|
+
Phase 3.1: create-component / create-screen → full implementation by the enabled stack plugin (evolve-component for an existing one)
|
|
179
|
+
Phase 3.2: figma-commit → 14-item review + commit + PR
|
|
180
|
+
Phase 3.3: figma-iteration-commit → Batch iteration commit (if iterating)
|
|
181
181
|
```
|
|
182
182
|
|
|
183
|
+
Skill names are the enabled stack plugin's (`ai-ios-toolkit:*` / `ai-android-toolkit:*`); the dispatch contract is `multi-agent-refs/component-dispatch.md`.
|
|
184
|
+
|
|
183
185
|
### Abstraction Layers
|
|
184
186
|
|
|
185
187
|
| Layer | Provider Options | Config |
|
|
@@ -219,70 +221,14 @@ Provider interfaces are defined inline in the plugin skill sets - see the `ai-
|
|
|
219
221
|
13. Build verification (xcodebuild)
|
|
220
222
|
14. Test verification (ViewInspector + Snapshot)
|
|
221
223
|
|
|
222
|
-
###
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
|
227
|
-
|
|
228
|
-
| `/figma-
|
|
229
|
-
| `/
|
|
230
|
-
| `/figma-to-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
| `/figma-to-swift-ui-code-connect <url>` | Phase 6 only (Code Connect) |
|
|
234
|
-
| `/figma-to-swift-ui-confluence-sync` | Sync wiki → Confluence |
|
|
235
|
-
| `/figma-to-swift-ui-status-update` | Refresh Confluence dashboard |
|
|
236
|
-
|
|
237
|
-
**Issue & Board:**
|
|
238
|
-
|
|
239
|
-
| Command | What It Does |
|
|
240
|
-
|---------|-------------|
|
|
241
|
-
| `/figma-issue open <url>` | Create GitHub Issue + optional Jira |
|
|
242
|
-
| `/figma-review <Name>` | Interactive review (approve/bug) |
|
|
243
|
-
| `/figma-validate <url>` | Pre-implementation validation |
|
|
244
|
-
|
|
245
|
-
**Commit & PR:**
|
|
246
|
-
|
|
247
|
-
| Command | What It Does |
|
|
248
|
-
|---------|-------------|
|
|
249
|
-
| `/figma-commit <Name>` | 14-item review + commit + PR |
|
|
250
|
-
| `/figma-iteration-commit <Name>` | Commit to iteration/develop + PR |
|
|
251
|
-
|
|
252
|
-
**Iteration & Batch:**
|
|
253
|
-
|
|
254
|
-
| Command | What It Does |
|
|
255
|
-
|---------|-------------|
|
|
256
|
-
| `/figma-iterate` | Auto loop: pick → implement → commit |
|
|
257
|
-
| `/figma-cli-iterate` | Full CLI iterate (all phases) |
|
|
258
|
-
| `/figma-cli-lean-iterate` | Lean iterate (skip test+wiki) |
|
|
259
|
-
| `/figma-cli-iterate-mend` | Re-implement discarded components |
|
|
260
|
-
| `/figma-cli-skip` | Mark component as skipped |
|
|
261
|
-
| `/figma-skip` | Skip + update Confluence |
|
|
262
|
-
|
|
263
|
-
**Bugfix:**
|
|
264
|
-
|
|
265
|
-
| Command | What It Does |
|
|
266
|
-
|---------|-------------|
|
|
267
|
-
| `/figma-fix` | Apply targeted bug fixes |
|
|
268
|
-
| `/figma-mend` | Re-implement from scratch |
|
|
269
|
-
|
|
270
|
-
**Setup & Utility:**
|
|
271
|
-
|
|
272
|
-
| Command | What It Does |
|
|
273
|
-
|---------|-------------|
|
|
274
|
-
| `/figma-setup` | Environment setup wizard |
|
|
275
|
-
| `/figma-utility` | Figma data fetch (screenshots, metadata) |
|
|
276
|
-
| `/figma-remote-mcp-auth` | Figma MCP OAuth flow |
|
|
277
|
-
| `/figma-ui-patterns` | UI pattern library index |
|
|
278
|
-
| `/figma-price-integration` | Price protocol adoption guide |
|
|
279
|
-
|
|
280
|
-
**Performance (batch production):**
|
|
281
|
-
|
|
282
|
-
| Command | What It Does |
|
|
283
|
-
|---------|-------------|
|
|
284
|
-
| `/performance-start #N` | Start component with perf tracking |
|
|
285
|
-
| `/performance-swiftui` | Perf-optimized pipeline |
|
|
286
|
-
| `/performance-tour` | Batch produce multiple components |
|
|
287
|
-
| `/performance-review-next` | Interactive batch review |
|
|
288
|
-
| `/performance-iteration-commit-all` | Batch validate + push all |
|
|
224
|
+
### Figma Skill Surfaces (current)
|
|
225
|
+
|
|
226
|
+
The old 31-command personal catalog is retired; none of those `/figma-*` commands ships with the pipeline. Figma capabilities live on three surfaces. Screens are ALWAYS drawn 1:1 from the Figma design context resolved in analysis, and a Code Connect-mapped component is ALWAYS used when one exists (see the fallback chain + Code Connect rules above).
|
|
227
|
+
|
|
228
|
+
| Surface | What it carries | When to use |
|
|
229
|
+
|---------|-----------------|-------------|
|
|
230
|
+
| `ai-ios-toolkit` / `ai-android-toolkit` (marketplace stack plugins, enabled per repo by `/multi-agent:stack`; skill names below are the iOS plugin's, the Android plugin carries the Compose equivalents) | Component skills: `create-component`, `create-screen`, `evolve-component`, `figma-validate`, `figma-review`, `figma-commit`, `figma-iteration-commit`, `figma-component-start`, `figma-utility`, `figma-setup`, `code-connect`, `component-docs`, `component-wiki`, plus the `navigation` / `overlays` / `bottom-sheets` / `ui-patterns` reference skills | Direct component work in a repo where the stack plugin is enabled; Phase 3 dispatches here for `taskType === component` |
|
|
231
|
+
| Pipeline commands | `/multi-agent:analysis` (the only phase allowed to fetch Figma), `/multi-agent:design-check` (mock-mode vs Figma conformance, Phase 4 Step 2.8), `/multi-agent:review` (cites the analysis doc, never Figma) | Inside multi-agent pipeline runs |
|
|
232
|
+
| Legacy `/figma-to-swiftui` (single standalone command) | The old full 8-phase flow | Kept for the old flow only; prefer the plugin's `create-component` for new work |
|
|
233
|
+
|
|
234
|
+
The pipeline keeps no Figma command catalog of its own: the routing table for component skills is maintained inside each stack plugin's `index` skill, and `/multi-agent:stack` decides which plugin is active for a repo.
|
|
@@ -21,6 +21,7 @@ available. Read the effective `enabledPlugins` and load each enabled toolkit's
|
|
|
21
21
|
`ai-common-toolkit` and `ai-analyst-toolkit` are on everywhere. Nothing enabled
|
|
22
22
|
is a normal state.
|
|
23
23
|
|
|
24
|
-
**multi-agent-toolkit MCP.** 80+ tools for a running app: `
|
|
25
|
-
`
|
|
26
|
-
|
|
24
|
+
**multi-agent-toolkit MCP.** 80+ tools for a running app: `ios_get_ui_tree` /
|
|
25
|
+
`android_get_ui_tree`, `ios_list_crashes` / `android_list_crashes`, `design_*`,
|
|
26
|
+
`ios_app_store_audit`, `ios_testflight_validate`. Use them instead of guessing
|
|
27
|
+
about on-screen state. Not registered is a silent no-op.
|
|
@@ -107,7 +107,7 @@
|
|
|
107
107
|
"siblings": {
|
|
108
108
|
"type": "array",
|
|
109
109
|
"maxItems": 10,
|
|
110
|
-
"description": "Repos the dev-context picker offered that this run does not modify: read-only siblings, plus any extra the user selected. Persisted at Phase 0 because the phases that consume them run much later - Phase 4's platform-parity cross-check reads this and
|
|
110
|
+
"description": "Repos the dev-context picker offered that this run does not modify: read-only siblings, plus any extra the user selected. Persisted at Phase 0 because the phases that consume them run much later - Phase 4's platform-parity cross-check reads this as the fourth of its four counterpart sources (after --with, prefs.projects[<slug>].counterpartRoots[] and the primary checkout's sibling directories; see multi-agent-refs/platform-parity.md), so a picker result that is not written here is a candidate the check can never see.",
|
|
111
111
|
"items": {
|
|
112
112
|
"type": "object",
|
|
113
113
|
"additionalProperties": false,
|
|
@@ -650,6 +650,90 @@
|
|
|
650
650
|
"type": "string",
|
|
651
651
|
"enum": ["fix", "accept", "escalate"]
|
|
652
652
|
},
|
|
653
|
+
"delta": {
|
|
654
|
+
"type": "object",
|
|
655
|
+
"additionalProperties": true,
|
|
656
|
+
"description": "Written by Phase 4 Step 3.8 from review-delta.mjs: how this round's accepted findings relate to the previous round's rework mandate. Absent on iteration 1 and on runs from before the delta existed.",
|
|
657
|
+
"properties": {
|
|
658
|
+
"previousIteration": { "type": "integer", "minimum": 1 },
|
|
659
|
+
"new": {
|
|
660
|
+
"type": "array",
|
|
661
|
+
"items": {
|
|
662
|
+
"type": "object",
|
|
663
|
+
"additionalProperties": true,
|
|
664
|
+
"properties": {
|
|
665
|
+
"fingerprint": { "type": "string" },
|
|
666
|
+
"severity": { "type": ["string", "null"] },
|
|
667
|
+
"file": { "type": ["string", "null"] },
|
|
668
|
+
"issue": { "type": ["string", "null"] }
|
|
669
|
+
}
|
|
670
|
+
}
|
|
671
|
+
},
|
|
672
|
+
"stillPresent": {
|
|
673
|
+
"type": "array",
|
|
674
|
+
"items": {
|
|
675
|
+
"type": "object",
|
|
676
|
+
"additionalProperties": true,
|
|
677
|
+
"properties": {
|
|
678
|
+
"fingerprint": { "type": "string" },
|
|
679
|
+
"severity": { "type": ["string", "null"] },
|
|
680
|
+
"file": { "type": ["string", "null"] },
|
|
681
|
+
"issue": { "type": ["string", "null"] }
|
|
682
|
+
}
|
|
683
|
+
}
|
|
684
|
+
},
|
|
685
|
+
"resolved": {
|
|
686
|
+
"type": "array",
|
|
687
|
+
"items": {
|
|
688
|
+
"type": "object",
|
|
689
|
+
"additionalProperties": true,
|
|
690
|
+
"properties": {
|
|
691
|
+
"fingerprint": { "type": "string" },
|
|
692
|
+
"severity": { "type": ["string", "null"] },
|
|
693
|
+
"file": { "type": ["string", "null"] },
|
|
694
|
+
"issue": { "type": ["string", "null"] }
|
|
695
|
+
}
|
|
696
|
+
}
|
|
697
|
+
},
|
|
698
|
+
"downgraded": {
|
|
699
|
+
"type": "array",
|
|
700
|
+
"items": {
|
|
701
|
+
"type": "object",
|
|
702
|
+
"additionalProperties": true,
|
|
703
|
+
"properties": {
|
|
704
|
+
"fingerprint": { "type": "string" },
|
|
705
|
+
"severity": { "type": ["string", "null"] },
|
|
706
|
+
"file": { "type": ["string", "null"] },
|
|
707
|
+
"issue": { "type": ["string", "null"] }
|
|
708
|
+
}
|
|
709
|
+
}
|
|
710
|
+
},
|
|
711
|
+
"stillPresentBlocking": {
|
|
712
|
+
"type": "array",
|
|
713
|
+
"items": {
|
|
714
|
+
"type": "object",
|
|
715
|
+
"additionalProperties": true,
|
|
716
|
+
"properties": {
|
|
717
|
+
"fingerprint": { "type": "string" },
|
|
718
|
+
"severity": { "type": ["string", "null"] },
|
|
719
|
+
"file": { "type": ["string", "null"] },
|
|
720
|
+
"issue": { "type": ["string", "null"] }
|
|
721
|
+
}
|
|
722
|
+
}
|
|
723
|
+
},
|
|
724
|
+
"recurrence": {
|
|
725
|
+
"type": "object",
|
|
726
|
+
"additionalProperties": { "type": "integer", "minimum": 1 },
|
|
727
|
+
"description": "fingerprint -> consecutive rework cycles the finding has survived."
|
|
728
|
+
},
|
|
729
|
+
"plateau": {
|
|
730
|
+
"type": "boolean",
|
|
731
|
+
"description": "The still-present set is unchanged from the previous delta and non-empty."
|
|
732
|
+
},
|
|
733
|
+
"tripped": { "type": "boolean" },
|
|
734
|
+
"computedAt": { "type": "string", "format": "date-time" }
|
|
735
|
+
}
|
|
736
|
+
},
|
|
653
737
|
"reviewers": {
|
|
654
738
|
"type": "array",
|
|
655
739
|
"description": "One entry per reviewer dispatch that RETURNED. Typed because two consumers depend on the shape: anonymize-findings.mjs needs model+findings to build the label map, and run-metrics.mjs reports acceptedRatio per reviewer. Extra keys are allowed; nothing is required, so a run written before this shape existed still validates and surfaces as model \"unknown\" rather than failing.",
|
|
@@ -699,6 +783,51 @@
|
|
|
699
783
|
}
|
|
700
784
|
}
|
|
701
785
|
},
|
|
786
|
+
"circuitBreaker": {
|
|
787
|
+
"type": "object",
|
|
788
|
+
"additionalProperties": false,
|
|
789
|
+
"description": "Autopilot circuit-breaker record (refs/features/autopilot-circuit-breaker.md). Written only when a trigger trips: trigger 2 by Phase 4 Step 3.8 (a mandate finding survived identicalFindingCycles rework cycles), trigger 3 by the Phase 3 re-entry hard-kill. /multi-agent:resume clears tripped and keeps counters.",
|
|
790
|
+
"required": ["tripped"],
|
|
791
|
+
"properties": {
|
|
792
|
+
"tripped": { "type": "boolean" },
|
|
793
|
+
"trigger": { "type": ["integer", "null"], "minimum": 1, "maximum": 5 },
|
|
794
|
+
"detail": { "type": "string" },
|
|
795
|
+
"checkpoint": {
|
|
796
|
+
"type": "object",
|
|
797
|
+
"additionalProperties": false,
|
|
798
|
+
"properties": {
|
|
799
|
+
"phase": { "type": "integer", "minimum": 0, "maximum": 7 },
|
|
800
|
+
"step": { "type": "string" },
|
|
801
|
+
"iteration": { "type": "integer", "minimum": 1 }
|
|
802
|
+
}
|
|
803
|
+
},
|
|
804
|
+
"trippedAt": { "type": "string", "format": "date-time" },
|
|
805
|
+
"counters": {
|
|
806
|
+
"type": "object",
|
|
807
|
+
"additionalProperties": false,
|
|
808
|
+
"properties": {
|
|
809
|
+
"identicalFindingCycles": { "type": "integer", "minimum": 0 },
|
|
810
|
+
"reworkCycles": { "type": "integer", "minimum": 0 }
|
|
811
|
+
}
|
|
812
|
+
}
|
|
813
|
+
}
|
|
814
|
+
},
|
|
815
|
+
"diffRisk": {
|
|
816
|
+
"type": "object",
|
|
817
|
+
"additionalProperties": true,
|
|
818
|
+
"description": "Totals from diff-risk-score.mjs, persisted by Phase 4 Step 1.75 so Phase 6 (PR risk section), Phase 7 and run-metrics.mjs read the same numbers the review scope decision used.",
|
|
819
|
+
"properties": {
|
|
820
|
+
"files": { "type": "integer", "minimum": 0 },
|
|
821
|
+
"loc_added": { "type": "integer", "minimum": 0 },
|
|
822
|
+
"loc_removed": { "type": "integer", "minimum": 0 },
|
|
823
|
+
"max_score": { "type": "number" },
|
|
824
|
+
"signals": {
|
|
825
|
+
"type": "array",
|
|
826
|
+
"items": { "type": "string" },
|
|
827
|
+
"description": "Distinct signal names seen across files (security_path, migration, public_api, no_test_change, test_lines_removed, ...)."
|
|
828
|
+
}
|
|
829
|
+
}
|
|
830
|
+
},
|
|
702
831
|
"confluenceSpace": {
|
|
703
832
|
"type": ["string", "null"],
|
|
704
833
|
"description": "Cached Confluence space key for the project - avoids re-asking on every run."
|
|
@@ -100,6 +100,11 @@
|
|
|
100
100
|
"description": "Citation. Format: 'rules/<file>.md#<anchor>' or 'gate/<name>'."
|
|
101
101
|
},
|
|
102
102
|
"issue": { "type": "string", "description": "Short, one-sentence description." },
|
|
103
|
+
"fingerprint": {
|
|
104
|
+
"type": "string",
|
|
105
|
+
"pattern": "^F:[0-9a-f]{8}$",
|
|
106
|
+
"description": "Stable id computed by finding-fingerprint.mjs --kind dev-critic. Round-2 findings carry the round-1 fingerprint; a round-2 finding with none is the scope creep the loop contract forbids."
|
|
107
|
+
},
|
|
103
108
|
"fix": {
|
|
104
109
|
"type": "string",
|
|
105
110
|
"description": "Concrete suggestion the generator can act on without re-reading the rule."
|