@mmerterden/multi-agent-pipeline 18.0.0 → 19.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (234) hide show
  1. package/CHANGELOG.md +287 -0
  2. package/README.md +36 -20
  3. package/README.tr.md +14 -16
  4. package/docs/adr/0002-instruction-driven-flag.md +1 -0
  5. package/docs/adr/0005-lazy-phase-docs.md +11 -1
  6. package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
  7. package/docs/adr/0010-own-code-graph.md +1 -0
  8. package/docs/adr/0014-six-phase-consolidation.md +134 -0
  9. package/docs/adr/README.md +2 -1
  10. package/docs/architecture.md +37 -38
  11. package/docs/best-practices.md +1 -1
  12. package/docs/ecosystem.md +46 -27
  13. package/docs/engineering.md +1 -1
  14. package/docs/facts.json +61 -0
  15. package/docs/features.md +55 -54
  16. package/docs/performance.md +5 -5
  17. package/docs/recovery-guide.md +17 -17
  18. package/docs/token-budget-history.md +3 -1
  19. package/index.js +2 -2
  20. package/install/_codex-agents.mjs +1 -1
  21. package/install/templates/claude-hooks.json +1 -1
  22. package/install/templates/codex-instructions.md +1 -1
  23. package/install/templates/copilot-instructions.md +28 -28
  24. package/manifest.json +234 -216
  25. package/package.json +2 -2
  26. package/pipeline/agents/dev-critic.md +7 -7
  27. package/pipeline/commands/figma-to-swiftui.md +1 -1
  28. package/pipeline/commands/multi-agent/SKILL.md +9 -9
  29. package/pipeline/commands/multi-agent/analysis/SKILL.md +15 -15
  30. package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +2 -2
  31. package/pipeline/commands/multi-agent/autopilot/SKILL.md +7 -7
  32. package/pipeline/commands/multi-agent/channels/SKILL.md +15 -15
  33. package/pipeline/commands/multi-agent/diff-explain/SKILL.md +6 -6
  34. package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
  35. package/pipeline/commands/multi-agent/graph/SKILL.md +1 -1
  36. package/pipeline/commands/multi-agent/help/SKILL.md +62 -62
  37. package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
  38. package/pipeline/commands/multi-agent/local/SKILL.md +11 -11
  39. package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +13 -13
  40. package/pipeline/commands/multi-agent/log/SKILL.md +2 -2
  41. package/pipeline/commands/multi-agent/manual-test/SKILL.md +9 -9
  42. package/pipeline/commands/multi-agent/model/SKILL.md +69 -0
  43. package/pipeline/commands/multi-agent/refactor/SKILL.md +3 -3
  44. package/pipeline/commands/multi-agent/resume/SKILL.md +4 -4
  45. package/pipeline/commands/multi-agent/resume-local/SKILL.md +19 -17
  46. package/pipeline/commands/multi-agent/review/SKILL.md +2 -2
  47. package/pipeline/commands/multi-agent/review-analysis/SKILL.md +1 -1
  48. package/pipeline/commands/multi-agent/route-off/SKILL.md +36 -0
  49. package/pipeline/commands/multi-agent/route-on/SKILL.md +74 -0
  50. package/pipeline/commands/multi-agent/route-status/SKILL.md +56 -0
  51. package/pipeline/commands/multi-agent/setup/SKILL.md +2 -2
  52. package/pipeline/commands/multi-agent/status/SKILL.md +5 -5
  53. package/pipeline/commands/multi-agent/steer/SKILL.md +2 -2
  54. package/pipeline/commands/multi-agent/sync/SKILL.md +12 -13
  55. package/pipeline/commands/multi-agent/test/SKILL.md +1 -1
  56. package/pipeline/lib/credential-inventory.sh +1 -1
  57. package/pipeline/lib/fetch-fortify.sh +1 -1
  58. package/pipeline/lib/model-dispatch.sh +140 -0
  59. package/pipeline/lib/model-rung.sh +142 -0
  60. package/pipeline/lib/outbound-gate.mjs +14 -0
  61. package/pipeline/lib/phase-schema.mjs +88 -0
  62. package/pipeline/lib/plan-todos.sh +5 -5
  63. package/pipeline/lib/route-state.sh +161 -0
  64. package/pipeline/lib/run-paths.sh +2 -2
  65. package/pipeline/multi-agent-refs/_account-picker.md +1 -1
  66. package/pipeline/multi-agent-refs/_dev-context.md +6 -6
  67. package/pipeline/multi-agent-refs/_input-parser.md +1 -1
  68. package/pipeline/multi-agent-refs/analysis/evidence.md +2 -11
  69. package/pipeline/multi-agent-refs/analysis/intake.md +7 -7
  70. package/pipeline/multi-agent-refs/analysis/locked.md +48 -22
  71. package/pipeline/multi-agent-refs/analysis/redesign.md +1 -1
  72. package/pipeline/multi-agent-refs/analysis/render.md +10 -10
  73. package/pipeline/multi-agent-refs/analysis/resolve.md +1 -1
  74. package/pipeline/multi-agent-refs/analysis/review.md +2 -2
  75. package/pipeline/multi-agent-refs/analysis/synthesis.md +13 -7
  76. package/pipeline/multi-agent-refs/analysis-template-corporate.md +9 -9
  77. package/pipeline/multi-agent-refs/analysis-template.md +19 -19
  78. package/pipeline/multi-agent-refs/android-guide.md +1 -1
  79. package/pipeline/multi-agent-refs/audit-guide.md +13 -13
  80. package/pipeline/multi-agent-refs/channels/issue-comment.md +2 -2
  81. package/pipeline/multi-agent-refs/channels/jira.md +3 -3
  82. package/pipeline/multi-agent-refs/channels/pr.md +4 -4
  83. package/pipeline/multi-agent-refs/channels/wiki.md +1 -1
  84. package/pipeline/multi-agent-refs/component-dispatch.md +8 -8
  85. package/pipeline/multi-agent-refs/conventions-defaults.md +2 -2
  86. package/pipeline/multi-agent-refs/cross-cli-contract.md +31 -6
  87. package/pipeline/multi-agent-refs/features/analysis-jira.md +1 -1
  88. package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +4 -4
  89. package/pipeline/multi-agent-refs/features/code-graph.md +5 -5
  90. package/pipeline/multi-agent-refs/features/design-conformance.md +1 -1
  91. package/pipeline/multi-agent-refs/features/dev-critic.md +3 -3
  92. package/pipeline/multi-agent-refs/features/doctor.md +3 -3
  93. package/pipeline/multi-agent-refs/features/external-context-injection.md +3 -3
  94. package/pipeline/multi-agent-refs/features/maturity-followup.md +3 -3
  95. package/pipeline/multi-agent-refs/features/model-fallback.md +41 -5
  96. package/pipeline/multi-agent-refs/features/plan-todos.md +1 -1
  97. package/pipeline/multi-agent-refs/features/repo-map.md +1 -1
  98. package/pipeline/multi-agent-refs/features/review-delta.md +3 -3
  99. package/pipeline/multi-agent-refs/features/review-multi-repo.md +2 -2
  100. package/pipeline/multi-agent-refs/features/scope-check.md +4 -4
  101. package/pipeline/multi-agent-refs/features/skill-conformance.md +2 -2
  102. package/pipeline/multi-agent-refs/features/stack-skill-routing.md +1 -1
  103. package/pipeline/multi-agent-refs/features/url-enrichment.md +1 -1
  104. package/pipeline/multi-agent-refs/features/verify-by-test.md +4 -4
  105. package/pipeline/multi-agent-refs/features/visual-evidence.md +19 -19
  106. package/pipeline/multi-agent-refs/features/worktree-finalize.md +6 -6
  107. package/pipeline/multi-agent-refs/issue-jira-triad.md +10 -10
  108. package/pipeline/multi-agent-refs/knowledge.md +11 -11
  109. package/pipeline/multi-agent-refs/multi-repo-integration-build.md +13 -13
  110. package/pipeline/multi-agent-refs/payload-contracts.md +8 -8
  111. package/pipeline/multi-agent-refs/phases/log-format.md +10 -10
  112. package/pipeline/multi-agent-refs/phases/modes.md +30 -30
  113. package/pipeline/multi-agent-refs/phases/operations.md +8 -8
  114. package/pipeline/multi-agent-refs/phases/phase-0-init.md +24 -24
  115. package/pipeline/multi-agent-refs/phases/phase-1-plan.md +599 -0
  116. package/pipeline/multi-agent-refs/phases/{phase-3-dev.md → phase-2-dev.md} +129 -49
  117. package/pipeline/multi-agent-refs/phases/{phase-4-review.md → phase-3-review.md} +225 -107
  118. package/pipeline/multi-agent-refs/phases/{phase-6-commit.md → phase-4-commit.md} +23 -23
  119. package/pipeline/multi-agent-refs/phases/{phase-7-report.md → phase-5-report.md} +29 -29
  120. package/pipeline/multi-agent-refs/phases.md +44 -48
  121. package/pipeline/multi-agent-refs/picker-contract.md +1 -1
  122. package/pipeline/multi-agent-refs/progress-contract.md +6 -6
  123. package/pipeline/multi-agent-refs/readiness-review.md +1 -1
  124. package/pipeline/multi-agent-refs/rules.md +7 -7
  125. package/pipeline/multi-agent-refs/swiftui-guide.md +2 -2
  126. package/pipeline/multi-agent-refs/tracker-contract.md +31 -32
  127. package/pipeline/multi-agent-refs/wiki-capture.md +14 -14
  128. package/pipeline/preferences-template.json +9 -1
  129. package/pipeline/rules/figma-pipeline.md +8 -8
  130. package/pipeline/rules/outside-the-pipeline.md +1 -1
  131. package/pipeline/schemas/agent-state.schema.json +50 -50
  132. package/pipeline/schemas/analysis-output.schema.json +3 -3
  133. package/pipeline/schemas/analysis-spec.schema.json +2 -2
  134. package/pipeline/schemas/autopilot-config.schema.json +1 -1
  135. package/pipeline/schemas/code-graph.schema.json +1 -1
  136. package/pipeline/schemas/criteria-manifest.schema.json +1 -1
  137. package/pipeline/schemas/dev-critic-output.schema.json +1 -1
  138. package/pipeline/schemas/diff-risk.schema.json +1 -1
  139. package/pipeline/schemas/figma-project-config.schema.json +1 -1
  140. package/pipeline/schemas/migrations/prefs-2.4.0-to-2.5.0.mjs +2 -2
  141. package/pipeline/schemas/migrations/prefs-2.6.0-to-2.7.0.mjs +31 -0
  142. package/pipeline/schemas/migrations/state-2.1.0-to-2.2.0.mjs +129 -0
  143. package/pipeline/schemas/phases.json +105 -0
  144. package/pipeline/schemas/plan-todos.schema.json +5 -5
  145. package/pipeline/schemas/planning-output.schema.json +1 -1
  146. package/pipeline/schemas/prefs.schema.json +102 -58
  147. package/pipeline/schemas/reviewer-output.schema.json +3 -3
  148. package/pipeline/schemas/route-config.schema.json +74 -0
  149. package/pipeline/schemas/scope-check.schema.json +1 -1
  150. package/pipeline/schemas/secret-patterns.json +124 -0
  151. package/pipeline/schemas/test-gap.schema.json +1 -1
  152. package/pipeline/schemas/token-budget.json +12 -18
  153. package/pipeline/schemas/triage-output.schema.json +6 -6
  154. package/pipeline/scripts/README.md +3 -3
  155. package/pipeline/scripts/_code-graph.mjs +2 -2
  156. package/pipeline/scripts/_run-paths.mjs +2 -2
  157. package/pipeline/scripts/_smoke-root.sh +1 -1
  158. package/pipeline/scripts/aggregate-metrics.mjs +1 -1
  159. package/pipeline/scripts/build-references.mjs +2 -2
  160. package/pipeline/scripts/bulk-read.sh +10 -1
  161. package/pipeline/scripts/capture-flush.sh +8 -8
  162. package/pipeline/scripts/capture-resume.sh +3 -3
  163. package/pipeline/scripts/classify-plan-safety.mjs +1 -1
  164. package/pipeline/scripts/cost-table.json +8 -1
  165. package/pipeline/scripts/diff-explain.mjs +1 -1
  166. package/pipeline/scripts/doctor.mjs +3 -3
  167. package/pipeline/scripts/gc-abandoned.sh +3 -3
  168. package/pipeline/scripts/gc-tmp.sh +1 -1
  169. package/pipeline/scripts/gc-worktrees.sh +1 -1
  170. package/pipeline/scripts/gen-facts.mjs +280 -0
  171. package/pipeline/scripts/gen-mode-dispatch.mjs +32 -37
  172. package/pipeline/scripts/gen-ref-toc.mjs +1 -1
  173. package/pipeline/scripts/graph-report.mjs +1 -1
  174. package/pipeline/scripts/jira-attach.sh +1 -1
  175. package/pipeline/scripts/learn-from-transcripts.mjs +1 -1
  176. package/pipeline/scripts/learning-curve.mjs +2 -2
  177. package/pipeline/scripts/log-metric.sh +17 -4
  178. package/pipeline/scripts/memory-save.sh +1 -1
  179. package/pipeline/scripts/migrate-prefs.mjs +22 -5
  180. package/pipeline/scripts/phase-banner.sh +20 -20
  181. package/pipeline/scripts/phase-tracker.sh +12 -12
  182. package/pipeline/scripts/plan-coverage-gate.mjs +2 -2
  183. package/pipeline/scripts/pre-commit-check.sh +30 -1
  184. package/pipeline/scripts/render-agent-log-cost.sh +1 -1
  185. package/pipeline/scripts/render-work-summary.sh +3 -3
  186. package/pipeline/scripts/review-file-filter.mjs +1 -1
  187. package/pipeline/scripts/run-aggregator.mjs +13 -6
  188. package/pipeline/scripts/run-metrics.mjs +1 -1
  189. package/pipeline/scripts/runs-index.mjs +11 -1
  190. package/pipeline/scripts/scan-skills.sh +26 -0
  191. package/pipeline/scripts/smoke-cross-cli-behavior.sh +6 -6
  192. package/pipeline/scripts/smoke-schema-validation.sh +26 -7
  193. package/pipeline/scripts/token-budget-report.mjs +13 -2
  194. package/pipeline/scripts/triage-memory.mjs +2 -2
  195. package/pipeline/scripts/validate-analysis-doc.mjs +274 -43
  196. package/pipeline/scripts/validate-planning.mjs +1 -1
  197. package/pipeline/scripts/validate-reviewer.mjs +1 -1
  198. package/pipeline/scripts/validate-state.mjs +45 -5
  199. package/pipeline/scripts/validate-triage.mjs +3 -3
  200. package/pipeline/scripts/verify-citations.mjs +1 -1
  201. package/pipeline/scripts/worktree-finalize.sh +5 -5
  202. package/pipeline/scripts/write-state.mjs +32 -0
  203. package/pipeline/skills/.skill-manifest.json +38 -22
  204. package/pipeline/skills/.skills-index.json +49 -5
  205. package/pipeline/skills/shared/README.md +10 -6
  206. package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +8 -8
  207. package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +8 -8
  208. package/pipeline/skills/shared/core/multi-agent/SKILL.md +81 -82
  209. package/pipeline/skills/shared/core/multi-agent-autopilot/SKILL.md +3 -3
  210. package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +14 -14
  211. package/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md +5 -5
  212. package/pipeline/skills/shared/core/multi-agent-graph/SKILL.md +1 -1
  213. package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +25 -23
  214. package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
  215. package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +2 -2
  216. package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +8 -8
  217. package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +6 -6
  218. package/pipeline/skills/shared/core/multi-agent-model/SKILL.md +71 -0
  219. package/pipeline/skills/shared/core/multi-agent-refactor/SKILL.md +3 -3
  220. package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +1 -1
  221. package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +7 -7
  222. package/pipeline/skills/shared/core/multi-agent-route-off/SKILL.md +39 -0
  223. package/pipeline/skills/shared/core/multi-agent-route-on/SKILL.md +76 -0
  224. package/pipeline/skills/shared/core/multi-agent-route-status/SKILL.md +59 -0
  225. package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +1 -1
  226. package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +5 -5
  227. package/pipeline/skills/shared/core/multi-agent-steer/SKILL.md +2 -2
  228. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +6 -5
  229. package/pipeline/skills/shared/external/NOTICE-swift-ios-skills.md +1 -1
  230. package/pipeline/skills/shared/external/signal-community/SKILL.md +8 -1
  231. package/pipeline/skills/skills-index.md +8 -4
  232. package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +0 -263
  233. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +0 -344
  234. package/pipeline/multi-agent-refs/phases/phase-5-test.md +0 -182
@@ -1,6 +1,6 @@
1
1
  ---
2
- description: "Continue already-done LOCAL work through the pipeline tail: Review → Build+Test → Commit/PR → Report (technical analysis + Jira test-scenario comment). No dev phase. Use when local work is already done and only review, build, commit and reporting remain."
3
- description-tr: "Halihazırda bitmiş LOKAL işi pipeline kuyruğundan geçirir: Review → Build+Test → Commit/PR → Report (teknik analiz + Jira test-senaryosu yorumu). Dev fazı yok."
2
+ description: "Continue already-done LOCAL work through the pipeline tail: Review (with its build gate) → Commit/PR → Report (technical analysis + Jira test-scenario comment). No dev phase. Use when local work is already done and only review, build, commit and reporting remain."
3
+ description-tr: "Halihazırda bitmiş LOKAL işi pipeline kuyruğundan geçirir: Review (build kapısıyla) → Commit/PR → Report (teknik analiz + Jira test-senaryosu yorumu). Dev fazı yok."
4
4
  allowed-tools: Agent, Bash, Read, Write, Edit, Glob, Grep, TaskCreate, TaskUpdate, TaskList, TaskGet, AskUserQuestion, WebFetch, WebSearch, Skill
5
5
  ---
6
6
 
@@ -25,7 +25,7 @@ You already did the work locally - wrote code on the current branch and maybe
25
25
 
26
26
  ```bash
27
27
  /multi-agent:resume-local # current branch vs its base; resolve Jira id from branch name
28
- /multi-agent:resume-local PROJ-12345 # bind to an explicit Jira id for the Phase 7 comment
28
+ /multi-agent:resume-local PROJ-12345 # bind to an explicit Jira id for the Phase 5 comment
29
29
  /multi-agent:resume-local --base develop # override the base branch for the diff
30
30
  /multi-agent:resume-local autopilot # no gate prompts: auto-fix blocking findings, auto-PR, auto-comment
31
31
  ```
@@ -34,13 +34,15 @@ You already did the work locally - wrote code on the current branch and maybe
34
34
 
35
35
  ```
36
36
  Phase 0: Init → project/branch detect, resolve base + diff (work-already-done), Jira id, state (NO worktree)
37
- Phase 4: Review → deterministic gates + parallel review (Fable + Opus + Sonnet) + Fable triage
38
- Phase 5: Build+Test → stack-aware build gate + run existing tests; SUCCESS required (automated, not the interactive user-test)
39
- Phase 6: Commit → commit remaining local changes + push + open PR if none exists
40
- Phase 7: Report → technical analysis + Jira comment with test scenarios (channels: Jira / PR / Confluence / Wiki)
37
+ Phase 3: Review → the Verify gate first (stack-aware build + existing tests; SUCCESS required), then
38
+ parallel review (Fable + Opus + Sonnet) + Fable triage
39
+ Phase 4: Commit → commit remaining local changes + push + open PR if none exists
40
+ Phase 5: Report → technical analysis + Jira comment with test scenarios (channels: Jira / PR / Confluence / Wiki)
41
41
  ```
42
42
 
43
- Phases 1-3 (Analysis / Planning / Dev) are skipped by design - `ship` treats the current branch's local changes as the Phase 3 output.
43
+ Phases 1-2 (Plan / Dev) are skipped by design - `ship` treats the current branch's local changes as the Phase 2 output.
44
+
45
+ **Why Review runs the build gate here.** Verify is Phase 2 Dev's exit gate since v19.0.0, and this mode has no Dev: the work arrived already written. The gate still has to run, so Review runs it before dispatching reviewers, which is what Review Stage 1 did before the consolidation. Reviewing a branch whose build was never checked is the failure this ordering prevents.
44
46
 
45
47
  ## Phase 0 - Context resolution (finish-specific)
46
48
 
@@ -52,12 +54,12 @@ Phases 1-3 (Analysis / Planning / Dev) are skipped by design - `ship` treats t
52
54
 
53
55
  ## Phase execution (reuse the existing phase contracts)
54
56
 
55
- - **Phase 4 Review** - run per `$HOME/.claude/multi-agent-refs/phases/phase-4-review.md` against the resolved diff: deterministic gates (Step 1.x), stack-specific parallel reviewers (Fable + Opus + Sonnet on Claude Code; GPT + Opus + Sonnet on Copilot CLI), Fable triage → `triage.accepted`. Blocking/important accepted findings:
57
+ - **Phase 3 Review** - run per `$HOME/.claude/multi-agent-refs/phases/phase-3-review.md` against the resolved diff: deterministic gates (Step 1.x), stack-specific parallel reviewers (Fable + Opus + Sonnet on Claude Code; GPT + Opus + Sonnet on Copilot CLI), Fable triage → `triage.accepted`. Blocking/important accepted findings:
56
58
  - interactive: present them and ask (`AskUserQuestion`) whether to fix now (loop back through a minimal Phase-3-style TDD fix) or proceed;
57
59
  - `autopilot` (or `prefs.global.resumeLocal.autoFix == true`): auto-fix accepted blocking/important findings, then re-review the fix, before advancing.
58
- - **Phase 5 Build+Test** - the **automated success gate** (this is what "build+test success" means here; the interactive device user-test is `/multi-agent:manual-test`). Stack-aware: build via `figma-config.build` (iOS scheme / Android gradle / detected backend/web build) and run the existing test suite if present (`swift test` / `xcodebuild test` / `./gradlew test` / `pytest` / `npm test` / `vitest`). Require success to advance; on failure, surface logs and (interactive) stop or (autopilot) attempt a bounded fix loop. **If the repo has no tests, report "no tests present" - never fabricate test results.**
59
- - **Phase 6 Commit/PR** - per `$HOME/.claude/multi-agent-refs/phases/phase-6-commit.md`: stage + commit any remaining local changes with a conventional message (`{type}(scope): desc [{JIRA_KEY}-{id}]`), push, and open a PR **only if one does not already exist** for the branch. PR body per `$HOME/.claude/multi-agent-refs/rules.md "External System Outputs"` and `$HOME/.claude/rules/git-conventions.md` - `Ref: #N` / `Related: #N`, never `Closes/Fixes/Resolves`; NO AI/bot attribution anywhere.
60
- - **Phase 7 Report** - per `$HOME/.claude/multi-agent-refs/phases/phase-7-report.md` + `channels.md`: produce the **technical analysis** and **test scenarios**, then post to the configured channels. Default content for `ship`: a Jira **comment** carrying the technical analysis + the test scenarios (and, when the PR was opened, the PR description). Every body runs through the humanizer; bot/tool/AI signatures are FORBIDDEN in comments.
60
+ - **Phase 3 Verify gate** - the **automated success gate** (this is what "build+test success" means here; the interactive device user-test is `/multi-agent:manual-test`). Stack-aware: build via `figma-config.build` (iOS scheme / Android gradle / detected backend/web build) and run the existing test suite if present (`swift test` / `xcodebuild test` / `./gradlew test` / `pytest` / `npm test` / `vitest`). Require success to advance; on failure, surface logs and (interactive) stop or (autopilot) attempt a bounded fix loop. **If the repo has no tests, report "no tests present" - never fabricate test results.**
61
+ - **Phase 4 Commit/PR** - per `$HOME/.claude/multi-agent-refs/phases/phase-4-commit.md`: stage + commit any remaining local changes with a conventional message (`{type}(scope): desc [{JIRA_KEY}-{id}]`), push, and open a PR **only if one does not already exist** for the branch. PR body per `$HOME/.claude/multi-agent-refs/rules.md "External System Outputs"` and `$HOME/.claude/rules/git-conventions.md` - `Ref: #N` / `Related: #N`, never `Closes/Fixes/Resolves`; NO AI/bot attribution anywhere.
62
+ - **Phase 5 Report** - per `$HOME/.claude/multi-agent-refs/phases/phase-5-report.md` + `channels.md`: produce the **technical analysis** and **test scenarios**, then post to the configured channels. Default content for `ship`: a Jira **comment** carrying the technical analysis + the test scenarios (and, when the PR was opened, the PR description). Every body runs through the humanizer; bot/tool/AI signatures are FORBIDDEN in comments.
61
63
 
62
64
  ## Modes
63
65
 
@@ -70,17 +72,17 @@ Before writing anything outward-facing - PR body, Jira comment, Confluence pag
70
72
 
71
73
  ## Required: Phase Tracker Contract
72
74
 
73
- **The phase tracker is required.** Full spec: [`$HOME/.claude/multi-agent-refs/tracker-contract.md`]($HOME/.claude/multi-agent-refs/tracker-contract.md). `ship` registers its active phase set - `0:Init 4:Review 5:Build+Test 6:Commit 7:Report` - but NEVER resets a tracker that an earlier run already built (load-or-continue; contract section "Continuation runs"):
75
+ **The phase tracker is required.** Full spec: [`$HOME/.claude/multi-agent-refs/tracker-contract.md`]($HOME/.claude/multi-agent-refs/tracker-contract.md). `ship` registers its active phase set - `0:Init 3:Review 4:Commit 5:Report` - but NEVER resets a tracker that an earlier run already built (load-or-continue; contract section "Continuation runs"):
74
76
 
75
77
  ```bash
76
78
  # Phase 0, first shell call (every CLI). init ONLY when no prior state exists -
77
- # a task handed off from Phase 5 ("awaiting local test") keeps its full
78
- # phase 0-3 history (elapsed, tokens, USD).
79
+ # a task handed off from the user test inside Phase 3 ("awaiting local test")
80
+ # keeps its full phase 0-2 history (elapsed, tokens, USD).
79
81
  STATE="$HOME/.claude/logs/multi-agent/${TASK_ID}/tracker-state.json"
80
82
  if [ ! -f "$STATE" ]; then
81
83
  bash $HOME/.claude/scripts/phase-tracker.sh init "$TASK_ID"
82
84
  fi
83
- for p in "0:Init" "4:Review" "5:Build+Test" "6:Commit" "7:Report"; do
85
+ for p in "0:Init" "3:Review" "4:Commit" "5:Report"; do
84
86
  bash $HOME/.claude/scripts/phase-tracker.sh add "${p%%:*}" "${p#*:}" # idempotent: existing phases keep their history
85
87
  done
86
88
  bash $HOME/.claude/scripts/phase-tracker.sh update 0 in_progress
@@ -91,7 +93,7 @@ bash $HOME/.claude/scripts/phase-tracker.sh update <N> in_progress|completed|fai
91
93
  bash $HOME/.claude/scripts/phase-tracker.sh tokens <N> <in> <out> [cached]
92
94
  ```
93
95
 
94
- **Continuation path (prior state existed):** (1) if Phase 5 was left `in_progress` with `Now: awaiting local test (user)`, mark it `update 5 completed` + `meta 5 Result "local test done (user)"` before finish's own work (finish re-opens it with `update 5 in_progress` when its Build+Test gate runs; elapsed keeps the original `started_at`, which is expected); (2) print ONE line in `outputLanguage` summarizing the inherited history, e.g. `Continuing PROJ-12345: phases 0-3 finished earlier (12m, 38.4k tok, ~$0.74)` (USD via `phase-tracker.sh cost total`); (3) `render`.
96
+ **Continuation path (prior state existed):** (1) if Phase 3 was left `in_progress` with `Now: awaiting local test (user)`, mark it `update 3 completed` + `meta 3 Result "local test done (user)"` before finish's own work (finish re-opens it with `update 3 in_progress` when its Verify gate runs; elapsed keeps the original `started_at`, which is expected); (2) print ONE line in `outputLanguage` summarizing the inherited history, e.g. `Continuing PROJ-12345: phases 0-3 finished earlier (12m, 38.4k tok, ~$0.74)` (USD via `phase-tracker.sh cost total`); (3) `render`.
95
97
 
96
98
  ### Visual channel - Claude Code (native TaskList widget, required)
97
99
 
@@ -194,7 +194,7 @@ Every host runs three reviewers; only the second slot differs, because GPT-5.4 e
194
194
 
195
195
  With the `fable` rung disabled by prefs the Claude Code panel is two reviewers (Opus + Sonnet); see `$HOME/.claude/multi-agent-refs/features/model-fallback.md`.
196
196
 
197
- Each reviewer receives the diff, the module review guides from Step 2b (when any were found), plus the standard reviewer system prompt (see `$HOME/.claude/multi-agent-refs/phases/phase-4-review.md` for the prompt contract). Output: structured `findings[]` per reviewer.
197
+ Each reviewer receives the diff, the module review guides from Step 2b (when any were found), plus the standard reviewer system prompt (see `$HOME/.claude/multi-agent-refs/phases/phase-3-review.md` for the prompt contract). Output: structured `findings[]` per reviewer.
198
198
 
199
199
  ### 4. Store-compliance cross-reference
200
200
 
@@ -312,7 +312,7 @@ Per the `$HOME/.claude/multi-agent-refs/tracker-contract.md` tokens contract, af
312
312
  bash $HOME/.claude/scripts/phase-tracker.sh tokens 4 <input_count> <output_count>
313
313
  ```
314
314
 
315
- Standalone `/multi-agent:review` runs use phase id 4 (matches Phase 4 Review in the full pipeline) so cost aggregation stays consistent.
315
+ Standalone `/multi-agent:review` runs use phase id 3 (matches Phase 3 Review in the full pipeline) so cost aggregation stays consistent.
316
316
 
317
317
  ## Input matrix
318
318
 
@@ -23,7 +23,7 @@ Read `$HOME/.claude/multi-agent-refs/analysis/review.md` and execute it:
23
23
 
24
24
  ## Why it cites Locked decisions
25
25
 
26
- The analysis flow already declares 36 Locked decisions and two deterministic validators. A reviewer that says "I would have written this differently" gives the author nothing to act on; one that says "Locked 34: the Confluence page is in the evidence record but not in Section 21" gives them a fix and a reason. Findings that map to no rule are still allowed, but they are marked as judgement, not dressed up as a violation.
26
+ The analysis flow already declares 36 Locked decisions and two deterministic validators. A reviewer that says "I would have written this differently" gives the author nothing to act on; one that says "Locked 33: the Confluence page is in the evidence record but not in Section 21" gives them a fix and a reason. Findings that map to no rule are still allowed, but they are marked as judgement, not dressed up as a violation.
27
27
 
28
28
  ## What it never does
29
29
 
@@ -0,0 +1,36 @@
1
+ ---
2
+ description: "Disable model routing and keep its rules, so turning it back on does not re-ask for the same configuration. Use when asked to turn model routing off."
3
+ description-tr: "Model yönlendirmesini kapatır ve kurallarını saklar; tekrar açıldığında aynı yapılandırma yeniden sorulmaz."
4
+ ---
5
+
6
+ # multi-agent route-off - disarm routing, keep the configuration
7
+
8
+ ```bash
9
+ bash "$HOME/.claude/lib/route-state.sh" off
10
+ ```
11
+
12
+ Sets `prefs.global.modelRouting.enabled` to `false`. Dispatch returns to the
13
+ plain ladder immediately: every persona takes its `preferredModel`, and the
14
+ `modelFallback` rules are the only thing that can move it.
15
+
16
+ ## The rules are not deleted
17
+
18
+ `strategy`, `scope`, `rules[]` and `budgetCeilingUsd` all survive. `route-on`
19
+ brings back exactly what was configured, without asking again.
20
+
21
+ This is the same contract `autopilot-off` follows with its repo selection, for
22
+ the same reason: a command named after a toggle that quietly discards
23
+ configuration is a destructive action in disguise. To actually remove the rules,
24
+ replace them with an empty array:
25
+
26
+ ```bash
27
+ echo '[]' > /tmp/none.json
28
+ bash "$HOME/.claude/lib/route-state.sh" set-rules /tmp/none.json
29
+ ```
30
+
31
+ ## What stays behind
32
+
33
+ Routing decisions already written to the cost ledger stay there. They are a
34
+ record of what happened on past runs, and deleting them would make a run's cost
35
+ unexplainable after the fact - which is the one thing the ledger exists to
36
+ prevent.
@@ -0,0 +1,74 @@
1
+ ---
2
+ description: "Enable policy-driven model routing: pick a strategy, a scope, and the rules that say which rung a call lands on. Use when asked to turn model routing on."
3
+ description-tr: "Politika güdümlü model yönlendirmesini açar: strateji, kapsam ve hangi çağrının hangi basamağa düşeceğini söyleyen kuralları sorar."
4
+ argument-hint: "[--strategy=manual|task-fit|cost-ceiling] [--scope=subagent,bulk-read,research]"
5
+ ---
6
+
7
+ # multi-agent route-on - arm model routing
8
+
9
+ ```bash
10
+ bash "$HOME/.claude/lib/route-state.sh" on ${ARGUMENTS}
11
+ ```
12
+
13
+ Routing ships **off**. This is the command that turns it on, and it writes to
14
+ `prefs.global.modelRouting`, validated against the repo's `schemas/route-config.schema.json`.
15
+
16
+ ## What a rule is
17
+
18
+ ```json
19
+ { "when": { "persona": "code-reviewer" }, "prefer": ["opus", "sonnet"] }
20
+ { "when": { "phase": 2 }, "prefer": ["sonnet", "haiku"] }
21
+ ```
22
+
23
+ Ordered, first match wins. `when` matches on `persona`, `phase` (0..5) or
24
+ `taskKind`; `prefer` lists rungs in descending preference.
25
+
26
+ **Rung names are the contract; model ids are not.** `opus` is a rung here and a
27
+ model id in `cost-table.json`, and the second can change without anyone editing
28
+ a rule. Writing `claude-opus-5` into a rule pins a decision to a string that will
29
+ go stale.
30
+
31
+ Set rules with:
32
+
33
+ ```bash
34
+ bash "$HOME/.claude/lib/route-state.sh" set-rules path/to/rules.json
35
+ ```
36
+
37
+ ## Strategies
38
+
39
+ | Strategy | What it does |
40
+ |---|---|
41
+ | `manual` | only the explicit rules apply, nothing is inferred. The default, because a router that guesses is a router nobody can predict |
42
+ | `task-fit` | a rule may match on `taskKind`, and the cheapest rung clearing it is chosen |
43
+ | `cost-ceiling` | rungs downgrade as the run approaches `budgetCeilingUsd` |
44
+
45
+ ## Scope, and the value that is not in it
46
+
47
+ `scope` names the call sites routing may act on: `subagent`, `bulk-read`,
48
+ `research`. Every one of them is a call **this pipeline makes itself**.
49
+
50
+ There is no `host-session` value, and that absence is enforced by that schema
51
+ rather than written as advice. Routing a host session means rewriting the CLI's
52
+ base URL to point at a local gateway - which sends the user's *entire* session
53
+ through a third layer, including work that has nothing to do with this pipeline,
54
+ breaks the subscription's auth model, and silently changes which model answered.
55
+ Passing `--scope=host-session` is refused with that reason, not ignored.
56
+
57
+ ## The limit this command prints every time
58
+
59
+ On Claude Code a subagent cannot be dispatched to a non-Anthropic model: subagent
60
+ dispatch belongs to the host, not to us. So Phase 1, 2 and 3 personas stay inside
61
+ the Anthropic ladder no matter what the rules say, and external providers apply
62
+ only at `bulk-read` and `research`, where the pipeline makes the HTTP call.
63
+
64
+ This is printed by `route-status` on every invocation instead of living in a doc,
65
+ because the question it answers - "routing is on, why is the reviewer still on
66
+ Opus" - otherwise arrives days later as a bug report.
67
+
68
+ ## Related
69
+
70
+ - `/multi-agent:route-off` - disables routing and **keeps** the rules
71
+ - `/multi-agent:route-status` - what is active, and what it costs
72
+ - `/multi-agent:model` - whether the top rung exists at all. That is a different
73
+ question: this command decides which rung a call picks, that one decides
74
+ whether the top one is in play
@@ -0,0 +1,56 @@
1
+ ---
2
+ description: "Report model routing: whether it is armed, which rule applies where, which rung the last dispatches took, and what this run has cost. Use when asked which model is being used or why."
3
+ description-tr: "Model yönlendirmesinin durumunu raporlar: açık mı, hangi kural nerede geçerli, son çağrılar hangi basamağa gitti, bu koşu ne tuttu."
4
+ ---
5
+
6
+ # multi-agent route-status - what is actually routing
7
+
8
+ ```bash
9
+ bash "$HOME/.claude/lib/route-state.sh" status
10
+ ```
11
+
12
+ Reports the stored policy, then the part that matters more: **what it can and
13
+ cannot reach.**
14
+
15
+ ## Three states, told apart
16
+
17
+ | Output | Meaning |
18
+ |---|---|
19
+ | `routing: false`, rules present | configured and disarmed. `route-on` restores it as-is; nothing is lost |
20
+ | `routing: true`, `rules: 0` | armed with nothing to match. Not an error - a configuration state, and the most common "I turned it on and nothing changed" |
21
+ | `routing: true` with rules listed | live. Each rule is printed as `when <key>=<value> -> rung > rung` |
22
+
23
+ A disabled router with rules is deliberately not reported as "off" alone: that
24
+ reads as "unconfigured" and sends the user through `route-on`'s questions a
25
+ second time.
26
+
27
+ ## The limit, printed every time
28
+
29
+ On Claude Code a subagent cannot be sent to a non-Anthropic model. Subagent
30
+ dispatch belongs to the host; the pipeline asks for a persona and the host
31
+ decides what answers. So Phase 1, 2 and 3 personas stay inside the Anthropic
32
+ ladder whatever the rules say, and an external provider is only reachable where
33
+ the pipeline makes the HTTP call itself - `bulk-read.sh` and `research_ask`.
34
+
35
+ This paragraph is output, not documentation, because the alternative is the
36
+ question arriving later as "routing is on but the reviewer is still on Opus, is
37
+ it broken". It is not broken; it is the seam.
38
+
39
+ ## Where decisions are recorded
40
+
41
+ With `recordDecisions: true` (the default) every routing decision is written to
42
+ the cost ledger: which rule matched, which rung it chose, and why. That is what
43
+ makes a run's cost explainable after it finished - a router whose choices are not
44
+ recorded cannot be audited, and the cost question always arrives after the run,
45
+ never during it.
46
+
47
+ Per-run cost and the model breakdown come from the same ledger:
48
+
49
+ ```bash
50
+ node "$HOME/.claude/scripts/token-budget-report.mjs" --json
51
+ ```
52
+
53
+ ## Related
54
+
55
+ - `/multi-agent:route-on` / `/multi-agent:route-off` - arm and disarm
56
+ - `/multi-agent:model` - whether the top rung exists at all
@@ -558,7 +558,7 @@ Re-run scan from Step 1. Show final status:
558
558
  All tokens present. Pipeline ready to use.
559
559
 
560
560
  Optional per-project features (configured on first use, nothing to do now):
561
- • Phase 7 Report Step 2 Wiki - auto-generates component wiki pages + Figma
561
+ • Phase 5 Report Step 2 Wiki - auto-generates component wiki pages + Figma
562
562
  screenshots. Activates when (a) task is a component AND (b) a Figma token
563
563
  is in Keychain. Four adapters supported: submodule / in-repo / github-wiki
564
564
  / separate-repo. First run asks: use auto-detected path, use a custom
@@ -803,7 +803,7 @@ All tokens are optional in the sense that every service can be answered with Ski
803
803
  Offer to merge `$HOME/.claude/templates/claude-hooks.json`: three `PreToolUse` gates that block on a non-zero exit (secret scan, agent-guard, read-size) plus three capture hooks that block nothing (`PreCompact`, `SessionEnd`, `SessionStart`). What each does: `$HOME/.claude/multi-agent-refs/picker-contract.md`.
804
804
 
805
805
  - Ask (picker): "Install the pipeline's hooks into `~/.claude/settings.json`?" Default Yes.
806
- - On Yes, deep-merge EVERY event in the template's `hooks` object, not `PreToolUse` alone - merging one event silently drops the capture hooks, and a run killed before Phase 7 then loses its findings exactly as it did before they existed. Preserve existing hooks; never duplicate a matcher already calling the same script.
806
+ - On Yes, deep-merge EVERY event in the template's `hooks` object, not `PreToolUse` alone - merging one event silently drops the capture hooks, and a run killed before Phase 5 then loses its findings exactly as it did before they existed. Preserve existing hooks; never duplicate a matcher already calling the same script.
807
807
  - Say what the merge does NOT cover: only the three gates need no run-specific arguments, so only they are hookable; the rest are phase-enforced.
808
808
  - Say what it does not turn on: the read-size gate is inert until `prefs.global.bulkRead.mode` is set. Recommend `observe` first. Why, and the Phase 3 exemption: `$HOME/.claude/multi-agent-refs/picker-contract.md`.
809
809
 
@@ -19,7 +19,7 @@ Show every active and completed task as a table.
19
19
 
20
20
  `runs-index.mjs` resolves through `lib/run-paths.sh` / `scripts/_run-paths.mjs`,
21
21
  so it sees both directory layouts (`<project>/<id>/` and the flat `<id>/`),
22
- the salvaged `artifacts/` copy Phase 6 leaves behind, and every spelling of a
22
+ the salvaged `artifacts/` copy Phase 4 leaves behind, and every spelling of a
23
23
  task id - and it counts a run that exists in both layouts once. Do NOT
24
24
  re-scan the tree by hand: the earlier instruction here listed three
25
25
  hard-coded `.worktrees/` paths and a single `find` depth, and on a real
@@ -33,7 +33,7 @@ Show every active and completed task as a table.
33
33
  2. **Fields per run** (already in the output): `taskId`, `project`, `branch`,
34
34
  `currentPhase`, `status`, `startedAt`, `worktreePath`, `prUrl`, `autopilot`,
35
35
  `phases[]`, `tokens`, `estUsd`, `group`, plus `duplicateOf` when the run also
36
- exists in the other layout and `salvaged` when its state is the Phase 6 copy.
36
+ exists in the other layout and `salvaged` when its state is the Phase 4 copy.
37
37
 
38
38
  3. **Groups are computed, not judged.** `runs-index.mjs` assigns `group` by the
39
39
  table below; report what it returns rather than re-deriving it. `in_progress`
@@ -42,7 +42,7 @@ Show every active and completed task as a table.
42
42
 
43
43
  | Group | Test | Action offered |
44
44
  |---|---|---|
45
- | `waiting` - Waiting on you | `status == "awaiting_input"`, or a `pr.url`, or phase >= 6 | `resume #N` - the work landed, it needs your answer |
45
+ | `waiting` - Waiting on you | `status == "awaiting_input"`, or a `pr.url`, or phase >= 4 | `resume #N` - the work landed, it needs your answer |
46
46
  | `stopped` - Stopped mid-development | anything else past phase 0 | `resume #N` or `kill #N` |
47
47
  | `question` - Left at a question | phase 0 | `garbage-collect --abandoned` - nothing was built |
48
48
  | `unknown` - Status not recorded | no `status` field | say so; offer nothing |
@@ -57,8 +57,8 @@ Show every active and completed task as a table.
57
57
 
58
58
  | ID | Jira/Task | Branch | Phase | Status | Duration |
59
59
  |----|-----------|--------|-------|--------|----------|
60
- | #1 | {JIRA-KEY}-12345 | feature/{JIRA-KEY}-12345-... | 7/7 DONE | ✅ Complete | 12m |
61
- | #3 | {JIRA-KEY}-12345 | feature/{JIRA-KEY}-12345-... | 7/7 DONE | ✅ Complete | 8m |
60
+ | #1 | {JIRA-KEY}-12345 | feature/{JIRA-KEY}-12345-... | 5/5 DONE | ✅ Complete | 12m |
61
+ | #3 | {JIRA-KEY}-12345 | feature/{JIRA-KEY}-12345-... | 5/5 DONE | ✅ Complete | 8m |
62
62
 
63
63
  💡 log #1 | resume #N | kill #N
64
64
  ```
@@ -41,7 +41,7 @@ is being replaced.
41
41
  | `paused` / `failed` | Say the task is not running, and that `/multi-agent:resume #N` will re-enter with the instruction applied at that phase's entry. Queue it. |
42
42
  | `complete` | Refuse. Nothing will read it. Point at `/multi-agent` for a follow-up run. |
43
43
 
44
- `currentPhase` is 7 and status is `in_progress` → warn that Phase 7 is the
44
+ `currentPhase` is 7 and status is `in_progress` → warn that Phase 5 is the
45
45
  last one, so an instruction queued now may never be consumed.
46
46
 
47
47
  3. **Read the instruction** - from the argument, or ask for it when the
@@ -52,7 +52,7 @@ is being replaced.
52
52
  4. **Show what will be queued, and ask**:
53
53
 
54
54
  ```
55
- Steer #3 ({JIRA-KEY}-12345, Phase 3 Dev, in_progress)
55
+ Steer #3 ({JIRA-KEY}-12345, Phase 2 Dev, in_progress)
56
56
 
57
57
  "the field is called web, not frontend"
58
58
 
@@ -65,8 +65,8 @@ Run every step automatically:
65
65
  ```
66
66
  Step 0: DOCTOR doctor.mjs - exit 2 or 4 stops the sync
67
67
  Step 1.5: DETECT Compare timestamps, find stale targets
68
- Step 2: COPILOT Claude Code -> Copilot CLI (instructions + 56 sub-command skills)
69
- Step 2b: CODEX Claude Code -> Codex CLI (1 router skill + 56 specs as refs + 8 agent TOML)
68
+ Step 2: COPILOT Claude Code -> Copilot CLI (instructions + 60 sub-command skills)
69
+ Step 2b: CODEX Claude Code -> Codex CLI (1 router skill + 60 specs as refs + 8 agent TOML)
70
70
  Step 3: REPO Claude Code -> pipeline repo (genericized, personal data scrub, bash -n on all sh)
71
71
  Step 3c: PLUGINS pipeline shared/external -> multi-agent-plugins marketplace (rebuild knowledge/,
72
72
  bump changed plugins' patch version, commit + push the plugins repo)
@@ -494,19 +494,18 @@ same 56 specs as reference files rather than as peer skills, via Step 2b - see
494
494
  |-------------|-------------|
495
495
  | `~/.claude/commands/multi-agent/{cmd}/SKILL.md` | `~/.copilot/skills/multi-agent-{cmd}/SKILL.md` |
496
496
 
497
- **56 commands are synced** (canonical inventory - must match `cross-cli-contract.md` section 1; drift = contract violation):
497
+ **60 commands are synced** (canonical inventory - must match `cross-cli-contract.md` section 1; drift = contract violation):
498
498
 
499
499
  ```
500
- analysis, analysis-jira, analysis-resolve, autopilot, autopilot-off,
501
- autopilot-on, autopilot-status, build-optimize, channels, complaint-analysis,
502
- create-jira, design-check, diff-explain, doctor, feedback, forget,
503
- garbage-collect, graph, help, ios-coding-standard, issue, jira, kill,
504
- language, local, local-autopilot, log, manual-test, prune-logs,
505
- prune-prompts, purge, refactor, resume, resume-local, review,
506
- review-analysis, review-issue, review-jira, routines, save, scan, search,
507
- setup, stack, status, steer, store-ready, sync, test, test-accessibility,
508
- test-dark-mode, test-dynamic-type, test-screenshots, testflight-validation,
509
- uninstall, update
500
+ analysis, analysis-jira, analysis-resolve, autopilot, autopilot-off, autopilot-on,
501
+ autopilot-status, build-optimize, channels, complaint-analysis, create-jira,
502
+ design-check, diff-explain, doctor, feedback, forget, garbage-collect, graph, help,
503
+ ios-coding-standard, issue, jira, kill, language, local, local-autopilot, log,
504
+ manual-test, model, prune-logs, prune-prompts, purge, refactor, resume, resume-local,
505
+ review, review-analysis, review-issue, review-jira, route-off, route-on, route-status,
506
+ routines, save, scan, search, setup, stack, status, steer, store-ready, sync, test,
507
+ test-accessibility, test-dark-mode, test-dynamic-type, test-screenshots,
508
+ testflight-validation, uninstall, update
510
509
  ```
511
510
 
512
511
  **NOT synced**: `$HOME/.claude/multi-agent-refs/*` - lazy-load references, Claude Code specific
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  description: "UI Bug Hunter - iOS Simulator (simctl) + Android Emulator (adb). Auto-detects platform. Screenshot + tap + analyze on the booted device. The Phase 5 Manual Test flow lives at /multi-agent:manual-test. Use when a running app should be driven on a simulator or emulator to hunt UI bugs."
3
- description-tr: "UI Bug Hunter - iOS Simulator (simctl) + Android Emulator (adb). Platformu otomatik algılar. Açık cihazda screenshot + tap + analiz. Faz 5 Manuel Test akışı /multi-agent:manual-test'te."
3
+ description-tr: "UI Bug Hunter - iOS Simulator (simctl) + Android Emulator (adb). Platformu otomatik algılar. Açık cihazda screenshot + tap + analiz. Faz 3 Manuel Test akışı /multi-agent:manual-test'te."
4
4
  argument-hint: '[scenario] - e.g. "dark mode" | "accessibility" | "dynamic type" | "screenshot tr" | "store-ready" | (empty = full sweep)'
5
5
  ---
6
6
 
@@ -345,7 +345,7 @@ capability_of() {
345
345
  esac
346
346
  fi
347
347
  case "$1" in
348
- jira) echo "read the ticket, its comments and its linked issues; post the Phase 7 comment" ;;
348
+ jira) echo "read the ticket, its comments and its linked issues; post the Phase 5 comment" ;;
349
349
  bitbucket_token) echo "read the repo, open and update pull requests" ;;
350
350
  bitbucket_user) echo "identify the PR author (paired with bitbucket_token)" ;;
351
351
  github) echo "read issues, open pull requests, read Actions runs" ;;
@@ -49,7 +49,7 @@
49
49
  # 6 not configured (prefs.global.hosts.fortify empty, or an instance-id-only
50
50
  # lookup with no prefs.global.fortify.versionIds to search)
51
51
  #
52
- # Phase 4 review gate contract:
52
+ # Phase 3 review gate contract:
53
53
  # Critical > 0 → blocking=true, reason="critical-findings"
54
54
  # High > 0 → blocking=false, reason="high-findings-warning"
55
55
  # else → blocking=false, reason="clean"
@@ -0,0 +1,140 @@
1
+ #!/usr/bin/env bash
2
+ # model-dispatch.sh - which rung answers this call.
3
+ #
4
+ # The one place `prefs.global.modelRouting` turns into a rung. Every call site
5
+ # that may be routed asks here, so the policy has a single implementation and
6
+ # `/multi-agent:route-status` describes something real.
7
+ #
8
+ # Usage:
9
+ # model-dispatch.sh <call-site> [--persona P] [--phase N] [--task-kind K]
10
+ # [--default RUNG] [--task-id ID]
11
+ #
12
+ # <call-site> is one of the `scope` members: subagent | bulk-read | research.
13
+ # stdout is a single rung NAME. Rung names are the contract; the model id behind
14
+ # one lives in cost-table.json and moves without a config edit.
15
+ #
16
+ # Exit is always 0 and stdout is always a usable rung. A router that can fail
17
+ # turns every call site into a place the run can die, for a feature that ships
18
+ # disabled; when anything is missing, unparseable or out of scope, the caller's
19
+ # default comes back unchanged.
20
+ #
21
+ # Two layers, and the difference is the safety argument the schema spells out:
22
+ #
23
+ # Layer 1 an Anthropic rung. Reachable from every call site, because it
24
+ # writes to a seam that already exists and opens no new network path.
25
+ # Layer 2 a non-Anthropic rung. Reachable ONLY from a call site the pipeline
26
+ # makes itself. A subagent is dispatched by the HOST, so a rule that
27
+ # prefers an external rung for a subagent cannot be honoured - and
28
+ # this script says so on stderr rather than pretending it was.
29
+
30
+ set -uo pipefail
31
+
32
+ CALL_SITE="${1:-}"
33
+ shift || true
34
+
35
+ PERSONA=""; PHASE=""; TASK_KIND=""; DEFAULT_RUNG=""; TASK_ID="${MULTI_AGENT_TASK_ID:-unknown}"
36
+ while [ $# -gt 0 ]; do
37
+ case "$1" in
38
+ --persona) PERSONA="${2:-}"; shift 2 ;;
39
+ --phase) PHASE="${2:-}"; shift 2 ;;
40
+ --task-kind) TASK_KIND="${2:-}"; shift 2 ;;
41
+ --default) DEFAULT_RUNG="${2:-}"; shift 2 ;;
42
+ --task-id) TASK_ID="${2:-}"; shift 2 ;;
43
+ *) shift ;;
44
+ esac
45
+ done
46
+
47
+ emit() { printf '%s\n' "$1"; exit 0; }
48
+
49
+ case "$CALL_SITE" in
50
+ subagent|bulk-read|research) ;;
51
+ *) emit "$DEFAULT_RUNG" ;;
52
+ esac
53
+
54
+ [ -n "$DEFAULT_RUNG" ] || DEFAULT_RUNG="sonnet"
55
+
56
+ command -v jq >/dev/null 2>&1 || emit "$DEFAULT_RUNG"
57
+
58
+ PREFS=""
59
+ for candidate in \
60
+ "${MULTI_AGENT_PREFS:-}" \
61
+ "$HOME/.claude/multi-agent-preferences.json" \
62
+ "$HOME/.config/multi-agent-pipeline/multi-agent-preferences.json"
63
+ do
64
+ [ -n "$candidate" ] && [ -f "$candidate" ] && { PREFS="$candidate"; break; }
65
+ done
66
+ [ -n "$PREFS" ] || emit "$DEFAULT_RUNG"
67
+
68
+ ENABLED=$(jq -r '.global.modelRouting.enabled // false' "$PREFS" 2>/dev/null) || emit "$DEFAULT_RUNG"
69
+ [ "$ENABLED" = "true" ] || emit "$DEFAULT_RUNG"
70
+
71
+ # Out of scope is not a failure and not a warning. The user named the call sites
72
+ # routing may touch; the ones they left out keep their existing behaviour, which
73
+ # is the point of naming them.
74
+ IN_SCOPE=$(jq -r --arg cs "$CALL_SITE" \
75
+ '((.global.modelRouting.scope // ["subagent"]) | index($cs)) != null' "$PREFS" 2>/dev/null)
76
+ [ "$IN_SCOPE" = "true" ] || emit "$DEFAULT_RUNG"
77
+
78
+ # First rule whose every stated condition matches. A `when` with three keys has
79
+ # to match on all three - the schema requires at least one, so an always-matching
80
+ # rule cannot be written by accident.
81
+ MATCH=$(jq -r \
82
+ --arg persona "$PERSONA" --arg phase "$PHASE" --arg kind "$TASK_KIND" \
83
+ '[ (.global.modelRouting.rules // [])[]
84
+ | select(
85
+ ((.when.persona // null) as $p | $p == null or $p == $persona)
86
+ and ((.when.phase // null) as $h | $h == null or ($phase != "" and ($h | tostring) == $phase))
87
+ and ((.when.taskKind // null) as $k | $k == null or $k == $kind)
88
+ )
89
+ ] | first | (.prefer // []) | join(" ")' "$PREFS" 2>/dev/null) || emit "$DEFAULT_RUNG"
90
+ [ -n "$MATCH" ] && [ "$MATCH" != "null" ] || emit "$DEFAULT_RUNG"
91
+
92
+ HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
93
+ COST_TABLE="$HERE/../scripts/cost-table.json"
94
+
95
+ # The fable rung answers only when the fable switch is on. Routing is a policy
96
+ # over what is available; it is not a second way to turn a rung on, or
97
+ # `/multi-agent:model off` would stop meaning anything the moment a rule named
98
+ # fable.
99
+ FABLE_ON=$(jq -r '.global.modelFallback.fableEnabled // false' "$PREFS" 2>/dev/null)
100
+
101
+ record() {
102
+ [ -x "$HERE/../scripts/log-metric.sh" ] || return 0
103
+ local rec
104
+ rec=$(jq -r '.global.modelRouting.recordDecisions // true' "$PREFS" 2>/dev/null)
105
+ [ "$rec" = "false" ] && return 0
106
+ "$HERE/../scripts/log-metric.sh" "$TASK_ID" "${PHASE:-0}" model_routing.decision \
107
+ call_site="$CALL_SITE" rung="$1" reason="$2" >/dev/null 2>&1 || true
108
+ }
109
+
110
+ for rung in $MATCH; do
111
+ if [ ! -f "$COST_TABLE" ]; then
112
+ record "$rung" "no cost table; rung taken on trust"
113
+ emit "$rung"
114
+ fi
115
+ PROVIDER=$(jq -r --arg r "$rung" '.prices[$r].provider // ""' "$COST_TABLE" 2>/dev/null)
116
+ # A rung nobody priced is a typo in a config far more often than it is a new
117
+ # model. Skipping it silently would route to the next one and leave the user
118
+ # certain their rule applied.
119
+ if [ -z "$PROVIDER" ]; then
120
+ echo "model-dispatch: rung '$rung' is not in cost-table.json; skipping it" >&2
121
+ continue
122
+ fi
123
+ if [ "$rung" = "fable" ] && [ "$FABLE_ON" != "true" ]; then
124
+ echo "model-dispatch: rule prefers 'fable' but modelFallback.fableEnabled is false; skipping it" >&2
125
+ continue
126
+ fi
127
+ if [ "$PROVIDER" != "anthropic" ] && [ "$CALL_SITE" = "subagent" ]; then
128
+ # The honest limit, said out loud at the moment it bites. Subagent dispatch
129
+ # belongs to the host; we cannot send one to another provider, and a router
130
+ # that quietly downgraded to an Anthropic rung here would leave the user
131
+ # believing a rule worked that never could.
132
+ echo "model-dispatch: rung '$rung' is $PROVIDER and subagent dispatch belongs to the host; skipping it" >&2
133
+ continue
134
+ fi
135
+ record "$rung" "rule matched"
136
+ emit "$rung"
137
+ done
138
+
139
+ record "$DEFAULT_RUNG" "every preferred rung was unavailable"
140
+ emit "$DEFAULT_RUNG"