@mmerterden/multi-agent-pipeline 18.0.0 → 19.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (209) hide show
  1. package/CHANGELOG.md +183 -0
  2. package/README.md +34 -18
  3. package/README.tr.md +14 -16
  4. package/docs/adr/0002-instruction-driven-flag.md +1 -0
  5. package/docs/adr/0005-lazy-phase-docs.md +11 -1
  6. package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
  7. package/docs/adr/0010-own-code-graph.md +1 -0
  8. package/docs/adr/0014-six-phase-consolidation.md +134 -0
  9. package/docs/adr/README.md +2 -1
  10. package/docs/architecture.md +37 -38
  11. package/docs/best-practices.md +1 -1
  12. package/docs/ecosystem.md +37 -26
  13. package/docs/engineering.md +1 -1
  14. package/docs/facts.json +45 -0
  15. package/docs/features.md +54 -53
  16. package/docs/performance.md +5 -5
  17. package/docs/recovery-guide.md +9 -9
  18. package/docs/token-budget-history.md +3 -1
  19. package/index.js +2 -2
  20. package/install/_codex-agents.mjs +1 -1
  21. package/install/templates/claude-hooks.json +1 -1
  22. package/install/templates/codex-instructions.md +1 -1
  23. package/install/templates/copilot-instructions.md +28 -28
  24. package/manifest.json +209 -193
  25. package/package.json +2 -2
  26. package/pipeline/agents/dev-critic.md +3 -3
  27. package/pipeline/commands/figma-to-swiftui.md +1 -1
  28. package/pipeline/commands/multi-agent/SKILL.md +8 -8
  29. package/pipeline/commands/multi-agent/analysis/SKILL.md +9 -9
  30. package/pipeline/commands/multi-agent/autopilot/SKILL.md +7 -7
  31. package/pipeline/commands/multi-agent/channels/SKILL.md +15 -15
  32. package/pipeline/commands/multi-agent/diff-explain/SKILL.md +6 -6
  33. package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
  34. package/pipeline/commands/multi-agent/graph/SKILL.md +1 -1
  35. package/pipeline/commands/multi-agent/help/SKILL.md +62 -62
  36. package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
  37. package/pipeline/commands/multi-agent/local/SKILL.md +11 -11
  38. package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +13 -13
  39. package/pipeline/commands/multi-agent/log/SKILL.md +2 -2
  40. package/pipeline/commands/multi-agent/manual-test/SKILL.md +9 -9
  41. package/pipeline/commands/multi-agent/model/SKILL.md +69 -0
  42. package/pipeline/commands/multi-agent/refactor/SKILL.md +3 -3
  43. package/pipeline/commands/multi-agent/resume/SKILL.md +4 -4
  44. package/pipeline/commands/multi-agent/resume-local/SKILL.md +19 -17
  45. package/pipeline/commands/multi-agent/review/SKILL.md +1 -1
  46. package/pipeline/commands/multi-agent/route-off/SKILL.md +36 -0
  47. package/pipeline/commands/multi-agent/route-on/SKILL.md +74 -0
  48. package/pipeline/commands/multi-agent/route-status/SKILL.md +56 -0
  49. package/pipeline/commands/multi-agent/setup/SKILL.md +2 -2
  50. package/pipeline/commands/multi-agent/status/SKILL.md +5 -5
  51. package/pipeline/commands/multi-agent/steer/SKILL.md +2 -2
  52. package/pipeline/commands/multi-agent/sync/SKILL.md +12 -13
  53. package/pipeline/commands/multi-agent/test/SKILL.md +1 -1
  54. package/pipeline/lib/credential-inventory.sh +1 -1
  55. package/pipeline/lib/fetch-fortify.sh +1 -1
  56. package/pipeline/lib/model-rung.sh +142 -0
  57. package/pipeline/lib/phase-schema.mjs +88 -0
  58. package/pipeline/lib/plan-todos.sh +5 -5
  59. package/pipeline/lib/route-state.sh +161 -0
  60. package/pipeline/lib/run-paths.sh +2 -2
  61. package/pipeline/multi-agent-refs/_account-picker.md +1 -1
  62. package/pipeline/multi-agent-refs/_dev-context.md +1 -1
  63. package/pipeline/multi-agent-refs/_input-parser.md +1 -1
  64. package/pipeline/multi-agent-refs/analysis/evidence.md +0 -9
  65. package/pipeline/multi-agent-refs/analysis/intake.md +1 -1
  66. package/pipeline/multi-agent-refs/analysis/locked.md +21 -22
  67. package/pipeline/multi-agent-refs/analysis/render.md +1 -1
  68. package/pipeline/multi-agent-refs/analysis/synthesis.md +12 -6
  69. package/pipeline/multi-agent-refs/android-guide.md +1 -1
  70. package/pipeline/multi-agent-refs/audit-guide.md +13 -13
  71. package/pipeline/multi-agent-refs/channels/issue-comment.md +2 -2
  72. package/pipeline/multi-agent-refs/channels/jira.md +3 -3
  73. package/pipeline/multi-agent-refs/channels/pr.md +4 -4
  74. package/pipeline/multi-agent-refs/channels/wiki.md +1 -1
  75. package/pipeline/multi-agent-refs/component-dispatch.md +3 -3
  76. package/pipeline/multi-agent-refs/cross-cli-contract.md +31 -6
  77. package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +4 -4
  78. package/pipeline/multi-agent-refs/features/code-graph.md +5 -5
  79. package/pipeline/multi-agent-refs/features/design-conformance.md +1 -1
  80. package/pipeline/multi-agent-refs/features/dev-critic.md +3 -3
  81. package/pipeline/multi-agent-refs/features/doctor.md +2 -2
  82. package/pipeline/multi-agent-refs/features/external-context-injection.md +3 -3
  83. package/pipeline/multi-agent-refs/features/maturity-followup.md +3 -3
  84. package/pipeline/multi-agent-refs/features/model-fallback.md +5 -5
  85. package/pipeline/multi-agent-refs/features/plan-todos.md +1 -1
  86. package/pipeline/multi-agent-refs/features/repo-map.md +1 -1
  87. package/pipeline/multi-agent-refs/features/review-delta.md +3 -3
  88. package/pipeline/multi-agent-refs/features/review-multi-repo.md +1 -1
  89. package/pipeline/multi-agent-refs/features/scope-check.md +4 -4
  90. package/pipeline/multi-agent-refs/features/skill-conformance.md +2 -2
  91. package/pipeline/multi-agent-refs/features/stack-skill-routing.md +1 -1
  92. package/pipeline/multi-agent-refs/features/verify-by-test.md +4 -4
  93. package/pipeline/multi-agent-refs/features/visual-evidence.md +19 -19
  94. package/pipeline/multi-agent-refs/features/worktree-finalize.md +6 -6
  95. package/pipeline/multi-agent-refs/issue-jira-triad.md +10 -10
  96. package/pipeline/multi-agent-refs/knowledge.md +11 -11
  97. package/pipeline/multi-agent-refs/multi-repo-integration-build.md +13 -13
  98. package/pipeline/multi-agent-refs/payload-contracts.md +8 -8
  99. package/pipeline/multi-agent-refs/phases/log-format.md +10 -10
  100. package/pipeline/multi-agent-refs/phases/modes.md +30 -30
  101. package/pipeline/multi-agent-refs/phases/operations.md +8 -8
  102. package/pipeline/multi-agent-refs/phases/phase-0-init.md +24 -24
  103. package/pipeline/multi-agent-refs/phases/phase-1-plan.md +599 -0
  104. package/pipeline/multi-agent-refs/phases/{phase-3-dev.md → phase-2-dev.md} +129 -49
  105. package/pipeline/multi-agent-refs/phases/{phase-4-review.md → phase-3-review.md} +225 -107
  106. package/pipeline/multi-agent-refs/phases/{phase-6-commit.md → phase-4-commit.md} +23 -23
  107. package/pipeline/multi-agent-refs/phases/{phase-7-report.md → phase-5-report.md} +29 -29
  108. package/pipeline/multi-agent-refs/phases.md +44 -48
  109. package/pipeline/multi-agent-refs/picker-contract.md +1 -1
  110. package/pipeline/multi-agent-refs/progress-contract.md +6 -6
  111. package/pipeline/multi-agent-refs/readiness-review.md +1 -1
  112. package/pipeline/multi-agent-refs/rules.md +7 -7
  113. package/pipeline/multi-agent-refs/swiftui-guide.md +2 -2
  114. package/pipeline/multi-agent-refs/tracker-contract.md +31 -32
  115. package/pipeline/multi-agent-refs/wiki-capture.md +14 -14
  116. package/pipeline/preferences-template.json +9 -1
  117. package/pipeline/rules/outside-the-pipeline.md +1 -1
  118. package/pipeline/schemas/agent-state.schema.json +50 -50
  119. package/pipeline/schemas/analysis-output.schema.json +2 -2
  120. package/pipeline/schemas/autopilot-config.schema.json +1 -1
  121. package/pipeline/schemas/code-graph.schema.json +1 -1
  122. package/pipeline/schemas/criteria-manifest.schema.json +1 -1
  123. package/pipeline/schemas/dev-critic-output.schema.json +1 -1
  124. package/pipeline/schemas/diff-risk.schema.json +1 -1
  125. package/pipeline/schemas/migrations/prefs-2.4.0-to-2.5.0.mjs +2 -2
  126. package/pipeline/schemas/migrations/prefs-2.6.0-to-2.7.0.mjs +31 -0
  127. package/pipeline/schemas/migrations/state-2.1.0-to-2.2.0.mjs +129 -0
  128. package/pipeline/schemas/phases.json +105 -0
  129. package/pipeline/schemas/plan-todos.schema.json +5 -5
  130. package/pipeline/schemas/planning-output.schema.json +1 -1
  131. package/pipeline/schemas/prefs.schema.json +100 -56
  132. package/pipeline/schemas/reviewer-output.schema.json +3 -3
  133. package/pipeline/schemas/route-config.schema.json +74 -0
  134. package/pipeline/schemas/scope-check.schema.json +1 -1
  135. package/pipeline/schemas/test-gap.schema.json +1 -1
  136. package/pipeline/schemas/token-budget.json +12 -18
  137. package/pipeline/schemas/triage-output.schema.json +6 -6
  138. package/pipeline/scripts/README.md +3 -3
  139. package/pipeline/scripts/_code-graph.mjs +2 -2
  140. package/pipeline/scripts/_run-paths.mjs +2 -2
  141. package/pipeline/scripts/_smoke-root.sh +1 -1
  142. package/pipeline/scripts/aggregate-metrics.mjs +1 -1
  143. package/pipeline/scripts/capture-flush.sh +8 -8
  144. package/pipeline/scripts/capture-resume.sh +3 -3
  145. package/pipeline/scripts/classify-plan-safety.mjs +1 -1
  146. package/pipeline/scripts/diff-explain.mjs +1 -1
  147. package/pipeline/scripts/doctor.mjs +2 -2
  148. package/pipeline/scripts/gc-abandoned.sh +3 -3
  149. package/pipeline/scripts/gc-tmp.sh +1 -1
  150. package/pipeline/scripts/gc-worktrees.sh +1 -1
  151. package/pipeline/scripts/gen-facts.mjs +175 -0
  152. package/pipeline/scripts/gen-mode-dispatch.mjs +32 -37
  153. package/pipeline/scripts/gen-ref-toc.mjs +1 -1
  154. package/pipeline/scripts/graph-report.mjs +1 -1
  155. package/pipeline/scripts/jira-attach.sh +1 -1
  156. package/pipeline/scripts/learn-from-transcripts.mjs +1 -1
  157. package/pipeline/scripts/learning-curve.mjs +2 -2
  158. package/pipeline/scripts/log-metric.sh +17 -4
  159. package/pipeline/scripts/memory-save.sh +1 -1
  160. package/pipeline/scripts/migrate-prefs.mjs +22 -5
  161. package/pipeline/scripts/phase-banner.sh +20 -20
  162. package/pipeline/scripts/phase-tracker.sh +7 -7
  163. package/pipeline/scripts/plan-coverage-gate.mjs +2 -2
  164. package/pipeline/scripts/render-agent-log-cost.sh +1 -1
  165. package/pipeline/scripts/render-work-summary.sh +3 -3
  166. package/pipeline/scripts/review-file-filter.mjs +1 -1
  167. package/pipeline/scripts/run-aggregator.mjs +13 -6
  168. package/pipeline/scripts/run-metrics.mjs +1 -1
  169. package/pipeline/scripts/runs-index.mjs +11 -1
  170. package/pipeline/scripts/smoke-cross-cli-behavior.sh +6 -6
  171. package/pipeline/scripts/smoke-schema-validation.sh +26 -7
  172. package/pipeline/scripts/token-budget-report.mjs +13 -2
  173. package/pipeline/scripts/triage-memory.mjs +2 -2
  174. package/pipeline/scripts/validate-analysis-doc.mjs +73 -17
  175. package/pipeline/scripts/validate-planning.mjs +1 -1
  176. package/pipeline/scripts/validate-reviewer.mjs +1 -1
  177. package/pipeline/scripts/validate-state.mjs +45 -5
  178. package/pipeline/scripts/validate-triage.mjs +3 -3
  179. package/pipeline/scripts/worktree-finalize.sh +5 -5
  180. package/pipeline/skills/.skill-manifest.json +37 -21
  181. package/pipeline/skills/.skills-index.json +49 -5
  182. package/pipeline/skills/shared/README.md +10 -6
  183. package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +2 -2
  184. package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +2 -2
  185. package/pipeline/skills/shared/core/multi-agent/SKILL.md +69 -71
  186. package/pipeline/skills/shared/core/multi-agent-autopilot/SKILL.md +3 -3
  187. package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +14 -14
  188. package/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md +5 -5
  189. package/pipeline/skills/shared/core/multi-agent-graph/SKILL.md +1 -1
  190. package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +25 -23
  191. package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
  192. package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +2 -2
  193. package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +8 -8
  194. package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +6 -6
  195. package/pipeline/skills/shared/core/multi-agent-model/SKILL.md +71 -0
  196. package/pipeline/skills/shared/core/multi-agent-refactor/SKILL.md +3 -3
  197. package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +1 -1
  198. package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +7 -7
  199. package/pipeline/skills/shared/core/multi-agent-route-off/SKILL.md +39 -0
  200. package/pipeline/skills/shared/core/multi-agent-route-on/SKILL.md +76 -0
  201. package/pipeline/skills/shared/core/multi-agent-route-status/SKILL.md +59 -0
  202. package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +1 -1
  203. package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +5 -5
  204. package/pipeline/skills/shared/core/multi-agent-steer/SKILL.md +2 -2
  205. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +6 -5
  206. package/pipeline/skills/skills-index.md +8 -4
  207. package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +0 -263
  208. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +0 -344
  209. package/pipeline/multi-agent-refs/phases/phase-5-test.md +0 -182
@@ -1,6 +1,6 @@
1
1
  ---
2
- description: "Continue already-done LOCAL work through the pipeline tail: Review → Build+Test → Commit/PR → Report (technical analysis + Jira test-scenario comment). No dev phase. Use when local work is already done and only review, build, commit and reporting remain."
3
- description-tr: "Halihazırda bitmiş LOKAL işi pipeline kuyruğundan geçirir: Review → Build+Test → Commit/PR → Report (teknik analiz + Jira test-senaryosu yorumu). Dev fazı yok."
2
+ description: "Continue already-done LOCAL work through the pipeline tail: Review (with its build gate) → Commit/PR → Report (technical analysis + Jira test-scenario comment). No dev phase. Use when local work is already done and only review, build, commit and reporting remain."
3
+ description-tr: "Halihazırda bitmiş LOKAL işi pipeline kuyruğundan geçirir: Review (build kapısıyla) → Commit/PR → Report (teknik analiz + Jira test-senaryosu yorumu). Dev fazı yok."
4
4
  allowed-tools: Agent, Bash, Read, Write, Edit, Glob, Grep, TaskCreate, TaskUpdate, TaskList, TaskGet, AskUserQuestion, WebFetch, WebSearch, Skill
5
5
  ---
6
6
 
@@ -25,7 +25,7 @@ You already did the work locally - wrote code on the current branch and maybe
25
25
 
26
26
  ```bash
27
27
  /multi-agent:resume-local # current branch vs its base; resolve Jira id from branch name
28
- /multi-agent:resume-local PROJ-12345 # bind to an explicit Jira id for the Phase 7 comment
28
+ /multi-agent:resume-local PROJ-12345 # bind to an explicit Jira id for the Phase 5 comment
29
29
  /multi-agent:resume-local --base develop # override the base branch for the diff
30
30
  /multi-agent:resume-local autopilot # no gate prompts: auto-fix blocking findings, auto-PR, auto-comment
31
31
  ```
@@ -34,13 +34,15 @@ You already did the work locally - wrote code on the current branch and maybe
34
34
 
35
35
  ```
36
36
  Phase 0: Init → project/branch detect, resolve base + diff (work-already-done), Jira id, state (NO worktree)
37
- Phase 4: Review → deterministic gates + parallel review (Fable + Opus + Sonnet) + Fable triage
38
- Phase 5: Build+Test → stack-aware build gate + run existing tests; SUCCESS required (automated, not the interactive user-test)
39
- Phase 6: Commit → commit remaining local changes + push + open PR if none exists
40
- Phase 7: Report → technical analysis + Jira comment with test scenarios (channels: Jira / PR / Confluence / Wiki)
37
+ Phase 3: Review → the Verify gate first (stack-aware build + existing tests; SUCCESS required), then
38
+ parallel review (Fable + Opus + Sonnet) + Fable triage
39
+ Phase 4: Commit → commit remaining local changes + push + open PR if none exists
40
+ Phase 5: Report → technical analysis + Jira comment with test scenarios (channels: Jira / PR / Confluence / Wiki)
41
41
  ```
42
42
 
43
- Phases 1-3 (Analysis / Planning / Dev) are skipped by design - `ship` treats the current branch's local changes as the Phase 3 output.
43
+ Phases 1-2 (Plan / Dev) are skipped by design - `ship` treats the current branch's local changes as the Phase 2 output.
44
+
45
+ **Why Review runs the build gate here.** Verify is Phase 2 Dev's exit gate since v19.0.0, and this mode has no Dev: the work arrived already written. The gate still has to run, so Review runs it before dispatching reviewers, which is what Review Stage 1 did before the consolidation. Reviewing a branch whose build was never checked is the failure this ordering prevents.
44
46
 
45
47
  ## Phase 0 - Context resolution (finish-specific)
46
48
 
@@ -52,12 +54,12 @@ Phases 1-3 (Analysis / Planning / Dev) are skipped by design - `ship` treats t
52
54
 
53
55
  ## Phase execution (reuse the existing phase contracts)
54
56
 
55
- - **Phase 4 Review** - run per `$HOME/.claude/multi-agent-refs/phases/phase-4-review.md` against the resolved diff: deterministic gates (Step 1.x), stack-specific parallel reviewers (Fable + Opus + Sonnet on Claude Code; GPT + Opus + Sonnet on Copilot CLI), Fable triage → `triage.accepted`. Blocking/important accepted findings:
57
+ - **Phase 4 Review** - run per `$HOME/.claude/multi-agent-refs/phases/phase-3-review.md` against the resolved diff: deterministic gates (Step 1.x), stack-specific parallel reviewers (Fable + Opus + Sonnet on Claude Code; GPT + Opus + Sonnet on Copilot CLI), Fable triage → `triage.accepted`. Blocking/important accepted findings:
56
58
  - interactive: present them and ask (`AskUserQuestion`) whether to fix now (loop back through a minimal Phase-3-style TDD fix) or proceed;
57
59
  - `autopilot` (or `prefs.global.resumeLocal.autoFix == true`): auto-fix accepted blocking/important findings, then re-review the fix, before advancing.
58
- - **Phase 5 Build+Test** - the **automated success gate** (this is what "build+test success" means here; the interactive device user-test is `/multi-agent:manual-test`). Stack-aware: build via `figma-config.build` (iOS scheme / Android gradle / detected backend/web build) and run the existing test suite if present (`swift test` / `xcodebuild test` / `./gradlew test` / `pytest` / `npm test` / `vitest`). Require success to advance; on failure, surface logs and (interactive) stop or (autopilot) attempt a bounded fix loop. **If the repo has no tests, report "no tests present" - never fabricate test results.**
59
- - **Phase 6 Commit/PR** - per `$HOME/.claude/multi-agent-refs/phases/phase-6-commit.md`: stage + commit any remaining local changes with a conventional message (`{type}(scope): desc [{JIRA_KEY}-{id}]`), push, and open a PR **only if one does not already exist** for the branch. PR body per `$HOME/.claude/multi-agent-refs/rules.md "External System Outputs"` and `$HOME/.claude/rules/git-conventions.md` - `Ref: #N` / `Related: #N`, never `Closes/Fixes/Resolves`; NO AI/bot attribution anywhere.
60
- - **Phase 7 Report** - per `$HOME/.claude/multi-agent-refs/phases/phase-7-report.md` + `channels.md`: produce the **technical analysis** and **test scenarios**, then post to the configured channels. Default content for `ship`: a Jira **comment** carrying the technical analysis + the test scenarios (and, when the PR was opened, the PR description). Every body runs through the humanizer; bot/tool/AI signatures are FORBIDDEN in comments.
60
+ - **Phase 3 Verify gate** - the **automated success gate** (this is what "build+test success" means here; the interactive device user-test is `/multi-agent:manual-test`). Stack-aware: build via `figma-config.build` (iOS scheme / Android gradle / detected backend/web build) and run the existing test suite if present (`swift test` / `xcodebuild test` / `./gradlew test` / `pytest` / `npm test` / `vitest`). Require success to advance; on failure, surface logs and (interactive) stop or (autopilot) attempt a bounded fix loop. **If the repo has no tests, report "no tests present" - never fabricate test results.**
61
+ - **Phase 4 Commit/PR** - per `$HOME/.claude/multi-agent-refs/phases/phase-4-commit.md`: stage + commit any remaining local changes with a conventional message (`{type}(scope): desc [{JIRA_KEY}-{id}]`), push, and open a PR **only if one does not already exist** for the branch. PR body per `$HOME/.claude/multi-agent-refs/rules.md "External System Outputs"` and `$HOME/.claude/rules/git-conventions.md` - `Ref: #N` / `Related: #N`, never `Closes/Fixes/Resolves`; NO AI/bot attribution anywhere.
62
+ - **Phase 5 Report** - per `$HOME/.claude/multi-agent-refs/phases/phase-5-report.md` + `channels.md`: produce the **technical analysis** and **test scenarios**, then post to the configured channels. Default content for `ship`: a Jira **comment** carrying the technical analysis + the test scenarios (and, when the PR was opened, the PR description). Every body runs through the humanizer; bot/tool/AI signatures are FORBIDDEN in comments.
61
63
 
62
64
  ## Modes
63
65
 
@@ -70,17 +72,17 @@ Before writing anything outward-facing - PR body, Jira comment, Confluence pag
70
72
 
71
73
  ## Required: Phase Tracker Contract
72
74
 
73
- **The phase tracker is required.** Full spec: [`$HOME/.claude/multi-agent-refs/tracker-contract.md`]($HOME/.claude/multi-agent-refs/tracker-contract.md). `ship` registers its active phase set - `0:Init 4:Review 5:Build+Test 6:Commit 7:Report` - but NEVER resets a tracker that an earlier run already built (load-or-continue; contract section "Continuation runs"):
75
+ **The phase tracker is required.** Full spec: [`$HOME/.claude/multi-agent-refs/tracker-contract.md`]($HOME/.claude/multi-agent-refs/tracker-contract.md). `ship` registers its active phase set - `0:Init 3:Review 4:Commit 5:Report` - but NEVER resets a tracker that an earlier run already built (load-or-continue; contract section "Continuation runs"):
74
76
 
75
77
  ```bash
76
78
  # Phase 0, first shell call (every CLI). init ONLY when no prior state exists -
77
- # a task handed off from Phase 5 ("awaiting local test") keeps its full
78
- # phase 0-3 history (elapsed, tokens, USD).
79
+ # a task handed off from the user test inside Phase 3 ("awaiting local test")
80
+ # keeps its full phase 0-2 history (elapsed, tokens, USD).
79
81
  STATE="$HOME/.claude/logs/multi-agent/${TASK_ID}/tracker-state.json"
80
82
  if [ ! -f "$STATE" ]; then
81
83
  bash $HOME/.claude/scripts/phase-tracker.sh init "$TASK_ID"
82
84
  fi
83
- for p in "0:Init" "4:Review" "5:Build+Test" "6:Commit" "7:Report"; do
85
+ for p in "0:Init" "3:Review" "4:Commit" "5:Report"; do
84
86
  bash $HOME/.claude/scripts/phase-tracker.sh add "${p%%:*}" "${p#*:}" # idempotent: existing phases keep their history
85
87
  done
86
88
  bash $HOME/.claude/scripts/phase-tracker.sh update 0 in_progress
@@ -91,7 +93,7 @@ bash $HOME/.claude/scripts/phase-tracker.sh update <N> in_progress|completed|fai
91
93
  bash $HOME/.claude/scripts/phase-tracker.sh tokens <N> <in> <out> [cached]
92
94
  ```
93
95
 
94
- **Continuation path (prior state existed):** (1) if Phase 5 was left `in_progress` with `Now: awaiting local test (user)`, mark it `update 5 completed` + `meta 5 Result "local test done (user)"` before finish's own work (finish re-opens it with `update 5 in_progress` when its Build+Test gate runs; elapsed keeps the original `started_at`, which is expected); (2) print ONE line in `outputLanguage` summarizing the inherited history, e.g. `Continuing PROJ-12345: phases 0-3 finished earlier (12m, 38.4k tok, ~$0.74)` (USD via `phase-tracker.sh cost total`); (3) `render`.
96
+ **Continuation path (prior state existed):** (1) if Phase 3 was left `in_progress` with `Now: awaiting local test (user)`, mark it `update 3 completed` + `meta 3 Result "local test done (user)"` before finish's own work (finish re-opens it with `update 3 in_progress` when its Verify gate runs; elapsed keeps the original `started_at`, which is expected); (2) print ONE line in `outputLanguage` summarizing the inherited history, e.g. `Continuing PROJ-12345: phases 0-3 finished earlier (12m, 38.4k tok, ~$0.74)` (USD via `phase-tracker.sh cost total`); (3) `render`.
95
97
 
96
98
  ### Visual channel - Claude Code (native TaskList widget, required)
97
99
 
@@ -194,7 +194,7 @@ Every host runs three reviewers; only the second slot differs, because GPT-5.4 e
194
194
 
195
195
  With the `fable` rung disabled by prefs the Claude Code panel is two reviewers (Opus + Sonnet); see `$HOME/.claude/multi-agent-refs/features/model-fallback.md`.
196
196
 
197
- Each reviewer receives the diff, the module review guides from Step 2b (when any were found), plus the standard reviewer system prompt (see `$HOME/.claude/multi-agent-refs/phases/phase-4-review.md` for the prompt contract). Output: structured `findings[]` per reviewer.
197
+ Each reviewer receives the diff, the module review guides from Step 2b (when any were found), plus the standard reviewer system prompt (see `$HOME/.claude/multi-agent-refs/phases/phase-3-review.md` for the prompt contract). Output: structured `findings[]` per reviewer.
198
198
 
199
199
  ### 4. Store-compliance cross-reference
200
200
 
@@ -0,0 +1,36 @@
1
+ ---
2
+ description: "Disable model routing and keep its rules, so turning it back on does not re-ask for the same configuration. Use when asked to turn model routing off."
3
+ description-tr: "Model yönlendirmesini kapatır ve kurallarını saklar; tekrar açıldığında aynı yapılandırma yeniden sorulmaz."
4
+ ---
5
+
6
+ # multi-agent route-off - disarm routing, keep the configuration
7
+
8
+ ```bash
9
+ bash "$HOME/.claude/lib/route-state.sh" off
10
+ ```
11
+
12
+ Sets `prefs.global.modelRouting.enabled` to `false`. Dispatch returns to the
13
+ plain ladder immediately: every persona takes its `preferredModel`, and the
14
+ `modelFallback` rules are the only thing that can move it.
15
+
16
+ ## The rules are not deleted
17
+
18
+ `strategy`, `scope`, `rules[]` and `budgetCeilingUsd` all survive. `route-on`
19
+ brings back exactly what was configured, without asking again.
20
+
21
+ This is the same contract `autopilot-off` follows with its repo selection, for
22
+ the same reason: a command named after a toggle that quietly discards
23
+ configuration is a destructive action in disguise. To actually remove the rules,
24
+ replace them with an empty array:
25
+
26
+ ```bash
27
+ echo '[]' > /tmp/none.json
28
+ bash "$HOME/.claude/lib/route-state.sh" set-rules /tmp/none.json
29
+ ```
30
+
31
+ ## What stays behind
32
+
33
+ Routing decisions already written to the cost ledger stay there. They are a
34
+ record of what happened on past runs, and deleting them would make a run's cost
35
+ unexplainable after the fact - which is the one thing the ledger exists to
36
+ prevent.
@@ -0,0 +1,74 @@
1
+ ---
2
+ description: "Enable policy-driven model routing: pick a strategy, a scope, and the rules that say which rung a call lands on. Use when asked to turn model routing on."
3
+ description-tr: "Politika güdümlü model yönlendirmesini açar: strateji, kapsam ve hangi çağrının hangi basamağa düşeceğini söyleyen kuralları sorar."
4
+ argument-hint: "[--strategy=manual|task-fit|cost-ceiling] [--scope=subagent,bulk-read,research]"
5
+ ---
6
+
7
+ # multi-agent route-on - arm model routing
8
+
9
+ ```bash
10
+ bash "$HOME/.claude/lib/route-state.sh" on ${ARGUMENTS}
11
+ ```
12
+
13
+ Routing ships **off**. This is the command that turns it on, and it writes to
14
+ `prefs.global.modelRouting`, validated against the repo's `schemas/route-config.schema.json`.
15
+
16
+ ## What a rule is
17
+
18
+ ```json
19
+ { "when": { "persona": "code-reviewer" }, "prefer": ["opus", "sonnet"] }
20
+ { "when": { "phase": 2 }, "prefer": ["sonnet", "haiku"] }
21
+ ```
22
+
23
+ Ordered, first match wins. `when` matches on `persona`, `phase` (0..5) or
24
+ `taskKind`; `prefer` lists rungs in descending preference.
25
+
26
+ **Rung names are the contract; model ids are not.** `opus` is a rung here and a
27
+ model id in `cost-table.json`, and the second can change without anyone editing
28
+ a rule. Writing `claude-opus-5` into a rule pins a decision to a string that will
29
+ go stale.
30
+
31
+ Set rules with:
32
+
33
+ ```bash
34
+ bash "$HOME/.claude/lib/route-state.sh" set-rules path/to/rules.json
35
+ ```
36
+
37
+ ## Strategies
38
+
39
+ | Strategy | What it does |
40
+ |---|---|
41
+ | `manual` | only the explicit rules apply, nothing is inferred. The default, because a router that guesses is a router nobody can predict |
42
+ | `task-fit` | a rule may match on `taskKind`, and the cheapest rung clearing it is chosen |
43
+ | `cost-ceiling` | rungs downgrade as the run approaches `budgetCeilingUsd` |
44
+
45
+ ## Scope, and the value that is not in it
46
+
47
+ `scope` names the call sites routing may act on: `subagent`, `bulk-read`,
48
+ `research`. Every one of them is a call **this pipeline makes itself**.
49
+
50
+ There is no `host-session` value, and that absence is enforced by that schema
51
+ rather than written as advice. Routing a host session means rewriting the CLI's
52
+ base URL to point at a local gateway - which sends the user's *entire* session
53
+ through a third layer, including work that has nothing to do with this pipeline,
54
+ breaks the subscription's auth model, and silently changes which model answered.
55
+ Passing `--scope=host-session` is refused with that reason, not ignored.
56
+
57
+ ## The limit this command prints every time
58
+
59
+ On Claude Code a subagent cannot be dispatched to a non-Anthropic model: subagent
60
+ dispatch belongs to the host, not to us. So Phase 1, 2 and 3 personas stay inside
61
+ the Anthropic ladder no matter what the rules say, and external providers apply
62
+ only at `bulk-read` and `research`, where the pipeline makes the HTTP call.
63
+
64
+ This is printed by `route-status` on every invocation instead of living in a doc,
65
+ because the question it answers - "routing is on, why is the reviewer still on
66
+ Opus" - otherwise arrives days later as a bug report.
67
+
68
+ ## Related
69
+
70
+ - `/multi-agent:route-off` - disables routing and **keeps** the rules
71
+ - `/multi-agent:route-status` - what is active, and what it costs
72
+ - `/multi-agent:model` - whether the top rung exists at all. That is a different
73
+ question: this command decides which rung a call picks, that one decides
74
+ whether the top one is in play
@@ -0,0 +1,56 @@
1
+ ---
2
+ description: "Report model routing: whether it is armed, which rule applies where, which rung the last dispatches took, and what this run has cost. Use when asked which model is being used or why."
3
+ description-tr: "Model yönlendirmesinin durumunu raporlar: açık mı, hangi kural nerede geçerli, son çağrılar hangi basamağa gitti, bu koşu ne tuttu."
4
+ ---
5
+
6
+ # multi-agent route-status - what is actually routing
7
+
8
+ ```bash
9
+ bash "$HOME/.claude/lib/route-state.sh" status
10
+ ```
11
+
12
+ Reports the stored policy, then the part that matters more: **what it can and
13
+ cannot reach.**
14
+
15
+ ## Three states, told apart
16
+
17
+ | Output | Meaning |
18
+ |---|---|
19
+ | `routing: false`, rules present | configured and disarmed. `route-on` restores it as-is; nothing is lost |
20
+ | `routing: true`, `rules: 0` | armed with nothing to match. Not an error - a configuration state, and the most common "I turned it on and nothing changed" |
21
+ | `routing: true` with rules listed | live. Each rule is printed as `when <key>=<value> -> rung > rung` |
22
+
23
+ A disabled router with rules is deliberately not reported as "off" alone: that
24
+ reads as "unconfigured" and sends the user through `route-on`'s questions a
25
+ second time.
26
+
27
+ ## The limit, printed every time
28
+
29
+ On Claude Code a subagent cannot be sent to a non-Anthropic model. Subagent
30
+ dispatch belongs to the host; the pipeline asks for a persona and the host
31
+ decides what answers. So Phase 1, 2 and 3 personas stay inside the Anthropic
32
+ ladder whatever the rules say, and an external provider is only reachable where
33
+ the pipeline makes the HTTP call itself - `bulk-read.sh` and `research_ask`.
34
+
35
+ This paragraph is output, not documentation, because the alternative is the
36
+ question arriving later as "routing is on but the reviewer is still on Opus, is
37
+ it broken". It is not broken; it is the seam.
38
+
39
+ ## Where decisions are recorded
40
+
41
+ With `recordDecisions: true` (the default) every routing decision is written to
42
+ the cost ledger: which rule matched, which rung it chose, and why. That is what
43
+ makes a run's cost explainable after it finished - a router whose choices are not
44
+ recorded cannot be audited, and the cost question always arrives after the run,
45
+ never during it.
46
+
47
+ Per-run cost and the model breakdown come from the same ledger:
48
+
49
+ ```bash
50
+ node "$HOME/.claude/scripts/token-budget-report.mjs" --json
51
+ ```
52
+
53
+ ## Related
54
+
55
+ - `/multi-agent:route-on` / `/multi-agent:route-off` - arm and disarm
56
+ - `/multi-agent:model` - whether the top rung exists at all
@@ -558,7 +558,7 @@ Re-run scan from Step 1. Show final status:
558
558
  All tokens present. Pipeline ready to use.
559
559
 
560
560
  Optional per-project features (configured on first use, nothing to do now):
561
- • Phase 7 Report Step 2 Wiki - auto-generates component wiki pages + Figma
561
+ • Phase 5 Report Step 2 Wiki - auto-generates component wiki pages + Figma
562
562
  screenshots. Activates when (a) task is a component AND (b) a Figma token
563
563
  is in Keychain. Four adapters supported: submodule / in-repo / github-wiki
564
564
  / separate-repo. First run asks: use auto-detected path, use a custom
@@ -803,7 +803,7 @@ All tokens are optional in the sense that every service can be answered with Ski
803
803
  Offer to merge `$HOME/.claude/templates/claude-hooks.json`: three `PreToolUse` gates that block on a non-zero exit (secret scan, agent-guard, read-size) plus three capture hooks that block nothing (`PreCompact`, `SessionEnd`, `SessionStart`). What each does: `$HOME/.claude/multi-agent-refs/picker-contract.md`.
804
804
 
805
805
  - Ask (picker): "Install the pipeline's hooks into `~/.claude/settings.json`?" Default Yes.
806
- - On Yes, deep-merge EVERY event in the template's `hooks` object, not `PreToolUse` alone - merging one event silently drops the capture hooks, and a run killed before Phase 7 then loses its findings exactly as it did before they existed. Preserve existing hooks; never duplicate a matcher already calling the same script.
806
+ - On Yes, deep-merge EVERY event in the template's `hooks` object, not `PreToolUse` alone - merging one event silently drops the capture hooks, and a run killed before Phase 5 then loses its findings exactly as it did before they existed. Preserve existing hooks; never duplicate a matcher already calling the same script.
807
807
  - Say what the merge does NOT cover: only the three gates need no run-specific arguments, so only they are hookable; the rest are phase-enforced.
808
808
  - Say what it does not turn on: the read-size gate is inert until `prefs.global.bulkRead.mode` is set. Recommend `observe` first. Why, and the Phase 3 exemption: `$HOME/.claude/multi-agent-refs/picker-contract.md`.
809
809
 
@@ -19,7 +19,7 @@ Show every active and completed task as a table.
19
19
 
20
20
  `runs-index.mjs` resolves through `lib/run-paths.sh` / `scripts/_run-paths.mjs`,
21
21
  so it sees both directory layouts (`<project>/<id>/` and the flat `<id>/`),
22
- the salvaged `artifacts/` copy Phase 6 leaves behind, and every spelling of a
22
+ the salvaged `artifacts/` copy Phase 4 leaves behind, and every spelling of a
23
23
  task id - and it counts a run that exists in both layouts once. Do NOT
24
24
  re-scan the tree by hand: the earlier instruction here listed three
25
25
  hard-coded `.worktrees/` paths and a single `find` depth, and on a real
@@ -33,7 +33,7 @@ Show every active and completed task as a table.
33
33
  2. **Fields per run** (already in the output): `taskId`, `project`, `branch`,
34
34
  `currentPhase`, `status`, `startedAt`, `worktreePath`, `prUrl`, `autopilot`,
35
35
  `phases[]`, `tokens`, `estUsd`, `group`, plus `duplicateOf` when the run also
36
- exists in the other layout and `salvaged` when its state is the Phase 6 copy.
36
+ exists in the other layout and `salvaged` when its state is the Phase 4 copy.
37
37
 
38
38
  3. **Groups are computed, not judged.** `runs-index.mjs` assigns `group` by the
39
39
  table below; report what it returns rather than re-deriving it. `in_progress`
@@ -42,7 +42,7 @@ Show every active and completed task as a table.
42
42
 
43
43
  | Group | Test | Action offered |
44
44
  |---|---|---|
45
- | `waiting` - Waiting on you | `status == "awaiting_input"`, or a `pr.url`, or phase >= 6 | `resume #N` - the work landed, it needs your answer |
45
+ | `waiting` - Waiting on you | `status == "awaiting_input"`, or a `pr.url`, or phase >= 4 | `resume #N` - the work landed, it needs your answer |
46
46
  | `stopped` - Stopped mid-development | anything else past phase 0 | `resume #N` or `kill #N` |
47
47
  | `question` - Left at a question | phase 0 | `garbage-collect --abandoned` - nothing was built |
48
48
  | `unknown` - Status not recorded | no `status` field | say so; offer nothing |
@@ -57,8 +57,8 @@ Show every active and completed task as a table.
57
57
 
58
58
  | ID | Jira/Task | Branch | Phase | Status | Duration |
59
59
  |----|-----------|--------|-------|--------|----------|
60
- | #1 | {JIRA-KEY}-12345 | feature/{JIRA-KEY}-12345-... | 7/7 DONE | ✅ Complete | 12m |
61
- | #3 | {JIRA-KEY}-12345 | feature/{JIRA-KEY}-12345-... | 7/7 DONE | ✅ Complete | 8m |
60
+ | #1 | {JIRA-KEY}-12345 | feature/{JIRA-KEY}-12345-... | 5/5 DONE | ✅ Complete | 12m |
61
+ | #3 | {JIRA-KEY}-12345 | feature/{JIRA-KEY}-12345-... | 5/5 DONE | ✅ Complete | 8m |
62
62
 
63
63
  💡 log #1 | resume #N | kill #N
64
64
  ```
@@ -41,7 +41,7 @@ is being replaced.
41
41
  | `paused` / `failed` | Say the task is not running, and that `/multi-agent:resume #N` will re-enter with the instruction applied at that phase's entry. Queue it. |
42
42
  | `complete` | Refuse. Nothing will read it. Point at `/multi-agent` for a follow-up run. |
43
43
 
44
- `currentPhase` is 7 and status is `in_progress` → warn that Phase 7 is the
44
+ `currentPhase` is 7 and status is `in_progress` → warn that Phase 5 is the
45
45
  last one, so an instruction queued now may never be consumed.
46
46
 
47
47
  3. **Read the instruction** - from the argument, or ask for it when the
@@ -52,7 +52,7 @@ is being replaced.
52
52
  4. **Show what will be queued, and ask**:
53
53
 
54
54
  ```
55
- Steer #3 ({JIRA-KEY}-12345, Phase 3 Dev, in_progress)
55
+ Steer #3 ({JIRA-KEY}-12345, Phase 2 Dev, in_progress)
56
56
 
57
57
  "the field is called web, not frontend"
58
58
 
@@ -65,8 +65,8 @@ Run every step automatically:
65
65
  ```
66
66
  Step 0: DOCTOR doctor.mjs - exit 2 or 4 stops the sync
67
67
  Step 1.5: DETECT Compare timestamps, find stale targets
68
- Step 2: COPILOT Claude Code -> Copilot CLI (instructions + 56 sub-command skills)
69
- Step 2b: CODEX Claude Code -> Codex CLI (1 router skill + 56 specs as refs + 8 agent TOML)
68
+ Step 2: COPILOT Claude Code -> Copilot CLI (instructions + 60 sub-command skills)
69
+ Step 2b: CODEX Claude Code -> Codex CLI (1 router skill + 60 specs as refs + 8 agent TOML)
70
70
  Step 3: REPO Claude Code -> pipeline repo (genericized, personal data scrub, bash -n on all sh)
71
71
  Step 3c: PLUGINS pipeline shared/external -> multi-agent-plugins marketplace (rebuild knowledge/,
72
72
  bump changed plugins' patch version, commit + push the plugins repo)
@@ -494,19 +494,18 @@ same 56 specs as reference files rather than as peer skills, via Step 2b - see
494
494
  |-------------|-------------|
495
495
  | `~/.claude/commands/multi-agent/{cmd}/SKILL.md` | `~/.copilot/skills/multi-agent-{cmd}/SKILL.md` |
496
496
 
497
- **56 commands are synced** (canonical inventory - must match `cross-cli-contract.md` section 1; drift = contract violation):
497
+ **60 commands are synced** (canonical inventory - must match `cross-cli-contract.md` section 1; drift = contract violation):
498
498
 
499
499
  ```
500
- analysis, analysis-jira, analysis-resolve, autopilot, autopilot-off,
501
- autopilot-on, autopilot-status, build-optimize, channels, complaint-analysis,
502
- create-jira, design-check, diff-explain, doctor, feedback, forget,
503
- garbage-collect, graph, help, ios-coding-standard, issue, jira, kill,
504
- language, local, local-autopilot, log, manual-test, prune-logs,
505
- prune-prompts, purge, refactor, resume, resume-local, review,
506
- review-analysis, review-issue, review-jira, routines, save, scan, search,
507
- setup, stack, status, steer, store-ready, sync, test, test-accessibility,
508
- test-dark-mode, test-dynamic-type, test-screenshots, testflight-validation,
509
- uninstall, update
500
+ analysis, analysis-jira, analysis-resolve, autopilot, autopilot-off, autopilot-on,
501
+ autopilot-status, build-optimize, channels, complaint-analysis, create-jira,
502
+ design-check, diff-explain, doctor, feedback, forget, garbage-collect, graph, help,
503
+ ios-coding-standard, issue, jira, kill, language, local, local-autopilot, log,
504
+ manual-test, model, prune-logs, prune-prompts, purge, refactor, resume, resume-local,
505
+ review, review-analysis, review-issue, review-jira, route-off, route-on, route-status,
506
+ routines, save, scan, search, setup, stack, status, steer, store-ready, sync, test,
507
+ test-accessibility, test-dark-mode, test-dynamic-type, test-screenshots,
508
+ testflight-validation, uninstall, update
510
509
  ```
511
510
 
512
511
  **NOT synced**: `$HOME/.claude/multi-agent-refs/*` - lazy-load references, Claude Code specific
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  description: "UI Bug Hunter - iOS Simulator (simctl) + Android Emulator (adb). Auto-detects platform. Screenshot + tap + analyze on the booted device. The Phase 5 Manual Test flow lives at /multi-agent:manual-test. Use when a running app should be driven on a simulator or emulator to hunt UI bugs."
3
- description-tr: "UI Bug Hunter - iOS Simulator (simctl) + Android Emulator (adb). Platformu otomatik algılar. Açık cihazda screenshot + tap + analiz. Faz 5 Manuel Test akışı /multi-agent:manual-test'te."
3
+ description-tr: "UI Bug Hunter - iOS Simulator (simctl) + Android Emulator (adb). Platformu otomatik algılar. Açık cihazda screenshot + tap + analiz. Faz 3 Manuel Test akışı /multi-agent:manual-test'te."
4
4
  argument-hint: '[scenario] - e.g. "dark mode" | "accessibility" | "dynamic type" | "screenshot tr" | "store-ready" | (empty = full sweep)'
5
5
  ---
6
6
 
@@ -345,7 +345,7 @@ capability_of() {
345
345
  esac
346
346
  fi
347
347
  case "$1" in
348
- jira) echo "read the ticket, its comments and its linked issues; post the Phase 7 comment" ;;
348
+ jira) echo "read the ticket, its comments and its linked issues; post the Phase 5 comment" ;;
349
349
  bitbucket_token) echo "read the repo, open and update pull requests" ;;
350
350
  bitbucket_user) echo "identify the PR author (paired with bitbucket_token)" ;;
351
351
  github) echo "read issues, open pull requests, read Actions runs" ;;
@@ -49,7 +49,7 @@
49
49
  # 6 not configured (prefs.global.hosts.fortify empty, or an instance-id-only
50
50
  # lookup with no prefs.global.fortify.versionIds to search)
51
51
  #
52
- # Phase 4 review gate contract:
52
+ # Phase 3 review gate contract:
53
53
  # Critical > 0 → blocking=true, reason="critical-findings"
54
54
  # High > 0 → blocking=false, reason="high-findings-warning"
55
55
  # else → blocking=false, reason="clean"
@@ -0,0 +1,142 @@
1
+ #!/usr/bin/env bash
2
+ # model-rung.sh - read or flip the top model rung, and keep cost pricing with it.
3
+ #
4
+ # Backs /multi-agent:model. Two preference keys move as ONE write:
5
+ # prefs.global.modelFallback.fableEnabled the rung itself
6
+ # prefs.global.costBudget.pricingModel what the ledger prices against
7
+ #
8
+ # They were always meant to move together - model-fallback.md says so in prose -
9
+ # but nothing enforced it, so a hand edit that flipped one left the estimate
10
+ # pricing every call at a rung that was not in play. With the rung off and
11
+ # pricing still `fable`, the budget ceiling trips early and triggers a downgrade
12
+ # nothing needed. One write, or none.
13
+ #
14
+ # Usage:
15
+ # model-rung.sh report current state (exit 0)
16
+ # model-rung.sh on enable the fable rung
17
+ # model-rung.sh off disable it
18
+ #
19
+ # Exit codes:
20
+ # 0 - reported or applied
21
+ # 2 - bad argument
22
+ # 3 - preferences file missing or unparseable (nothing written)
23
+
24
+ set -uo pipefail
25
+
26
+ PREFS="${MULTI_AGENT_PREFS:-$HOME/.claude/multi-agent-preferences.json}"
27
+ ACTION="${1:-report}"
28
+
29
+ case "$ACTION" in
30
+ report | on | off) ;;
31
+ *)
32
+ echo "usage: model-rung.sh [on|off]" >&2
33
+ exit 2
34
+ ;;
35
+ esac
36
+
37
+ # A missing `jq` must not look like a missing setting. Without this the reads
38
+ # below return empty, the command reports the rung as already off, and a user
39
+ # who asked to turn it off is told it was never on.
40
+ if ! command -v jq >/dev/null 2>&1; then
41
+ echo "model-rung: jq not found - cannot read or write preferences (install: brew install jq)" >&2
42
+ exit 3
43
+ fi
44
+
45
+ if [ ! -f "$PREFS" ]; then
46
+ echo "model-rung: preferences not found at $PREFS - run /multi-agent:setup first" >&2
47
+ exit 3
48
+ fi
49
+
50
+ if ! jq -e . "$PREFS" >/dev/null 2>&1; then
51
+ echo "model-rung: $PREFS is not valid JSON - refusing to write over it" >&2
52
+ exit 3
53
+ fi
54
+
55
+ CUR=$(jq -r '.global.modelFallback.fableEnabled // false' "$PREFS")
56
+ PRICING=$(jq -r '.global.costBudget.pricingModel // "fable"' "$PREFS")
57
+
58
+ host_note() {
59
+ # Which host this is decides whether the switch is a switch or a status line.
60
+ # Saying "enabled" on a host where the rung means something else entirely is
61
+ # the surprise this block exists to avoid.
62
+ case "${MULTI_AGENT_HOST:-claude}" in
63
+ claude)
64
+ echo "Claude Code: the fable rung is Fable 5. This switch is live here."
65
+ ;;
66
+ copilot)
67
+ echo "Copilot CLI: Fable 5 is not offered; its personas never sat on this rung."
68
+ echo "This setting is recorded but changes nothing on this host."
69
+ ;;
70
+ codex)
71
+ echo "Codex CLI: the fable rung means gpt-5.6 @ xhigh, a different model on a"
72
+ echo "different account. This switch deliberately does NOT touch it - a knob"
73
+ echo "named after an Anthropic model must not silently retune a Codex run."
74
+ ;;
75
+ *)
76
+ echo "Unknown host '${MULTI_AGENT_HOST}'; reporting the stored value only."
77
+ ;;
78
+ esac
79
+ }
80
+
81
+ if [ "$ACTION" = "report" ]; then
82
+ echo "fable rung: $CUR"
83
+ echo "cost pricing: $PRICING"
84
+ if [ "$CUR" = "true" ] && [ "$PRICING" != "fable" ]; then
85
+ echo "MISMATCH: rung is on but the ledger prices against '$PRICING' - run 'model on' to realign"
86
+ elif [ "$CUR" != "true" ] && [ "$PRICING" = "fable" ]; then
87
+ echo "MISMATCH: rung is off but the ledger still prices against fable - this trips"
88
+ echo " the budget ceiling early. Run 'model off' to realign."
89
+ fi
90
+ echo ""
91
+ host_note
92
+ exit 0
93
+ fi
94
+
95
+ if [ "$ACTION" = "on" ]; then
96
+ WANT=true
97
+ WANT_PRICING=fable
98
+ else
99
+ WANT=false
100
+ WANT_PRICING=opus
101
+ fi
102
+
103
+ TMP=$(mktemp) || { echo "model-rung: mktemp failed" >&2; exit 3; }
104
+ trap 'rm -f "$TMP"' EXIT
105
+
106
+ # Both keys in one filter. A two-step write can leave the pair half-applied if
107
+ # the second step fails, which is the exact state this command exists to prevent.
108
+ if ! jq --argjson want "$WANT" --arg pricing "$WANT_PRICING" '
109
+ .global.modelFallback //= {}
110
+ | .global.modelFallback.fableEnabled = $want
111
+ | .global.costBudget //= {}
112
+ | .global.costBudget.pricingModel = $pricing
113
+ ' "$PREFS" > "$TMP"; then
114
+ echo "model-rung: jq failed - preferences left untouched" >&2
115
+ exit 3
116
+ fi
117
+
118
+ if ! jq -e . "$TMP" >/dev/null 2>&1; then
119
+ echo "model-rung: produced invalid JSON - preferences left untouched" >&2
120
+ exit 3
121
+ fi
122
+
123
+ mv "$TMP" "$PREFS"
124
+ trap - EXIT
125
+
126
+ echo "fable rung: $CUR -> $WANT"
127
+ echo "cost pricing: $PRICING -> $WANT_PRICING"
128
+ echo ""
129
+
130
+ if [ "$WANT" = "false" ]; then
131
+ # Said at the moment of the change rather than left in a doc: the panel
132
+ # shrinking is the part people discover later and read as a bug.
133
+ echo "Phase 3 reviewer panel: 3 models -> 2 (opus + sonnet)."
134
+ echo "Reviewer 1 lands on opus, which Reviewer 2 already holds, and dispatching"
135
+ echo "one model twice is not cross-model review. consensus.reviewerCount records 2."
136
+ echo "Triage also runs on opus, so it shares a model with Reviewer 1 - the Phase 3"
137
+ echo "Step 3 anonymisation requirement is not optional while the rung is off."
138
+ echo ""
139
+ fi
140
+
141
+ host_note
142
+ exit 0