@mmerterden/multi-agent-pipeline 19.1.4 → 20.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/CHANGELOG.md +94 -0
  2. package/README.md +19 -36
  3. package/README.tr.md +18 -35
  4. package/docs/adr/0002-instruction-driven-flag.md +6 -5
  5. package/docs/adr/0005-lazy-phase-docs.md +2 -2
  6. package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
  7. package/docs/adr/0009-claude-stack-skills-plugin-only.md +1 -1
  8. package/docs/adr/0010-own-code-graph.md +5 -4
  9. package/docs/adr/0012-macos-only.md +2 -2
  10. package/docs/adr/0013-lsp-code-intelligence.md +2 -2
  11. package/docs/adr/0014-six-phase-consolidation.md +9 -9
  12. package/docs/adr/0015-one-pipeline-no-depth-answer.md +83 -0
  13. package/docs/adr/0016-the-run-shape-is-asked-not-typed.md +69 -0
  14. package/docs/adr/README.md +18 -16
  15. package/docs/architecture.md +2 -2
  16. package/docs/ecosystem.md +5 -5
  17. package/docs/facts.json +3 -6
  18. package/docs/features.md +4 -5
  19. package/docs/token-budget-history.md +1 -1
  20. package/install/_common.mjs +9 -1
  21. package/install/templates/copilot-instructions.md +7 -16
  22. package/manifest.json +111 -114
  23. package/package.json +1 -1
  24. package/pipeline/commands/multi-agent/SKILL.md +6 -8
  25. package/pipeline/commands/multi-agent/analysis/SKILL.md +2 -0
  26. package/pipeline/commands/multi-agent/analysis-jira/SKILL.md +2 -0
  27. package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +2 -0
  28. package/pipeline/commands/multi-agent/autopilot/SKILL.md +2 -0
  29. package/pipeline/commands/multi-agent/autopilot-on/SKILL.md +2 -0
  30. package/pipeline/commands/multi-agent/autopilot-status/SKILL.md +1 -1
  31. package/pipeline/commands/multi-agent/build-optimize/SKILL.md +2 -0
  32. package/pipeline/commands/multi-agent/channels/SKILL.md +1 -1
  33. package/pipeline/commands/multi-agent/create-jira/SKILL.md +2 -0
  34. package/pipeline/commands/multi-agent/design-check/SKILL.md +1 -1
  35. package/pipeline/commands/multi-agent/forget/SKILL.md +2 -0
  36. package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +4 -2
  37. package/pipeline/commands/multi-agent/help/SKILL.md +21 -27
  38. package/pipeline/commands/multi-agent/ios-coding-standard/SKILL.md +5 -4
  39. package/pipeline/commands/multi-agent/issue/SKILL.md +2 -0
  40. package/pipeline/commands/multi-agent/jira/SKILL.md +2 -0
  41. package/pipeline/commands/multi-agent/language/SKILL.md +2 -0
  42. package/pipeline/commands/multi-agent/prune-logs/SKILL.md +2 -0
  43. package/pipeline/commands/multi-agent/purge/SKILL.md +2 -0
  44. package/pipeline/commands/multi-agent/resume/SKILL.md +177 -48
  45. package/pipeline/commands/multi-agent/save/SKILL.md +2 -0
  46. package/pipeline/commands/multi-agent/stack/SKILL.md +2 -0
  47. package/pipeline/commands/multi-agent/sync/SKILL.md +6 -7
  48. package/pipeline/commands/multi-agent/test-screenshots/SKILL.md +2 -0
  49. package/pipeline/commands/multi-agent/uninstall/SKILL.md +2 -0
  50. package/pipeline/lib/repo-hygiene.sh +1 -1
  51. package/pipeline/multi-agent-refs/analysis/render.md +1 -1
  52. package/pipeline/multi-agent-refs/analysis/resolve.md +1 -1
  53. package/pipeline/multi-agent-refs/analysis/synthesis.md +1 -1
  54. package/pipeline/multi-agent-refs/analysis-template.md +1 -1
  55. package/pipeline/multi-agent-refs/component-dispatch.md +0 -8
  56. package/pipeline/multi-agent-refs/cross-cli-contract.md +10 -11
  57. package/pipeline/multi-agent-refs/features/external-context-injection.md +2 -0
  58. package/pipeline/multi-agent-refs/features/review-delta.md +1 -1
  59. package/pipeline/multi-agent-refs/features/review-multi-repo.md +3 -3
  60. package/pipeline/multi-agent-refs/features/skill-conformance.md +1 -1
  61. package/pipeline/multi-agent-refs/features/visual-evidence.md +2 -1
  62. package/pipeline/multi-agent-refs/features/worktree-finalize.md +1 -1
  63. package/pipeline/multi-agent-refs/generate-issue.md +2 -0
  64. package/pipeline/multi-agent-refs/issue-jira-triad.md +2 -0
  65. package/pipeline/multi-agent-refs/keychain.md +2 -0
  66. package/pipeline/multi-agent-refs/knowledge.md +0 -7
  67. package/pipeline/multi-agent-refs/outside-the-pipeline.md +1 -1
  68. package/pipeline/multi-agent-refs/payload-contracts.md +1 -1
  69. package/pipeline/multi-agent-refs/phases/modes.md +32 -108
  70. package/pipeline/multi-agent-refs/phases/operations.md +2 -0
  71. package/pipeline/multi-agent-refs/phases/phase-0-init.md +23 -42
  72. package/pipeline/multi-agent-refs/phases/phase-1-plan.md +7 -18
  73. package/pipeline/multi-agent-refs/phases/phase-2-dev.md +13 -44
  74. package/pipeline/multi-agent-refs/phases/phase-3-review.md +19 -22
  75. package/pipeline/multi-agent-refs/phases/phase-4-commit.md +6 -6
  76. package/pipeline/multi-agent-refs/phases/phase-5-report.md +2 -2
  77. package/pipeline/multi-agent-refs/phases.md +9 -11
  78. package/pipeline/multi-agent-refs/progress-contract.md +1 -1
  79. package/pipeline/multi-agent-refs/readiness-review.md +2 -0
  80. package/pipeline/multi-agent-refs/rules.md +1 -1
  81. package/pipeline/multi-agent-refs/tracker-contract.md +9 -40
  82. package/pipeline/multi-agent-refs/wiki-capture.md +3 -2
  83. package/pipeline/preferences-template.json +2 -2
  84. package/pipeline/rules/figma-pipeline.md +1 -1
  85. package/pipeline/schemas/agent-state.schema.json +5 -10
  86. package/pipeline/schemas/migrations/prefs-2.7.0-to-2.8.0.mjs +33 -0
  87. package/pipeline/schemas/phases.json +3 -24
  88. package/pipeline/schemas/prefs.schema.json +5 -5
  89. package/pipeline/scripts/cost-table.json +1 -1
  90. package/pipeline/scripts/gc-refs.sh +1 -1
  91. package/pipeline/scripts/gen-mode-dispatch.mjs +11 -41
  92. package/pipeline/scripts/migrate-prefs.mjs +18 -17
  93. package/pipeline/scripts/phase-tracker.sh +2 -2
  94. package/pipeline/scripts/phase0-exit-gate.mjs +1 -1
  95. package/pipeline/scripts/plan-coverage-gate.mjs +3 -3
  96. package/pipeline/scripts/run-aggregator.mjs +1 -1
  97. package/pipeline/scripts/usage-report.mjs +0 -2
  98. package/pipeline/scripts/worktree-finalize.sh +2 -2
  99. package/pipeline/skills/.skill-manifest.json +8 -20
  100. package/pipeline/skills/.skills-index.json +6 -39
  101. package/pipeline/skills/shared/README.md +5 -8
  102. package/pipeline/skills/shared/core/multi-agent/SKILL.md +8 -11
  103. package/pipeline/skills/shared/core/multi-agent-autopilot-status/SKILL.md +1 -1
  104. package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +13 -16
  105. package/pipeline/skills/shared/core/multi-agent-ios-coding-standard/SKILL.md +2 -3
  106. package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +51 -15
  107. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +6 -6
  108. package/pipeline/skills/skills-index.md +3 -6
  109. package/pipeline/commands/multi-agent/local/SKILL.md +0 -132
  110. package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +0 -142
  111. package/pipeline/commands/multi-agent/resume-local/SKILL.md +0 -114
  112. package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +0 -41
  113. package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +0 -55
  114. package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +0 -51
@@ -104,13 +104,13 @@ Persist the totals as `state.diffRisk` (Phase 4 `risk` section, Phase 5, `run-me
104
104
  **Gate behavior**: this step is **never blocking**. If risk scoring fails (git error, parse error, validator rejection), continue with no priority hint - reviewers receive the full diff in their default order. Failures are logged via metrics:
105
105
 
106
106
  ```bash
107
- [ -z "$RISK_JSON" ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 review.diff_risk_skipped reason=$REASON
107
+ [ -z "$RISK_JSON" ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.diff_risk_skipped reason=$REASON
108
108
  ```
109
109
 
110
110
  On success, emit a single summary metric:
111
111
 
112
112
  ```bash
113
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 review.diff_risk \
113
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.diff_risk \
114
114
  top_files=$(jq '.files | length' <<< "$RISK_JSON") \
115
115
  max_score=$(jq '.totals.max_score' <<< "$RISK_JSON") \
116
116
  loc_added=$(jq '.totals.loc_added' <<< "$RISK_JSON") \
@@ -127,7 +127,7 @@ Step 1.75 uses `test_lines_removed` as an advisory hint only - too weak for wh
127
127
  ```bash
128
128
  TEST_INTEGRITY_JSON=$(printf '%s' "$RISK_FULL" | node $HOME/.claude/scripts/test-integrity-gate.mjs 2>/dev/null || echo "")
129
129
  TI_COUNT=$(jq -r '.count // 0' <<< "${TEST_INTEGRITY_JSON:-{\}}" 2>/dev/null || echo 0)
130
- [ "$TI_COUNT" -gt 0 ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 review.test_integrity findings="$TI_COUNT"
130
+ [ "$TI_COUNT" -gt 0 ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.test_integrity findings="$TI_COUNT"
131
131
  ```
132
132
 
133
133
  `findings[]` are reviewer-shaped (`test_integrity`, `blocking`), so they merge into the reviewer findings at Step 3.0 and need no triage-prompt or `validate-triage.mjs` change. Triage keeps each blocking unless the removal is justified per the immutable-test rule (spec changed AND commit body names the test) → `deferred[]`.
@@ -142,7 +142,7 @@ On a trivial diff every reviewer agrees and the extra models plus triage are pai
142
142
  SCOPE_JSON=$(printf '%s' "$RISK_FULL" | node $HOME/.claude/scripts/review-scope.mjs 2>/dev/null \
143
143
  || echo '{"scope":"full","reason":"no risk report - failing safe"}')
144
144
  REVIEW_SCOPE=$(jq -r '.scope // "full"' <<< "$SCOPE_JSON")
145
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 review.scope scope="$REVIEW_SCOPE"
145
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.scope scope="$REVIEW_SCOPE"
146
146
  ```
147
147
 
148
148
  `single` (Reviewer 1 only) requires **all** of: churn <= 20 lines, `totals.max_score` < 3.0, and no `security_path` / `migration` / `public_api` / `no_test_change` / `test_lines_removed` on any file. Anything else → `full`.
@@ -159,7 +159,7 @@ node $HOME/.claude/scripts/skill-conformance.mjs \
159
159
  --repo "$WORKTREE" --out "$WORKTREE/.pipeline/criteria-manifest.json"
160
160
  CRIT_RC=$?
161
161
  CRITERIA=$(cat "$WORKTREE/.pipeline/criteria-manifest.json" 2>/dev/null || echo '{}')
162
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 review.criteria \
162
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.criteria \
163
163
  rules="$(jq -r '.selectedRuleCount // 0' <<< "$CRITERIA")" \
164
164
  ledger="$(jq -r '.ledger.source // "derived"' <<< "$CRITERIA")"
165
165
  [ "$CRIT_RC" != "0" ] && HALT "criteria could not be resolved (rc=$CRIT_RC) - see resolutionFailure / unparseableRegistries"
@@ -175,7 +175,7 @@ Output conforms to `$HOME/.claude/schemas/criteria-manifest.schema.json`. The fo
175
175
 
176
176
  #### Step 1.8 - Figma visual-fidelity context (when task carries a Figma reference)
177
177
 
178
- **Short-run inputs.** Phases 1 and 2 do not run in a Short run, so three inputs this phase was written around are absent. Substitute them and RECORD the substitution - a step that could not run and a step that passed must not read the same, or the completeness claim cannot be checked:
178
+ **Missing inputs.** A section the evidence did not support is absent from the analysis document, so three inputs this phase was written around can be missing. Substitute them and RECORD the substitution - a step that could not run and one that passed must not read the same, or the completeness claim cannot be checked:
179
179
 
180
180
  | Absent input | Substitute |
181
181
  |---|---|
@@ -426,8 +426,8 @@ Exit 2 (empty ledger) skips silently. The `## Rejected review preferences` secti
426
426
  **Recall telemetry.** Log what was injected, then what triage cited. Zero cited is a legitimate answer; `learning-curve.mjs` trends the ratio:
427
427
 
428
428
  ```bash
429
- bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 memory.injected kind=prior-art rows=$N
430
- bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 memory.hit rows=$CITED_COUNT
429
+ bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 memory.injected kind=prior-art rows=$N
430
+ bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 memory.hit rows=$CITED_COUNT
431
431
  ```
432
432
 
433
433
  **Bulky payloads (opt-in via `prefs.global.contextOffload.enabled`).** Test output and whole-file diffs go through the offload filter, which leaves a `[[ref:<node_id>]]` line plus the tail in context and the full text under `.multi-agent/refs/`. Read that file when the tail is not enough; with the pref off it is a pass-through.
@@ -513,7 +513,7 @@ One `review.reviewer_call` per dispatched reviewer, one `review.triage_call`, on
513
513
  ```bash
514
514
  M=$HOME/.claude/scripts/log-metric.sh
515
515
  emit() { # $1=event $2=model $3=duration $4=tokens_in $5=tokens_out
516
- LOG_METRIC_FORWARD_TO_TRACKER=1 bash "$M" "$TASK_ID" 4 "$1" \
516
+ LOG_METRIC_FORWARD_TO_TRACKER=1 bash "$M" "$TASK_ID" 3 "$1" \
517
517
  model="$2" duration_ms="$3" tokens_in="$4" tokens_out="$5"
518
518
  }
519
519
  emit review.reviewer_call fable "$R1_DURATION" "$R1_IN" "$R1_OUT" # opus on Copilot CLI
@@ -525,7 +525,7 @@ else
525
525
  fi
526
526
  emit review.reviewer_call sonnet "$SONNET_DURATION" "$SONNET_IN" "$SONNET_OUT"
527
527
  emit review.triage_call fable "$TRIAGE_DURATION" "$TRIAGE_IN" "$TRIAGE_OUT"
528
- bash "$M" "$TASK_ID" 4 review.completed raw_count=$RAW accepted=$ACC \
528
+ bash "$M" "$TASK_ID" 3 review.completed raw_count=$RAW accepted=$ACC \
529
529
  deferred=$DEF rejected=$REJ approved=$APPROVED duration_ms=$DURATION
530
530
  ```
531
531
 
@@ -614,7 +614,7 @@ Log: "Phase 3: Review - raw={N1+N2+N3} accepted={Na} deferred={Nd} rejected={N
614
614
  ## Token telemetry - invoke after every LLM call
615
615
 
616
616
  ```bash
617
- bash $HOME/.claude/scripts/phase-tracker.sh tokens 4 <input_count> <output_count> [cached_count]
617
+ bash $HOME/.claude/scripts/phase-tracker.sh tokens 3 <input_count> <output_count> [cached_count]
618
618
  ```
619
619
 
620
620
  The optional 4th `cached_count` is the prompt-cache-read token count when the host reports it (Anthropic `cache_read_input_tokens`); it defaults to 0 and is priced at the cheaper `cacheReadPerMtok` rate in the Phase 5 cost ledger. The tracker accumulates the totals additively, so multiple calls in the same phase compound. The render output then shows live cost on the active phase tile (e.g. `Phase 2 Dev 2m 14s · 12.4k tok`). This satisfies the contract in `$HOME/.claude/multi-agent-refs/tracker-contract.md` and the `smoke-tracker-tokens-invocation.sh` enforcement gate. Skipping this call is the #1 cause of "I can't see how much it cost" complaints.
@@ -623,15 +623,12 @@ Contract and rationale: `progress-contract.md` -> Token telemetry forwarding.
623
623
 
624
624
  ---
625
625
 
626
- ## User test (was Phase 3 until v19.0.0)
626
+ ## User test
627
627
 
628
- Optional test gate, now the tail of Review rather than a phase of its own: it
629
- judges work that already exists, which is what Review does. Needs an interactive
630
- prompt AND a worktree checkout, so it is skipped where either is missing. If
631
- issues are found, the run returns to Phase 1 Dev.
628
+ The tail of Review rather than a phase of its own: it judges work that already
629
+ exists, which is what Review does.
632
630
 
633
-
634
- > **TLDR** - Optional test gate. Offers to boot the simulator/emulator (UI Bug Hunter) or hand off to the user for manual QA. Needs an interactive prompt AND a worktree checkout, so it is in the phase set of `/multi-agent` alone and dropped by every `autopilot` or `--local` entry. Depth does not affect it: a Short run still reaches Phase 3. If issues found, loops back to Phase 2.
631
+ > **TLDR** - Optional test gate. Offers to boot the simulator/emulator (UI Bug Hunter) or hand off to the user for manual QA. Needs an interactive prompt AND a worktree checkout, so it runs on an attended `/multi-agent` whose workspace is a worktree, and is skipped on `autopilot` or when the user chose to work locally. If issues found, loops back to Phase 2.
635
632
 
636
633
  <!-- progress-contract: applied -->
637
634
  Progress emission per `$HOME/.claude/multi-agent-refs/progress-contract.md` - lines for local-test prompt render, user-answer capture, repo checkout (if selected).
@@ -682,7 +679,7 @@ fi
682
679
  **Telemetry**:
683
680
 
684
681
  ```bash
685
- LOG_METRIC_FORWARD_TO_TRACKER=0 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 5 test_gap.scanned \
682
+ LOG_METRIC_FORWARD_TO_TRACKER=0 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 test_gap.scanned \
686
683
  stack=$SCAN_STACK \
687
684
  sources=$(jq '.totals.sourcesScanned' <<< "$GAP_JSON") \
688
685
  gaps=$(jq '.totals.gapCount' <<< "$GAP_JSON")
@@ -702,7 +699,7 @@ Figma evidence (tier=<n>):
702
699
 
703
700
  Tier 1 / Tier 2 records print `screenshotUrl` from the captured evidence (Tier 2 URLs expire after 30 days, re-fetch on the spot if needed). Tier 3 records print the local path to the user-attached screenshot. The block is informational; it never blocks the prompt.
704
701
 
705
- 1. Ask with a native `AskUserQuestion` picker (never a typed y/N prompt). The options MUST make the local-checkout side effect explicit - testing removes the worktree and checks the branch out into the main repo:
702
+ 1. Ask with a native `AskUserQuestion` picker (never a typed y/N prompt), per `$HOME/.claude/multi-agent-refs/picker-contract.md`. The options MUST make the local-checkout side effect explicit - testing removes the worktree and checks the branch out into the main repo:
706
703
  - `question`: "Check out locally to test now?" (rendered in `outputLanguage`)
707
704
  - `header`: "Test" (English, <=12 chars)
708
705
  - `options`:
@@ -733,7 +730,7 @@ Tier 1 / Tier 2 records print `screenshotUrl` from the captured evidence (Tier 2
733
730
  bash $HOME/.claude/scripts/phase-tracker.sh now 3 "awaiting local test (user)"
734
731
  bash $HOME/.claude/scripts/phase-tracker.sh render
735
732
  ```
736
- The waiting state persists in `tracker-state.json` across the handoff; `/multi-agent:resume-local` and `/multi-agent:manual-test` CONTINUE this state file and never re-init it (`$HOME/.claude/multi-agent-refs/tracker-contract.md` "Continuation runs").
733
+ The waiting state persists in `tracker-state.json` across the handoff; `/multi-agent:resume` and `/multi-agent:manual-test` CONTINUE this state file and never re-init it (`$HOME/.claude/multi-agent-refs/tracker-contract.md` "Continuation runs").
737
734
 
738
735
  **"ok" is a structured result, not a word.** Before "ok" is accepted, the run writes `$WORKTREE/.pipeline/manual-test.json`: one entry per acceptance criterion, the criteria taken from the analysis doc test plan (Section 15 / 20), the plan tasks, and the user's own words in the reply. Every criterion records what was seen; a criterion that was not tried says so with a reason.
739
736
  ```json
@@ -806,7 +803,7 @@ Returns severity-tagged findings (Critical / High / Medium). Critical items bloc
806
803
  When the security-auditor or any other Phase 3 sub-agent runs, forward its token totals so Phase 5's Cost Breakdown captures Phase 3:
807
804
 
808
805
  ```bash
809
- LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 5 audit.completed \
806
+ LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 audit.completed \
810
807
  model=opus tokens_in=$IN tokens_out=$OUT duration_ms=$DUR
811
808
  ```
812
809
 
@@ -354,19 +354,19 @@ A task ending with one or more `pushStatus === "skipped"` repos does NOT bump `r
354
354
 
355
355
  ```bash
356
356
  for proj in $(jq -r '.projects[].name' "$STATE_FILE"); do
357
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 6 commit.created repo=$proj sha=$SHA
358
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 6 push.attempted repo=$proj attempts=$N status=$STATUS
359
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 6 pr.opened repo=$proj url=$URL number=$N
357
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 commit.created repo=$proj sha=$SHA
358
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 push.attempted repo=$proj attempts=$N status=$STATUS
359
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 pr.opened repo=$proj url=$URL number=$N
360
360
  done
361
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 6 multi_repo.completed repos=$REPOS skipped=$SKIPPED
361
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 multi_repo.completed repos=$REPOS skipped=$SKIPPED
362
362
  ```
363
363
 
364
364
  **Token forwarding:** the commit-message and PR-body generators run on a model. Forward those calls into the tracker so Phase 5's Cost Breakdown captures Phase 4:
365
365
 
366
366
  ```bash
367
- LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 6 commit.message_generated \
367
+ LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 commit.message_generated \
368
368
  model=<sonnet|opus> tokens_in=$IN tokens_out=$OUT duration_ms=$DUR
369
- LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 6 pr.body_generated \
369
+ LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 pr.body_generated \
370
370
  model=<sonnet|opus> tokens_in=$IN tokens_out=$OUT duration_ms=$DUR
371
371
  ```
372
372
 
@@ -188,9 +188,9 @@ Skipped sections: when `planTodos.enabled` is false or no `plan.todos[]` was emi
188
188
  **Telemetry emission** (mandatory): forward the phase's own LLM spend (humanizer + report compose calls) to the tracker, then emit the final event:
189
189
 
190
190
  ```bash
191
- LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 7 report.compose \
191
+ LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 5 report.compose \
192
192
  model=$REPORT_MODEL tokens_in=$R_IN tokens_out=$R_OUT duration_ms=$R_DUR
193
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 7 task.completed \
193
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 5 task.completed \
194
194
  phases=$PHASE_COUNT review_cycles=$CYCLES lang=$PROMPT_LANG \
195
195
  channels_pr=$PR_STATUS channels_jira=$JIRA_STATUS \
196
196
  channels_confluence=$CONF_STATUS channels_wiki=$WIKI_STATUS \
@@ -15,7 +15,7 @@
15
15
 
16
16
  | Phase | File |
17
17
  | --------------------------------- | -------------------------------------------------------------------- |
18
- | Modes (depth picker, autopilot, --local) | `$HOME/.claude/multi-agent-refs/phases/modes.md` |
18
+ | Modes (autopilot, analysis) | `$HOME/.claude/multi-agent-refs/phases/modes.md` |
19
19
  | Operations (kill, purge, resume) | `$HOME/.claude/multi-agent-refs/phases/operations.md` |
20
20
  | Phase 0: Init | `$HOME/.claude/multi-agent-refs/phases/phase-0-init.md` |
21
21
  | Phase 1: Plan | `$HOME/.claude/multi-agent-refs/phases/phase-1-plan.md` |
@@ -28,11 +28,11 @@
28
28
  ## Pipeline Flow
29
29
 
30
30
  ```
31
- Full: 0-Init -> 1-Plan -> 2-Dev -> 3-Review -> 4-Commit -> 5-Report
32
- Short: 0-Init -> (1 skipped) -> 2-Dev -> 3-Review -> 4-Commit -> 5-Report
33
- --local: Either of the above with no worktree - works directly on a local branch
31
+ 0-Init -> 1-Plan -> 2-Dev -> 3-Review -> 4-Commit -> 5-Report
32
+ Local: The same set with no worktree - Phase 0 Step 5b answered local, so
33
+ work happens directly on a branch in the project root
34
34
 
35
- Full or Short is the Phase 0 Step 7.5 question, not a command name. Autopilot never asks and always runs Full.
35
+ One pipeline: every mode runs its whole phase set, and no answer during the run adds or removes a phase.
36
36
  ```
37
37
 
38
38
  ## Phase entry - pending steer (every phase, every mode)
@@ -99,10 +99,8 @@ done
99
99
 
100
100
  This produces an initial card stack printed by both CLIs.
101
101
 
102
- `/multi-agent` and `/multi-agent:local` are the exception: they do not know their
103
- set here, because depth decides it at Step 7.5. They register `0:Init` alone, then
104
- the rest once the answer lands, and call `tiles --new` for the second batch. Full
105
- contract: `tracker-contract.md`, "Deferred registration".
102
+ No mode is an exception: the phase set is a property of the command, so every
103
+ tile is created in this one batch. Full contract: `tracker-contract.md`.
106
104
 
107
105
  ### Tracker updates (every phase boundary)
108
106
 
@@ -114,7 +112,7 @@ $HOME/.claude/scripts/phase-tracker.sh update <N> in_progress # phase starts
114
112
  $HOME/.claude/scripts/phase-tracker.sh update <N> completed # phase ends OK
115
113
  # or:
116
114
  $HOME/.claude/scripts/phase-tracker.sh update <N> failed # phase failed
117
- $HOME/.claude/scripts/phase-tracker.sh update <N> skipped # e.g. 1 and 2 in a Short run
115
+ $HOME/.claude/scripts/phase-tracker.sh update <N> skipped # e.g. 2 in analysis mode
118
116
  ```
119
117
 
120
118
  After every LLM call (counts are additive; skipping this is why runs end with durations but no cost - nothing reconstructs spend afterwards):
@@ -165,7 +163,7 @@ TaskUpdate({ taskId: <saved>, status: "completed" })
165
163
  bash phase-tracker.sh update <N> completed
166
164
  ```
167
165
 
168
- A phase outside the command's set gets no TaskCreate at all. Depth is different: it is not known at registration time, because the tracker boots at Step -1 and the depth question runs at Step 7.5. So registration splits - Phase 0 alone at Step -1, the rest at Step 7.5 once the answer says which phases the run has. A Short run never draws a Plan tile it will not use. Registering every phase and flipping the skipped one to `skipped` is what this replaced, in v17.5.0: it put a full-width widget on screen beside the question asking whether to run part of it - see the ordering rule below.
166
+ A phase outside the command's set gets no TaskCreate at all, and the set is known before the tracker boots.
169
167
 
170
168
  **(strict) TaskCreate ordering**: All TaskCreate calls MUST fire in strict phase-number order BEFORE any TaskUpdate is applied. The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks (e.g. `1 ✓ · 3 ✓ · 4 ✓ · 0 ▶ · 2 ☐`) even when the underlying state is correct. Pre-marking phases as completed/skipped before Phase 0 starts is FORBIDDEN - register the tile in order with default `pending` status, then flip status via TaskUpdate when the phase actually short-circuits. Full contract in `$HOME/.claude/multi-agent-refs/tracker-contract.md` section "TaskCreate ordering (strict)".
171
169
 
@@ -69,7 +69,7 @@ Emit a progress line **at least** at every one of these moments:
69
69
 
70
70
  ### Phase 2 (Planning)
71
71
  - plan draft start, plan render, user-approval prompt.
72
- - **v5.3.0 Plan Approval Gate (Full + interactive only - a Short run has no plan, autopilot may not ask):**
72
+ - **v5.3.0 Plan Approval Gate (interactive only - autopilot may not ask):**
73
73
  - `clarification-ask` per round - orchestrator writes structured questions when Phase 1 flagged ambiguity (missing acceptance criteria, no Figma/endpoint link, vague language, parent-story scope drift)
74
74
  - `clarification-answer` per round - user reply captured into `state.phases["2"].clarificationAnswers`
75
75
  - `plan-edit-request` per free-text edit - user-supplied revision instruction captured into `state.phases["2"].planEditRequests`
@@ -1,5 +1,7 @@
1
1
  # Readiness Review (shared flow for review-jira + review-issue)
2
2
 
3
+ > **Pickers follow** `$HOME/.claude/multi-agent-refs/picker-contract.md`: never a one-option call, and branch on the option selected, not on its text.
4
+
3
5
  Assess whether a tracker item (Jira issue or GitHub issue) is READY to hand to the multi-agent pipeline, list the concrete gaps, and - after confirmation - post them back as a comment on the item so the reporter can fix it. Read-only on code: no worktree, no branch, no commits, no dev chaining. This is the inverse of `/multi-agent:create-jira` (which authors a well-formed item); here we grade an existing one.
4
6
 
5
7
  Both `/multi-agent:review-jira` and `/multi-agent:review-issue` execute this flow; only the provider (fetch + comment endpoint + picker) differs.
@@ -44,7 +44,7 @@ This is the single source of truth. When a contributor or model is unsure where
44
44
  3. `AskUserQuestion` `question`, `options[].label` and `options[].description` all follow `OUTPUT_LANG`. Only `header` stays English: a <=12-char chip Turkish overflows. Callers branch on which option was picked, never on its literal text, and pass `default` / `ASK_CHOICE_DEFAULT` as a 1-based index. The host's own **Other** row is always English. Caller rules: `picker-contract.md`.
45
45
  4. Always English regardless of either axis: commit messages, PR titles, branch names, code identifiers, agent-state.json values, agent-log.md, reviewer/triage system prompts.
46
46
 
47
- **Failure mode this prevents.** Entering `/multi-agent`, `/multi-agent:autopilot`, `/multi-agent:local`, etc. and switching the assistant's conversational text or picker question copy to English while `outputLanguage="tr"` is set. The user sees a half-English half-Turkish dialogue, flagged as a pipeline bug, not a stylistic choice.
47
+ **Failure mode this prevents.** Entering `/multi-agent`, `/multi-agent:autopilot`, etc. and switching the assistant's conversational text or picker question copy to English while `outputLanguage="tr"` is set. The user sees a half-English half-Turkish dialogue, flagged as a pipeline bug, not a stylistic choice.
48
48
 
49
49
  ## Code & Commit Rules
50
50
 
@@ -90,11 +90,11 @@ Phases by mode:
90
90
  | Mode | Phases |
91
91
  |---|---|
92
92
  | `/multi-agent` | 0,1,2,3,4,5 |
93
- | `/multi-agent:local` | 0,1,2,3,4,5 (the user test needs a worktree checkout and local has none, so that STEP inside Review is skipped - the phase is not) |
94
- | `/multi-agent:autopilot`, `/multi-agent:local-autopilot` | 0,1,2,3,4,5 (always Full; autopilot drops the interactive user test inside Review) |
93
+ | answered local at Step 5b | 0,1,2,3,4,5 (the user test needs a worktree checkout and local has none, so that STEP inside Review is skipped - the phase is not) |
94
+ | `/multi-agent:autopilot` | 0,1,2,3,4,5 (autopilot drops the interactive user test inside Review) |
95
95
  | `/multi-agent:analysis` | 0,1,3,4,5 (no code is written, so Dev is not in the set) |
96
96
 
97
- The two picker entries (`/multi-agent`, `:local`) register in two batches, because at Step -1 they do not yet know which set is theirs: see "Deferred registration" below. Every other mode registers its whole set at Step -1.
97
+ Every entry registers its whole set in one batch: the set is a property of the command, known before the tracker boots.
98
98
 
99
99
  Register each phase:
100
100
 
@@ -171,7 +171,7 @@ report's cost-unavailable list.
171
171
 
172
172
  > **All TaskCreate calls for the active mode's phase set MUST fire in strict phase-number order BEFORE any TaskUpdate is applied. No "pre-mark skipped phases as completed before Phase 0" reasoning is permitted - even when the agent knows in advance that a phase will be skipped.**
173
173
 
174
- Why: an agent reasoning "the user picked Short, so phases 1/2 will be skipped - let me TaskCreate them as completed first" produces tile IDs `1, 2, 3, ...` for phases 1/2/4, then the Phase 0 tile gets ID `4` and visually drops below them. The user sees `1, 2, 4 ✓ · 0 ▶ · 3 ☐ · ...` instead of `0 ▶ · 1 ✓ · 2 ✓ · 3 ☐ · 4 ✓ · ...`.
174
+ Why: an agent reasoning "analysis mode skips Dev - let me TaskCreate it as completed first" produces tile IDs `1, 2, 3, ...` for phases 1/2/4, then the Phase 0 tile gets ID `4` and visually drops below them. The user sees `1, 2, 4 ✓ · 0 ▶ · 3 ☐ · ...` instead of `0 ▶ · 1 ✓ · 2 ✓ · 3 ☐ · 4 ✓ · ...`.
175
175
 
176
176
  The correct sequence is **always**:
177
177
 
@@ -197,44 +197,13 @@ Mode-specific phase sets:
197
197
 
198
198
  | Mode | TaskCreate set (in order) |
199
199
  |---|---|
200
- | `/multi-agent` | 0 at Step -1; then at Step 7.5 either 1 → 2 → 3 → 4 → 5 (Full) or 2 → 3 → 4 → 5 (Short) |
201
- | `:local` | same two batches; nothing is dropped, because the user test that needs a worktree checkout is a step inside Review rather than a phase of its own |
202
- | `:autopilot`, `:local-autopilot` | 0 → 1 → 2 → 3 → 4 → 5 (6 phases - always Full; the user test is inside Review now, so no phase is dropped) |
200
+ | `/multi-agent` | 0 → 1 → 2 → 3 → 4 → 5 (6 phases) |
201
+ | answered local | the same set; nothing is dropped, because the user test that needs a worktree checkout is a step inside Review rather than a phase of its own |
202
+ | `:autopilot` | 0 → 1 → 2 → 3 → 4 → 5 (6 phases - the user test is inside Review now, so no phase is dropped) |
203
203
  | `:analysis` | 0 → 1 → 3 → 4 → 5 (5 phases - no code is written, so Dev is not in the set) |
204
204
 
205
205
  A phase outside the mode's set gets no TaskCreate at all; the `[SKIPPED]` pattern applies only to a phase that IS in the set and short-circuits at runtime. Phase 3 is in every mode's set as of v14.0.0. The authoritative per-mode set is the `for p in ...` init block in each mode's own entry doc, generated by `gen-mode-dispatch.mjs`; this table mirrors those blocks.
206
206
 
207
- #### Deferred registration - the depth picker
208
-
209
- `/multi-agent` and `:local` cannot know their phase set at Step -1. Depth decides it, and the depth picker cannot run before Step 7.5: its recommendation needs `taskType`, which needs the fetched issue and the branch.
210
-
211
- Until v17.5.0 they registered all eight anyway and flipped 1 and 2 to `skipped` at 7.5. That put a widget reading "8 tasks, 7 open - Phase 1 Plan, Phase 1 Plan, ..." on screen *beside* the question asking whether to run Analysis and Planning at all, and a Short answer then contradicted a list the user had just been shown. The widget was asserting a shape the run had not chosen.
212
-
213
- So registration splits at the moment the shape is known:
214
-
215
- ```text
216
- # Step -1, first thing in the run: Phase 0 only. It is the one phase that is
217
- # certain, and the run is never silent while Phase 0 does its work.
218
- bash $HOME/.claude/scripts/phase-tracker.sh add 0 Init
219
- bash $HOME/.claude/scripts/phase-tracker.sh tiles # -> TaskCreate(Phase 0)
220
- bash $HOME/.claude/scripts/phase-tracker.sh update 0 in_progress
221
-
222
- # Step 7.5, immediately after the depth answer:
223
- # Full -> 1 2 3 4 5 Short -> 2 3 4 5
224
- # :local drops nothing - the user test lives inside Review now, so there is
225
- # no separate phase for it to skip
226
- for p in "1:Plan" "2:Dev" "3:Review" "4:Commit" "5:Report"; do
227
- bash $HOME/.claude/scripts/phase-tracker.sh add "${p%%:*}" "${p#*:}"
228
- done
229
- bash $HOME/.claude/scripts/phase-tracker.sh tiles --new # -> TaskCreate for the new tiles only
230
- ```
231
-
232
- `tiles --new` emits `TaskCreate` only for phases that carry no `tasklist_id` yet, so the Phase 0 tile is not created twice. It is the same ordering rule, applied per batch: every tile in a batch is created in ascending phase order, and a deferred batch only ever appends phases numbered above everything already registered. Nothing is pre-marked, and a phase the run will not execute never gets a tile at all.
233
-
234
- A phase that IS registered and short-circuits later still flips with `[SKIPPED]` - autopilot suppressing the Phase 3 user test, for instance. That is a runtime outcome, not an unknown set.
235
-
236
- **Enforcement**: `smoke-tasklist-ordering.sh` scans the dispatcher (`commands/multi-agent/SKILL.md`) and every mode entry point doc (`commands/multi-agent/{autopilot,local,local-autopilot,analysis,resume-local}/SKILL.md` + the Copilot full-inline orchestrator mirror) for the explicit "in phase-number order" rule. Inventory drift fails the smoke.
237
-
238
207
  ### Other CLIs - call render after every state change
239
208
 
240
209
  There is no TaskList outside Claude Code. Instead, after each state change the agent calls:
@@ -299,7 +268,7 @@ Throttling rules: mirror only canonical-set lines (`verbose`-tier internals are
299
268
 
300
269
  ### Delegated phases - mirror limitation + chunked dispatch (required)
301
270
 
302
- When a phase's work is delegated to a subagent (Phase 2 Dev on Opus in a Short run, `create-component` plugin dispatch, Phase 1 explorers, Phase 3 reviewers), the visual channel freezes for the duration of the Agent call: the orchestrator is blocked while the call is in flight, so it cannot fire `TaskUpdate` / `now` / `tokens`, and a subagent cannot drive the parent session's TaskList (its own TaskCreate/TaskUpdate calls land on an invisible child list). The progress-line mirror above can therefore only fire while the orchestrator holds control. Rules:
271
+ When a phase's work is delegated to a subagent (`create-component` plugin dispatch, Phase 1 explorers, Phase 3 reviewers), the visual channel freezes for the duration of the Agent call: the orchestrator is blocked while the call is in flight, so it cannot fire `TaskUpdate` / `now` / `tokens`, and a subagent cannot drive the parent session's TaskList (its own TaskCreate/TaskUpdate calls land on an invisible child list). The progress-line mirror above can therefore only fire while the orchestrator holds control. Rules:
303
272
 
304
273
  1. **Pre-dispatch marker.** Immediately before every Agent call, set the active-phase line to the delegation itself, so the frozen interval at least states what is running and on which model:
305
274
  - Claude Code: `TaskUpdate({activeForm: "Dev subagent (opus): <task subject>"})`
@@ -434,7 +403,7 @@ The `tasklist_id` meta from the previous session is replaced with the new IDs du
434
403
 
435
404
  ## Continuation runs (finish / manual-test)
436
405
 
437
- A pre-existing `tracker-state.json` for the task is never re-initialized. Rules for any command that continues an earlier run (`/multi-agent:resume-local`, `/multi-agent:manual-test`, resume):
406
+ A pre-existing `tracker-state.json` for the task is never re-initialized. Rules for any command that continues an earlier run (`/multi-agent:resume`, `/multi-agent:manual-test`):
438
407
 
439
408
  1. `init` runs ONLY when no state file exists for the task. Otherwise the existing file is kept - phase history (elapsed, tokens, model, meta) survives.
440
409
  2. The continuing command re-declares its phase set with `add` - `add` is idempotent, so existing phases keep their name, status, and token history; only genuinely new phases are appended. The card renders phases sorted by numeric id, so mixed sets stay in order.
@@ -11,6 +11,8 @@
11
11
  - [Cross-CLI parity](#cross-cli-parity)
12
12
  <!-- /toc -->
13
13
 
14
+ > **Pickers follow** `$HOME/.claude/multi-agent-refs/picker-contract.md`: never a one-option call, and branch on the option selected, not on its text.
15
+
14
16
  > **TLDR** - Component tasks can auto-generate wiki docs + Figma screenshots. The Wiki adapter is invoked from `/multi-agent:channels` (Phase 5 delegates, or user invokes post-hoc). Four layouts supported (`submodule`, `in-repo`, `github-wiki`, `separate-repo`) - adapter picked from `figmaConfig.wiki.mode`. Non-blocking: failures log a warning and channels continues to other adapters. The Wiki adapter supports scope multi-select (Case A) and a precondition-failure menu (Case B) - see below.
15
17
 
16
18
  This doc is referenced from `commands/multi-agent/channels/SKILL.md` (Wiki adapter) and indirectly from `$HOME/.claude/multi-agent-refs/phases/phase-5-report.md` (which delegates all external delivery to channels). Keeping it separate keeps both files under their token budgets and gives the contract a stable location for Claude-side + Copilot-side implementations.
@@ -75,7 +77,7 @@ Autopilot in Phase 5 pauses at the channels menu (per modes.md contract) - if
75
77
 
76
78
  ## Legacy prompt + preference flow (pre-v5.7, still supported for backward compat)
77
79
 
78
- Interactive path (any interactive run, Full or Short), ONLY when the schema lacks `wikiScope` - ask with a native `AskUserQuestion` picker (never a typed y/n):
80
+ Interactive path (any interactive run), ONLY when the schema lacks `wikiScope` - ask with a native `AskUserQuestion` picker (never a typed y/n):
79
81
 
80
82
  - `question`: "Generate component wiki docs?" (rendered in `outputLanguage`)
81
83
  - `header`: "Wiki" (English, <=12 chars)
@@ -103,7 +105,6 @@ Explicit logs help the developer understand why wiki did or did not run:
103
105
  - `figmaConfig` missing entirely → log `Phase 5: wiki skipped (no figma-config for this project)`.
104
106
  - User declined at prompt → log `Phase 5: wiki skipped by user`.
105
107
  - Autopilot with `wikiDefault=false` → log `Phase 5: wiki skipped (autopilot + wikiDefault=false)`.
106
- - Short run: DO prompt - wiki is cheap and keeps docs fresh on the fast path; skip only if the user says no.
107
108
 
108
109
  ## Success log
109
110
 
@@ -1,5 +1,5 @@
1
1
  {
2
- "schemaVersion": "2.7.0",
2
+ "schemaVersion": "2.8.0",
3
3
  "global": {
4
4
  "identities": [],
5
5
  "keychainMapping": {
@@ -101,7 +101,7 @@
101
101
  "skillConformance": {
102
102
  "blockOnCoverageGap": false
103
103
  },
104
- "resumeLocal": {
104
+ "resume": {
105
105
  "autoFix": false
106
106
  }
107
107
  },
@@ -75,7 +75,7 @@ The 3-tier fallback chain above governs Figma access in `/multi-agent:analysis`
75
75
  | Phase | Sub-command examples | Figma MCP allowed | Figma REST allowed | Reason |
76
76
  |---|---|---|---|---|
77
77
  | Analysis Phase 1 | `/multi-agent:analysis` Phase 1 fetch | yes | yes (Tier 2 fallback) | Single source of design ground truth |
78
- | Plan (Phase 1) | `/multi-agent`, `/multi-agent:local`, `/multi-agent:autopilot`, `/multi-agent:local-autopilot` | no | no | Plan reads analysis doc Section 14 + Section 6 |
78
+ | Plan (Phase 1) | `/multi-agent`, `/multi-agent:autopilot` | no | no | Plan reads analysis doc Section 14 + Section 6 |
79
79
  | Dev (Phase 2) | every mode that runs Phase 2 (8 modes total) | no | no | Reads analysis doc + Code Connect mapping |
80
80
  | Review (Phase 3) | `/multi-agent:review`, every full-pipeline mode | no | no | Reviewer cites analysis doc Section 21 References |
81
81
  | Test (inside Phase 3) | `/multi-agent:test`, `/multi-agent:manual-test`, every full mode | no | no | Variant list comes from analysis Section 13.6 + 15.2 |
@@ -66,7 +66,7 @@
66
66
  },
67
67
  "worktreePath": {
68
68
  "type": ["string", "null"],
69
- "description": "Absolute path to the task worktree. Null in --local mode."
69
+ "description": "Absolute path to the task worktree. Null when the Step 5b picker answered local."
70
70
  },
71
71
  "branch": {
72
72
  "type": "string",
@@ -274,7 +274,7 @@
274
274
  "workspaceSource": {
275
275
  "type": "string",
276
276
  "enum": ["asked", "command", "autopilot"],
277
- "description": "Who decided where the branch lives. asked = the user answered the Step 5b workspace picker; command = :local / --local / :local-autopilot stated it up front, or a flow that only ever builds worktrees; autopilot = resolved to a worktree without asking, because an unattended commit in the user's own checkout is what worktrees prevent. localMode alone cannot say: false is both a chosen worktree and one nothing asked about."
277
+ "description": "Who decided where the branch lives. asked = the user answered the Step 5b workspace picker; command = a flow that only ever builds worktrees stated it up front; autopilot = resolved to a worktree without asking, because an unattended commit in the user's own checkout is what worktrees prevent. localMode alone cannot say: false is both a chosen worktree and one nothing asked about."
278
278
  },
279
279
  "remoteType": {
280
280
  "type": "string",
@@ -476,15 +476,10 @@
476
476
  "type": "boolean",
477
477
  "default": false
478
478
  },
479
- "onlyDevelop": {
480
- "type": "boolean",
481
- "default": false,
482
- "description": "Short pipeline - phases 1 and 2 are skipped. Set by the Phase 0 Step 7.5 depth picker as of v16.0.0; the key and every reader of it are unchanged. Phase 3 Review still runs (v14.0.0+). Always false in an autopilot run, which never asks the depth question."
483
- },
484
479
  "localMode": {
485
480
  "type": "boolean",
486
481
  "default": false,
487
- "description": "--local flag - no worktree, direct branch in projectRoot."
482
+ "description": "Step 5b answered local - no worktree, direct branch in projectRoot."
488
483
  },
489
484
  "instructionDriven": {
490
485
  "type": "boolean",
@@ -498,7 +493,7 @@
498
493
  "analysis": {
499
494
  "type": ["object", "null"],
500
495
  "additionalProperties": true,
501
- "description": "Phase 1 analysis-document outcome. Absent in Short runs, which produce no document by design.",
496
+ "description": "Phase 1 analysis-document outcome. Phase 1 runs in every mode, so a document is always produced; what varies is how many of its sections the evidence supported.",
502
497
  "properties": {
503
498
  "docStatus": {
504
499
  "type": "string",
@@ -1530,7 +1525,7 @@
1530
1525
  }
1531
1526
  }
1532
1527
  },
1533
- "description": "Captured in Phase 2 after the build+test gate, because the user test is dropped by every autopilot and --local entry."
1528
+ "description": "Captured in Phase 2 after the build+test gate, because the user test is dropped by every autopilot entry and by any run whose workspace is local."
1534
1529
  },
1535
1530
  "host": {
1536
1531
  "type": ["string", "null"],
@@ -0,0 +1,33 @@
1
+ /**
2
+ * prefs-2.7.0-to-2.8.0.mjs - preferences migration
3
+ *
4
+ * v20.0.0 folds `/multi-agent:resume-local` into `/multi-agent:resume`. One
5
+ * command now asks which unfinished work to pick up: a run that stopped
6
+ * mid-phase, or a branch carrying work no run ever produced. The preference
7
+ * that decides whether triage-accepted findings are auto-fixed applies to the
8
+ * second path, so it moves with it: `global.resumeLocal` -> `global.resume`.
9
+ *
10
+ * The value is carried, never reset. Someone who turned auto-fix on for the
11
+ * tail meant it for exactly the work this path still handles.
12
+ *
13
+ * @param {object} data - parsed multi-agent-preferences.json
14
+ * @returns {object} - migrated data with schemaVersion "2.8.0"
15
+ */
16
+ export default function migrate(data) {
17
+ const out = JSON.parse(JSON.stringify(data));
18
+
19
+ out.schemaVersion = "2.8.0";
20
+
21
+ if (!out.global || typeof out.global !== "object") return out;
22
+
23
+ const legacy = out.global.resumeLocal;
24
+ if (!out.global.resume || typeof out.global.resume !== "object") {
25
+ out.global.resume = {};
26
+ }
27
+ if (typeof out.global.resume.autoFix !== "boolean") {
28
+ out.global.resume.autoFix = typeof legacy?.autoFix === "boolean" ? legacy.autoFix : false;
29
+ }
30
+ delete out.global.resumeLocal;
31
+
32
+ return out;
33
+ }
@@ -6,7 +6,6 @@
6
6
  "phaseSchema": 2,
7
7
  "phaseSchemaNote": "The phase vocabulary generation. metrics.jsonl is append-only and v19.0.0 renumbered the phases, so every line written from v19.0.0 on carries this number and a line without the field is generation 1. Aggregators pick their name table from it; without it phase 3 means Dev in old rows and Review in new ones and no reader can tell them apart.",
8
8
  "legacyPhaseNames": {
9
- "note": "Generation 1 phase names, kept so an aggregator can label a historical metrics.jsonl row correctly instead of printing the current name for a number that meant something else. Read-only history: nothing emits these any more. The generation 1 -> 2 number map is not repeated here, it is the `was` array on each phase above.",
10
9
  "0": "Init",
11
10
  "1": "Analysis",
12
11
  "2": "Planning",
@@ -14,7 +13,8 @@
14
13
  "4": "Review",
15
14
  "5": "Test",
16
15
  "6": "Commit",
17
- "7": "Report"
16
+ "7": "Report",
17
+ "note": "Generation 1 phase names, kept so an aggregator can label a historical metrics.jsonl row correctly instead of printing the current name for a number that meant something else. Read-only history: nothing emits these any more. The generation 1 -> 2 number map is not repeated here, it is the `was` array on each phase above."
18
18
  },
19
19
  "phases": [
20
20
  {
@@ -64,35 +64,14 @@
64
64
  "modes": {
65
65
  "full": {
66
66
  "phases": [0, 1, 2, 3, 4, 5],
67
- "local": false,
68
- "autopilot": false,
69
- "depth": true
67
+ "autopilot": false
70
68
  },
71
69
  "autopilot": {
72
70
  "phases": [0, 1, 2, 3, 4, 5],
73
- "local": false,
74
- "autopilot": true
75
- },
76
- "local": {
77
- "phases": [0, 1, 2, 3, 4, 5],
78
- "local": true,
79
- "autopilot": false,
80
- "depth": true
81
- },
82
- "local-autopilot": {
83
- "phases": [0, 1, 2, 3, 4, 5],
84
- "local": true,
85
71
  "autopilot": true
86
72
  },
87
- "full-local": {
88
- "phases": [0, 1, 2, 3, 4, 5],
89
- "local": true,
90
- "autopilot": false,
91
- "depth": true
92
- },
93
73
  "analysis": {
94
74
  "phases": [0, 1, 3, 4, 5],
95
- "local": false,
96
75
  "autopilot": false
97
76
  }
98
77
  },
@@ -14,8 +14,8 @@
14
14
  "properties": {
15
15
  "schemaVersion": {
16
16
  "type": "string",
17
- "enum": ["2.0.0", "2.1.0", "2.2.0", "2.3.0", "2.4.0", "2.5.0", "2.6.0", "2.7.0"],
18
- "description": "v2.0.0: pre-v3.7. v2.1.0: v3.7+ adds identities[].servicePatMap, platformIdentityRouting, recentGroups, recentBranches, serviceStatus, settings, expanded keychainMapping. v2.2.0: v6.0.0 formalizes v5.7 / v5.8 additions (reportChannels, reportContent with technicalAnalysis, wikiScope, autopilotReportTimeoutSeconds) that had been running as 2.1.0 sub-migrations without a proper version bump. v2.5.0: v14.0.0 adds global.skillConformance (Phase 4 criteria resolution) and declares global.ship.autoFix, which the tail command's spec had referenced as global.finish.autoFix without ever declaring it. v2.6.0: v15.0.0 renames global.ship to global.resumeLocal (/multi-agent:ship -> :resume-local). v2.7.0: v19.0.0 drops the analysis Lite mode (global.analysisPhase.mode keeps a single value 'full') and adds global.modelRouting, which ships disabled."
17
+ "enum": ["2.0.0", "2.1.0", "2.2.0", "2.3.0", "2.4.0", "2.5.0", "2.6.0", "2.7.0", "2.8.0"],
18
+ "description": "v2.0.0: pre-v3.7. v2.1.0: v3.7+ adds identities[].servicePatMap, platformIdentityRouting, recentGroups, recentBranches, serviceStatus, settings, expanded keychainMapping. v2.2.0: v6.0.0 formalizes v5.7 / v5.8 additions (reportChannels, reportContent with technicalAnalysis, wikiScope, autopilotReportTimeoutSeconds) that had been running as 2.1.0 sub-migrations without a proper version bump. v2.5.0: v14.0.0 adds global.skillConformance (Phase 4 criteria resolution) and declares global.ship.autoFix, which the tail command's spec had referenced as global.finish.autoFix without ever declaring it. v2.6.0: v15.0.0 renames global.ship to global.resumeLocal (/multi-agent:ship -> :resume-local). v2.7.0: v19.0.0 drops the analysis Lite mode (global.analysisPhase.mode keeps a single value 'full') and adds global.modelRouting, which ships disabled. v2.8.0: v20.0.0 folds :resume-local into :resume and renames global.resumeLocal to global.resume."
19
19
  },
20
20
  "global": {
21
21
  "type": "object",
@@ -1923,15 +1923,15 @@
1923
1923
  }
1924
1924
  }
1925
1925
  },
1926
- "resumeLocal": {
1926
+ "resume": {
1927
1927
  "type": "object",
1928
1928
  "additionalProperties": false,
1929
- "description": "/multi-agent:resume-local (formerly :ship, :finish) - the tail that runs review + build/test + PR + report over work already on the branch.",
1929
+ "description": "/multi-agent:resume - picking up unfinished work, either a run that stopped mid-phase or a branch carrying work no run produced. These keys apply to the second path, which runs review + build/test + PR + report over the branch diff.",
1930
1930
  "properties": {
1931
1931
  "autoFix": {
1932
1932
  "type": "boolean",
1933
1933
  "default": false,
1934
- "description": "When true, resume-local auto-fixes triage-accepted blocking/important findings and re-reviews instead of asking. Equivalent to passing `autopilot` on every run. Default false: resume-local operates on work the user wrote by hand, so silently rewriting it is the surprising option."
1934
+ "description": "When true, the tail path auto-fixes triage-accepted blocking/important findings and re-reviews instead of asking. Equivalent to passing `autopilot` on every run. Default false: this path operates on work the user wrote by hand, so silently rewriting it is the surprising option."
1935
1935
  }
1936
1936
  }
1937
1937
  },
@@ -16,7 +16,7 @@
16
16
  "cacheReadPerMtok": 0.5,
17
17
  "modelId": "claude-opus-5",
18
18
  "provider": "anthropic",
19
- "note": "Second tier - dev phase on a Short run, Reviewer 1 and triage on Copilot CLI, and the opus rung of the fable -> opus -> sonnet fallback ladder. Same rate as the Opus 4.8 it replaces, so the ledger needed no reprice on the generation move. Claude Opus 5 draws on a rate-limit pool SEPARATE from the combined Opus 4.x pool - moving traffic here neither frees headroom on the old bucket nor inherits it."
19
+ "note": "Second tier - Reviewer 1 and triage on Copilot CLI, and the opus rung of the fable -> opus -> sonnet fallback ladder. Same rate as the Opus 4.8 it replaces, so the ledger needed no reprice on the generation move. Claude Opus 5 draws on a rate-limit pool SEPARATE from the combined Opus 4.x pool - moving traffic here neither frees headroom on the old bucket nor inherits it."
20
20
  },
21
21
  "sonnet": {
22
22
  "inPerMtok": 3.0,