@mmerterden/multi-agent-pipeline 19.1.4 → 20.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. package/CHANGELOG.md +123 -0
  2. package/README.md +19 -36
  3. package/README.tr.md +18 -35
  4. package/SECURITY.md +3 -3
  5. package/docs/adr/0002-instruction-driven-flag.md +6 -5
  6. package/docs/adr/0005-lazy-phase-docs.md +2 -2
  7. package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
  8. package/docs/adr/0009-claude-stack-skills-plugin-only.md +1 -1
  9. package/docs/adr/0010-own-code-graph.md +5 -4
  10. package/docs/adr/0011-dormant-ci.md +10 -1
  11. package/docs/adr/0012-macos-only.md +2 -2
  12. package/docs/adr/0013-lsp-code-intelligence.md +2 -2
  13. package/docs/adr/0014-six-phase-consolidation.md +9 -9
  14. package/docs/adr/0015-one-pipeline-no-depth-answer.md +83 -0
  15. package/docs/adr/0016-the-run-shape-is-asked-not-typed.md +69 -0
  16. package/docs/adr/README.md +18 -16
  17. package/docs/architecture.md +2 -2
  18. package/docs/ecosystem.md +5 -5
  19. package/docs/facts.json +7 -9
  20. package/docs/features.md +4 -5
  21. package/docs/token-budget-history.md +1 -1
  22. package/install/_codex-agents.mjs +1 -1
  23. package/install/_common.mjs +9 -1
  24. package/install/templates/copilot-instructions.md +7 -16
  25. package/manifest.json +133 -129
  26. package/package.json +1 -1
  27. package/pipeline/agents/code-reviewer.md +2 -2
  28. package/pipeline/agents/dev-critic.md +5 -5
  29. package/pipeline/agents/security-auditor.md +80 -72
  30. package/pipeline/commands/figma-to-swiftui.md +1 -1
  31. package/pipeline/commands/multi-agent/SKILL.md +7 -9
  32. package/pipeline/commands/multi-agent/analysis/SKILL.md +2 -0
  33. package/pipeline/commands/multi-agent/analysis-jira/SKILL.md +2 -0
  34. package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +2 -0
  35. package/pipeline/commands/multi-agent/autopilot/SKILL.md +2 -0
  36. package/pipeline/commands/multi-agent/autopilot-on/SKILL.md +2 -0
  37. package/pipeline/commands/multi-agent/autopilot-status/SKILL.md +1 -1
  38. package/pipeline/commands/multi-agent/build-optimize/SKILL.md +2 -0
  39. package/pipeline/commands/multi-agent/channels/SKILL.md +2 -2
  40. package/pipeline/commands/multi-agent/create-jira/SKILL.md +2 -0
  41. package/pipeline/commands/multi-agent/design-check/SKILL.md +1 -1
  42. package/pipeline/commands/multi-agent/diff-explain/SKILL.md +1 -1
  43. package/pipeline/commands/multi-agent/forget/SKILL.md +2 -0
  44. package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +4 -2
  45. package/pipeline/commands/multi-agent/help/SKILL.md +23 -27
  46. package/pipeline/commands/multi-agent/ios-coding-standard/SKILL.md +5 -4
  47. package/pipeline/commands/multi-agent/issue/SKILL.md +2 -0
  48. package/pipeline/commands/multi-agent/jira/SKILL.md +2 -0
  49. package/pipeline/commands/multi-agent/language/SKILL.md +2 -0
  50. package/pipeline/commands/multi-agent/prune-logs/SKILL.md +2 -0
  51. package/pipeline/commands/multi-agent/purge/SKILL.md +2 -0
  52. package/pipeline/commands/multi-agent/resume/SKILL.md +177 -48
  53. package/pipeline/commands/multi-agent/save/SKILL.md +2 -0
  54. package/pipeline/commands/multi-agent/scan/SKILL.md +2 -2
  55. package/pipeline/commands/multi-agent/security-review/SKILL.md +52 -0
  56. package/pipeline/commands/multi-agent/stack/SKILL.md +2 -0
  57. package/pipeline/commands/multi-agent/sync/SKILL.md +7 -8
  58. package/pipeline/commands/multi-agent/test-screenshots/SKILL.md +2 -0
  59. package/pipeline/commands/multi-agent/uninstall/SKILL.md +2 -0
  60. package/pipeline/lib/repo-hygiene.sh +1 -1
  61. package/pipeline/multi-agent-refs/analysis/render.md +1 -1
  62. package/pipeline/multi-agent-refs/analysis/resolve.md +1 -1
  63. package/pipeline/multi-agent-refs/analysis/synthesis.md +1 -1
  64. package/pipeline/multi-agent-refs/analysis-template.md +1 -1
  65. package/pipeline/multi-agent-refs/component-dispatch.md +5 -13
  66. package/pipeline/multi-agent-refs/cross-cli-contract.md +14 -15
  67. package/pipeline/multi-agent-refs/features/external-context-injection.md +2 -0
  68. package/pipeline/multi-agent-refs/features/review-delta.md +1 -1
  69. package/pipeline/multi-agent-refs/features/review-multi-repo.md +3 -3
  70. package/pipeline/multi-agent-refs/features/security-audit.md +55 -0
  71. package/pipeline/multi-agent-refs/features/skill-conformance.md +1 -1
  72. package/pipeline/multi-agent-refs/features/visual-evidence.md +2 -1
  73. package/pipeline/multi-agent-refs/features/worktree-finalize.md +1 -1
  74. package/pipeline/multi-agent-refs/generate-issue.md +2 -0
  75. package/pipeline/multi-agent-refs/issue-jira-triad.md +2 -0
  76. package/pipeline/multi-agent-refs/keychain.md +2 -0
  77. package/pipeline/multi-agent-refs/knowledge.md +0 -7
  78. package/pipeline/multi-agent-refs/outside-the-pipeline.md +1 -1
  79. package/pipeline/multi-agent-refs/payload-contracts.md +1 -1
  80. package/pipeline/multi-agent-refs/phases/modes.md +33 -109
  81. package/pipeline/multi-agent-refs/phases/operations.md +2 -0
  82. package/pipeline/multi-agent-refs/phases/phase-0-init.md +23 -42
  83. package/pipeline/multi-agent-refs/phases/phase-1-plan.md +7 -18
  84. package/pipeline/multi-agent-refs/phases/phase-2-dev.md +13 -44
  85. package/pipeline/multi-agent-refs/phases/phase-3-review.md +28 -37
  86. package/pipeline/multi-agent-refs/phases/phase-4-commit.md +6 -6
  87. package/pipeline/multi-agent-refs/phases/phase-5-report.md +3 -3
  88. package/pipeline/multi-agent-refs/phases.md +9 -11
  89. package/pipeline/multi-agent-refs/progress-contract.md +1 -1
  90. package/pipeline/multi-agent-refs/readiness-review.md +2 -0
  91. package/pipeline/multi-agent-refs/rules.md +1 -1
  92. package/pipeline/multi-agent-refs/threat-model.md +39 -0
  93. package/pipeline/multi-agent-refs/tracker-contract.md +9 -40
  94. package/pipeline/multi-agent-refs/wiki-capture.md +3 -2
  95. package/pipeline/preferences-template.json +2 -2
  96. package/pipeline/rules/figma-pipeline.md +1 -1
  97. package/pipeline/schemas/agent-state.schema.json +28 -10
  98. package/pipeline/schemas/migrations/prefs-2.7.0-to-2.8.0.mjs +33 -0
  99. package/pipeline/schemas/phases.json +4 -26
  100. package/pipeline/schemas/prefs.schema.json +5 -9
  101. package/pipeline/schemas/reviewer-output.schema.json +99 -2
  102. package/pipeline/schemas/security-finding.schema.json +144 -0
  103. package/pipeline/scripts/_stack-routing.mjs +1 -0
  104. package/pipeline/scripts/cost-table.json +1 -1
  105. package/pipeline/scripts/gc-abandoned.sh +16 -9
  106. package/pipeline/scripts/gc-refs.sh +1 -1
  107. package/pipeline/scripts/gen-mode-dispatch.mjs +11 -41
  108. package/pipeline/scripts/migrate-prefs.mjs +18 -17
  109. package/pipeline/scripts/phase-tracker.sh +2 -2
  110. package/pipeline/scripts/phase0-exit-gate.mjs +1 -1
  111. package/pipeline/scripts/plan-coverage-gate.mjs +3 -3
  112. package/pipeline/scripts/render-work-summary.sh +7 -4
  113. package/pipeline/scripts/run-aggregator.mjs +1 -1
  114. package/pipeline/scripts/usage-report.mjs +0 -2
  115. package/pipeline/scripts/worktree-finalize.sh +2 -2
  116. package/pipeline/skills/.skill-manifest.json +17 -21
  117. package/pipeline/skills/.skills-index.json +6 -39
  118. package/pipeline/skills/shared/README.md +5 -8
  119. package/pipeline/skills/shared/core/multi-agent/SKILL.md +11 -15
  120. package/pipeline/skills/shared/core/multi-agent-autopilot-status/SKILL.md +1 -1
  121. package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +13 -16
  122. package/pipeline/skills/shared/core/multi-agent-ios-coding-standard/SKILL.md +2 -3
  123. package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +51 -15
  124. package/pipeline/skills/shared/core/multi-agent-scan/SKILL.md +2 -2
  125. package/pipeline/skills/shared/core/multi-agent-security-review/SKILL.md +29 -0
  126. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +7 -7
  127. package/pipeline/skills/shared/external/security-review/SKILL.md +64 -0
  128. package/pipeline/skills/shared/external/security-review/references/owasp-mobile-top10-2024.md +53 -0
  129. package/pipeline/skills/shared/external/security-review/references/owasp-web-api-top10-2021.md +56 -0
  130. package/pipeline/skills/skills-index.md +3 -6
  131. package/pipeline/commands/multi-agent/local/SKILL.md +0 -132
  132. package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +0 -142
  133. package/pipeline/commands/multi-agent/resume-local/SKILL.md +0 -114
  134. package/pipeline/commands/security-review.md +0 -6
  135. package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +0 -41
  136. package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +0 -55
  137. package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +0 -51
@@ -104,13 +104,13 @@ Persist the totals as `state.diffRisk` (Phase 4 `risk` section, Phase 5, `run-me
104
104
  **Gate behavior**: this step is **never blocking**. If risk scoring fails (git error, parse error, validator rejection), continue with no priority hint - reviewers receive the full diff in their default order. Failures are logged via metrics:
105
105
 
106
106
  ```bash
107
- [ -z "$RISK_JSON" ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 review.diff_risk_skipped reason=$REASON
107
+ [ -z "$RISK_JSON" ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.diff_risk_skipped reason=$REASON
108
108
  ```
109
109
 
110
110
  On success, emit a single summary metric:
111
111
 
112
112
  ```bash
113
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 review.diff_risk \
113
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.diff_risk \
114
114
  top_files=$(jq '.files | length' <<< "$RISK_JSON") \
115
115
  max_score=$(jq '.totals.max_score' <<< "$RISK_JSON") \
116
116
  loc_added=$(jq '.totals.loc_added' <<< "$RISK_JSON") \
@@ -127,7 +127,7 @@ Step 1.75 uses `test_lines_removed` as an advisory hint only - too weak for wh
127
127
  ```bash
128
128
  TEST_INTEGRITY_JSON=$(printf '%s' "$RISK_FULL" | node $HOME/.claude/scripts/test-integrity-gate.mjs 2>/dev/null || echo "")
129
129
  TI_COUNT=$(jq -r '.count // 0' <<< "${TEST_INTEGRITY_JSON:-{\}}" 2>/dev/null || echo 0)
130
- [ "$TI_COUNT" -gt 0 ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 review.test_integrity findings="$TI_COUNT"
130
+ [ "$TI_COUNT" -gt 0 ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.test_integrity findings="$TI_COUNT"
131
131
  ```
132
132
 
133
133
  `findings[]` are reviewer-shaped (`test_integrity`, `blocking`), so they merge into the reviewer findings at Step 3.0 and need no triage-prompt or `validate-triage.mjs` change. Triage keeps each blocking unless the removal is justified per the immutable-test rule (spec changed AND commit body names the test) → `deferred[]`.
@@ -142,7 +142,7 @@ On a trivial diff every reviewer agrees and the extra models plus triage are pai
142
142
  SCOPE_JSON=$(printf '%s' "$RISK_FULL" | node $HOME/.claude/scripts/review-scope.mjs 2>/dev/null \
143
143
  || echo '{"scope":"full","reason":"no risk report - failing safe"}')
144
144
  REVIEW_SCOPE=$(jq -r '.scope // "full"' <<< "$SCOPE_JSON")
145
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 review.scope scope="$REVIEW_SCOPE"
145
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.scope scope="$REVIEW_SCOPE"
146
146
  ```
147
147
 
148
148
  `single` (Reviewer 1 only) requires **all** of: churn <= 20 lines, `totals.max_score` < 3.0, and no `security_path` / `migration` / `public_api` / `no_test_change` / `test_lines_removed` on any file. Anything else → `full`.
@@ -159,7 +159,7 @@ node $HOME/.claude/scripts/skill-conformance.mjs \
159
159
  --repo "$WORKTREE" --out "$WORKTREE/.pipeline/criteria-manifest.json"
160
160
  CRIT_RC=$?
161
161
  CRITERIA=$(cat "$WORKTREE/.pipeline/criteria-manifest.json" 2>/dev/null || echo '{}')
162
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 review.criteria \
162
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.criteria \
163
163
  rules="$(jq -r '.selectedRuleCount // 0' <<< "$CRITERIA")" \
164
164
  ledger="$(jq -r '.ledger.source // "derived"' <<< "$CRITERIA")"
165
165
  [ "$CRIT_RC" != "0" ] && HALT "criteria could not be resolved (rc=$CRIT_RC) - see resolutionFailure / unparseableRegistries"
@@ -175,7 +175,7 @@ Output conforms to `$HOME/.claude/schemas/criteria-manifest.schema.json`. The fo
175
175
 
176
176
  #### Step 1.8 - Figma visual-fidelity context (when task carries a Figma reference)
177
177
 
178
- **Short-run inputs.** Phases 1 and 2 do not run in a Short run, so three inputs this phase was written around are absent. Substitute them and RECORD the substitution - a step that could not run and a step that passed must not read the same, or the completeness claim cannot be checked:
178
+ **Missing inputs.** A section the evidence did not support is absent from the analysis document, so three inputs this phase was written around can be missing. Substitute them and RECORD the substitution - a step that could not run and one that passed must not read the same, or the completeness claim cannot be checked:
179
179
 
180
180
  | Absent input | Substitute |
181
181
  |---|---|
@@ -260,7 +260,7 @@ Each reviewer inherits the `code-reviewer` agent's focus areas (Security, Archit
260
260
  | Python | `ai-backend-toolkit:api-security-best-practices` | `ai-backend-toolkit:fastapi-pro` | `ai-backend-toolkit:python-patterns` |
261
261
  | Node.js | `ai-backend-toolkit:api-security-best-practices` | `ai-backend-toolkit:nodejs-backend-patterns` | `ai-frontend-toolkit:typescript-patterns` |
262
262
  | Docker | `ai-backend-toolkit:docker-expert` | `ai-backend-toolkit:docker-expert` | `ai-backend-toolkit:ci-cd-pipelines` |
263
- | Generic | `security-review` | `ai-backend-toolkit:clean-code` | `ai-backend-toolkit:clean-code` |
263
+ | Generic | `ai-common-toolkit:security-review` | `ai-backend-toolkit:clean-code` | `ai-backend-toolkit:clean-code` |
264
264
 
265
265
  ##### 2.1 Previous-round findings (iteration >= 2) and 2.2 scope self-check (every iteration)
266
266
 
@@ -298,6 +298,10 @@ Skip only when the diff has no UI change. Record the outcome in
298
298
 
299
299
  Step 2 produces N reviewer-output objects (one per dispatched reviewer), each conforming to `$HOME/.claude/schemas/reviewer-output.schema.json`. They are persisted to `state.reviewIterations[<iteration>].reviewers[]` and consumed by Step 3 (Fable triage) - never by Phase 4 directly. The triage step (below) is the producer of the only review artifact Phase 4 reads, conforming to `$HOME/.claude/schemas/triage-output.schema.json`.
300
300
 
301
+ #### Step 2.7 - Security audit (conditional, produces reviewer-shaped findings)
302
+
303
+ Runs when Step 1.75 scored `security_path`, on a release branch, or when `/multi-agent:security-review` dispatches here. The `security-auditor` returns reviewer `findings[]`, each with a `security` envelope (`security-finding.schema.json`) and severity from the CVSS band, held in `$SECURITY_AUDIT_JSON` for the merge so a `blocking` one reaches triage and blocks Phase 4. Mechanics: `~/.claude/multi-agent-refs/features/security-audit.md`.
304
+
301
305
  **Subagent return format** - each reviewer returns JSON conforming to `$HOME/.claude/schemas/reviewer-output.schema.json`:
302
306
 
303
307
  ```json
@@ -371,14 +375,14 @@ ANON=$(jq -n --argjson r "$REVIEWERS_JSON" --arg t "$TASK_ID" --argjson i "$ITER
371
375
 
372
376
  `$REVIEWERS_JSON` is `state.reviewIterations[i].reviewers`. Findings come back with `foundBy: "Source A|B|C"` and every identity key removed. Persist the map to `state.reviewIterations[i].anonymizationMap` for Phase 5 per-reviewer telemetry, and **never put the map in a prompt**.
373
377
 
374
- Then append the Step 1.76 test-integrity findings, so they are adjudicated rather than never seen:
378
+ Then append the Step 1.76 test-integrity and Step 2.7 security-audit findings, so they are adjudicated rather than never seen:
375
379
 
376
380
  ```bash
377
- MERGED=$(jq -s '.[0] + (.[1].findings // [])' \
378
- <(printf '%s' "$ANON") <(printf '%s' "${TEST_INTEGRITY_JSON:-{\}}"))
381
+ MERGED=$(jq -s '.[0] + (.[1].findings // []) + (.[2].findings // [])' \
382
+ <(printf '%s' "$ANON") <(printf '%s' "${TEST_INTEGRITY_JSON:-{\}}") <(printf '%s' "${SECURITY_AUDIT_JSON:-{\}}"))
379
383
  ```
380
384
 
381
- Deterministic findings keep `tag: test_integrity` and carry no `foundBy`: a reviewer finding may be a hallucination, a gate finding is a fact.
385
+ Deterministic findings keep `tag: test_integrity` and no `foundBy`: a reviewer finding may hallucinate, a gate finding is a fact. Security-audit findings carry their `security` envelope and `foundBy: "security-auditor"`; an empty `$SECURITY_AUDIT_JSON` contributes nothing.
382
386
 
383
387
  ##### 3.1 Short-circuit: no findings
384
388
 
@@ -426,8 +430,8 @@ Exit 2 (empty ledger) skips silently. The `## Rejected review preferences` secti
426
430
  **Recall telemetry.** Log what was injected, then what triage cited. Zero cited is a legitimate answer; `learning-curve.mjs` trends the ratio:
427
431
 
428
432
  ```bash
429
- bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 memory.injected kind=prior-art rows=$N
430
- bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 memory.hit rows=$CITED_COUNT
433
+ bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 memory.injected kind=prior-art rows=$N
434
+ bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 memory.hit rows=$CITED_COUNT
431
435
  ```
432
436
 
433
437
  **Bulky payloads (opt-in via `prefs.global.contextOffload.enabled`).** Test output and whole-file diffs go through the offload filter, which leaves a `[[ref:<node_id>]]` line plus the tail in context and the full text under `.multi-agent/refs/`. Read that file when the tail is not enough; with the pref off it is a pass-through.
@@ -513,7 +517,7 @@ One `review.reviewer_call` per dispatched reviewer, one `review.triage_call`, on
513
517
  ```bash
514
518
  M=$HOME/.claude/scripts/log-metric.sh
515
519
  emit() { # $1=event $2=model $3=duration $4=tokens_in $5=tokens_out
516
- LOG_METRIC_FORWARD_TO_TRACKER=1 bash "$M" "$TASK_ID" 4 "$1" \
520
+ LOG_METRIC_FORWARD_TO_TRACKER=1 bash "$M" "$TASK_ID" 3 "$1" \
517
521
  model="$2" duration_ms="$3" tokens_in="$4" tokens_out="$5"
518
522
  }
519
523
  emit review.reviewer_call fable "$R1_DURATION" "$R1_IN" "$R1_OUT" # opus on Copilot CLI
@@ -525,7 +529,7 @@ else
525
529
  fi
526
530
  emit review.reviewer_call sonnet "$SONNET_DURATION" "$SONNET_IN" "$SONNET_OUT"
527
531
  emit review.triage_call fable "$TRIAGE_DURATION" "$TRIAGE_IN" "$TRIAGE_OUT"
528
- bash "$M" "$TASK_ID" 4 review.completed raw_count=$RAW accepted=$ACC \
532
+ bash "$M" "$TASK_ID" 3 review.completed raw_count=$RAW accepted=$ACC \
529
533
  deferred=$DEF rejected=$REJ approved=$APPROVED duration_ms=$DURATION
530
534
  ```
531
535
 
@@ -614,7 +618,7 @@ Log: "Phase 3: Review - raw={N1+N2+N3} accepted={Na} deferred={Nd} rejected={N
614
618
  ## Token telemetry - invoke after every LLM call
615
619
 
616
620
  ```bash
617
- bash $HOME/.claude/scripts/phase-tracker.sh tokens 4 <input_count> <output_count> [cached_count]
621
+ bash $HOME/.claude/scripts/phase-tracker.sh tokens 3 <input_count> <output_count> [cached_count]
618
622
  ```
619
623
 
620
624
  The optional 4th `cached_count` is the prompt-cache-read token count when the host reports it (Anthropic `cache_read_input_tokens`); it defaults to 0 and is priced at the cheaper `cacheReadPerMtok` rate in the Phase 5 cost ledger. The tracker accumulates the totals additively, so multiple calls in the same phase compound. The render output then shows live cost on the active phase tile (e.g. `Phase 2 Dev 2m 14s · 12.4k tok`). This satisfies the contract in `$HOME/.claude/multi-agent-refs/tracker-contract.md` and the `smoke-tracker-tokens-invocation.sh` enforcement gate. Skipping this call is the #1 cause of "I can't see how much it cost" complaints.
@@ -623,15 +627,12 @@ Contract and rationale: `progress-contract.md` -> Token telemetry forwarding.
623
627
 
624
628
  ---
625
629
 
626
- ## User test (was Phase 3 until v19.0.0)
627
-
628
- Optional test gate, now the tail of Review rather than a phase of its own: it
629
- judges work that already exists, which is what Review does. Needs an interactive
630
- prompt AND a worktree checkout, so it is skipped where either is missing. If
631
- issues are found, the run returns to Phase 1 Dev.
630
+ ## User test
632
631
 
632
+ The tail of Review rather than a phase of its own: it judges work that already
633
+ exists, which is what Review does.
633
634
 
634
- > **TLDR** - Optional test gate. Offers to boot the simulator/emulator (UI Bug Hunter) or hand off to the user for manual QA. Needs an interactive prompt AND a worktree checkout, so it is in the phase set of `/multi-agent` alone and dropped by every `autopilot` or `--local` entry. Depth does not affect it: a Short run still reaches Phase 3. If issues found, loops back to Phase 2.
635
+ > **TLDR** - Optional test gate. Offers to boot the simulator/emulator (UI Bug Hunter) or hand off to the user for manual QA. Needs an interactive prompt AND a worktree checkout, so it runs on an attended `/multi-agent` whose workspace is a worktree, and is skipped on `autopilot` or when the user chose to work locally. If issues found, loops back to Phase 2.
635
636
 
636
637
  <!-- progress-contract: applied -->
637
638
  Progress emission per `$HOME/.claude/multi-agent-refs/progress-contract.md` - lines for local-test prompt render, user-answer capture, repo checkout (if selected).
@@ -682,7 +683,7 @@ fi
682
683
  **Telemetry**:
683
684
 
684
685
  ```bash
685
- LOG_METRIC_FORWARD_TO_TRACKER=0 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 5 test_gap.scanned \
686
+ LOG_METRIC_FORWARD_TO_TRACKER=0 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 test_gap.scanned \
686
687
  stack=$SCAN_STACK \
687
688
  sources=$(jq '.totals.sourcesScanned' <<< "$GAP_JSON") \
688
689
  gaps=$(jq '.totals.gapCount' <<< "$GAP_JSON")
@@ -702,7 +703,7 @@ Figma evidence (tier=<n>):
702
703
 
703
704
  Tier 1 / Tier 2 records print `screenshotUrl` from the captured evidence (Tier 2 URLs expire after 30 days, re-fetch on the spot if needed). Tier 3 records print the local path to the user-attached screenshot. The block is informational; it never blocks the prompt.
704
705
 
705
- 1. Ask with a native `AskUserQuestion` picker (never a typed y/N prompt). The options MUST make the local-checkout side effect explicit - testing removes the worktree and checks the branch out into the main repo:
706
+ 1. Ask with a native `AskUserQuestion` picker (never a typed y/N prompt), per `$HOME/.claude/multi-agent-refs/picker-contract.md`. The options MUST make the local-checkout side effect explicit - testing removes the worktree and checks the branch out into the main repo:
706
707
  - `question`: "Check out locally to test now?" (rendered in `outputLanguage`)
707
708
  - `header`: "Test" (English, <=12 chars)
708
709
  - `options`:
@@ -733,7 +734,7 @@ Tier 1 / Tier 2 records print `screenshotUrl` from the captured evidence (Tier 2
733
734
  bash $HOME/.claude/scripts/phase-tracker.sh now 3 "awaiting local test (user)"
734
735
  bash $HOME/.claude/scripts/phase-tracker.sh render
735
736
  ```
736
- The waiting state persists in `tracker-state.json` across the handoff; `/multi-agent:resume-local` and `/multi-agent:manual-test` CONTINUE this state file and never re-init it (`$HOME/.claude/multi-agent-refs/tracker-contract.md` "Continuation runs").
737
+ The waiting state persists in `tracker-state.json` across the handoff; `/multi-agent:resume` and `/multi-agent:manual-test` CONTINUE this state file and never re-init it (`$HOME/.claude/multi-agent-refs/tracker-contract.md` "Continuation runs").
737
738
 
738
739
  **"ok" is a structured result, not a word.** Before "ok" is accepted, the run writes `$WORKTREE/.pipeline/manual-test.json`: one entry per acceptance criterion, the criteria taken from the analysis doc test plan (Section 15 / 20), the plan tasks, and the user's own words in the reply. Every criterion records what was seen; a criterion that was not tried says so with a reason.
739
740
  ```json
@@ -791,22 +792,12 @@ Results included in Phase 5 report. MCP tools preferred when available - conci
791
792
 
792
793
  **Snapshot regression flow (optional):** when the task changes a stable component, capture a screenshot before the change (baseline) and after (current), then call `ios_visual_diff({baseline, current, max_diff_pct: 1.0})`. Threshold can be relaxed for animated / non-deterministic regions - keep `max_diff_pct ≤ 1.0` for static layouts.
793
794
 
794
- #### Security Audit (store-readiness)
795
-
796
- When the task touches authentication, keychain, network, or is scheduled for an imminent release, launch the `security-auditor` subagent to run an OWASP Mobile Top 10 pass plus App Store / Play Store compliance checks:
797
-
798
- ```
799
- Agent(subagent_type: "security-auditor", prompt: "<diff + context>")
800
- ```
801
-
802
- Returns severity-tagged findings (Critical / High / Medium). Critical items block Phase 4 just like Phase 3 blockers; High items are logged and surfaced in Phase 5 report. Skipped by default - opt-in for release branches or on explicit `/multi-agent "<task>" --audit` flag.
803
-
804
795
  #### Telemetry - token forwarding
805
796
 
806
797
  When the security-auditor or any other Phase 3 sub-agent runs, forward its token totals so Phase 5's Cost Breakdown captures Phase 3:
807
798
 
808
799
  ```bash
809
- LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 5 audit.completed \
800
+ LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 audit.completed \
810
801
  model=opus tokens_in=$IN tokens_out=$OUT duration_ms=$DUR
811
802
  ```
812
803
 
@@ -354,19 +354,19 @@ A task ending with one or more `pushStatus === "skipped"` repos does NOT bump `r
354
354
 
355
355
  ```bash
356
356
  for proj in $(jq -r '.projects[].name' "$STATE_FILE"); do
357
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 6 commit.created repo=$proj sha=$SHA
358
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 6 push.attempted repo=$proj attempts=$N status=$STATUS
359
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 6 pr.opened repo=$proj url=$URL number=$N
357
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 commit.created repo=$proj sha=$SHA
358
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 push.attempted repo=$proj attempts=$N status=$STATUS
359
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 pr.opened repo=$proj url=$URL number=$N
360
360
  done
361
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 6 multi_repo.completed repos=$REPOS skipped=$SKIPPED
361
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 multi_repo.completed repos=$REPOS skipped=$SKIPPED
362
362
  ```
363
363
 
364
364
  **Token forwarding:** the commit-message and PR-body generators run on a model. Forward those calls into the tracker so Phase 5's Cost Breakdown captures Phase 4:
365
365
 
366
366
  ```bash
367
- LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 6 commit.message_generated \
367
+ LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 commit.message_generated \
368
368
  model=<sonnet|opus> tokens_in=$IN tokens_out=$OUT duration_ms=$DUR
369
- LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 6 pr.body_generated \
369
+ LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 pr.body_generated \
370
370
  model=<sonnet|opus> tokens_in=$IN tokens_out=$OUT duration_ms=$DUR
371
371
  ```
372
372
 
@@ -25,7 +25,7 @@ On entry: `phase-tracker.sh sub 5 <N> "<name>" in_progress`. On exit: `completed
25
25
  Phase 5 is the single exception to the autopilot zero-interaction rule: every mode, Full or Short, attended or autopilot, pauses at the channels multi-select menu. Full contract in `$HOME/.claude/multi-agent-refs/phases/modes.md`:
26
26
 
27
27
  - **30-min timeout** - if user does not respond, session ends cleanly. External delivery is aborted (no silent apply of defaults - prevents accidental Jira comments / Confluence pages). Internal capture (Steps 2 + 3 below) STILL runs so `agent-log.md` + knowledge base are persisted.
28
- - **Resumable** - state written as `{status: "awaiting_input", phase: 7, waitingFor: "user-channels-choice", channelsInput: <state-bundle>}`. User can `/multi-agent:resume <task-id>` any time later; channels menu re-opens with same inputs.
28
+ - **Resumable** - state written as `{status: "awaiting_input", phase: 5, waitingFor: "user-channels-choice", channelsInput: <state-bundle>}`. User can `/multi-agent:resume <task-id>` any time later; channels menu re-opens with same inputs.
29
29
  - **Timeout log line:** `Phase 5: channels menu timeout (30 min) - session ended, resume with /multi-agent:resume {taskId}`.
30
30
 
31
31
  ---
@@ -188,9 +188,9 @@ Skipped sections: when `planTodos.enabled` is false or no `plan.todos[]` was emi
188
188
  **Telemetry emission** (mandatory): forward the phase's own LLM spend (humanizer + report compose calls) to the tracker, then emit the final event:
189
189
 
190
190
  ```bash
191
- LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 7 report.compose \
191
+ LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 5 report.compose \
192
192
  model=$REPORT_MODEL tokens_in=$R_IN tokens_out=$R_OUT duration_ms=$R_DUR
193
- $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 7 task.completed \
193
+ $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 5 task.completed \
194
194
  phases=$PHASE_COUNT review_cycles=$CYCLES lang=$PROMPT_LANG \
195
195
  channels_pr=$PR_STATUS channels_jira=$JIRA_STATUS \
196
196
  channels_confluence=$CONF_STATUS channels_wiki=$WIKI_STATUS \
@@ -15,7 +15,7 @@
15
15
 
16
16
  | Phase | File |
17
17
  | --------------------------------- | -------------------------------------------------------------------- |
18
- | Modes (depth picker, autopilot, --local) | `$HOME/.claude/multi-agent-refs/phases/modes.md` |
18
+ | Modes (autopilot, analysis) | `$HOME/.claude/multi-agent-refs/phases/modes.md` |
19
19
  | Operations (kill, purge, resume) | `$HOME/.claude/multi-agent-refs/phases/operations.md` |
20
20
  | Phase 0: Init | `$HOME/.claude/multi-agent-refs/phases/phase-0-init.md` |
21
21
  | Phase 1: Plan | `$HOME/.claude/multi-agent-refs/phases/phase-1-plan.md` |
@@ -28,11 +28,11 @@
28
28
  ## Pipeline Flow
29
29
 
30
30
  ```
31
- Full: 0-Init -> 1-Plan -> 2-Dev -> 3-Review -> 4-Commit -> 5-Report
32
- Short: 0-Init -> (1 skipped) -> 2-Dev -> 3-Review -> 4-Commit -> 5-Report
33
- --local: Either of the above with no worktree - works directly on a local branch
31
+ 0-Init -> 1-Plan -> 2-Dev -> 3-Review -> 4-Commit -> 5-Report
32
+ Local: The same set with no worktree - Phase 0 Step 5b answered local, so
33
+ work happens directly on a branch in the project root
34
34
 
35
- Full or Short is the Phase 0 Step 7.5 question, not a command name. Autopilot never asks and always runs Full.
35
+ One pipeline: every mode runs its whole phase set, and no answer during the run adds or removes a phase.
36
36
  ```
37
37
 
38
38
  ## Phase entry - pending steer (every phase, every mode)
@@ -99,10 +99,8 @@ done
99
99
 
100
100
  This produces an initial card stack printed by both CLIs.
101
101
 
102
- `/multi-agent` and `/multi-agent:local` are the exception: they do not know their
103
- set here, because depth decides it at Step 7.5. They register `0:Init` alone, then
104
- the rest once the answer lands, and call `tiles --new` for the second batch. Full
105
- contract: `tracker-contract.md`, "Deferred registration".
102
+ No mode is an exception: the phase set is a property of the command, so every
103
+ tile is created in this one batch. Full contract: `tracker-contract.md`.
106
104
 
107
105
  ### Tracker updates (every phase boundary)
108
106
 
@@ -114,7 +112,7 @@ $HOME/.claude/scripts/phase-tracker.sh update <N> in_progress # phase starts
114
112
  $HOME/.claude/scripts/phase-tracker.sh update <N> completed # phase ends OK
115
113
  # or:
116
114
  $HOME/.claude/scripts/phase-tracker.sh update <N> failed # phase failed
117
- $HOME/.claude/scripts/phase-tracker.sh update <N> skipped # e.g. 1 and 2 in a Short run
115
+ $HOME/.claude/scripts/phase-tracker.sh update <N> skipped # e.g. 2 in analysis mode
118
116
  ```
119
117
 
120
118
  After every LLM call (counts are additive; skipping this is why runs end with durations but no cost - nothing reconstructs spend afterwards):
@@ -165,7 +163,7 @@ TaskUpdate({ taskId: <saved>, status: "completed" })
165
163
  bash phase-tracker.sh update <N> completed
166
164
  ```
167
165
 
168
- A phase outside the command's set gets no TaskCreate at all. Depth is different: it is not known at registration time, because the tracker boots at Step -1 and the depth question runs at Step 7.5. So registration splits - Phase 0 alone at Step -1, the rest at Step 7.5 once the answer says which phases the run has. A Short run never draws a Plan tile it will not use. Registering every phase and flipping the skipped one to `skipped` is what this replaced, in v17.5.0: it put a full-width widget on screen beside the question asking whether to run part of it - see the ordering rule below.
166
+ A phase outside the command's set gets no TaskCreate at all, and the set is known before the tracker boots.
169
167
 
170
168
  **(strict) TaskCreate ordering**: All TaskCreate calls MUST fire in strict phase-number order BEFORE any TaskUpdate is applied. The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks (e.g. `1 ✓ · 3 ✓ · 4 ✓ · 0 ▶ · 2 ☐`) even when the underlying state is correct. Pre-marking phases as completed/skipped before Phase 0 starts is FORBIDDEN - register the tile in order with default `pending` status, then flip status via TaskUpdate when the phase actually short-circuits. Full contract in `$HOME/.claude/multi-agent-refs/tracker-contract.md` section "TaskCreate ordering (strict)".
171
169
 
@@ -69,7 +69,7 @@ Emit a progress line **at least** at every one of these moments:
69
69
 
70
70
  ### Phase 2 (Planning)
71
71
  - plan draft start, plan render, user-approval prompt.
72
- - **v5.3.0 Plan Approval Gate (Full + interactive only - a Short run has no plan, autopilot may not ask):**
72
+ - **v5.3.0 Plan Approval Gate (interactive only - autopilot may not ask):**
73
73
  - `clarification-ask` per round - orchestrator writes structured questions when Phase 1 flagged ambiguity (missing acceptance criteria, no Figma/endpoint link, vague language, parent-story scope drift)
74
74
  - `clarification-answer` per round - user reply captured into `state.phases["2"].clarificationAnswers`
75
75
  - `plan-edit-request` per free-text edit - user-supplied revision instruction captured into `state.phases["2"].planEditRequests`
@@ -1,5 +1,7 @@
1
1
  # Readiness Review (shared flow for review-jira + review-issue)
2
2
 
3
+ > **Pickers follow** `$HOME/.claude/multi-agent-refs/picker-contract.md`: never a one-option call, and branch on the option selected, not on its text.
4
+
3
5
  Assess whether a tracker item (Jira issue or GitHub issue) is READY to hand to the multi-agent pipeline, list the concrete gaps, and - after confirmation - post them back as a comment on the item so the reporter can fix it. Read-only on code: no worktree, no branch, no commits, no dev chaining. This is the inverse of `/multi-agent:create-jira` (which authors a well-formed item); here we grade an existing one.
4
6
 
5
7
  Both `/multi-agent:review-jira` and `/multi-agent:review-issue` execute this flow; only the provider (fetch + comment endpoint + picker) differs.
@@ -44,7 +44,7 @@ This is the single source of truth. When a contributor or model is unsure where
44
44
  3. `AskUserQuestion` `question`, `options[].label` and `options[].description` all follow `OUTPUT_LANG`. Only `header` stays English: a <=12-char chip Turkish overflows. Callers branch on which option was picked, never on its literal text, and pass `default` / `ASK_CHOICE_DEFAULT` as a 1-based index. The host's own **Other** row is always English. Caller rules: `picker-contract.md`.
45
45
  4. Always English regardless of either axis: commit messages, PR titles, branch names, code identifiers, agent-state.json values, agent-log.md, reviewer/triage system prompts.
46
46
 
47
- **Failure mode this prevents.** Entering `/multi-agent`, `/multi-agent:autopilot`, `/multi-agent:local`, etc. and switching the assistant's conversational text or picker question copy to English while `outputLanguage="tr"` is set. The user sees a half-English half-Turkish dialogue, flagged as a pipeline bug, not a stylistic choice.
47
+ **Failure mode this prevents.** Entering `/multi-agent`, `/multi-agent:autopilot`, etc. and switching the assistant's conversational text or picker question copy to English while `outputLanguage="tr"` is set. The user sees a half-English half-Turkish dialogue, flagged as a pipeline bug, not a stylistic choice.
48
48
 
49
49
  ## Code & Commit Rules
50
50
 
@@ -0,0 +1,39 @@
1
+ # Threat model contract
2
+
3
+ The run-scoped threat model is the frame every security finding is calibrated against. It is produced by the security-auditor at Phase 3 Step 2.7 (or by `/multi-agent:security-review` running standalone), read by any later step that revisits security, and mirrored to `state.threatModel.path` so a resume re-uses it instead of re-deriving it.
4
+
5
+ ## Where it lives
6
+
7
+ `.pipeline/threat-model.md` in the run's worktree, keyed to repo + branch. One per run. If a fresh one already exists for this repo+branch (a prior step or a Phase 1 pass wrote it), read it; do not overwrite. If it is absent, produce it before emitting any finding.
8
+
9
+ Mirror the resolved path and a content hash to `state.threatModel = { path, sha, producedAt, producedBy }` so resume and Phase 5 can find it without re-reading the tree.
10
+
11
+ ## The four sections (all required, in order)
12
+
13
+ A threat model with a missing section is not a threat model; the auditor treats a missing section as "produce it," not "skip it."
14
+
15
+ ### 1. Attacker
16
+
17
+ The one realistic adversary for THIS change. Name it concretely: an unauthenticated internet caller, an authenticated low-privilege user, a malicious or compromised dependency, a co-located app on the device, a user with physical access. Not "attackers" in the abstract - the specific actor whose capability makes this diff interesting. If the diff has no plausible attacker, say so; that is a valid, short threat model and most findings then cap at `suggestion`.
18
+
19
+ ### 2. Trust boundaries
20
+
21
+ Where untrusted data crosses into trusted code within the changed surface: a request body or query parameter, a deep link or universal link, a WebView `postMessage`, a file the app did not write, an environment variable an attacker can set, a response from a third-party service treated as safe. List the boundaries the diff touches, each with the `file:line` where the crossing happens.
22
+
23
+ ### 3. Attack surface
24
+
25
+ What this diff actually added or touched: a new endpoint, a new query or ORM call, a new deserialization, a new permission, a new dependency, a new crypto usage, a new storage write. A finding outside this surface is out of scope unless the diff made it reachable - and if it did, say how. This section is what keeps the audit anchored to the change instead of drifting into a whole-repo review.
26
+
27
+ ### 4. Severity calibration
28
+
29
+ The assumption each severity rests on, stated so a reader disputes the assumption rather than the number. "Critical assumes this route is unauthenticated in production; if it is admin-only the same finding is medium." "High assumes the secret reaches a log that ships off-device." Every `blocking` finding must trace to an assumption named here; a blocker resting on an unstated assumption is miscalibrated.
30
+
31
+ ## Shape
32
+
33
+ Plain Markdown, four `##` sections with those names, human-readable. It is evidence for a person and context for the auditor, not a machine artifact - no schema. Keep it short: a page, not a report. Cite `file:line` where a boundary or surface item has one.
34
+
35
+ ## What it is not
36
+
37
+ - Not a whole-repo model. It is scoped to the diff under review.
38
+ - Not a dynamic test plan. This is a static, read-only posture: no live target, no payloads, no exploitation. `fixVerification` on a finding names the empirical check a human or a later dynamic pass would run; the threat model does not run it.
39
+ - Not embedded in the analysis document. The security-review command must run standalone, so the model lives in its own file and is produced on demand.
@@ -90,11 +90,11 @@ Phases by mode:
90
90
  | Mode | Phases |
91
91
  |---|---|
92
92
  | `/multi-agent` | 0,1,2,3,4,5 |
93
- | `/multi-agent:local` | 0,1,2,3,4,5 (the user test needs a worktree checkout and local has none, so that STEP inside Review is skipped - the phase is not) |
94
- | `/multi-agent:autopilot`, `/multi-agent:local-autopilot` | 0,1,2,3,4,5 (always Full; autopilot drops the interactive user test inside Review) |
93
+ | answered local at Step 5b | 0,1,2,3,4,5 (the user test needs a worktree checkout and local has none, so that STEP inside Review is skipped - the phase is not) |
94
+ | `/multi-agent:autopilot` | 0,1,2,3,4,5 (autopilot drops the interactive user test inside Review) |
95
95
  | `/multi-agent:analysis` | 0,1,3,4,5 (no code is written, so Dev is not in the set) |
96
96
 
97
- The two picker entries (`/multi-agent`, `:local`) register in two batches, because at Step -1 they do not yet know which set is theirs: see "Deferred registration" below. Every other mode registers its whole set at Step -1.
97
+ Every entry registers its whole set in one batch: the set is a property of the command, known before the tracker boots.
98
98
 
99
99
  Register each phase:
100
100
 
@@ -171,7 +171,7 @@ report's cost-unavailable list.
171
171
 
172
172
  > **All TaskCreate calls for the active mode's phase set MUST fire in strict phase-number order BEFORE any TaskUpdate is applied. No "pre-mark skipped phases as completed before Phase 0" reasoning is permitted - even when the agent knows in advance that a phase will be skipped.**
173
173
 
174
- Why: an agent reasoning "the user picked Short, so phases 1/2 will be skipped - let me TaskCreate them as completed first" produces tile IDs `1, 2, 3, ...` for phases 1/2/4, then the Phase 0 tile gets ID `4` and visually drops below them. The user sees `1, 2, 4 ✓ · 0 ▶ · 3 ☐ · ...` instead of `0 ▶ · 1 ✓ · 2 ✓ · 3 ☐ · 4 ✓ · ...`.
174
+ Why: an agent reasoning "analysis mode skips Dev - let me TaskCreate it as completed first" produces tile IDs `1, 2, 3, ...` for phases 1/2/4, then the Phase 0 tile gets ID `4` and visually drops below them. The user sees `1, 2, 4 ✓ · 0 ▶ · 3 ☐ · ...` instead of `0 ▶ · 1 ✓ · 2 ✓ · 3 ☐ · 4 ✓ · ...`.
175
175
 
176
176
  The correct sequence is **always**:
177
177
 
@@ -197,44 +197,13 @@ Mode-specific phase sets:
197
197
 
198
198
  | Mode | TaskCreate set (in order) |
199
199
  |---|---|
200
- | `/multi-agent` | 0 at Step -1; then at Step 7.5 either 1 → 2 → 3 → 4 → 5 (Full) or 2 → 3 → 4 → 5 (Short) |
201
- | `:local` | same two batches; nothing is dropped, because the user test that needs a worktree checkout is a step inside Review rather than a phase of its own |
202
- | `:autopilot`, `:local-autopilot` | 0 → 1 → 2 → 3 → 4 → 5 (6 phases - always Full; the user test is inside Review now, so no phase is dropped) |
200
+ | `/multi-agent` | 0 → 1 → 2 → 3 → 4 → 5 (6 phases) |
201
+ | answered local | the same set; nothing is dropped, because the user test that needs a worktree checkout is a step inside Review rather than a phase of its own |
202
+ | `:autopilot` | 0 → 1 → 2 → 3 → 4 → 5 (6 phases - the user test is inside Review now, so no phase is dropped) |
203
203
  | `:analysis` | 0 → 1 → 3 → 4 → 5 (5 phases - no code is written, so Dev is not in the set) |
204
204
 
205
205
  A phase outside the mode's set gets no TaskCreate at all; the `[SKIPPED]` pattern applies only to a phase that IS in the set and short-circuits at runtime. Phase 3 is in every mode's set as of v14.0.0. The authoritative per-mode set is the `for p in ...` init block in each mode's own entry doc, generated by `gen-mode-dispatch.mjs`; this table mirrors those blocks.
206
206
 
207
- #### Deferred registration - the depth picker
208
-
209
- `/multi-agent` and `:local` cannot know their phase set at Step -1. Depth decides it, and the depth picker cannot run before Step 7.5: its recommendation needs `taskType`, which needs the fetched issue and the branch.
210
-
211
- Until v17.5.0 they registered all eight anyway and flipped 1 and 2 to `skipped` at 7.5. That put a widget reading "8 tasks, 7 open - Phase 1 Plan, Phase 1 Plan, ..." on screen *beside* the question asking whether to run Analysis and Planning at all, and a Short answer then contradicted a list the user had just been shown. The widget was asserting a shape the run had not chosen.
212
-
213
- So registration splits at the moment the shape is known:
214
-
215
- ```text
216
- # Step -1, first thing in the run: Phase 0 only. It is the one phase that is
217
- # certain, and the run is never silent while Phase 0 does its work.
218
- bash $HOME/.claude/scripts/phase-tracker.sh add 0 Init
219
- bash $HOME/.claude/scripts/phase-tracker.sh tiles # -> TaskCreate(Phase 0)
220
- bash $HOME/.claude/scripts/phase-tracker.sh update 0 in_progress
221
-
222
- # Step 7.5, immediately after the depth answer:
223
- # Full -> 1 2 3 4 5 Short -> 2 3 4 5
224
- # :local drops nothing - the user test lives inside Review now, so there is
225
- # no separate phase for it to skip
226
- for p in "1:Plan" "2:Dev" "3:Review" "4:Commit" "5:Report"; do
227
- bash $HOME/.claude/scripts/phase-tracker.sh add "${p%%:*}" "${p#*:}"
228
- done
229
- bash $HOME/.claude/scripts/phase-tracker.sh tiles --new # -> TaskCreate for the new tiles only
230
- ```
231
-
232
- `tiles --new` emits `TaskCreate` only for phases that carry no `tasklist_id` yet, so the Phase 0 tile is not created twice. It is the same ordering rule, applied per batch: every tile in a batch is created in ascending phase order, and a deferred batch only ever appends phases numbered above everything already registered. Nothing is pre-marked, and a phase the run will not execute never gets a tile at all.
233
-
234
- A phase that IS registered and short-circuits later still flips with `[SKIPPED]` - autopilot suppressing the Phase 3 user test, for instance. That is a runtime outcome, not an unknown set.
235
-
236
- **Enforcement**: `smoke-tasklist-ordering.sh` scans the dispatcher (`commands/multi-agent/SKILL.md`) and every mode entry point doc (`commands/multi-agent/{autopilot,local,local-autopilot,analysis,resume-local}/SKILL.md` + the Copilot full-inline orchestrator mirror) for the explicit "in phase-number order" rule. Inventory drift fails the smoke.
237
-
238
207
  ### Other CLIs - call render after every state change
239
208
 
240
209
  There is no TaskList outside Claude Code. Instead, after each state change the agent calls:
@@ -299,7 +268,7 @@ Throttling rules: mirror only canonical-set lines (`verbose`-tier internals are
299
268
 
300
269
  ### Delegated phases - mirror limitation + chunked dispatch (required)
301
270
 
302
- When a phase's work is delegated to a subagent (Phase 2 Dev on Opus in a Short run, `create-component` plugin dispatch, Phase 1 explorers, Phase 3 reviewers), the visual channel freezes for the duration of the Agent call: the orchestrator is blocked while the call is in flight, so it cannot fire `TaskUpdate` / `now` / `tokens`, and a subagent cannot drive the parent session's TaskList (its own TaskCreate/TaskUpdate calls land on an invisible child list). The progress-line mirror above can therefore only fire while the orchestrator holds control. Rules:
271
+ When a phase's work is delegated to a subagent (`create-component` plugin dispatch, Phase 1 explorers, Phase 3 reviewers), the visual channel freezes for the duration of the Agent call: the orchestrator is blocked while the call is in flight, so it cannot fire `TaskUpdate` / `now` / `tokens`, and a subagent cannot drive the parent session's TaskList (its own TaskCreate/TaskUpdate calls land on an invisible child list). The progress-line mirror above can therefore only fire while the orchestrator holds control. Rules:
303
272
 
304
273
  1. **Pre-dispatch marker.** Immediately before every Agent call, set the active-phase line to the delegation itself, so the frozen interval at least states what is running and on which model:
305
274
  - Claude Code: `TaskUpdate({activeForm: "Dev subagent (opus): <task subject>"})`
@@ -434,7 +403,7 @@ The `tasklist_id` meta from the previous session is replaced with the new IDs du
434
403
 
435
404
  ## Continuation runs (finish / manual-test)
436
405
 
437
- A pre-existing `tracker-state.json` for the task is never re-initialized. Rules for any command that continues an earlier run (`/multi-agent:resume-local`, `/multi-agent:manual-test`, resume):
406
+ A pre-existing `tracker-state.json` for the task is never re-initialized. Rules for any command that continues an earlier run (`/multi-agent:resume`, `/multi-agent:manual-test`):
438
407
 
439
408
  1. `init` runs ONLY when no state file exists for the task. Otherwise the existing file is kept - phase history (elapsed, tokens, model, meta) survives.
440
409
  2. The continuing command re-declares its phase set with `add` - `add` is idempotent, so existing phases keep their name, status, and token history; only genuinely new phases are appended. The card renders phases sorted by numeric id, so mixed sets stay in order.
@@ -11,6 +11,8 @@
11
11
  - [Cross-CLI parity](#cross-cli-parity)
12
12
  <!-- /toc -->
13
13
 
14
+ > **Pickers follow** `$HOME/.claude/multi-agent-refs/picker-contract.md`: never a one-option call, and branch on the option selected, not on its text.
15
+
14
16
  > **TLDR** - Component tasks can auto-generate wiki docs + Figma screenshots. The Wiki adapter is invoked from `/multi-agent:channels` (Phase 5 delegates, or user invokes post-hoc). Four layouts supported (`submodule`, `in-repo`, `github-wiki`, `separate-repo`) - adapter picked from `figmaConfig.wiki.mode`. Non-blocking: failures log a warning and channels continues to other adapters. The Wiki adapter supports scope multi-select (Case A) and a precondition-failure menu (Case B) - see below.
15
17
 
16
18
  This doc is referenced from `commands/multi-agent/channels/SKILL.md` (Wiki adapter) and indirectly from `$HOME/.claude/multi-agent-refs/phases/phase-5-report.md` (which delegates all external delivery to channels). Keeping it separate keeps both files under their token budgets and gives the contract a stable location for Claude-side + Copilot-side implementations.
@@ -75,7 +77,7 @@ Autopilot in Phase 5 pauses at the channels menu (per modes.md contract) - if
75
77
 
76
78
  ## Legacy prompt + preference flow (pre-v5.7, still supported for backward compat)
77
79
 
78
- Interactive path (any interactive run, Full or Short), ONLY when the schema lacks `wikiScope` - ask with a native `AskUserQuestion` picker (never a typed y/n):
80
+ Interactive path (any interactive run), ONLY when the schema lacks `wikiScope` - ask with a native `AskUserQuestion` picker (never a typed y/n):
79
81
 
80
82
  - `question`: "Generate component wiki docs?" (rendered in `outputLanguage`)
81
83
  - `header`: "Wiki" (English, <=12 chars)
@@ -103,7 +105,6 @@ Explicit logs help the developer understand why wiki did or did not run:
103
105
  - `figmaConfig` missing entirely → log `Phase 5: wiki skipped (no figma-config for this project)`.
104
106
  - User declined at prompt → log `Phase 5: wiki skipped by user`.
105
107
  - Autopilot with `wikiDefault=false` → log `Phase 5: wiki skipped (autopilot + wikiDefault=false)`.
106
- - Short run: DO prompt - wiki is cheap and keeps docs fresh on the fast path; skip only if the user says no.
107
108
 
108
109
  ## Success log
109
110
 
@@ -1,5 +1,5 @@
1
1
  {
2
- "schemaVersion": "2.7.0",
2
+ "schemaVersion": "2.8.0",
3
3
  "global": {
4
4
  "identities": [],
5
5
  "keychainMapping": {
@@ -101,7 +101,7 @@
101
101
  "skillConformance": {
102
102
  "blockOnCoverageGap": false
103
103
  },
104
- "resumeLocal": {
104
+ "resume": {
105
105
  "autoFix": false
106
106
  }
107
107
  },
@@ -75,7 +75,7 @@ The 3-tier fallback chain above governs Figma access in `/multi-agent:analysis`
75
75
  | Phase | Sub-command examples | Figma MCP allowed | Figma REST allowed | Reason |
76
76
  |---|---|---|---|---|
77
77
  | Analysis Phase 1 | `/multi-agent:analysis` Phase 1 fetch | yes | yes (Tier 2 fallback) | Single source of design ground truth |
78
- | Plan (Phase 1) | `/multi-agent`, `/multi-agent:local`, `/multi-agent:autopilot`, `/multi-agent:local-autopilot` | no | no | Plan reads analysis doc Section 14 + Section 6 |
78
+ | Plan (Phase 1) | `/multi-agent`, `/multi-agent:autopilot` | no | no | Plan reads analysis doc Section 14 + Section 6 |
79
79
  | Dev (Phase 2) | every mode that runs Phase 2 (8 modes total) | no | no | Reads analysis doc + Code Connect mapping |
80
80
  | Review (Phase 3) | `/multi-agent:review`, every full-pipeline mode | no | no | Reviewer cites analysis doc Section 21 References |
81
81
  | Test (inside Phase 3) | `/multi-agent:test`, `/multi-agent:manual-test`, every full mode | no | no | Variant list comes from analysis Section 13.6 + 15.2 |
@@ -66,7 +66,7 @@
66
66
  },
67
67
  "worktreePath": {
68
68
  "type": ["string", "null"],
69
- "description": "Absolute path to the task worktree. Null in --local mode."
69
+ "description": "Absolute path to the task worktree. Null when the Step 5b picker answered local."
70
70
  },
71
71
  "branch": {
72
72
  "type": "string",
@@ -274,7 +274,7 @@
274
274
  "workspaceSource": {
275
275
  "type": "string",
276
276
  "enum": ["asked", "command", "autopilot"],
277
- "description": "Who decided where the branch lives. asked = the user answered the Step 5b workspace picker; command = :local / --local / :local-autopilot stated it up front, or a flow that only ever builds worktrees; autopilot = resolved to a worktree without asking, because an unattended commit in the user's own checkout is what worktrees prevent. localMode alone cannot say: false is both a chosen worktree and one nothing asked about."
277
+ "description": "Who decided where the branch lives. asked = the user answered the Step 5b workspace picker; command = a flow that only ever builds worktrees stated it up front; autopilot = resolved to a worktree without asking, because an unattended commit in the user's own checkout is what worktrees prevent. localMode alone cannot say: false is both a chosen worktree and one nothing asked about."
278
278
  },
279
279
  "remoteType": {
280
280
  "type": "string",
@@ -476,15 +476,10 @@
476
476
  "type": "boolean",
477
477
  "default": false
478
478
  },
479
- "onlyDevelop": {
480
- "type": "boolean",
481
- "default": false,
482
- "description": "Short pipeline - phases 1 and 2 are skipped. Set by the Phase 0 Step 7.5 depth picker as of v16.0.0; the key and every reader of it are unchanged. Phase 3 Review still runs (v14.0.0+). Always false in an autopilot run, which never asks the depth question."
483
- },
484
479
  "localMode": {
485
480
  "type": "boolean",
486
481
  "default": false,
487
- "description": "--local flag - no worktree, direct branch in projectRoot."
482
+ "description": "Step 5b answered local - no worktree, direct branch in projectRoot."
488
483
  },
489
484
  "instructionDriven": {
490
485
  "type": "boolean",
@@ -498,7 +493,7 @@
498
493
  "analysis": {
499
494
  "type": ["object", "null"],
500
495
  "additionalProperties": true,
501
- "description": "Phase 1 analysis-document outcome. Absent in Short runs, which produce no document by design.",
496
+ "description": "Phase 1 analysis-document outcome. Phase 1 runs in every mode, so a document is always produced; what varies is how many of its sections the evidence supported.",
502
497
  "properties": {
503
498
  "docStatus": {
504
499
  "type": "string",
@@ -862,6 +857,29 @@
862
857
  }
863
858
  }
864
859
  },
860
+ "threatModel": {
861
+ "type": "object",
862
+ "additionalProperties": true,
863
+ "description": "The run-scoped threat model produced at Phase 3 Step 2.7 (or by /multi-agent:security-review), mirrored here so a resume reuses it instead of re-deriving it. Contract: multi-agent-refs/threat-model.md.",
864
+ "properties": {
865
+ "path": {
866
+ "type": "string",
867
+ "description": "Path to .pipeline/threat-model.md in the worktree."
868
+ },
869
+ "sha": {
870
+ "type": "string",
871
+ "description": "Content hash, so a later step can tell whether the model changed."
872
+ },
873
+ "producedAt": {
874
+ "type": "string",
875
+ "format": "date-time"
876
+ },
877
+ "producedBy": {
878
+ "type": "string",
879
+ "description": "Which step or command wrote it (e.g. 'phase-3-step-2.7', 'security-review')."
880
+ }
881
+ }
882
+ },
865
883
  "reviewIterations": {
866
884
  "type": "array",
867
885
  "items": {
@@ -1530,7 +1548,7 @@
1530
1548
  }
1531
1549
  }
1532
1550
  },
1533
- "description": "Captured in Phase 2 after the build+test gate, because the user test is dropped by every autopilot and --local entry."
1551
+ "description": "Captured in Phase 2 after the build+test gate, because the user test is dropped by every autopilot entry and by any run whose workspace is local."
1534
1552
  },
1535
1553
  "host": {
1536
1554
  "type": ["string", "null"],