@mmerterden/multi-agent-pipeline 19.1.4 → 20.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +123 -0
- package/README.md +19 -36
- package/README.tr.md +18 -35
- package/SECURITY.md +3 -3
- package/docs/adr/0002-instruction-driven-flag.md +6 -5
- package/docs/adr/0005-lazy-phase-docs.md +2 -2
- package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
- package/docs/adr/0009-claude-stack-skills-plugin-only.md +1 -1
- package/docs/adr/0010-own-code-graph.md +5 -4
- package/docs/adr/0011-dormant-ci.md +10 -1
- package/docs/adr/0012-macos-only.md +2 -2
- package/docs/adr/0013-lsp-code-intelligence.md +2 -2
- package/docs/adr/0014-six-phase-consolidation.md +9 -9
- package/docs/adr/0015-one-pipeline-no-depth-answer.md +83 -0
- package/docs/adr/0016-the-run-shape-is-asked-not-typed.md +69 -0
- package/docs/adr/README.md +18 -16
- package/docs/architecture.md +2 -2
- package/docs/ecosystem.md +5 -5
- package/docs/facts.json +7 -9
- package/docs/features.md +4 -5
- package/docs/token-budget-history.md +1 -1
- package/install/_codex-agents.mjs +1 -1
- package/install/_common.mjs +9 -1
- package/install/templates/copilot-instructions.md +7 -16
- package/manifest.json +133 -129
- package/package.json +1 -1
- package/pipeline/agents/code-reviewer.md +2 -2
- package/pipeline/agents/dev-critic.md +5 -5
- package/pipeline/agents/security-auditor.md +80 -72
- package/pipeline/commands/figma-to-swiftui.md +1 -1
- package/pipeline/commands/multi-agent/SKILL.md +7 -9
- package/pipeline/commands/multi-agent/analysis/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/analysis-jira/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/autopilot/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/autopilot-on/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/autopilot-status/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/build-optimize/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/channels/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/create-jira/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/design-check/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/diff-explain/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/forget/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +4 -2
- package/pipeline/commands/multi-agent/help/SKILL.md +23 -27
- package/pipeline/commands/multi-agent/ios-coding-standard/SKILL.md +5 -4
- package/pipeline/commands/multi-agent/issue/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/jira/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/language/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/prune-logs/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/purge/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/resume/SKILL.md +177 -48
- package/pipeline/commands/multi-agent/save/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/scan/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/security-review/SKILL.md +52 -0
- package/pipeline/commands/multi-agent/stack/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/sync/SKILL.md +7 -8
- package/pipeline/commands/multi-agent/test-screenshots/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/uninstall/SKILL.md +2 -0
- package/pipeline/lib/repo-hygiene.sh +1 -1
- package/pipeline/multi-agent-refs/analysis/render.md +1 -1
- package/pipeline/multi-agent-refs/analysis/resolve.md +1 -1
- package/pipeline/multi-agent-refs/analysis/synthesis.md +1 -1
- package/pipeline/multi-agent-refs/analysis-template.md +1 -1
- package/pipeline/multi-agent-refs/component-dispatch.md +5 -13
- package/pipeline/multi-agent-refs/cross-cli-contract.md +14 -15
- package/pipeline/multi-agent-refs/features/external-context-injection.md +2 -0
- package/pipeline/multi-agent-refs/features/review-delta.md +1 -1
- package/pipeline/multi-agent-refs/features/review-multi-repo.md +3 -3
- package/pipeline/multi-agent-refs/features/security-audit.md +55 -0
- package/pipeline/multi-agent-refs/features/skill-conformance.md +1 -1
- package/pipeline/multi-agent-refs/features/visual-evidence.md +2 -1
- package/pipeline/multi-agent-refs/features/worktree-finalize.md +1 -1
- package/pipeline/multi-agent-refs/generate-issue.md +2 -0
- package/pipeline/multi-agent-refs/issue-jira-triad.md +2 -0
- package/pipeline/multi-agent-refs/keychain.md +2 -0
- package/pipeline/multi-agent-refs/knowledge.md +0 -7
- package/pipeline/multi-agent-refs/outside-the-pipeline.md +1 -1
- package/pipeline/multi-agent-refs/payload-contracts.md +1 -1
- package/pipeline/multi-agent-refs/phases/modes.md +33 -109
- package/pipeline/multi-agent-refs/phases/operations.md +2 -0
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +23 -42
- package/pipeline/multi-agent-refs/phases/phase-1-plan.md +7 -18
- package/pipeline/multi-agent-refs/phases/phase-2-dev.md +13 -44
- package/pipeline/multi-agent-refs/phases/phase-3-review.md +28 -37
- package/pipeline/multi-agent-refs/phases/phase-4-commit.md +6 -6
- package/pipeline/multi-agent-refs/phases/phase-5-report.md +3 -3
- package/pipeline/multi-agent-refs/phases.md +9 -11
- package/pipeline/multi-agent-refs/progress-contract.md +1 -1
- package/pipeline/multi-agent-refs/readiness-review.md +2 -0
- package/pipeline/multi-agent-refs/rules.md +1 -1
- package/pipeline/multi-agent-refs/threat-model.md +39 -0
- package/pipeline/multi-agent-refs/tracker-contract.md +9 -40
- package/pipeline/multi-agent-refs/wiki-capture.md +3 -2
- package/pipeline/preferences-template.json +2 -2
- package/pipeline/rules/figma-pipeline.md +1 -1
- package/pipeline/schemas/agent-state.schema.json +28 -10
- package/pipeline/schemas/migrations/prefs-2.7.0-to-2.8.0.mjs +33 -0
- package/pipeline/schemas/phases.json +4 -26
- package/pipeline/schemas/prefs.schema.json +5 -9
- package/pipeline/schemas/reviewer-output.schema.json +99 -2
- package/pipeline/schemas/security-finding.schema.json +144 -0
- package/pipeline/scripts/_stack-routing.mjs +1 -0
- package/pipeline/scripts/cost-table.json +1 -1
- package/pipeline/scripts/gc-abandoned.sh +16 -9
- package/pipeline/scripts/gc-refs.sh +1 -1
- package/pipeline/scripts/gen-mode-dispatch.mjs +11 -41
- package/pipeline/scripts/migrate-prefs.mjs +18 -17
- package/pipeline/scripts/phase-tracker.sh +2 -2
- package/pipeline/scripts/phase0-exit-gate.mjs +1 -1
- package/pipeline/scripts/plan-coverage-gate.mjs +3 -3
- package/pipeline/scripts/render-work-summary.sh +7 -4
- package/pipeline/scripts/run-aggregator.mjs +1 -1
- package/pipeline/scripts/usage-report.mjs +0 -2
- package/pipeline/scripts/worktree-finalize.sh +2 -2
- package/pipeline/skills/.skill-manifest.json +17 -21
- package/pipeline/skills/.skills-index.json +6 -39
- package/pipeline/skills/shared/README.md +5 -8
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +11 -15
- package/pipeline/skills/shared/core/multi-agent-autopilot-status/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +13 -16
- package/pipeline/skills/shared/core/multi-agent-ios-coding-standard/SKILL.md +2 -3
- package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +51 -15
- package/pipeline/skills/shared/core/multi-agent-scan/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-security-review/SKILL.md +29 -0
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +7 -7
- package/pipeline/skills/shared/external/security-review/SKILL.md +64 -0
- package/pipeline/skills/shared/external/security-review/references/owasp-mobile-top10-2024.md +53 -0
- package/pipeline/skills/shared/external/security-review/references/owasp-web-api-top10-2021.md +56 -0
- package/pipeline/skills/skills-index.md +3 -6
- package/pipeline/commands/multi-agent/local/SKILL.md +0 -132
- package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +0 -142
- package/pipeline/commands/multi-agent/resume-local/SKILL.md +0 -114
- package/pipeline/commands/security-review.md +0 -6
- package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +0 -41
- package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +0 -55
- package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +0 -51
|
@@ -104,13 +104,13 @@ Persist the totals as `state.diffRisk` (Phase 4 `risk` section, Phase 5, `run-me
|
|
|
104
104
|
**Gate behavior**: this step is **never blocking**. If risk scoring fails (git error, parse error, validator rejection), continue with no priority hint - reviewers receive the full diff in their default order. Failures are logged via metrics:
|
|
105
105
|
|
|
106
106
|
```bash
|
|
107
|
-
[ -z "$RISK_JSON" ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
107
|
+
[ -z "$RISK_JSON" ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.diff_risk_skipped reason=$REASON
|
|
108
108
|
```
|
|
109
109
|
|
|
110
110
|
On success, emit a single summary metric:
|
|
111
111
|
|
|
112
112
|
```bash
|
|
113
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
113
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.diff_risk \
|
|
114
114
|
top_files=$(jq '.files | length' <<< "$RISK_JSON") \
|
|
115
115
|
max_score=$(jq '.totals.max_score' <<< "$RISK_JSON") \
|
|
116
116
|
loc_added=$(jq '.totals.loc_added' <<< "$RISK_JSON") \
|
|
@@ -127,7 +127,7 @@ Step 1.75 uses `test_lines_removed` as an advisory hint only - too weak for wh
|
|
|
127
127
|
```bash
|
|
128
128
|
TEST_INTEGRITY_JSON=$(printf '%s' "$RISK_FULL" | node $HOME/.claude/scripts/test-integrity-gate.mjs 2>/dev/null || echo "")
|
|
129
129
|
TI_COUNT=$(jq -r '.count // 0' <<< "${TEST_INTEGRITY_JSON:-{\}}" 2>/dev/null || echo 0)
|
|
130
|
-
[ "$TI_COUNT" -gt 0 ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
130
|
+
[ "$TI_COUNT" -gt 0 ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.test_integrity findings="$TI_COUNT"
|
|
131
131
|
```
|
|
132
132
|
|
|
133
133
|
`findings[]` are reviewer-shaped (`test_integrity`, `blocking`), so they merge into the reviewer findings at Step 3.0 and need no triage-prompt or `validate-triage.mjs` change. Triage keeps each blocking unless the removal is justified per the immutable-test rule (spec changed AND commit body names the test) → `deferred[]`.
|
|
@@ -142,7 +142,7 @@ On a trivial diff every reviewer agrees and the extra models plus triage are pai
|
|
|
142
142
|
SCOPE_JSON=$(printf '%s' "$RISK_FULL" | node $HOME/.claude/scripts/review-scope.mjs 2>/dev/null \
|
|
143
143
|
|| echo '{"scope":"full","reason":"no risk report - failing safe"}')
|
|
144
144
|
REVIEW_SCOPE=$(jq -r '.scope // "full"' <<< "$SCOPE_JSON")
|
|
145
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
145
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.scope scope="$REVIEW_SCOPE"
|
|
146
146
|
```
|
|
147
147
|
|
|
148
148
|
`single` (Reviewer 1 only) requires **all** of: churn <= 20 lines, `totals.max_score` < 3.0, and no `security_path` / `migration` / `public_api` / `no_test_change` / `test_lines_removed` on any file. Anything else → `full`.
|
|
@@ -159,7 +159,7 @@ node $HOME/.claude/scripts/skill-conformance.mjs \
|
|
|
159
159
|
--repo "$WORKTREE" --out "$WORKTREE/.pipeline/criteria-manifest.json"
|
|
160
160
|
CRIT_RC=$?
|
|
161
161
|
CRITERIA=$(cat "$WORKTREE/.pipeline/criteria-manifest.json" 2>/dev/null || echo '{}')
|
|
162
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
162
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.criteria \
|
|
163
163
|
rules="$(jq -r '.selectedRuleCount // 0' <<< "$CRITERIA")" \
|
|
164
164
|
ledger="$(jq -r '.ledger.source // "derived"' <<< "$CRITERIA")"
|
|
165
165
|
[ "$CRIT_RC" != "0" ] && HALT "criteria could not be resolved (rc=$CRIT_RC) - see resolutionFailure / unparseableRegistries"
|
|
@@ -175,7 +175,7 @@ Output conforms to `$HOME/.claude/schemas/criteria-manifest.schema.json`. The fo
|
|
|
175
175
|
|
|
176
176
|
#### Step 1.8 - Figma visual-fidelity context (when task carries a Figma reference)
|
|
177
177
|
|
|
178
|
-
**
|
|
178
|
+
**Missing inputs.** A section the evidence did not support is absent from the analysis document, so three inputs this phase was written around can be missing. Substitute them and RECORD the substitution - a step that could not run and one that passed must not read the same, or the completeness claim cannot be checked:
|
|
179
179
|
|
|
180
180
|
| Absent input | Substitute |
|
|
181
181
|
|---|---|
|
|
@@ -260,7 +260,7 @@ Each reviewer inherits the `code-reviewer` agent's focus areas (Security, Archit
|
|
|
260
260
|
| Python | `ai-backend-toolkit:api-security-best-practices` | `ai-backend-toolkit:fastapi-pro` | `ai-backend-toolkit:python-patterns` |
|
|
261
261
|
| Node.js | `ai-backend-toolkit:api-security-best-practices` | `ai-backend-toolkit:nodejs-backend-patterns` | `ai-frontend-toolkit:typescript-patterns` |
|
|
262
262
|
| Docker | `ai-backend-toolkit:docker-expert` | `ai-backend-toolkit:docker-expert` | `ai-backend-toolkit:ci-cd-pipelines` |
|
|
263
|
-
| Generic | `security-review` | `ai-backend-toolkit:clean-code` | `ai-backend-toolkit:clean-code` |
|
|
263
|
+
| Generic | `ai-common-toolkit:security-review` | `ai-backend-toolkit:clean-code` | `ai-backend-toolkit:clean-code` |
|
|
264
264
|
|
|
265
265
|
##### 2.1 Previous-round findings (iteration >= 2) and 2.2 scope self-check (every iteration)
|
|
266
266
|
|
|
@@ -298,6 +298,10 @@ Skip only when the diff has no UI change. Record the outcome in
|
|
|
298
298
|
|
|
299
299
|
Step 2 produces N reviewer-output objects (one per dispatched reviewer), each conforming to `$HOME/.claude/schemas/reviewer-output.schema.json`. They are persisted to `state.reviewIterations[<iteration>].reviewers[]` and consumed by Step 3 (Fable triage) - never by Phase 4 directly. The triage step (below) is the producer of the only review artifact Phase 4 reads, conforming to `$HOME/.claude/schemas/triage-output.schema.json`.
|
|
300
300
|
|
|
301
|
+
#### Step 2.7 - Security audit (conditional, produces reviewer-shaped findings)
|
|
302
|
+
|
|
303
|
+
Runs when Step 1.75 scored `security_path`, on a release branch, or when `/multi-agent:security-review` dispatches here. The `security-auditor` returns reviewer `findings[]`, each with a `security` envelope (`security-finding.schema.json`) and severity from the CVSS band, held in `$SECURITY_AUDIT_JSON` for the merge so a `blocking` one reaches triage and blocks Phase 4. Mechanics: `~/.claude/multi-agent-refs/features/security-audit.md`.
|
|
304
|
+
|
|
301
305
|
**Subagent return format** - each reviewer returns JSON conforming to `$HOME/.claude/schemas/reviewer-output.schema.json`:
|
|
302
306
|
|
|
303
307
|
```json
|
|
@@ -371,14 +375,14 @@ ANON=$(jq -n --argjson r "$REVIEWERS_JSON" --arg t "$TASK_ID" --argjson i "$ITER
|
|
|
371
375
|
|
|
372
376
|
`$REVIEWERS_JSON` is `state.reviewIterations[i].reviewers`. Findings come back with `foundBy: "Source A|B|C"` and every identity key removed. Persist the map to `state.reviewIterations[i].anonymizationMap` for Phase 5 per-reviewer telemetry, and **never put the map in a prompt**.
|
|
373
377
|
|
|
374
|
-
Then append the Step 1.76 test-integrity findings, so they are adjudicated rather than never seen:
|
|
378
|
+
Then append the Step 1.76 test-integrity and Step 2.7 security-audit findings, so they are adjudicated rather than never seen:
|
|
375
379
|
|
|
376
380
|
```bash
|
|
377
|
-
MERGED=$(jq -s '.[0] + (.[1].findings // [])' \
|
|
378
|
-
<(printf '%s' "$ANON") <(printf '%s' "${TEST_INTEGRITY_JSON:-{\}}"))
|
|
381
|
+
MERGED=$(jq -s '.[0] + (.[1].findings // []) + (.[2].findings // [])' \
|
|
382
|
+
<(printf '%s' "$ANON") <(printf '%s' "${TEST_INTEGRITY_JSON:-{\}}") <(printf '%s' "${SECURITY_AUDIT_JSON:-{\}}"))
|
|
379
383
|
```
|
|
380
384
|
|
|
381
|
-
Deterministic findings keep `tag: test_integrity` and
|
|
385
|
+
Deterministic findings keep `tag: test_integrity` and no `foundBy`: a reviewer finding may hallucinate, a gate finding is a fact. Security-audit findings carry their `security` envelope and `foundBy: "security-auditor"`; an empty `$SECURITY_AUDIT_JSON` contributes nothing.
|
|
382
386
|
|
|
383
387
|
##### 3.1 Short-circuit: no findings
|
|
384
388
|
|
|
@@ -426,8 +430,8 @@ Exit 2 (empty ledger) skips silently. The `## Rejected review preferences` secti
|
|
|
426
430
|
**Recall telemetry.** Log what was injected, then what triage cited. Zero cited is a legitimate answer; `learning-curve.mjs` trends the ratio:
|
|
427
431
|
|
|
428
432
|
```bash
|
|
429
|
-
bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
430
|
-
bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
433
|
+
bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 memory.injected kind=prior-art rows=$N
|
|
434
|
+
bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 memory.hit rows=$CITED_COUNT
|
|
431
435
|
```
|
|
432
436
|
|
|
433
437
|
**Bulky payloads (opt-in via `prefs.global.contextOffload.enabled`).** Test output and whole-file diffs go through the offload filter, which leaves a `[[ref:<node_id>]]` line plus the tail in context and the full text under `.multi-agent/refs/`. Read that file when the tail is not enough; with the pref off it is a pass-through.
|
|
@@ -513,7 +517,7 @@ One `review.reviewer_call` per dispatched reviewer, one `review.triage_call`, on
|
|
|
513
517
|
```bash
|
|
514
518
|
M=$HOME/.claude/scripts/log-metric.sh
|
|
515
519
|
emit() { # $1=event $2=model $3=duration $4=tokens_in $5=tokens_out
|
|
516
|
-
LOG_METRIC_FORWARD_TO_TRACKER=1 bash "$M" "$TASK_ID"
|
|
520
|
+
LOG_METRIC_FORWARD_TO_TRACKER=1 bash "$M" "$TASK_ID" 3 "$1" \
|
|
517
521
|
model="$2" duration_ms="$3" tokens_in="$4" tokens_out="$5"
|
|
518
522
|
}
|
|
519
523
|
emit review.reviewer_call fable "$R1_DURATION" "$R1_IN" "$R1_OUT" # opus on Copilot CLI
|
|
@@ -525,7 +529,7 @@ else
|
|
|
525
529
|
fi
|
|
526
530
|
emit review.reviewer_call sonnet "$SONNET_DURATION" "$SONNET_IN" "$SONNET_OUT"
|
|
527
531
|
emit review.triage_call fable "$TRIAGE_DURATION" "$TRIAGE_IN" "$TRIAGE_OUT"
|
|
528
|
-
bash "$M" "$TASK_ID"
|
|
532
|
+
bash "$M" "$TASK_ID" 3 review.completed raw_count=$RAW accepted=$ACC \
|
|
529
533
|
deferred=$DEF rejected=$REJ approved=$APPROVED duration_ms=$DURATION
|
|
530
534
|
```
|
|
531
535
|
|
|
@@ -614,7 +618,7 @@ Log: "Phase 3: Review - raw={N1+N2+N3} accepted={Na} deferred={Nd} rejected={N
|
|
|
614
618
|
## Token telemetry - invoke after every LLM call
|
|
615
619
|
|
|
616
620
|
```bash
|
|
617
|
-
bash $HOME/.claude/scripts/phase-tracker.sh tokens
|
|
621
|
+
bash $HOME/.claude/scripts/phase-tracker.sh tokens 3 <input_count> <output_count> [cached_count]
|
|
618
622
|
```
|
|
619
623
|
|
|
620
624
|
The optional 4th `cached_count` is the prompt-cache-read token count when the host reports it (Anthropic `cache_read_input_tokens`); it defaults to 0 and is priced at the cheaper `cacheReadPerMtok` rate in the Phase 5 cost ledger. The tracker accumulates the totals additively, so multiple calls in the same phase compound. The render output then shows live cost on the active phase tile (e.g. `Phase 2 Dev 2m 14s · 12.4k tok`). This satisfies the contract in `$HOME/.claude/multi-agent-refs/tracker-contract.md` and the `smoke-tracker-tokens-invocation.sh` enforcement gate. Skipping this call is the #1 cause of "I can't see how much it cost" complaints.
|
|
@@ -623,15 +627,12 @@ Contract and rationale: `progress-contract.md` -> Token telemetry forwarding.
|
|
|
623
627
|
|
|
624
628
|
---
|
|
625
629
|
|
|
626
|
-
## User test
|
|
627
|
-
|
|
628
|
-
Optional test gate, now the tail of Review rather than a phase of its own: it
|
|
629
|
-
judges work that already exists, which is what Review does. Needs an interactive
|
|
630
|
-
prompt AND a worktree checkout, so it is skipped where either is missing. If
|
|
631
|
-
issues are found, the run returns to Phase 1 Dev.
|
|
630
|
+
## User test
|
|
632
631
|
|
|
632
|
+
The tail of Review rather than a phase of its own: it judges work that already
|
|
633
|
+
exists, which is what Review does.
|
|
633
634
|
|
|
634
|
-
> **TLDR** - Optional test gate. Offers to boot the simulator/emulator (UI Bug Hunter) or hand off to the user for manual QA. Needs an interactive prompt AND a worktree checkout, so it
|
|
635
|
+
> **TLDR** - Optional test gate. Offers to boot the simulator/emulator (UI Bug Hunter) or hand off to the user for manual QA. Needs an interactive prompt AND a worktree checkout, so it runs on an attended `/multi-agent` whose workspace is a worktree, and is skipped on `autopilot` or when the user chose to work locally. If issues found, loops back to Phase 2.
|
|
635
636
|
|
|
636
637
|
<!-- progress-contract: applied -->
|
|
637
638
|
Progress emission per `$HOME/.claude/multi-agent-refs/progress-contract.md` - lines for local-test prompt render, user-answer capture, repo checkout (if selected).
|
|
@@ -682,7 +683,7 @@ fi
|
|
|
682
683
|
**Telemetry**:
|
|
683
684
|
|
|
684
685
|
```bash
|
|
685
|
-
LOG_METRIC_FORWARD_TO_TRACKER=0 $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
686
|
+
LOG_METRIC_FORWARD_TO_TRACKER=0 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 test_gap.scanned \
|
|
686
687
|
stack=$SCAN_STACK \
|
|
687
688
|
sources=$(jq '.totals.sourcesScanned' <<< "$GAP_JSON") \
|
|
688
689
|
gaps=$(jq '.totals.gapCount' <<< "$GAP_JSON")
|
|
@@ -702,7 +703,7 @@ Figma evidence (tier=<n>):
|
|
|
702
703
|
|
|
703
704
|
Tier 1 / Tier 2 records print `screenshotUrl` from the captured evidence (Tier 2 URLs expire after 30 days, re-fetch on the spot if needed). Tier 3 records print the local path to the user-attached screenshot. The block is informational; it never blocks the prompt.
|
|
704
705
|
|
|
705
|
-
1. Ask with a native `AskUserQuestion` picker (never a typed y/N prompt). The options MUST make the local-checkout side effect explicit - testing removes the worktree and checks the branch out into the main repo:
|
|
706
|
+
1. Ask with a native `AskUserQuestion` picker (never a typed y/N prompt), per `$HOME/.claude/multi-agent-refs/picker-contract.md`. The options MUST make the local-checkout side effect explicit - testing removes the worktree and checks the branch out into the main repo:
|
|
706
707
|
- `question`: "Check out locally to test now?" (rendered in `outputLanguage`)
|
|
707
708
|
- `header`: "Test" (English, <=12 chars)
|
|
708
709
|
- `options`:
|
|
@@ -733,7 +734,7 @@ Tier 1 / Tier 2 records print `screenshotUrl` from the captured evidence (Tier 2
|
|
|
733
734
|
bash $HOME/.claude/scripts/phase-tracker.sh now 3 "awaiting local test (user)"
|
|
734
735
|
bash $HOME/.claude/scripts/phase-tracker.sh render
|
|
735
736
|
```
|
|
736
|
-
The waiting state persists in `tracker-state.json` across the handoff; `/multi-agent:resume
|
|
737
|
+
The waiting state persists in `tracker-state.json` across the handoff; `/multi-agent:resume` and `/multi-agent:manual-test` CONTINUE this state file and never re-init it (`$HOME/.claude/multi-agent-refs/tracker-contract.md` "Continuation runs").
|
|
737
738
|
|
|
738
739
|
**"ok" is a structured result, not a word.** Before "ok" is accepted, the run writes `$WORKTREE/.pipeline/manual-test.json`: one entry per acceptance criterion, the criteria taken from the analysis doc test plan (Section 15 / 20), the plan tasks, and the user's own words in the reply. Every criterion records what was seen; a criterion that was not tried says so with a reason.
|
|
739
740
|
```json
|
|
@@ -791,22 +792,12 @@ Results included in Phase 5 report. MCP tools preferred when available - conci
|
|
|
791
792
|
|
|
792
793
|
**Snapshot regression flow (optional):** when the task changes a stable component, capture a screenshot before the change (baseline) and after (current), then call `ios_visual_diff({baseline, current, max_diff_pct: 1.0})`. Threshold can be relaxed for animated / non-deterministic regions - keep `max_diff_pct ≤ 1.0` for static layouts.
|
|
793
794
|
|
|
794
|
-
#### Security Audit (store-readiness)
|
|
795
|
-
|
|
796
|
-
When the task touches authentication, keychain, network, or is scheduled for an imminent release, launch the `security-auditor` subagent to run an OWASP Mobile Top 10 pass plus App Store / Play Store compliance checks:
|
|
797
|
-
|
|
798
|
-
```
|
|
799
|
-
Agent(subagent_type: "security-auditor", prompt: "<diff + context>")
|
|
800
|
-
```
|
|
801
|
-
|
|
802
|
-
Returns severity-tagged findings (Critical / High / Medium). Critical items block Phase 4 just like Phase 3 blockers; High items are logged and surfaced in Phase 5 report. Skipped by default - opt-in for release branches or on explicit `/multi-agent "<task>" --audit` flag.
|
|
803
|
-
|
|
804
795
|
#### Telemetry - token forwarding
|
|
805
796
|
|
|
806
797
|
When the security-auditor or any other Phase 3 sub-agent runs, forward its token totals so Phase 5's Cost Breakdown captures Phase 3:
|
|
807
798
|
|
|
808
799
|
```bash
|
|
809
|
-
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
800
|
+
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 audit.completed \
|
|
810
801
|
model=opus tokens_in=$IN tokens_out=$OUT duration_ms=$DUR
|
|
811
802
|
```
|
|
812
803
|
|
|
@@ -354,19 +354,19 @@ A task ending with one or more `pushStatus === "skipped"` repos does NOT bump `r
|
|
|
354
354
|
|
|
355
355
|
```bash
|
|
356
356
|
for proj in $(jq -r '.projects[].name' "$STATE_FILE"); do
|
|
357
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
358
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
359
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
357
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 commit.created repo=$proj sha=$SHA
|
|
358
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 push.attempted repo=$proj attempts=$N status=$STATUS
|
|
359
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 pr.opened repo=$proj url=$URL number=$N
|
|
360
360
|
done
|
|
361
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
361
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 multi_repo.completed repos=$REPOS skipped=$SKIPPED
|
|
362
362
|
```
|
|
363
363
|
|
|
364
364
|
**Token forwarding:** the commit-message and PR-body generators run on a model. Forward those calls into the tracker so Phase 5's Cost Breakdown captures Phase 4:
|
|
365
365
|
|
|
366
366
|
```bash
|
|
367
|
-
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
367
|
+
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 commit.message_generated \
|
|
368
368
|
model=<sonnet|opus> tokens_in=$IN tokens_out=$OUT duration_ms=$DUR
|
|
369
|
-
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
369
|
+
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 pr.body_generated \
|
|
370
370
|
model=<sonnet|opus> tokens_in=$IN tokens_out=$OUT duration_ms=$DUR
|
|
371
371
|
```
|
|
372
372
|
|
|
@@ -25,7 +25,7 @@ On entry: `phase-tracker.sh sub 5 <N> "<name>" in_progress`. On exit: `completed
|
|
|
25
25
|
Phase 5 is the single exception to the autopilot zero-interaction rule: every mode, Full or Short, attended or autopilot, pauses at the channels multi-select menu. Full contract in `$HOME/.claude/multi-agent-refs/phases/modes.md`:
|
|
26
26
|
|
|
27
27
|
- **30-min timeout** - if user does not respond, session ends cleanly. External delivery is aborted (no silent apply of defaults - prevents accidental Jira comments / Confluence pages). Internal capture (Steps 2 + 3 below) STILL runs so `agent-log.md` + knowledge base are persisted.
|
|
28
|
-
- **Resumable** - state written as `{status: "awaiting_input", phase:
|
|
28
|
+
- **Resumable** - state written as `{status: "awaiting_input", phase: 5, waitingFor: "user-channels-choice", channelsInput: <state-bundle>}`. User can `/multi-agent:resume <task-id>` any time later; channels menu re-opens with same inputs.
|
|
29
29
|
- **Timeout log line:** `Phase 5: channels menu timeout (30 min) - session ended, resume with /multi-agent:resume {taskId}`.
|
|
30
30
|
|
|
31
31
|
---
|
|
@@ -188,9 +188,9 @@ Skipped sections: when `planTodos.enabled` is false or no `plan.todos[]` was emi
|
|
|
188
188
|
**Telemetry emission** (mandatory): forward the phase's own LLM spend (humanizer + report compose calls) to the tracker, then emit the final event:
|
|
189
189
|
|
|
190
190
|
```bash
|
|
191
|
-
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
191
|
+
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 5 report.compose \
|
|
192
192
|
model=$REPORT_MODEL tokens_in=$R_IN tokens_out=$R_OUT duration_ms=$R_DUR
|
|
193
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
193
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 5 task.completed \
|
|
194
194
|
phases=$PHASE_COUNT review_cycles=$CYCLES lang=$PROMPT_LANG \
|
|
195
195
|
channels_pr=$PR_STATUS channels_jira=$JIRA_STATUS \
|
|
196
196
|
channels_confluence=$CONF_STATUS channels_wiki=$WIKI_STATUS \
|
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
|
|
16
16
|
| Phase | File |
|
|
17
17
|
| --------------------------------- | -------------------------------------------------------------------- |
|
|
18
|
-
| Modes (
|
|
18
|
+
| Modes (autopilot, analysis) | `$HOME/.claude/multi-agent-refs/phases/modes.md` |
|
|
19
19
|
| Operations (kill, purge, resume) | `$HOME/.claude/multi-agent-refs/phases/operations.md` |
|
|
20
20
|
| Phase 0: Init | `$HOME/.claude/multi-agent-refs/phases/phase-0-init.md` |
|
|
21
21
|
| Phase 1: Plan | `$HOME/.claude/multi-agent-refs/phases/phase-1-plan.md` |
|
|
@@ -28,11 +28,11 @@
|
|
|
28
28
|
## Pipeline Flow
|
|
29
29
|
|
|
30
30
|
```
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
31
|
+
0-Init -> 1-Plan -> 2-Dev -> 3-Review -> 4-Commit -> 5-Report
|
|
32
|
+
Local: The same set with no worktree - Phase 0 Step 5b answered local, so
|
|
33
|
+
work happens directly on a branch in the project root
|
|
34
34
|
|
|
35
|
-
|
|
35
|
+
One pipeline: every mode runs its whole phase set, and no answer during the run adds or removes a phase.
|
|
36
36
|
```
|
|
37
37
|
|
|
38
38
|
## Phase entry - pending steer (every phase, every mode)
|
|
@@ -99,10 +99,8 @@ done
|
|
|
99
99
|
|
|
100
100
|
This produces an initial card stack printed by both CLIs.
|
|
101
101
|
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
the rest once the answer lands, and call `tiles --new` for the second batch. Full
|
|
105
|
-
contract: `tracker-contract.md`, "Deferred registration".
|
|
102
|
+
No mode is an exception: the phase set is a property of the command, so every
|
|
103
|
+
tile is created in this one batch. Full contract: `tracker-contract.md`.
|
|
106
104
|
|
|
107
105
|
### Tracker updates (every phase boundary)
|
|
108
106
|
|
|
@@ -114,7 +112,7 @@ $HOME/.claude/scripts/phase-tracker.sh update <N> in_progress # phase starts
|
|
|
114
112
|
$HOME/.claude/scripts/phase-tracker.sh update <N> completed # phase ends OK
|
|
115
113
|
# or:
|
|
116
114
|
$HOME/.claude/scripts/phase-tracker.sh update <N> failed # phase failed
|
|
117
|
-
$HOME/.claude/scripts/phase-tracker.sh update <N> skipped # e.g.
|
|
115
|
+
$HOME/.claude/scripts/phase-tracker.sh update <N> skipped # e.g. 2 in analysis mode
|
|
118
116
|
```
|
|
119
117
|
|
|
120
118
|
After every LLM call (counts are additive; skipping this is why runs end with durations but no cost - nothing reconstructs spend afterwards):
|
|
@@ -165,7 +163,7 @@ TaskUpdate({ taskId: <saved>, status: "completed" })
|
|
|
165
163
|
bash phase-tracker.sh update <N> completed
|
|
166
164
|
```
|
|
167
165
|
|
|
168
|
-
A phase outside the command's set gets no TaskCreate at all
|
|
166
|
+
A phase outside the command's set gets no TaskCreate at all, and the set is known before the tracker boots.
|
|
169
167
|
|
|
170
168
|
**(strict) TaskCreate ordering**: All TaskCreate calls MUST fire in strict phase-number order BEFORE any TaskUpdate is applied. The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks (e.g. `1 ✓ · 3 ✓ · 4 ✓ · 0 ▶ · 2 ☐`) even when the underlying state is correct. Pre-marking phases as completed/skipped before Phase 0 starts is FORBIDDEN - register the tile in order with default `pending` status, then flip status via TaskUpdate when the phase actually short-circuits. Full contract in `$HOME/.claude/multi-agent-refs/tracker-contract.md` section "TaskCreate ordering (strict)".
|
|
171
169
|
|
|
@@ -69,7 +69,7 @@ Emit a progress line **at least** at every one of these moments:
|
|
|
69
69
|
|
|
70
70
|
### Phase 2 (Planning)
|
|
71
71
|
- plan draft start, plan render, user-approval prompt.
|
|
72
|
-
- **v5.3.0 Plan Approval Gate (
|
|
72
|
+
- **v5.3.0 Plan Approval Gate (interactive only - autopilot may not ask):**
|
|
73
73
|
- `clarification-ask` per round - orchestrator writes structured questions when Phase 1 flagged ambiguity (missing acceptance criteria, no Figma/endpoint link, vague language, parent-story scope drift)
|
|
74
74
|
- `clarification-answer` per round - user reply captured into `state.phases["2"].clarificationAnswers`
|
|
75
75
|
- `plan-edit-request` per free-text edit - user-supplied revision instruction captured into `state.phases["2"].planEditRequests`
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# Readiness Review (shared flow for review-jira + review-issue)
|
|
2
2
|
|
|
3
|
+
> **Pickers follow** `$HOME/.claude/multi-agent-refs/picker-contract.md`: never a one-option call, and branch on the option selected, not on its text.
|
|
4
|
+
|
|
3
5
|
Assess whether a tracker item (Jira issue or GitHub issue) is READY to hand to the multi-agent pipeline, list the concrete gaps, and - after confirmation - post them back as a comment on the item so the reporter can fix it. Read-only on code: no worktree, no branch, no commits, no dev chaining. This is the inverse of `/multi-agent:create-jira` (which authors a well-formed item); here we grade an existing one.
|
|
4
6
|
|
|
5
7
|
Both `/multi-agent:review-jira` and `/multi-agent:review-issue` execute this flow; only the provider (fetch + comment endpoint + picker) differs.
|
|
@@ -44,7 +44,7 @@ This is the single source of truth. When a contributor or model is unsure where
|
|
|
44
44
|
3. `AskUserQuestion` `question`, `options[].label` and `options[].description` all follow `OUTPUT_LANG`. Only `header` stays English: a <=12-char chip Turkish overflows. Callers branch on which option was picked, never on its literal text, and pass `default` / `ASK_CHOICE_DEFAULT` as a 1-based index. The host's own **Other** row is always English. Caller rules: `picker-contract.md`.
|
|
45
45
|
4. Always English regardless of either axis: commit messages, PR titles, branch names, code identifiers, agent-state.json values, agent-log.md, reviewer/triage system prompts.
|
|
46
46
|
|
|
47
|
-
**Failure mode this prevents.** Entering `/multi-agent`, `/multi-agent:autopilot`,
|
|
47
|
+
**Failure mode this prevents.** Entering `/multi-agent`, `/multi-agent:autopilot`, etc. and switching the assistant's conversational text or picker question copy to English while `outputLanguage="tr"` is set. The user sees a half-English half-Turkish dialogue, flagged as a pipeline bug, not a stylistic choice.
|
|
48
48
|
|
|
49
49
|
## Code & Commit Rules
|
|
50
50
|
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# Threat model contract
|
|
2
|
+
|
|
3
|
+
The run-scoped threat model is the frame every security finding is calibrated against. It is produced by the security-auditor at Phase 3 Step 2.7 (or by `/multi-agent:security-review` running standalone), read by any later step that revisits security, and mirrored to `state.threatModel.path` so a resume re-uses it instead of re-deriving it.
|
|
4
|
+
|
|
5
|
+
## Where it lives
|
|
6
|
+
|
|
7
|
+
`.pipeline/threat-model.md` in the run's worktree, keyed to repo + branch. One per run. If a fresh one already exists for this repo+branch (a prior step or a Phase 1 pass wrote it), read it; do not overwrite. If it is absent, produce it before emitting any finding.
|
|
8
|
+
|
|
9
|
+
Mirror the resolved path and a content hash to `state.threatModel = { path, sha, producedAt, producedBy }` so resume and Phase 5 can find it without re-reading the tree.
|
|
10
|
+
|
|
11
|
+
## The four sections (all required, in order)
|
|
12
|
+
|
|
13
|
+
A threat model with a missing section is not a threat model; the auditor treats a missing section as "produce it," not "skip it."
|
|
14
|
+
|
|
15
|
+
### 1. Attacker
|
|
16
|
+
|
|
17
|
+
The one realistic adversary for THIS change. Name it concretely: an unauthenticated internet caller, an authenticated low-privilege user, a malicious or compromised dependency, a co-located app on the device, a user with physical access. Not "attackers" in the abstract - the specific actor whose capability makes this diff interesting. If the diff has no plausible attacker, say so; that is a valid, short threat model and most findings then cap at `suggestion`.
|
|
18
|
+
|
|
19
|
+
### 2. Trust boundaries
|
|
20
|
+
|
|
21
|
+
Where untrusted data crosses into trusted code within the changed surface: a request body or query parameter, a deep link or universal link, a WebView `postMessage`, a file the app did not write, an environment variable an attacker can set, a response from a third-party service treated as safe. List the boundaries the diff touches, each with the `file:line` where the crossing happens.
|
|
22
|
+
|
|
23
|
+
### 3. Attack surface
|
|
24
|
+
|
|
25
|
+
What this diff actually added or touched: a new endpoint, a new query or ORM call, a new deserialization, a new permission, a new dependency, a new crypto usage, a new storage write. A finding outside this surface is out of scope unless the diff made it reachable - and if it did, say how. This section is what keeps the audit anchored to the change instead of drifting into a whole-repo review.
|
|
26
|
+
|
|
27
|
+
### 4. Severity calibration
|
|
28
|
+
|
|
29
|
+
The assumption each severity rests on, stated so a reader disputes the assumption rather than the number. "Critical assumes this route is unauthenticated in production; if it is admin-only the same finding is medium." "High assumes the secret reaches a log that ships off-device." Every `blocking` finding must trace to an assumption named here; a blocker resting on an unstated assumption is miscalibrated.
|
|
30
|
+
|
|
31
|
+
## Shape
|
|
32
|
+
|
|
33
|
+
Plain Markdown, four `##` sections with those names, human-readable. It is evidence for a person and context for the auditor, not a machine artifact - no schema. Keep it short: a page, not a report. Cite `file:line` where a boundary or surface item has one.
|
|
34
|
+
|
|
35
|
+
## What it is not
|
|
36
|
+
|
|
37
|
+
- Not a whole-repo model. It is scoped to the diff under review.
|
|
38
|
+
- Not a dynamic test plan. This is a static, read-only posture: no live target, no payloads, no exploitation. `fixVerification` on a finding names the empirical check a human or a later dynamic pass would run; the threat model does not run it.
|
|
39
|
+
- Not embedded in the analysis document. The security-review command must run standalone, so the model lives in its own file and is produced on demand.
|
|
@@ -90,11 +90,11 @@ Phases by mode:
|
|
|
90
90
|
| Mode | Phases |
|
|
91
91
|
|---|---|
|
|
92
92
|
| `/multi-agent` | 0,1,2,3,4,5 |
|
|
93
|
-
|
|
|
94
|
-
| `/multi-agent:autopilot
|
|
93
|
+
| answered local at Step 5b | 0,1,2,3,4,5 (the user test needs a worktree checkout and local has none, so that STEP inside Review is skipped - the phase is not) |
|
|
94
|
+
| `/multi-agent:autopilot` | 0,1,2,3,4,5 (autopilot drops the interactive user test inside Review) |
|
|
95
95
|
| `/multi-agent:analysis` | 0,1,3,4,5 (no code is written, so Dev is not in the set) |
|
|
96
96
|
|
|
97
|
-
|
|
97
|
+
Every entry registers its whole set in one batch: the set is a property of the command, known before the tracker boots.
|
|
98
98
|
|
|
99
99
|
Register each phase:
|
|
100
100
|
|
|
@@ -171,7 +171,7 @@ report's cost-unavailable list.
|
|
|
171
171
|
|
|
172
172
|
> **All TaskCreate calls for the active mode's phase set MUST fire in strict phase-number order BEFORE any TaskUpdate is applied. No "pre-mark skipped phases as completed before Phase 0" reasoning is permitted - even when the agent knows in advance that a phase will be skipped.**
|
|
173
173
|
|
|
174
|
-
Why: an agent reasoning "
|
|
174
|
+
Why: an agent reasoning "analysis mode skips Dev - let me TaskCreate it as completed first" produces tile IDs `1, 2, 3, ...` for phases 1/2/4, then the Phase 0 tile gets ID `4` and visually drops below them. The user sees `1, 2, 4 ✓ · 0 ▶ · 3 ☐ · ...` instead of `0 ▶ · 1 ✓ · 2 ✓ · 3 ☐ · 4 ✓ · ...`.
|
|
175
175
|
|
|
176
176
|
The correct sequence is **always**:
|
|
177
177
|
|
|
@@ -197,44 +197,13 @@ Mode-specific phase sets:
|
|
|
197
197
|
|
|
198
198
|
| Mode | TaskCreate set (in order) |
|
|
199
199
|
|---|---|
|
|
200
|
-
| `/multi-agent` | 0
|
|
201
|
-
|
|
|
202
|
-
| `:autopilot
|
|
200
|
+
| `/multi-agent` | 0 → 1 → 2 → 3 → 4 → 5 (6 phases) |
|
|
201
|
+
| answered local | the same set; nothing is dropped, because the user test that needs a worktree checkout is a step inside Review rather than a phase of its own |
|
|
202
|
+
| `:autopilot` | 0 → 1 → 2 → 3 → 4 → 5 (6 phases - the user test is inside Review now, so no phase is dropped) |
|
|
203
203
|
| `:analysis` | 0 → 1 → 3 → 4 → 5 (5 phases - no code is written, so Dev is not in the set) |
|
|
204
204
|
|
|
205
205
|
A phase outside the mode's set gets no TaskCreate at all; the `[SKIPPED]` pattern applies only to a phase that IS in the set and short-circuits at runtime. Phase 3 is in every mode's set as of v14.0.0. The authoritative per-mode set is the `for p in ...` init block in each mode's own entry doc, generated by `gen-mode-dispatch.mjs`; this table mirrors those blocks.
|
|
206
206
|
|
|
207
|
-
#### Deferred registration - the depth picker
|
|
208
|
-
|
|
209
|
-
`/multi-agent` and `:local` cannot know their phase set at Step -1. Depth decides it, and the depth picker cannot run before Step 7.5: its recommendation needs `taskType`, which needs the fetched issue and the branch.
|
|
210
|
-
|
|
211
|
-
Until v17.5.0 they registered all eight anyway and flipped 1 and 2 to `skipped` at 7.5. That put a widget reading "8 tasks, 7 open - Phase 1 Plan, Phase 1 Plan, ..." on screen *beside* the question asking whether to run Analysis and Planning at all, and a Short answer then contradicted a list the user had just been shown. The widget was asserting a shape the run had not chosen.
|
|
212
|
-
|
|
213
|
-
So registration splits at the moment the shape is known:
|
|
214
|
-
|
|
215
|
-
```text
|
|
216
|
-
# Step -1, first thing in the run: Phase 0 only. It is the one phase that is
|
|
217
|
-
# certain, and the run is never silent while Phase 0 does its work.
|
|
218
|
-
bash $HOME/.claude/scripts/phase-tracker.sh add 0 Init
|
|
219
|
-
bash $HOME/.claude/scripts/phase-tracker.sh tiles # -> TaskCreate(Phase 0)
|
|
220
|
-
bash $HOME/.claude/scripts/phase-tracker.sh update 0 in_progress
|
|
221
|
-
|
|
222
|
-
# Step 7.5, immediately after the depth answer:
|
|
223
|
-
# Full -> 1 2 3 4 5 Short -> 2 3 4 5
|
|
224
|
-
# :local drops nothing - the user test lives inside Review now, so there is
|
|
225
|
-
# no separate phase for it to skip
|
|
226
|
-
for p in "1:Plan" "2:Dev" "3:Review" "4:Commit" "5:Report"; do
|
|
227
|
-
bash $HOME/.claude/scripts/phase-tracker.sh add "${p%%:*}" "${p#*:}"
|
|
228
|
-
done
|
|
229
|
-
bash $HOME/.claude/scripts/phase-tracker.sh tiles --new # -> TaskCreate for the new tiles only
|
|
230
|
-
```
|
|
231
|
-
|
|
232
|
-
`tiles --new` emits `TaskCreate` only for phases that carry no `tasklist_id` yet, so the Phase 0 tile is not created twice. It is the same ordering rule, applied per batch: every tile in a batch is created in ascending phase order, and a deferred batch only ever appends phases numbered above everything already registered. Nothing is pre-marked, and a phase the run will not execute never gets a tile at all.
|
|
233
|
-
|
|
234
|
-
A phase that IS registered and short-circuits later still flips with `[SKIPPED]` - autopilot suppressing the Phase 3 user test, for instance. That is a runtime outcome, not an unknown set.
|
|
235
|
-
|
|
236
|
-
**Enforcement**: `smoke-tasklist-ordering.sh` scans the dispatcher (`commands/multi-agent/SKILL.md`) and every mode entry point doc (`commands/multi-agent/{autopilot,local,local-autopilot,analysis,resume-local}/SKILL.md` + the Copilot full-inline orchestrator mirror) for the explicit "in phase-number order" rule. Inventory drift fails the smoke.
|
|
237
|
-
|
|
238
207
|
### Other CLIs - call render after every state change
|
|
239
208
|
|
|
240
209
|
There is no TaskList outside Claude Code. Instead, after each state change the agent calls:
|
|
@@ -299,7 +268,7 @@ Throttling rules: mirror only canonical-set lines (`verbose`-tier internals are
|
|
|
299
268
|
|
|
300
269
|
### Delegated phases - mirror limitation + chunked dispatch (required)
|
|
301
270
|
|
|
302
|
-
When a phase's work is delegated to a subagent (
|
|
271
|
+
When a phase's work is delegated to a subagent (`create-component` plugin dispatch, Phase 1 explorers, Phase 3 reviewers), the visual channel freezes for the duration of the Agent call: the orchestrator is blocked while the call is in flight, so it cannot fire `TaskUpdate` / `now` / `tokens`, and a subagent cannot drive the parent session's TaskList (its own TaskCreate/TaskUpdate calls land on an invisible child list). The progress-line mirror above can therefore only fire while the orchestrator holds control. Rules:
|
|
303
272
|
|
|
304
273
|
1. **Pre-dispatch marker.** Immediately before every Agent call, set the active-phase line to the delegation itself, so the frozen interval at least states what is running and on which model:
|
|
305
274
|
- Claude Code: `TaskUpdate({activeForm: "Dev subagent (opus): <task subject>"})`
|
|
@@ -434,7 +403,7 @@ The `tasklist_id` meta from the previous session is replaced with the new IDs du
|
|
|
434
403
|
|
|
435
404
|
## Continuation runs (finish / manual-test)
|
|
436
405
|
|
|
437
|
-
A pre-existing `tracker-state.json` for the task is never re-initialized. Rules for any command that continues an earlier run (`/multi-agent:resume
|
|
406
|
+
A pre-existing `tracker-state.json` for the task is never re-initialized. Rules for any command that continues an earlier run (`/multi-agent:resume`, `/multi-agent:manual-test`):
|
|
438
407
|
|
|
439
408
|
1. `init` runs ONLY when no state file exists for the task. Otherwise the existing file is kept - phase history (elapsed, tokens, model, meta) survives.
|
|
440
409
|
2. The continuing command re-declares its phase set with `add` - `add` is idempotent, so existing phases keep their name, status, and token history; only genuinely new phases are appended. The card renders phases sorted by numeric id, so mixed sets stay in order.
|
|
@@ -11,6 +11,8 @@
|
|
|
11
11
|
- [Cross-CLI parity](#cross-cli-parity)
|
|
12
12
|
<!-- /toc -->
|
|
13
13
|
|
|
14
|
+
> **Pickers follow** `$HOME/.claude/multi-agent-refs/picker-contract.md`: never a one-option call, and branch on the option selected, not on its text.
|
|
15
|
+
|
|
14
16
|
> **TLDR** - Component tasks can auto-generate wiki docs + Figma screenshots. The Wiki adapter is invoked from `/multi-agent:channels` (Phase 5 delegates, or user invokes post-hoc). Four layouts supported (`submodule`, `in-repo`, `github-wiki`, `separate-repo`) - adapter picked from `figmaConfig.wiki.mode`. Non-blocking: failures log a warning and channels continues to other adapters. The Wiki adapter supports scope multi-select (Case A) and a precondition-failure menu (Case B) - see below.
|
|
15
17
|
|
|
16
18
|
This doc is referenced from `commands/multi-agent/channels/SKILL.md` (Wiki adapter) and indirectly from `$HOME/.claude/multi-agent-refs/phases/phase-5-report.md` (which delegates all external delivery to channels). Keeping it separate keeps both files under their token budgets and gives the contract a stable location for Claude-side + Copilot-side implementations.
|
|
@@ -75,7 +77,7 @@ Autopilot in Phase 5 pauses at the channels menu (per modes.md contract) - if
|
|
|
75
77
|
|
|
76
78
|
## Legacy prompt + preference flow (pre-v5.7, still supported for backward compat)
|
|
77
79
|
|
|
78
|
-
Interactive path (any interactive run
|
|
80
|
+
Interactive path (any interactive run), ONLY when the schema lacks `wikiScope` - ask with a native `AskUserQuestion` picker (never a typed y/n):
|
|
79
81
|
|
|
80
82
|
- `question`: "Generate component wiki docs?" (rendered in `outputLanguage`)
|
|
81
83
|
- `header`: "Wiki" (English, <=12 chars)
|
|
@@ -103,7 +105,6 @@ Explicit logs help the developer understand why wiki did or did not run:
|
|
|
103
105
|
- `figmaConfig` missing entirely → log `Phase 5: wiki skipped (no figma-config for this project)`.
|
|
104
106
|
- User declined at prompt → log `Phase 5: wiki skipped by user`.
|
|
105
107
|
- Autopilot with `wikiDefault=false` → log `Phase 5: wiki skipped (autopilot + wikiDefault=false)`.
|
|
106
|
-
- Short run: DO prompt - wiki is cheap and keeps docs fresh on the fast path; skip only if the user says no.
|
|
107
108
|
|
|
108
109
|
## Success log
|
|
109
110
|
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"schemaVersion": "2.
|
|
2
|
+
"schemaVersion": "2.8.0",
|
|
3
3
|
"global": {
|
|
4
4
|
"identities": [],
|
|
5
5
|
"keychainMapping": {
|
|
@@ -101,7 +101,7 @@
|
|
|
101
101
|
"skillConformance": {
|
|
102
102
|
"blockOnCoverageGap": false
|
|
103
103
|
},
|
|
104
|
-
"
|
|
104
|
+
"resume": {
|
|
105
105
|
"autoFix": false
|
|
106
106
|
}
|
|
107
107
|
},
|
|
@@ -75,7 +75,7 @@ The 3-tier fallback chain above governs Figma access in `/multi-agent:analysis`
|
|
|
75
75
|
| Phase | Sub-command examples | Figma MCP allowed | Figma REST allowed | Reason |
|
|
76
76
|
|---|---|---|---|---|
|
|
77
77
|
| Analysis Phase 1 | `/multi-agent:analysis` Phase 1 fetch | yes | yes (Tier 2 fallback) | Single source of design ground truth |
|
|
78
|
-
| Plan (Phase 1) | `/multi-agent`, `/multi-agent:
|
|
78
|
+
| Plan (Phase 1) | `/multi-agent`, `/multi-agent:autopilot` | no | no | Plan reads analysis doc Section 14 + Section 6 |
|
|
79
79
|
| Dev (Phase 2) | every mode that runs Phase 2 (8 modes total) | no | no | Reads analysis doc + Code Connect mapping |
|
|
80
80
|
| Review (Phase 3) | `/multi-agent:review`, every full-pipeline mode | no | no | Reviewer cites analysis doc Section 21 References |
|
|
81
81
|
| Test (inside Phase 3) | `/multi-agent:test`, `/multi-agent:manual-test`, every full mode | no | no | Variant list comes from analysis Section 13.6 + 15.2 |
|
|
@@ -66,7 +66,7 @@
|
|
|
66
66
|
},
|
|
67
67
|
"worktreePath": {
|
|
68
68
|
"type": ["string", "null"],
|
|
69
|
-
"description": "Absolute path to the task worktree. Null
|
|
69
|
+
"description": "Absolute path to the task worktree. Null when the Step 5b picker answered local."
|
|
70
70
|
},
|
|
71
71
|
"branch": {
|
|
72
72
|
"type": "string",
|
|
@@ -274,7 +274,7 @@
|
|
|
274
274
|
"workspaceSource": {
|
|
275
275
|
"type": "string",
|
|
276
276
|
"enum": ["asked", "command", "autopilot"],
|
|
277
|
-
"description": "Who decided where the branch lives. asked = the user answered the Step 5b workspace picker; command =
|
|
277
|
+
"description": "Who decided where the branch lives. asked = the user answered the Step 5b workspace picker; command = a flow that only ever builds worktrees stated it up front; autopilot = resolved to a worktree without asking, because an unattended commit in the user's own checkout is what worktrees prevent. localMode alone cannot say: false is both a chosen worktree and one nothing asked about."
|
|
278
278
|
},
|
|
279
279
|
"remoteType": {
|
|
280
280
|
"type": "string",
|
|
@@ -476,15 +476,10 @@
|
|
|
476
476
|
"type": "boolean",
|
|
477
477
|
"default": false
|
|
478
478
|
},
|
|
479
|
-
"onlyDevelop": {
|
|
480
|
-
"type": "boolean",
|
|
481
|
-
"default": false,
|
|
482
|
-
"description": "Short pipeline - phases 1 and 2 are skipped. Set by the Phase 0 Step 7.5 depth picker as of v16.0.0; the key and every reader of it are unchanged. Phase 3 Review still runs (v14.0.0+). Always false in an autopilot run, which never asks the depth question."
|
|
483
|
-
},
|
|
484
479
|
"localMode": {
|
|
485
480
|
"type": "boolean",
|
|
486
481
|
"default": false,
|
|
487
|
-
"description": "
|
|
482
|
+
"description": "Step 5b answered local - no worktree, direct branch in projectRoot."
|
|
488
483
|
},
|
|
489
484
|
"instructionDriven": {
|
|
490
485
|
"type": "boolean",
|
|
@@ -498,7 +493,7 @@
|
|
|
498
493
|
"analysis": {
|
|
499
494
|
"type": ["object", "null"],
|
|
500
495
|
"additionalProperties": true,
|
|
501
|
-
"description": "Phase 1 analysis-document outcome.
|
|
496
|
+
"description": "Phase 1 analysis-document outcome. Phase 1 runs in every mode, so a document is always produced; what varies is how many of its sections the evidence supported.",
|
|
502
497
|
"properties": {
|
|
503
498
|
"docStatus": {
|
|
504
499
|
"type": "string",
|
|
@@ -862,6 +857,29 @@
|
|
|
862
857
|
}
|
|
863
858
|
}
|
|
864
859
|
},
|
|
860
|
+
"threatModel": {
|
|
861
|
+
"type": "object",
|
|
862
|
+
"additionalProperties": true,
|
|
863
|
+
"description": "The run-scoped threat model produced at Phase 3 Step 2.7 (or by /multi-agent:security-review), mirrored here so a resume reuses it instead of re-deriving it. Contract: multi-agent-refs/threat-model.md.",
|
|
864
|
+
"properties": {
|
|
865
|
+
"path": {
|
|
866
|
+
"type": "string",
|
|
867
|
+
"description": "Path to .pipeline/threat-model.md in the worktree."
|
|
868
|
+
},
|
|
869
|
+
"sha": {
|
|
870
|
+
"type": "string",
|
|
871
|
+
"description": "Content hash, so a later step can tell whether the model changed."
|
|
872
|
+
},
|
|
873
|
+
"producedAt": {
|
|
874
|
+
"type": "string",
|
|
875
|
+
"format": "date-time"
|
|
876
|
+
},
|
|
877
|
+
"producedBy": {
|
|
878
|
+
"type": "string",
|
|
879
|
+
"description": "Which step or command wrote it (e.g. 'phase-3-step-2.7', 'security-review')."
|
|
880
|
+
}
|
|
881
|
+
}
|
|
882
|
+
},
|
|
865
883
|
"reviewIterations": {
|
|
866
884
|
"type": "array",
|
|
867
885
|
"items": {
|
|
@@ -1530,7 +1548,7 @@
|
|
|
1530
1548
|
}
|
|
1531
1549
|
}
|
|
1532
1550
|
},
|
|
1533
|
-
"description": "Captured in Phase 2 after the build+test gate, because the user test is dropped by every autopilot and
|
|
1551
|
+
"description": "Captured in Phase 2 after the build+test gate, because the user test is dropped by every autopilot entry and by any run whose workspace is local."
|
|
1534
1552
|
},
|
|
1535
1553
|
"host": {
|
|
1536
1554
|
"type": ["string", "null"],
|