@mmerterden/multi-agent-pipeline 19.1.3 → 20.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +150 -7
- package/README.md +60 -50
- package/README.tr.md +55 -46
- package/docs/adr/0002-instruction-driven-flag.md +6 -5
- package/docs/adr/0005-lazy-phase-docs.md +2 -2
- package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
- package/docs/adr/0009-claude-stack-skills-plugin-only.md +1 -1
- package/docs/adr/0010-own-code-graph.md +5 -4
- package/docs/adr/0012-macos-only.md +2 -2
- package/docs/adr/0013-lsp-code-intelligence.md +2 -2
- package/docs/adr/0014-six-phase-consolidation.md +9 -9
- package/docs/adr/0015-one-pipeline-no-depth-answer.md +83 -0
- package/docs/adr/0016-the-run-shape-is-asked-not-typed.md +69 -0
- package/docs/adr/README.md +18 -16
- package/docs/architecture.md +2 -2
- package/docs/ecosystem.md +8 -9
- package/docs/facts.json +5 -8
- package/docs/features.md +4 -5
- package/docs/token-budget-history.md +1 -1
- package/install/_common.mjs +14 -6
- package/install/_mcp-register.mjs +1 -1
- package/install/_plugin-skills.mjs +3 -4
- package/install/copilot.mjs +5 -5
- package/install/templates/copilot-instructions.md +7 -16
- package/manifest.json +135 -138
- package/package.json +1 -1
- package/pipeline/commands/multi-agent/SKILL.md +6 -8
- package/pipeline/commands/multi-agent/analysis/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/analysis-jira/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/autopilot/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/autopilot-on/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/autopilot-status/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/build-optimize/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/channels/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/create-jira/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/design-check/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/forget/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +4 -2
- package/pipeline/commands/multi-agent/help/SKILL.md +21 -27
- package/pipeline/commands/multi-agent/ios-coding-standard/SKILL.md +5 -4
- package/pipeline/commands/multi-agent/issue/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/jira/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/language/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/prune-logs/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/purge/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/resume/SKILL.md +177 -48
- package/pipeline/commands/multi-agent/save/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/setup/SKILL.md +4 -4
- package/pipeline/commands/multi-agent/stack/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/sync/SKILL.md +6 -7
- package/pipeline/commands/multi-agent/test-screenshots/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/uninstall/SKILL.md +2 -0
- package/pipeline/commands/sim-test.md +4 -4
- package/pipeline/lib/repo-hygiene.sh +1 -1
- package/pipeline/multi-agent-refs/analysis/locked.md +2 -2
- package/pipeline/multi-agent-refs/analysis/render.md +1 -1
- package/pipeline/multi-agent-refs/analysis/resolve.md +1 -1
- package/pipeline/multi-agent-refs/analysis/synthesis.md +1 -1
- package/pipeline/multi-agent-refs/analysis-template.md +1 -1
- package/pipeline/multi-agent-refs/channels/jira.md +8 -8
- package/pipeline/multi-agent-refs/component-dispatch.md +0 -8
- package/pipeline/multi-agent-refs/cross-cli-contract.md +10 -11
- package/pipeline/multi-agent-refs/features/base-branch-evidence.md +2 -2
- package/pipeline/multi-agent-refs/features/external-context-injection.md +2 -0
- package/pipeline/multi-agent-refs/features/review-delta.md +1 -1
- package/pipeline/multi-agent-refs/features/review-multi-repo.md +3 -3
- package/pipeline/multi-agent-refs/features/scope-check.md +1 -1
- package/pipeline/multi-agent-refs/features/skill-conformance.md +1 -1
- package/pipeline/multi-agent-refs/features/stack-skill-routing.md +1 -1
- package/pipeline/multi-agent-refs/features/visual-evidence.md +2 -1
- package/pipeline/multi-agent-refs/features/worktree-finalize.md +1 -1
- package/pipeline/multi-agent-refs/generate-issue.md +2 -0
- package/pipeline/multi-agent-refs/issue-jira-triad.md +2 -0
- package/pipeline/multi-agent-refs/keychain.md +2 -0
- package/pipeline/multi-agent-refs/knowledge.md +0 -7
- package/pipeline/multi-agent-refs/outside-the-pipeline.md +1 -1
- package/pipeline/multi-agent-refs/payload-contracts.md +1 -1
- package/pipeline/multi-agent-refs/phases/modes.md +32 -108
- package/pipeline/multi-agent-refs/phases/operations.md +3 -1
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +23 -42
- package/pipeline/multi-agent-refs/phases/phase-1-plan.md +10 -21
- package/pipeline/multi-agent-refs/phases/phase-2-dev.md +13 -44
- package/pipeline/multi-agent-refs/phases/phase-3-review.md +19 -22
- package/pipeline/multi-agent-refs/phases/phase-4-commit.md +6 -6
- package/pipeline/multi-agent-refs/phases/phase-5-report.md +2 -2
- package/pipeline/multi-agent-refs/phases.md +9 -11
- package/pipeline/multi-agent-refs/progress-contract.md +1 -1
- package/pipeline/multi-agent-refs/readiness-review.md +2 -0
- package/pipeline/multi-agent-refs/rules.md +2 -2
- package/pipeline/multi-agent-refs/tracker-contract.md +9 -40
- package/pipeline/multi-agent-refs/wiki-capture.md +3 -2
- package/pipeline/preferences-template.json +2 -2
- package/pipeline/rules/figma-pipeline.md +1 -1
- package/pipeline/schemas/agent-state.schema.json +5 -10
- package/pipeline/schemas/migrations/prefs-2.7.0-to-2.8.0.mjs +33 -0
- package/pipeline/schemas/phases.json +3 -24
- package/pipeline/schemas/prefs.schema.json +5 -5
- package/pipeline/scripts/autopilot-runner.mjs +6 -7
- package/pipeline/scripts/build-references.mjs +3 -3
- package/pipeline/scripts/build-stack-plugins.mjs +1 -1
- package/pipeline/scripts/bulk-read.sh +6 -4
- package/pipeline/scripts/cost-table.json +1 -1
- package/pipeline/scripts/doctor.mjs +4 -4
- package/pipeline/scripts/gc-refs.sh +1 -1
- package/pipeline/scripts/gen-mode-dispatch.mjs +11 -41
- package/pipeline/scripts/learnings-ledger.mjs +1 -1
- package/pipeline/scripts/match-skills.mjs +4 -4
- package/pipeline/scripts/memory-load.sh +3 -3
- package/pipeline/scripts/migrate-prefs.mjs +18 -17
- package/pipeline/scripts/phase-tracker.sh +2 -2
- package/pipeline/scripts/phase0-exit-gate.mjs +1 -1
- package/pipeline/scripts/plan-coverage-gate.mjs +6 -6
- package/pipeline/scripts/run-aggregator.mjs +3 -3
- package/pipeline/scripts/runs-index.mjs +7 -7
- package/pipeline/scripts/scope-check-gate.mjs +1 -1
- package/pipeline/scripts/smoke-schema-validation.sh +9 -12
- package/pipeline/scripts/usage-report.mjs +5 -7
- package/pipeline/scripts/validate-analysis-doc.mjs +3 -3
- package/pipeline/scripts/worktree-finalize.sh +2 -2
- package/pipeline/scripts/write-state.mjs +22 -11
- package/pipeline/skills/.skill-manifest.json +9 -21
- package/pipeline/skills/.skills-index.json +6 -39
- package/pipeline/skills/shared/README.md +5 -8
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +10 -13
- package/pipeline/skills/shared/core/multi-agent-autopilot-status/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +4 -5
- package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +13 -16
- package/pipeline/skills/shared/core/multi-agent-ios-coding-standard/SKILL.md +2 -3
- package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +51 -15
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +6 -6
- package/pipeline/skills/skills-index.md +3 -6
- package/pipeline/commands/multi-agent/local/SKILL.md +0 -132
- package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +0 -142
- package/pipeline/commands/multi-agent/resume-local/SKILL.md +0 -114
- package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +0 -41
- package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +0 -55
- package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +0 -51
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
### Phase 2: Dev (Sonnet)
|
|
2
2
|
|
|
3
|
-
> **TLDR** - Sonnet executes the plan task-by-task with TDD (red→green→refactor). Required: issue-tracker status moved to "In Progress" before any code, with a post-mutation verify step (re-reads the field, retries once on silent VALIDATION failures). Build verification after each task (up to 3 retries). Build-queue lock serializes concurrent xcodebuild.
|
|
3
|
+
> **TLDR** - Sonnet executes the plan task-by-task with TDD (red→green→refactor). Required: issue-tracker status moved to "In Progress" before any code, with a post-mutation verify step (re-reads the field, retries once on silent VALIDATION failures). Build verification after each task (up to 3 retries). Build-queue lock serializes concurrent xcodebuild. `taskType === component` **short-circuits the TDD path** and delegates the whole phase to the enabled `ai-<platform>-toolkit` marketplace plugin's component skill (`create-component`, fallback `create-ui-component`) - see next subsection.
|
|
4
4
|
|
|
5
5
|
## Phase 2 Pre-flight (BLOCKING, v9.0.0)
|
|
6
6
|
|
|
@@ -8,11 +8,11 @@ Per Locked decision 30, Phase 2 Dev consumes the analysis document as the sole d
|
|
|
8
8
|
|
|
9
9
|
Pre-flight steps (run in order, abort on failure).
|
|
10
10
|
|
|
11
|
-
**Steps 1, 2, 3, 5 and 6
|
|
11
|
+
**Steps 1, 2, 3, 5 and 6 read the analysis document.** Phase 1 always runs, so the document always exists; what varies is how much evidence it carries. A section the evidence did not support is absent by the Locked 2 omission rule, and a step whose section is absent is recorded `not-applicable (no <section> in this document)` rather than aborting. Steps 4, 7, 8 and 9 read nothing from it and apply always.
|
|
12
12
|
|
|
13
13
|
1. **Analysis document presence** (Phase 1 modes only): read `state.analysis.docStatus` and `state.analysis.docPath[]`, both set by Phase 1 Step 4.
|
|
14
14
|
- `produced` | `reused` -> read the active platform's file; multi-repo runs need one per selected repo.
|
|
15
|
-
- `not-applicable` ->
|
|
15
|
+
- `not-applicable` -> the analysis mode exported a document this run does not consume; record it for steps 1, 2, 3, 5, 6 and skip them.
|
|
16
16
|
- **Abort**: `produced` but unreadable -> `ERR: analysis doc at <path> unreadable. Resume with /multi-agent:resume #N.` Producing it is Phase 1's job.
|
|
17
17
|
|
|
18
18
|
2. **Parse YAML front-matter** into `state.analysis.frontMatter`: `feature`, `platform`, `language`, `mode`, `evidence_digest`, `template_version`. **Abort** when `template_version` < `v3`; there is no degraded mode.
|
|
@@ -49,7 +49,7 @@ The analysis document is the SOLE design source in Phase 2. Variant choices, pad
|
|
|
49
49
|
|
|
50
50
|
#### Input contract
|
|
51
51
|
|
|
52
|
-
Phase 2 consumes the Phase 1 output object conforming to `$HOME/.claude/schemas/planning-output.schema.json` - the task graph (`tasks[]` with `id`, `title`, `type`, `files`, and optional `dependsOn` / `acceptanceCriteria`) plus the architecture review notes. Tasks execute in dependency order; the schema's `dependsOn` field drives the ready-task picker.
|
|
52
|
+
Phase 2 consumes the Phase 1 output object conforming to `$HOME/.claude/schemas/planning-output.schema.json` - the task graph (`tasks[]` with `id`, `title`, `type`, `files`, and optional `dependsOn` / `acceptanceCriteria`) plus the architecture review notes. Tasks execute in dependency order; the schema's `dependsOn` field drives the ready-task picker.
|
|
53
53
|
|
|
54
54
|
**Plan Todo iteration (opt-in)**: gated by `prefs.global.planTodos.enabled` (default: `false`). When enabled and Phase 1 Step 11 emitted a `plan.todos[]`, Phase 2 iterates via `$HOME/.claude/lib/plan-todos.sh next/start/complete/fail` instead of walking `tasks[]` directly. When disabled, the loop walks `tasks[]` from `planning-output` - TDD contract is unchanged. Full helper loop + state semantics: `$HOME/.claude/multi-agent-refs/features/plan-todos.md`. A todo with `sourceTag: Reuse` binds the file analysis already found; `Modify` edits in place. Writing a new file over a `Reuse` step is a Locked 11 violation and Phase 3 flags it.
|
|
55
55
|
|
|
@@ -57,7 +57,7 @@ Phase 2 consumes the Phase 1 output object conforming to `$HOME/.claude/schemas/
|
|
|
57
57
|
|
|
58
58
|
#### Component tasks - delegated dispatch (taskType === "component")
|
|
59
59
|
|
|
60
|
-
When Phase 0 Step 7 classified the task as `component`, Phase 2 delegates the entire phase to the enabled `ai-<platform>-toolkit` marketplace plugin's component skill (`create-component`, fallback `create-ui-component`) via the Skill tool and does NOT run the TDD loop below. The dispatch layer passes the plugin skill the analysis Section 6 (Bileşen Envanteri) entry + Section 13.1 conventions for the named component as context. Because plugin skills do not write pipeline state, the **dispatch layer** (not the skill) owns `state.phases["2"].subphases[]`, recording a coarse component-build row - multi-agent's `phase-tracker` reads that array with no special case. Plugin resolution (dual-name), dispatch call, failure/resume, multi-repo,
|
|
60
|
+
When Phase 0 Step 7 classified the task as `component`, Phase 2 delegates the entire phase to the enabled `ai-<platform>-toolkit` marketplace plugin's component skill (`create-component`, fallback `create-ui-component`) via the Skill tool and does NOT run the TDD loop below. The dispatch layer passes the plugin skill the analysis Section 6 (Bileşen Envanteri) entry + Section 13.1 conventions for the named component as context. Because plugin skills do not write pipeline state, the **dispatch layer** (not the skill) owns `state.phases["2"].subphases[]`, recording a coarse component-build row - multi-agent's `phase-tracker` reads that array with no special case. Plugin resolution (dual-name), dispatch call, failure/resume, multi-repo, and the intentional cross-CLI divergence live in `$HOME/.claude/multi-agent-refs/component-dispatch.md` - read it before editing component-task behaviour here. Phase 2 still owns: progress line `-> dispatching create-component <name>`, `retryCount` cap at 3, and fallthrough to the TDD path when dispatch prerequisites are missing (`taskType` absent OR the plugin is not enabled in this repo -> log anomaly, halt or run TDD per component-dispatch.md).
|
|
61
61
|
|
|
62
62
|
For non-component taskTypes (`bugfix`, `feature`, `refactor`, `chore`), continue with the standard TDD section below.
|
|
63
63
|
|
|
@@ -92,7 +92,7 @@ If the latest iteration has `triage.approved === true` AND `accepted === []`, Ph
|
|
|
92
92
|
**Telemetry**: at the start of every re-entry, emit:
|
|
93
93
|
|
|
94
94
|
```bash
|
|
95
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
95
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 2 rework.started \
|
|
96
96
|
iteration=$ITERATION accepted_blocking=$BLOCKING accepted_important=$IMPORTANT
|
|
97
97
|
```
|
|
98
98
|
|
|
@@ -275,12 +275,12 @@ After the build/test green step and BEFORE Phase 3 handoff, run one diff-shrink
|
|
|
275
275
|
5. **Record tokens in the cost ledger** so Phase 5's Cost Breakdown captures the pass:
|
|
276
276
|
|
|
277
277
|
```bash
|
|
278
|
-
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
278
|
+
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 2 dev.simplifier_pass \
|
|
279
279
|
model=sonnet tokens_in=$IN tokens_out=$OUT duration_ms=$DUR \
|
|
280
280
|
edits_returned=$RET edits_applied=$APPLIED edits_skipped=$SKIPPED
|
|
281
281
|
```
|
|
282
282
|
|
|
283
|
-
Scope guard: a single pass, never looped.
|
|
283
|
+
Scope guard: a single pass, never looped. Component tasks (`taskType === "component"`) skip it (the figma skill owns its own checklist).
|
|
284
284
|
|
|
285
285
|
---
|
|
286
286
|
|
|
@@ -297,37 +297,6 @@ Exit 1 lists `unjustified[]`: complete the record once and re-run. A second exit
|
|
|
297
297
|
|
|
298
298
|
---
|
|
299
299
|
|
|
300
|
-
#### Short pipeline (`state.onlyDevelop === true`)
|
|
301
|
-
|
|
302
|
-
Set by the Phase 0 Step 7.5 depth picker, or by autopilot never (autopilot always runs Full). When it is true, Phase 2 runs self-contained with **Opus** (not Sonnet). No Phase 1 plan exists - the agent creates its own scope.
|
|
303
|
-
|
|
304
|
-
**Flow:**
|
|
305
|
-
1. Read task description (from Jira, GitHub issue, or free-text)
|
|
306
|
-
2. Lightweight file scan - grep/glob for relevant code (not full Explore agents)
|
|
307
|
-
3. Determine scope autonomously - no task breakdown, no user confirmation
|
|
308
|
-
4. Implement with TDD cycle (same RED→GREEN→REFACTOR as normal mode)
|
|
309
|
-
5. Build verification (same lock, same retry logic)
|
|
310
|
-
6. Intermediate commits (same WIP pattern)
|
|
311
|
-
|
|
312
|
-
**Key differences from normal mode:**
|
|
313
|
-
|
|
314
|
-
| Aspect | Full | Short |
|
|
315
|
-
|--------|--------|----------|
|
|
316
|
-
| Model | Sonnet | **Opus** |
|
|
317
|
-
| Plan source | Phase 1 task list | Self-determined |
|
|
318
|
-
| Task granularity | Per-plan-item | Agent decides |
|
|
319
|
-
| Status updates | Per task item | Single in_progress → completed |
|
|
320
|
-
| Scope confirmation | Phase 1 user approval | None (agent autonomous) |
|
|
321
|
-
| Review of the result | Phase 3 | Phase 3 (same) |
|
|
322
|
-
|
|
323
|
-
Because the agent determines its own scope here, Phase 3 is the only place that checks the result against anything external. Record every skill, plugin skill and guide consulted during this phase into `state.telemetry.skillCalls[]` with the files it was applied to - Phase 3 resolves the criteria set independently, and this record is what lets it tell "applied and honoured" from "never opened".
|
|
324
|
-
|
|
325
|
-
**Never combined with autopilot.** Autopilot skips the depth question and runs Full, so `onlyDevelop` is false in every unattended run. "Fast plus unattended" was removed in v16.0.0 and no longer exists: something has to choose when nobody is asked, and unattended is the worst place to drop analysis and planning.
|
|
326
|
-
|
|
327
|
-
**Tracker visibility during Opus dispatch**: on Claude Code the model switch to Opus happens via subagent dispatch, and the parent widget cannot move while an Agent call is in flight. Dispatch per task from the self-generated task list (never one monolithic call for the whole phase), set the pre-dispatch `activeForm` marker, and record tokens between chunks - full rules in `$HOME/.claude/multi-agent-refs/tracker-contract.md` section "Delegated phases".
|
|
328
|
-
|
|
329
|
-
---
|
|
330
|
-
|
|
331
300
|
#### v2.1.0+ Multi-Repo Mode
|
|
332
301
|
|
|
333
302
|
Active when `state.projects[].length > 1` (set by Phase 0 multi-select). Single-repo flow above is preserved verbatim - this section adds the deltas.
|
|
@@ -372,26 +341,26 @@ This closes the gap where an agent records "built" without ever producing build
|
|
|
372
341
|
|
|
373
342
|
**Telemetry**: Per-repo metrics in addition to per-task metrics:
|
|
374
343
|
```bash
|
|
375
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
376
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
344
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 2 build.completed repo=common duration_ms=$D status=ok
|
|
345
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 2 build.completed repo=uicomponents duration_ms=$D status=ok
|
|
377
346
|
```
|
|
378
347
|
|
|
379
348
|
**Token forwarding:** every TDD round (red, green, refactor) that hits the dev model MUST forward token totals into the tracker so Phase 5's Cost Breakdown captures Phase 2:
|
|
380
349
|
|
|
381
350
|
```bash
|
|
382
|
-
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
351
|
+
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 2 dev.tdd_round \
|
|
383
352
|
model=<sonnet|opus> step=<red|green|refactor> \
|
|
384
353
|
tokens_in=$IN tokens_out=$OUT duration_ms=$DUR
|
|
385
354
|
```
|
|
386
355
|
|
|
387
|
-
Model
|
|
356
|
+
Model is Sonnet unless routing names another rung (`multi-agent-refs/features/model-fallback.md`). Best-effort. See `$HOME/.claude/multi-agent-refs/progress-contract.md#token-telemetry-forwarding`.
|
|
388
357
|
|
|
389
358
|
---
|
|
390
359
|
|
|
391
360
|
## Token telemetry - invoke after every LLM call
|
|
392
361
|
|
|
393
362
|
```bash
|
|
394
|
-
bash $HOME/.claude/scripts/phase-tracker.sh tokens
|
|
363
|
+
bash $HOME/.claude/scripts/phase-tracker.sh tokens 2 <input_count> <output_count>
|
|
395
364
|
```
|
|
396
365
|
|
|
397
366
|
Contract and rationale: `progress-contract.md` -> Token telemetry forwarding.
|
|
@@ -104,13 +104,13 @@ Persist the totals as `state.diffRisk` (Phase 4 `risk` section, Phase 5, `run-me
|
|
|
104
104
|
**Gate behavior**: this step is **never blocking**. If risk scoring fails (git error, parse error, validator rejection), continue with no priority hint - reviewers receive the full diff in their default order. Failures are logged via metrics:
|
|
105
105
|
|
|
106
106
|
```bash
|
|
107
|
-
[ -z "$RISK_JSON" ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
107
|
+
[ -z "$RISK_JSON" ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.diff_risk_skipped reason=$REASON
|
|
108
108
|
```
|
|
109
109
|
|
|
110
110
|
On success, emit a single summary metric:
|
|
111
111
|
|
|
112
112
|
```bash
|
|
113
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
113
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.diff_risk \
|
|
114
114
|
top_files=$(jq '.files | length' <<< "$RISK_JSON") \
|
|
115
115
|
max_score=$(jq '.totals.max_score' <<< "$RISK_JSON") \
|
|
116
116
|
loc_added=$(jq '.totals.loc_added' <<< "$RISK_JSON") \
|
|
@@ -127,7 +127,7 @@ Step 1.75 uses `test_lines_removed` as an advisory hint only - too weak for wh
|
|
|
127
127
|
```bash
|
|
128
128
|
TEST_INTEGRITY_JSON=$(printf '%s' "$RISK_FULL" | node $HOME/.claude/scripts/test-integrity-gate.mjs 2>/dev/null || echo "")
|
|
129
129
|
TI_COUNT=$(jq -r '.count // 0' <<< "${TEST_INTEGRITY_JSON:-{\}}" 2>/dev/null || echo 0)
|
|
130
|
-
[ "$TI_COUNT" -gt 0 ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
130
|
+
[ "$TI_COUNT" -gt 0 ] && $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.test_integrity findings="$TI_COUNT"
|
|
131
131
|
```
|
|
132
132
|
|
|
133
133
|
`findings[]` are reviewer-shaped (`test_integrity`, `blocking`), so they merge into the reviewer findings at Step 3.0 and need no triage-prompt or `validate-triage.mjs` change. Triage keeps each blocking unless the removal is justified per the immutable-test rule (spec changed AND commit body names the test) → `deferred[]`.
|
|
@@ -142,7 +142,7 @@ On a trivial diff every reviewer agrees and the extra models plus triage are pai
|
|
|
142
142
|
SCOPE_JSON=$(printf '%s' "$RISK_FULL" | node $HOME/.claude/scripts/review-scope.mjs 2>/dev/null \
|
|
143
143
|
|| echo '{"scope":"full","reason":"no risk report - failing safe"}')
|
|
144
144
|
REVIEW_SCOPE=$(jq -r '.scope // "full"' <<< "$SCOPE_JSON")
|
|
145
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
145
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.scope scope="$REVIEW_SCOPE"
|
|
146
146
|
```
|
|
147
147
|
|
|
148
148
|
`single` (Reviewer 1 only) requires **all** of: churn <= 20 lines, `totals.max_score` < 3.0, and no `security_path` / `migration` / `public_api` / `no_test_change` / `test_lines_removed` on any file. Anything else → `full`.
|
|
@@ -159,7 +159,7 @@ node $HOME/.claude/scripts/skill-conformance.mjs \
|
|
|
159
159
|
--repo "$WORKTREE" --out "$WORKTREE/.pipeline/criteria-manifest.json"
|
|
160
160
|
CRIT_RC=$?
|
|
161
161
|
CRITERIA=$(cat "$WORKTREE/.pipeline/criteria-manifest.json" 2>/dev/null || echo '{}')
|
|
162
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
162
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 review.criteria \
|
|
163
163
|
rules="$(jq -r '.selectedRuleCount // 0' <<< "$CRITERIA")" \
|
|
164
164
|
ledger="$(jq -r '.ledger.source // "derived"' <<< "$CRITERIA")"
|
|
165
165
|
[ "$CRIT_RC" != "0" ] && HALT "criteria could not be resolved (rc=$CRIT_RC) - see resolutionFailure / unparseableRegistries"
|
|
@@ -175,7 +175,7 @@ Output conforms to `$HOME/.claude/schemas/criteria-manifest.schema.json`. The fo
|
|
|
175
175
|
|
|
176
176
|
#### Step 1.8 - Figma visual-fidelity context (when task carries a Figma reference)
|
|
177
177
|
|
|
178
|
-
**
|
|
178
|
+
**Missing inputs.** A section the evidence did not support is absent from the analysis document, so three inputs this phase was written around can be missing. Substitute them and RECORD the substitution - a step that could not run and one that passed must not read the same, or the completeness claim cannot be checked:
|
|
179
179
|
|
|
180
180
|
| Absent input | Substitute |
|
|
181
181
|
|---|---|
|
|
@@ -426,8 +426,8 @@ Exit 2 (empty ledger) skips silently. The `## Rejected review preferences` secti
|
|
|
426
426
|
**Recall telemetry.** Log what was injected, then what triage cited. Zero cited is a legitimate answer; `learning-curve.mjs` trends the ratio:
|
|
427
427
|
|
|
428
428
|
```bash
|
|
429
|
-
bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
430
|
-
bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
429
|
+
bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 memory.injected kind=prior-art rows=$N
|
|
430
|
+
bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 memory.hit rows=$CITED_COUNT
|
|
431
431
|
```
|
|
432
432
|
|
|
433
433
|
**Bulky payloads (opt-in via `prefs.global.contextOffload.enabled`).** Test output and whole-file diffs go through the offload filter, which leaves a `[[ref:<node_id>]]` line plus the tail in context and the full text under `.multi-agent/refs/`. Read that file when the tail is not enough; with the pref off it is a pass-through.
|
|
@@ -513,7 +513,7 @@ One `review.reviewer_call` per dispatched reviewer, one `review.triage_call`, on
|
|
|
513
513
|
```bash
|
|
514
514
|
M=$HOME/.claude/scripts/log-metric.sh
|
|
515
515
|
emit() { # $1=event $2=model $3=duration $4=tokens_in $5=tokens_out
|
|
516
|
-
LOG_METRIC_FORWARD_TO_TRACKER=1 bash "$M" "$TASK_ID"
|
|
516
|
+
LOG_METRIC_FORWARD_TO_TRACKER=1 bash "$M" "$TASK_ID" 3 "$1" \
|
|
517
517
|
model="$2" duration_ms="$3" tokens_in="$4" tokens_out="$5"
|
|
518
518
|
}
|
|
519
519
|
emit review.reviewer_call fable "$R1_DURATION" "$R1_IN" "$R1_OUT" # opus on Copilot CLI
|
|
@@ -525,7 +525,7 @@ else
|
|
|
525
525
|
fi
|
|
526
526
|
emit review.reviewer_call sonnet "$SONNET_DURATION" "$SONNET_IN" "$SONNET_OUT"
|
|
527
527
|
emit review.triage_call fable "$TRIAGE_DURATION" "$TRIAGE_IN" "$TRIAGE_OUT"
|
|
528
|
-
bash "$M" "$TASK_ID"
|
|
528
|
+
bash "$M" "$TASK_ID" 3 review.completed raw_count=$RAW accepted=$ACC \
|
|
529
529
|
deferred=$DEF rejected=$REJ approved=$APPROVED duration_ms=$DURATION
|
|
530
530
|
```
|
|
531
531
|
|
|
@@ -614,7 +614,7 @@ Log: "Phase 3: Review - raw={N1+N2+N3} accepted={Na} deferred={Nd} rejected={N
|
|
|
614
614
|
## Token telemetry - invoke after every LLM call
|
|
615
615
|
|
|
616
616
|
```bash
|
|
617
|
-
bash $HOME/.claude/scripts/phase-tracker.sh tokens
|
|
617
|
+
bash $HOME/.claude/scripts/phase-tracker.sh tokens 3 <input_count> <output_count> [cached_count]
|
|
618
618
|
```
|
|
619
619
|
|
|
620
620
|
The optional 4th `cached_count` is the prompt-cache-read token count when the host reports it (Anthropic `cache_read_input_tokens`); it defaults to 0 and is priced at the cheaper `cacheReadPerMtok` rate in the Phase 5 cost ledger. The tracker accumulates the totals additively, so multiple calls in the same phase compound. The render output then shows live cost on the active phase tile (e.g. `Phase 2 Dev 2m 14s · 12.4k tok`). This satisfies the contract in `$HOME/.claude/multi-agent-refs/tracker-contract.md` and the `smoke-tracker-tokens-invocation.sh` enforcement gate. Skipping this call is the #1 cause of "I can't see how much it cost" complaints.
|
|
@@ -623,15 +623,12 @@ Contract and rationale: `progress-contract.md` -> Token telemetry forwarding.
|
|
|
623
623
|
|
|
624
624
|
---
|
|
625
625
|
|
|
626
|
-
## User test
|
|
626
|
+
## User test
|
|
627
627
|
|
|
628
|
-
|
|
629
|
-
|
|
630
|
-
prompt AND a worktree checkout, so it is skipped where either is missing. If
|
|
631
|
-
issues are found, the run returns to Phase 1 Dev.
|
|
628
|
+
The tail of Review rather than a phase of its own: it judges work that already
|
|
629
|
+
exists, which is what Review does.
|
|
632
630
|
|
|
633
|
-
|
|
634
|
-
> **TLDR** - Optional test gate. Offers to boot the simulator/emulator (UI Bug Hunter) or hand off to the user for manual QA. Needs an interactive prompt AND a worktree checkout, so it is in the phase set of `/multi-agent` alone and dropped by every `autopilot` or `--local` entry. Depth does not affect it: a Short run still reaches Phase 3. If issues found, loops back to Phase 2.
|
|
631
|
+
> **TLDR** - Optional test gate. Offers to boot the simulator/emulator (UI Bug Hunter) or hand off to the user for manual QA. Needs an interactive prompt AND a worktree checkout, so it runs on an attended `/multi-agent` whose workspace is a worktree, and is skipped on `autopilot` or when the user chose to work locally. If issues found, loops back to Phase 2.
|
|
635
632
|
|
|
636
633
|
<!-- progress-contract: applied -->
|
|
637
634
|
Progress emission per `$HOME/.claude/multi-agent-refs/progress-contract.md` - lines for local-test prompt render, user-answer capture, repo checkout (if selected).
|
|
@@ -682,7 +679,7 @@ fi
|
|
|
682
679
|
**Telemetry**:
|
|
683
680
|
|
|
684
681
|
```bash
|
|
685
|
-
LOG_METRIC_FORWARD_TO_TRACKER=0 $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
682
|
+
LOG_METRIC_FORWARD_TO_TRACKER=0 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 test_gap.scanned \
|
|
686
683
|
stack=$SCAN_STACK \
|
|
687
684
|
sources=$(jq '.totals.sourcesScanned' <<< "$GAP_JSON") \
|
|
688
685
|
gaps=$(jq '.totals.gapCount' <<< "$GAP_JSON")
|
|
@@ -702,7 +699,7 @@ Figma evidence (tier=<n>):
|
|
|
702
699
|
|
|
703
700
|
Tier 1 / Tier 2 records print `screenshotUrl` from the captured evidence (Tier 2 URLs expire after 30 days, re-fetch on the spot if needed). Tier 3 records print the local path to the user-attached screenshot. The block is informational; it never blocks the prompt.
|
|
704
701
|
|
|
705
|
-
1. Ask with a native `AskUserQuestion` picker (never a typed y/N prompt). The options MUST make the local-checkout side effect explicit - testing removes the worktree and checks the branch out into the main repo:
|
|
702
|
+
1. Ask with a native `AskUserQuestion` picker (never a typed y/N prompt), per `$HOME/.claude/multi-agent-refs/picker-contract.md`. The options MUST make the local-checkout side effect explicit - testing removes the worktree and checks the branch out into the main repo:
|
|
706
703
|
- `question`: "Check out locally to test now?" (rendered in `outputLanguage`)
|
|
707
704
|
- `header`: "Test" (English, <=12 chars)
|
|
708
705
|
- `options`:
|
|
@@ -733,7 +730,7 @@ Tier 1 / Tier 2 records print `screenshotUrl` from the captured evidence (Tier 2
|
|
|
733
730
|
bash $HOME/.claude/scripts/phase-tracker.sh now 3 "awaiting local test (user)"
|
|
734
731
|
bash $HOME/.claude/scripts/phase-tracker.sh render
|
|
735
732
|
```
|
|
736
|
-
The waiting state persists in `tracker-state.json` across the handoff; `/multi-agent:resume
|
|
733
|
+
The waiting state persists in `tracker-state.json` across the handoff; `/multi-agent:resume` and `/multi-agent:manual-test` CONTINUE this state file and never re-init it (`$HOME/.claude/multi-agent-refs/tracker-contract.md` "Continuation runs").
|
|
737
734
|
|
|
738
735
|
**"ok" is a structured result, not a word.** Before "ok" is accepted, the run writes `$WORKTREE/.pipeline/manual-test.json`: one entry per acceptance criterion, the criteria taken from the analysis doc test plan (Section 15 / 20), the plan tasks, and the user's own words in the reply. Every criterion records what was seen; a criterion that was not tried says so with a reason.
|
|
739
736
|
```json
|
|
@@ -806,7 +803,7 @@ Returns severity-tagged findings (Critical / High / Medium). Critical items bloc
|
|
|
806
803
|
When the security-auditor or any other Phase 3 sub-agent runs, forward its token totals so Phase 5's Cost Breakdown captures Phase 3:
|
|
807
804
|
|
|
808
805
|
```bash
|
|
809
|
-
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
806
|
+
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 3 audit.completed \
|
|
810
807
|
model=opus tokens_in=$IN tokens_out=$OUT duration_ms=$DUR
|
|
811
808
|
```
|
|
812
809
|
|
|
@@ -354,19 +354,19 @@ A task ending with one or more `pushStatus === "skipped"` repos does NOT bump `r
|
|
|
354
354
|
|
|
355
355
|
```bash
|
|
356
356
|
for proj in $(jq -r '.projects[].name' "$STATE_FILE"); do
|
|
357
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
358
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
359
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
357
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 commit.created repo=$proj sha=$SHA
|
|
358
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 push.attempted repo=$proj attempts=$N status=$STATUS
|
|
359
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 pr.opened repo=$proj url=$URL number=$N
|
|
360
360
|
done
|
|
361
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
361
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 multi_repo.completed repos=$REPOS skipped=$SKIPPED
|
|
362
362
|
```
|
|
363
363
|
|
|
364
364
|
**Token forwarding:** the commit-message and PR-body generators run on a model. Forward those calls into the tracker so Phase 5's Cost Breakdown captures Phase 4:
|
|
365
365
|
|
|
366
366
|
```bash
|
|
367
|
-
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
367
|
+
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 commit.message_generated \
|
|
368
368
|
model=<sonnet|opus> tokens_in=$IN tokens_out=$OUT duration_ms=$DUR
|
|
369
|
-
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
369
|
+
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 4 pr.body_generated \
|
|
370
370
|
model=<sonnet|opus> tokens_in=$IN tokens_out=$OUT duration_ms=$DUR
|
|
371
371
|
```
|
|
372
372
|
|
|
@@ -188,9 +188,9 @@ Skipped sections: when `planTodos.enabled` is false or no `plan.todos[]` was emi
|
|
|
188
188
|
**Telemetry emission** (mandatory): forward the phase's own LLM spend (humanizer + report compose calls) to the tracker, then emit the final event:
|
|
189
189
|
|
|
190
190
|
```bash
|
|
191
|
-
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
191
|
+
LOG_METRIC_FORWARD_TO_TRACKER=1 $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 5 report.compose \
|
|
192
192
|
model=$REPORT_MODEL tokens_in=$R_IN tokens_out=$R_OUT duration_ms=$R_DUR
|
|
193
|
-
$HOME/.claude/scripts/log-metric.sh "$TASK_ID"
|
|
193
|
+
$HOME/.claude/scripts/log-metric.sh "$TASK_ID" 5 task.completed \
|
|
194
194
|
phases=$PHASE_COUNT review_cycles=$CYCLES lang=$PROMPT_LANG \
|
|
195
195
|
channels_pr=$PR_STATUS channels_jira=$JIRA_STATUS \
|
|
196
196
|
channels_confluence=$CONF_STATUS channels_wiki=$WIKI_STATUS \
|
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
|
|
16
16
|
| Phase | File |
|
|
17
17
|
| --------------------------------- | -------------------------------------------------------------------- |
|
|
18
|
-
| Modes (
|
|
18
|
+
| Modes (autopilot, analysis) | `$HOME/.claude/multi-agent-refs/phases/modes.md` |
|
|
19
19
|
| Operations (kill, purge, resume) | `$HOME/.claude/multi-agent-refs/phases/operations.md` |
|
|
20
20
|
| Phase 0: Init | `$HOME/.claude/multi-agent-refs/phases/phase-0-init.md` |
|
|
21
21
|
| Phase 1: Plan | `$HOME/.claude/multi-agent-refs/phases/phase-1-plan.md` |
|
|
@@ -28,11 +28,11 @@
|
|
|
28
28
|
## Pipeline Flow
|
|
29
29
|
|
|
30
30
|
```
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
31
|
+
0-Init -> 1-Plan -> 2-Dev -> 3-Review -> 4-Commit -> 5-Report
|
|
32
|
+
Local: The same set with no worktree - Phase 0 Step 5b answered local, so
|
|
33
|
+
work happens directly on a branch in the project root
|
|
34
34
|
|
|
35
|
-
|
|
35
|
+
One pipeline: every mode runs its whole phase set, and no answer during the run adds or removes a phase.
|
|
36
36
|
```
|
|
37
37
|
|
|
38
38
|
## Phase entry - pending steer (every phase, every mode)
|
|
@@ -99,10 +99,8 @@ done
|
|
|
99
99
|
|
|
100
100
|
This produces an initial card stack printed by both CLIs.
|
|
101
101
|
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
the rest once the answer lands, and call `tiles --new` for the second batch. Full
|
|
105
|
-
contract: `tracker-contract.md`, "Deferred registration".
|
|
102
|
+
No mode is an exception: the phase set is a property of the command, so every
|
|
103
|
+
tile is created in this one batch. Full contract: `tracker-contract.md`.
|
|
106
104
|
|
|
107
105
|
### Tracker updates (every phase boundary)
|
|
108
106
|
|
|
@@ -114,7 +112,7 @@ $HOME/.claude/scripts/phase-tracker.sh update <N> in_progress # phase starts
|
|
|
114
112
|
$HOME/.claude/scripts/phase-tracker.sh update <N> completed # phase ends OK
|
|
115
113
|
# or:
|
|
116
114
|
$HOME/.claude/scripts/phase-tracker.sh update <N> failed # phase failed
|
|
117
|
-
$HOME/.claude/scripts/phase-tracker.sh update <N> skipped # e.g.
|
|
115
|
+
$HOME/.claude/scripts/phase-tracker.sh update <N> skipped # e.g. 2 in analysis mode
|
|
118
116
|
```
|
|
119
117
|
|
|
120
118
|
After every LLM call (counts are additive; skipping this is why runs end with durations but no cost - nothing reconstructs spend afterwards):
|
|
@@ -165,7 +163,7 @@ TaskUpdate({ taskId: <saved>, status: "completed" })
|
|
|
165
163
|
bash phase-tracker.sh update <N> completed
|
|
166
164
|
```
|
|
167
165
|
|
|
168
|
-
A phase outside the command's set gets no TaskCreate at all
|
|
166
|
+
A phase outside the command's set gets no TaskCreate at all, and the set is known before the tracker boots.
|
|
169
167
|
|
|
170
168
|
**(strict) TaskCreate ordering**: All TaskCreate calls MUST fire in strict phase-number order BEFORE any TaskUpdate is applied. The native widget renders by creation order, not by phase number - out-of-order calls produce visually scrambled tile stacks (e.g. `1 ✓ · 3 ✓ · 4 ✓ · 0 ▶ · 2 ☐`) even when the underlying state is correct. Pre-marking phases as completed/skipped before Phase 0 starts is FORBIDDEN - register the tile in order with default `pending` status, then flip status via TaskUpdate when the phase actually short-circuits. Full contract in `$HOME/.claude/multi-agent-refs/tracker-contract.md` section "TaskCreate ordering (strict)".
|
|
171
169
|
|
|
@@ -69,7 +69,7 @@ Emit a progress line **at least** at every one of these moments:
|
|
|
69
69
|
|
|
70
70
|
### Phase 2 (Planning)
|
|
71
71
|
- plan draft start, plan render, user-approval prompt.
|
|
72
|
-
- **v5.3.0 Plan Approval Gate (
|
|
72
|
+
- **v5.3.0 Plan Approval Gate (interactive only - autopilot may not ask):**
|
|
73
73
|
- `clarification-ask` per round - orchestrator writes structured questions when Phase 1 flagged ambiguity (missing acceptance criteria, no Figma/endpoint link, vague language, parent-story scope drift)
|
|
74
74
|
- `clarification-answer` per round - user reply captured into `state.phases["2"].clarificationAnswers`
|
|
75
75
|
- `plan-edit-request` per free-text edit - user-supplied revision instruction captured into `state.phases["2"].planEditRequests`
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# Readiness Review (shared flow for review-jira + review-issue)
|
|
2
2
|
|
|
3
|
+
> **Pickers follow** `$HOME/.claude/multi-agent-refs/picker-contract.md`: never a one-option call, and branch on the option selected, not on its text.
|
|
4
|
+
|
|
3
5
|
Assess whether a tracker item (Jira issue or GitHub issue) is READY to hand to the multi-agent pipeline, list the concrete gaps, and - after confirmation - post them back as a comment on the item so the reporter can fix it. Read-only on code: no worktree, no branch, no commits, no dev chaining. This is the inverse of `/multi-agent:create-jira` (which authors a well-formed item); here we grade an existing one.
|
|
4
6
|
|
|
5
7
|
Both `/multi-agent:review-jira` and `/multi-agent:review-issue` execute this flow; only the provider (fetch + comment endpoint + picker) differs.
|
|
@@ -44,7 +44,7 @@ This is the single source of truth. When a contributor or model is unsure where
|
|
|
44
44
|
3. `AskUserQuestion` `question`, `options[].label` and `options[].description` all follow `OUTPUT_LANG`. Only `header` stays English: a <=12-char chip Turkish overflows. Callers branch on which option was picked, never on its literal text, and pass `default` / `ASK_CHOICE_DEFAULT` as a 1-based index. The host's own **Other** row is always English. Caller rules: `picker-contract.md`.
|
|
45
45
|
4. Always English regardless of either axis: commit messages, PR titles, branch names, code identifiers, agent-state.json values, agent-log.md, reviewer/triage system prompts.
|
|
46
46
|
|
|
47
|
-
**Failure mode this prevents.** Entering `/multi-agent`, `/multi-agent:autopilot`,
|
|
47
|
+
**Failure mode this prevents.** Entering `/multi-agent`, `/multi-agent:autopilot`, etc. and switching the assistant's conversational text or picker question copy to English while `outputLanguage="tr"` is set. The user sees a half-English half-Turkish dialogue, flagged as a pipeline bug, not a stylistic choice.
|
|
48
48
|
|
|
49
49
|
## Code & Commit Rules
|
|
50
50
|
|
|
@@ -215,7 +215,7 @@ Per Locked decision 30 of `/multi-agent:analysis` and the parallel rule in `$HOM
|
|
|
215
215
|
|
|
216
216
|
The ban is **Figma**-shaped, not MCP-shaped: the multi-agent-toolkit MCP (simulator, screenshot, xcodebuild, accessibility) is unaffected in every phase, and `smoke-no-mcp-in-dev-phases.sh` agrees - it matches `figma` in the tool name and nothing else. A bare "MCP forbidden" has been read as banning the screenshot and UI-test tools, which is how a run reaches Phase 5 with no evidence.
|
|
217
217
|
|
|
218
|
-
The per-phase matrix (which phase may fetch Figma, and what its sole design source is instead) lives in `rules/figma-pipeline.md` "Phase access matrix". It
|
|
218
|
+
The per-phase matrix (which phase may fetch Figma, and what its sole design source is instead) lives in `rules/figma-pipeline.md` "Phase access matrix". It is not restated here, per this file's own instruction two paragraphs above not to duplicate that rule file. Violation of either copy: `smoke-no-mcp-in-dev-phases.sh` fails the run.
|
|
219
219
|
|
|
220
220
|
Memory: [[mcp-only-in-analysis]]
|
|
221
221
|
|
|
@@ -90,11 +90,11 @@ Phases by mode:
|
|
|
90
90
|
| Mode | Phases |
|
|
91
91
|
|---|---|
|
|
92
92
|
| `/multi-agent` | 0,1,2,3,4,5 |
|
|
93
|
-
|
|
|
94
|
-
| `/multi-agent:autopilot
|
|
93
|
+
| answered local at Step 5b | 0,1,2,3,4,5 (the user test needs a worktree checkout and local has none, so that STEP inside Review is skipped - the phase is not) |
|
|
94
|
+
| `/multi-agent:autopilot` | 0,1,2,3,4,5 (autopilot drops the interactive user test inside Review) |
|
|
95
95
|
| `/multi-agent:analysis` | 0,1,3,4,5 (no code is written, so Dev is not in the set) |
|
|
96
96
|
|
|
97
|
-
|
|
97
|
+
Every entry registers its whole set in one batch: the set is a property of the command, known before the tracker boots.
|
|
98
98
|
|
|
99
99
|
Register each phase:
|
|
100
100
|
|
|
@@ -171,7 +171,7 @@ report's cost-unavailable list.
|
|
|
171
171
|
|
|
172
172
|
> **All TaskCreate calls for the active mode's phase set MUST fire in strict phase-number order BEFORE any TaskUpdate is applied. No "pre-mark skipped phases as completed before Phase 0" reasoning is permitted - even when the agent knows in advance that a phase will be skipped.**
|
|
173
173
|
|
|
174
|
-
Why: an agent reasoning "
|
|
174
|
+
Why: an agent reasoning "analysis mode skips Dev - let me TaskCreate it as completed first" produces tile IDs `1, 2, 3, ...` for phases 1/2/4, then the Phase 0 tile gets ID `4` and visually drops below them. The user sees `1, 2, 4 ✓ · 0 ▶ · 3 ☐ · ...` instead of `0 ▶ · 1 ✓ · 2 ✓ · 3 ☐ · 4 ✓ · ...`.
|
|
175
175
|
|
|
176
176
|
The correct sequence is **always**:
|
|
177
177
|
|
|
@@ -197,44 +197,13 @@ Mode-specific phase sets:
|
|
|
197
197
|
|
|
198
198
|
| Mode | TaskCreate set (in order) |
|
|
199
199
|
|---|---|
|
|
200
|
-
| `/multi-agent` | 0
|
|
201
|
-
|
|
|
202
|
-
| `:autopilot
|
|
200
|
+
| `/multi-agent` | 0 → 1 → 2 → 3 → 4 → 5 (6 phases) |
|
|
201
|
+
| answered local | the same set; nothing is dropped, because the user test that needs a worktree checkout is a step inside Review rather than a phase of its own |
|
|
202
|
+
| `:autopilot` | 0 → 1 → 2 → 3 → 4 → 5 (6 phases - the user test is inside Review now, so no phase is dropped) |
|
|
203
203
|
| `:analysis` | 0 → 1 → 3 → 4 → 5 (5 phases - no code is written, so Dev is not in the set) |
|
|
204
204
|
|
|
205
205
|
A phase outside the mode's set gets no TaskCreate at all; the `[SKIPPED]` pattern applies only to a phase that IS in the set and short-circuits at runtime. Phase 3 is in every mode's set as of v14.0.0. The authoritative per-mode set is the `for p in ...` init block in each mode's own entry doc, generated by `gen-mode-dispatch.mjs`; this table mirrors those blocks.
|
|
206
206
|
|
|
207
|
-
#### Deferred registration - the depth picker
|
|
208
|
-
|
|
209
|
-
`/multi-agent` and `:local` cannot know their phase set at Step -1. Depth decides it, and the depth picker cannot run before Step 7.5: its recommendation needs `taskType`, which needs the fetched issue and the branch.
|
|
210
|
-
|
|
211
|
-
Until v17.5.0 they registered all eight anyway and flipped 1 and 2 to `skipped` at 7.5. That put a widget reading "8 tasks, 7 open - Phase 1 Plan, Phase 1 Plan, ..." on screen *beside* the question asking whether to run Analysis and Planning at all, and a Short answer then contradicted a list the user had just been shown. The widget was asserting a shape the run had not chosen.
|
|
212
|
-
|
|
213
|
-
So registration splits at the moment the shape is known:
|
|
214
|
-
|
|
215
|
-
```text
|
|
216
|
-
# Step -1, first thing in the run: Phase 0 only. It is the one phase that is
|
|
217
|
-
# certain, and the run is never silent while Phase 0 does its work.
|
|
218
|
-
bash $HOME/.claude/scripts/phase-tracker.sh add 0 Init
|
|
219
|
-
bash $HOME/.claude/scripts/phase-tracker.sh tiles # -> TaskCreate(Phase 0)
|
|
220
|
-
bash $HOME/.claude/scripts/phase-tracker.sh update 0 in_progress
|
|
221
|
-
|
|
222
|
-
# Step 7.5, immediately after the depth answer:
|
|
223
|
-
# Full -> 1 2 3 4 5 Short -> 2 3 4 5
|
|
224
|
-
# :local drops nothing - the user test lives inside Review now, so there is
|
|
225
|
-
# no separate phase for it to skip
|
|
226
|
-
for p in "1:Plan" "2:Dev" "3:Review" "4:Commit" "5:Report"; do
|
|
227
|
-
bash $HOME/.claude/scripts/phase-tracker.sh add "${p%%:*}" "${p#*:}"
|
|
228
|
-
done
|
|
229
|
-
bash $HOME/.claude/scripts/phase-tracker.sh tiles --new # -> TaskCreate for the new tiles only
|
|
230
|
-
```
|
|
231
|
-
|
|
232
|
-
`tiles --new` emits `TaskCreate` only for phases that carry no `tasklist_id` yet, so the Phase 0 tile is not created twice. It is the same ordering rule, applied per batch: every tile in a batch is created in ascending phase order, and a deferred batch only ever appends phases numbered above everything already registered. Nothing is pre-marked, and a phase the run will not execute never gets a tile at all.
|
|
233
|
-
|
|
234
|
-
A phase that IS registered and short-circuits later still flips with `[SKIPPED]` - autopilot suppressing the Phase 3 user test, for instance. That is a runtime outcome, not an unknown set.
|
|
235
|
-
|
|
236
|
-
**Enforcement**: `smoke-tasklist-ordering.sh` scans the dispatcher (`commands/multi-agent/SKILL.md`) and every mode entry point doc (`commands/multi-agent/{autopilot,local,local-autopilot,analysis,resume-local}/SKILL.md` + the Copilot full-inline orchestrator mirror) for the explicit "in phase-number order" rule. Inventory drift fails the smoke.
|
|
237
|
-
|
|
238
207
|
### Other CLIs - call render after every state change
|
|
239
208
|
|
|
240
209
|
There is no TaskList outside Claude Code. Instead, after each state change the agent calls:
|
|
@@ -299,7 +268,7 @@ Throttling rules: mirror only canonical-set lines (`verbose`-tier internals are
|
|
|
299
268
|
|
|
300
269
|
### Delegated phases - mirror limitation + chunked dispatch (required)
|
|
301
270
|
|
|
302
|
-
When a phase's work is delegated to a subagent (
|
|
271
|
+
When a phase's work is delegated to a subagent (`create-component` plugin dispatch, Phase 1 explorers, Phase 3 reviewers), the visual channel freezes for the duration of the Agent call: the orchestrator is blocked while the call is in flight, so it cannot fire `TaskUpdate` / `now` / `tokens`, and a subagent cannot drive the parent session's TaskList (its own TaskCreate/TaskUpdate calls land on an invisible child list). The progress-line mirror above can therefore only fire while the orchestrator holds control. Rules:
|
|
303
272
|
|
|
304
273
|
1. **Pre-dispatch marker.** Immediately before every Agent call, set the active-phase line to the delegation itself, so the frozen interval at least states what is running and on which model:
|
|
305
274
|
- Claude Code: `TaskUpdate({activeForm: "Dev subagent (opus): <task subject>"})`
|
|
@@ -434,7 +403,7 @@ The `tasklist_id` meta from the previous session is replaced with the new IDs du
|
|
|
434
403
|
|
|
435
404
|
## Continuation runs (finish / manual-test)
|
|
436
405
|
|
|
437
|
-
A pre-existing `tracker-state.json` for the task is never re-initialized. Rules for any command that continues an earlier run (`/multi-agent:resume
|
|
406
|
+
A pre-existing `tracker-state.json` for the task is never re-initialized. Rules for any command that continues an earlier run (`/multi-agent:resume`, `/multi-agent:manual-test`):
|
|
438
407
|
|
|
439
408
|
1. `init` runs ONLY when no state file exists for the task. Otherwise the existing file is kept - phase history (elapsed, tokens, model, meta) survives.
|
|
440
409
|
2. The continuing command re-declares its phase set with `add` - `add` is idempotent, so existing phases keep their name, status, and token history; only genuinely new phases are appended. The card renders phases sorted by numeric id, so mixed sets stay in order.
|
|
@@ -11,6 +11,8 @@
|
|
|
11
11
|
- [Cross-CLI parity](#cross-cli-parity)
|
|
12
12
|
<!-- /toc -->
|
|
13
13
|
|
|
14
|
+
> **Pickers follow** `$HOME/.claude/multi-agent-refs/picker-contract.md`: never a one-option call, and branch on the option selected, not on its text.
|
|
15
|
+
|
|
14
16
|
> **TLDR** - Component tasks can auto-generate wiki docs + Figma screenshots. The Wiki adapter is invoked from `/multi-agent:channels` (Phase 5 delegates, or user invokes post-hoc). Four layouts supported (`submodule`, `in-repo`, `github-wiki`, `separate-repo`) - adapter picked from `figmaConfig.wiki.mode`. Non-blocking: failures log a warning and channels continues to other adapters. The Wiki adapter supports scope multi-select (Case A) and a precondition-failure menu (Case B) - see below.
|
|
15
17
|
|
|
16
18
|
This doc is referenced from `commands/multi-agent/channels/SKILL.md` (Wiki adapter) and indirectly from `$HOME/.claude/multi-agent-refs/phases/phase-5-report.md` (which delegates all external delivery to channels). Keeping it separate keeps both files under their token budgets and gives the contract a stable location for Claude-side + Copilot-side implementations.
|
|
@@ -75,7 +77,7 @@ Autopilot in Phase 5 pauses at the channels menu (per modes.md contract) - if
|
|
|
75
77
|
|
|
76
78
|
## Legacy prompt + preference flow (pre-v5.7, still supported for backward compat)
|
|
77
79
|
|
|
78
|
-
Interactive path (any interactive run
|
|
80
|
+
Interactive path (any interactive run), ONLY when the schema lacks `wikiScope` - ask with a native `AskUserQuestion` picker (never a typed y/n):
|
|
79
81
|
|
|
80
82
|
- `question`: "Generate component wiki docs?" (rendered in `outputLanguage`)
|
|
81
83
|
- `header`: "Wiki" (English, <=12 chars)
|
|
@@ -103,7 +105,6 @@ Explicit logs help the developer understand why wiki did or did not run:
|
|
|
103
105
|
- `figmaConfig` missing entirely → log `Phase 5: wiki skipped (no figma-config for this project)`.
|
|
104
106
|
- User declined at prompt → log `Phase 5: wiki skipped by user`.
|
|
105
107
|
- Autopilot with `wikiDefault=false` → log `Phase 5: wiki skipped (autopilot + wikiDefault=false)`.
|
|
106
|
-
- Short run: DO prompt - wiki is cheap and keeps docs fresh on the fast path; skip only if the user says no.
|
|
107
108
|
|
|
108
109
|
## Success log
|
|
109
110
|
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"schemaVersion": "2.
|
|
2
|
+
"schemaVersion": "2.8.0",
|
|
3
3
|
"global": {
|
|
4
4
|
"identities": [],
|
|
5
5
|
"keychainMapping": {
|
|
@@ -101,7 +101,7 @@
|
|
|
101
101
|
"skillConformance": {
|
|
102
102
|
"blockOnCoverageGap": false
|
|
103
103
|
},
|
|
104
|
-
"
|
|
104
|
+
"resume": {
|
|
105
105
|
"autoFix": false
|
|
106
106
|
}
|
|
107
107
|
},
|
|
@@ -75,7 +75,7 @@ The 3-tier fallback chain above governs Figma access in `/multi-agent:analysis`
|
|
|
75
75
|
| Phase | Sub-command examples | Figma MCP allowed | Figma REST allowed | Reason |
|
|
76
76
|
|---|---|---|---|---|
|
|
77
77
|
| Analysis Phase 1 | `/multi-agent:analysis` Phase 1 fetch | yes | yes (Tier 2 fallback) | Single source of design ground truth |
|
|
78
|
-
| Plan (Phase 1) | `/multi-agent`, `/multi-agent:
|
|
78
|
+
| Plan (Phase 1) | `/multi-agent`, `/multi-agent:autopilot` | no | no | Plan reads analysis doc Section 14 + Section 6 |
|
|
79
79
|
| Dev (Phase 2) | every mode that runs Phase 2 (8 modes total) | no | no | Reads analysis doc + Code Connect mapping |
|
|
80
80
|
| Review (Phase 3) | `/multi-agent:review`, every full-pipeline mode | no | no | Reviewer cites analysis doc Section 21 References |
|
|
81
81
|
| Test (inside Phase 3) | `/multi-agent:test`, `/multi-agent:manual-test`, every full mode | no | no | Variant list comes from analysis Section 13.6 + 15.2 |
|