@mmerterden/multi-agent-pipeline 20.8.3 → 20.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/CHANGELOG.md +37 -0
  2. package/docs/facts.json +1 -1
  3. package/install/claude.mjs +1 -1
  4. package/manifest.json +37 -28
  5. package/package.json +1 -1
  6. package/pipeline/lib/claude-md-links.mjs +328 -0
  7. package/pipeline/lib/owned-path-gate.mjs +699 -0
  8. package/pipeline/lib/repo-profile-derive.mjs +1771 -0
  9. package/pipeline/lib/repo-profile.mjs +780 -0
  10. package/pipeline/lib/stack-detect.sh +59 -19
  11. package/pipeline/lib/unattended.mjs +17 -0
  12. package/pipeline/multi-agent-refs/features/repo-profile.md +96 -0
  13. package/pipeline/multi-agent-refs/features/review-decision.md +18 -13
  14. package/pipeline/multi-agent-refs/features/stack-skill-routing.md +179 -33
  15. package/pipeline/multi-agent-refs/outside-the-pipeline.md +33 -11
  16. package/pipeline/multi-agent-refs/phases/phase-1-plan.md +26 -12
  17. package/pipeline/multi-agent-refs/phases/phase-2-dev.md +24 -13
  18. package/pipeline/multi-agent-refs/phases/phase-3-review.md +16 -4
  19. package/pipeline/multi-agent-refs/phases/phase-4-commit.md +1 -1
  20. package/pipeline/multi-agent-refs/phases/phase-5-report.md +8 -0
  21. package/pipeline/rules/outside-the-pipeline.md +6 -1
  22. package/pipeline/schemas/agent-state.schema.json +66 -2
  23. package/pipeline/schemas/phases.json +4 -4
  24. package/pipeline/schemas/repo-profile.schema.json +1107 -0
  25. package/pipeline/schemas/token-budget.json +4 -4
  26. package/pipeline/scripts/agent-guard.py +30 -0
  27. package/pipeline/scripts/owned-path-gate.mjs +205 -0
  28. package/pipeline/scripts/pre-commit-check.sh +151 -1
  29. package/pipeline/scripts/repo-profile.mjs +244 -0
  30. package/pipeline/scripts/review-decision-gate.mjs +42 -18
  31. package/pipeline/scripts/skill-candidates.mjs +882 -0
  32. package/pipeline/scripts/unattended_policy.py +90 -0
  33. package/pipeline/scripts/usage-report.mjs +36 -6
  34. package/pipeline/skills/.skill-manifest.json +1 -1
@@ -1,5 +1,13 @@
1
1
  # Outside a pipeline run - the detail
2
2
 
3
+ <!-- toc -->
4
+ - [Resolving a service credential](#resolving-a-service-credential)
5
+ - [Keeping the value out of everything](#keeping-the-value-out-of-everything)
6
+ - [Read here, write through a command](#read-here-write-through-a-command)
7
+ - [Which stack skills apply](#which-stack-skills-apply)
8
+ - [multi-agent-toolkit MCP](#multi-agent-toolkit-mcp)
9
+ <!-- /toc -->
10
+
3
11
  > Loaded on demand by `rules/outside-the-pipeline.md`, which carries only the
4
12
  > pointer. Everything here costs nothing until something asks for it.
5
13
 
@@ -56,22 +64,36 @@ plain session that posts directly satisfies none of them.
56
64
 
57
65
  ## Which stack skills apply
58
66
 
67
+ Write what you are about to do to a scratch file with the Write tool (ticket or
68
+ issue text never goes into a shell string), then:
69
+
59
70
  ```bash
60
- # effective set: repo settings override the global ones
61
- jq -s '.[0].enabledPlugins * .[1].enabledPlugins | to_entries
62
- | map(select(.value)) | map(.key)' \
63
- "$HOME/.claude/settings.json" .claude/settings.json 2>/dev/null
71
+ node "$HOME/.claude/scripts/skill-candidates.mjs" resolve --dir . --task-file "$TASK_FILE"
64
72
  ```
65
73
 
66
- For each enabled `@multi-agent-plugins` toolkit, load its `index` skill and let it
67
- route. The intent-to-skill table lives in the plugin and is maintained beside the
68
- skills it points at; a copy here would be the stale one. `ai-common-toolkit` and
69
- `ai-analyst-toolkit` are on everywhere - the first for cross-stack work (humanizer,
70
- accessibility audit, Firebase), the second for outside facts (GitHub and registry
71
- evidence, community signal).
74
+ `--task -` reads the same text from stdin. It reads the effective `enabledPlugins`
75
+ (the user's `settings.json` and `settings.local.json`, then the repo's
76
+ `.claude/settings.json` and `.claude/settings.local.json`), drops a toolkit
77
+ inherited from user settings whose name marks a stack this repo is not built with,
78
+ keeps every toolkit the repo enabled itself, and lists the repo's own
79
+ `.claude/skills`. `excluded[]` and `hints[]` say what was dropped and which toolkit
80
+ to enable instead. An inherited toolkit that marks no stack and ships no `index` is
81
+ in `unscoped[]`: load it only when the user names it. `overlaps[]` says which
82
+ toolkit owns a skill name two of them ship.
83
+
84
+ For each kept toolkit, load its `index` skill and let it route. The intent-to-skill
85
+ table lives in the plugin and is maintained beside the skills it points at; a copy
86
+ here would be the stale one. A repo-local skill with the same name as a toolkit skill
87
+ wins for that repo; an explicit-only one is loaded only when invoked as `/name`,
88
+ `` `name` `` or "skill name". `ai-common-toolkit` and `ai-analyst-toolkit` are on
89
+ everywhere - the first for cross-stack work (humanizer, accessibility audit,
90
+ Firebase), the second for outside facts (GitHub and registry evidence, community
91
+ signal). The full contract, including precedence, is
92
+ `features/stack-skill-routing.md`.
72
93
 
73
94
  Nothing enabled is normal: a repo whose stack was never selected simply has no
74
- toolkit. Continue without one rather than guessing which might fit.
95
+ toolkit. Continue without one rather than guessing which might fit, and never use
96
+ another stack's toolkit in its place.
75
97
 
76
98
  ## multi-agent-toolkit MCP
77
99
 
@@ -28,6 +28,14 @@ Also read project-level CLAUDE.md if exists:
28
28
  - `$PROJECT_ROOT/CLAUDE.md`
29
29
  - `$PROJECT_ROOT/.claude/CLAUDE.md`
30
30
 
31
+ **Documents those files link to are project rules too.** A project CLAUDE.md is often an index ("naming rules live in `docs/style.md`"), so read the in-repo files it links to and treat them as authoritative for planning, with the same weight as CLAUDE.md itself:
32
+
33
+ ```bash
34
+ node $HOME/.claude/lib/claude-md-links.mjs "$PROJECT_ROOT"
35
+ ```
36
+
37
+ It follows in-repo links, inline-code doc paths and `@` imports one hop, authoritative ones first, bounded (10 files, 200 KB, repo only, no SKILL.md); a file cut at the budget is in `truncated[]`, a skipped link in `rejected[]` with its reason. If a rule the plan needs was cut or skipped, read that file directly and say so.
38
+
31
39
  **Per-repo memory injection (opt-in via `prefs.global.perRepoMemory`):**
32
40
 
33
41
  ```bash
@@ -53,6 +61,14 @@ bash $HOME/.claude/scripts/log-metric.sh "$TASK_ID" 1 memory.injected \
53
61
  kind=<profile|task-relevant> rows=$N chars=$C
54
62
  ```
55
63
 
64
+ **Repo profile (every run):** read how this repo works before planning against it.
65
+
66
+ ```bash
67
+ node $HOME/.claude/scripts/repo-profile.mjs ensure --repo "$PROJECT_ROOT" --state "$STATE_FILE"
68
+ ```
69
+
70
+ Missing or stale: it derives and saves the profile outside the repo. `needsConfirmation: true` (attended only) means ask the user once to confirm it, then run `repo-profile.mjs confirm`; unattended and autopilot runs never ask. Persist `state.repoProfile`. Cite `ownedPaths` and `generators` in the plan so no task writes into them. Contract, confidence policy, confirmation picker: [`features/repo-profile.md`]($HOME/.claude/multi-agent-refs/features/repo-profile.md).
71
+
56
72
  **If knowledge files exist and are fresh** (modified within last 90 days - see knowledge.md staleness rules):
57
73
 
58
74
  1. Read relevant knowledge files based on task description
@@ -80,9 +96,8 @@ Halt if all three tiers fail; never substitute primitives or invent layout from
80
96
 
81
97
  **Spacing goes in by token NAME, per atom - never a pixel number.** `tokens[]` must
82
98
  carry each frame's spacing/padding as Figma names them (`Spacing/12`, edge `4`), keyed
83
- to the atom. Phase 2 cannot call Figma, so what is missed here is gone: one run guessed
84
- `16` where the frame said `Spacing/12` and the sheet was rebuilt. A pixel number also
85
- cannot map back to a token. No spacing entries on a UI frame is a **capture failure**,
99
+ to the atom. Phase 2 cannot call Figma, so what is missed here is gone, and a pixel
100
+ number cannot map back to a token. No spacing entries on a UI frame is a **capture failure**,
86
101
  not an empty frame - Open Question and halt. Canonical chain reference: `$HOME/.claude/rules/figma-pipeline.md` "MUST: Figma access - 3-tier fallback chain".
87
102
 
88
103
  **Telemetry:** Tier 1 uses `mcp__claude_ai_Figma__*` tools. Every such MCP invocation MUST append an entry to `state.telemetry.mcpCalls[]` as `{ "tool": "<full mcp tool name>", "phase": 1, "timestamp": "<ISO-8601>" }`, `phase` always set. Only phases 0 and 1 may carry a Figma entry. The record is what the maintainer regression check `smoke-no-mcp-in-dev-phases.sh` audits; it flags a Figma entry at `phase >= 2` and rejects an entry without a phase. Nothing checks it during a run, so an unrecorded call goes unseen.
@@ -104,10 +119,10 @@ the generated call leaves optional); an **entity** for the same concept (module'
104
119
  shared entities first, then siblings); a **mapper** over the same response; a **screen**
105
120
  doing the same interaction.
106
121
 
107
- Why blocking: one run proposed a new repository over an endpoint a sibling already
108
- wrapped **with its country parameter**, called the generated method without it, and
109
- re-invented an entity the module had. Half that branch's commits went to converging
110
- back. "Copy X and rename it" is the reuse answer, not a hint - name X's files.
122
+ Why blocking: a parallel repository over an endpoint a sibling already wraps drops the
123
+ parameters the sibling learned to pass and duplicates its entity, and converging back
124
+ costs more than the task. "Copy X and rename it" is the reuse answer, not a hint - name
125
+ X's files.
111
126
 
112
127
  #### Step 1.5 - External Context Injection (`state.contextLinks[]`)
113
128
 
@@ -137,11 +152,10 @@ repo as a language-free one. `state.stacks[]` routes toolkits, through
137
152
  `pluginsForStacks()` in `$HOME/.claude/scripts/_stack-routing.mjs`. Do not derive
138
153
  plugin names here.
139
154
 
140
- The table that used to live here scanned at depth 2, and `AndroidManifest.xml`
141
- sits at `<module>/src/main/` - depth 4 in every multi-module app - so **Android
142
- was never detected**. The owner goes to depth 5 for that one file, prunes
143
- submodules (a vendored checkout ships its own `Package.swift`, which reported a
144
- Compose app as iOS) and checks the root first so traversal order cannot decide.
155
+ `AndroidManifest.xml` sits at `<module>/src/main/`, depth 4 in a multi-module app,
156
+ so the owner goes to depth 5 for that one file. It prunes submodules (a vendored
157
+ checkout ships its own `Package.swift`) and checks the root first so traversal
158
+ order cannot decide.
145
159
 
146
160
  **What the code is WRITTEN IN** - a different axis with its own field.
147
161
  `state.detectedStack[]` stays the language answer (`ios`, `python`, `node`, `go`,
@@ -39,7 +39,7 @@ Pre-flight steps (run in order, abort on failure).
39
39
 
40
40
  `targetFiles` is required - without it a skill applied to the wrong files still reads as "applied". Append at the moment of consultation, not at the end of the phase. Phase 3 Step 1.78 treats this as self-report only and resolves criteria independently; it is the one signal separating "applied to the wrong files" from "never opened".
41
41
 
42
- 9. **Stack skill routing (every `taskType`, when a stack toolkit plugin is enabled)**: ask each enabled toolkit's own `index` skill which skills govern this task, load them BEFORE writing code, and record each into `state.telemetry.skillCalls[]` with `routedBy: "<toolkit>:index@<version>"`. Candidates are the effective `enabledPlugins`, not a stack table. The routing table stays in the plugin - a copy here would be the stale one, and `rules/outside-the-pipeline.md` runs the same routing outside a run. A screen-creation task loads the routed toolkit's `workflow/create-screen` when one exists. No toolkit, or none enabled, is a recorded no-op, not a halt. Contract: [`features/stack-skill-routing.md`]($HOME/.claude/multi-agent-refs/features/stack-skill-routing.md).
42
+ 9. **Stack skill routing (every `taskType`, always)**: write the task title + intent to `$TASK_FILE` with the Write tool (never a shell string), run `node $HOME/.claude/scripts/skill-candidates.mjs resolve --dir "$PROJECT_ROOT" --state "$STATE_FILE" --task-file "$TASK_FILE"`, and store its `mode`, `toolkits`, `excluded`, `unscoped`, `hints` and `fallbacks` in `state.telemetry.skillRouting`. Ask each kept toolkit's own `index` which skills govern this task, load them BEFORE writing code, and record each into `state.telemetry.skillCalls[]` with `phase: 2` and `routedBy: "<toolkit>:index@<version>"`. Precedence: repo-local skill (`source: "repo"`) > detected-stack toolkit skill > common toolkit; `overlaps[]` names the owner of a skill two toolkits share. Unattended or autopilot runs load `fallbacks[]` read-only (`source: "marketplace-fallback"`) and never route to `unscoped[]`; a missing stack toolkit is a hint, never another stack's toolkit. The routing table stays in the plugin. A screen-creation task loads the routed toolkit's `workflow/create-screen` when one exists. No toolkit is a recorded no-op, not a halt. Contract: [`features/stack-skill-routing.md`]($HOME/.claude/multi-agent-refs/features/stack-skill-routing.md).
43
43
 
44
44
  10. **Autopilot** (or `MULTI_AGENT_UNATTENDED=1`): a run that Phase 1's `open-questions-gate.mjs` parked (`waitingFor: "question"`), or that `spec-consistency-gate.mjs` or `plan-critique-gate.mjs` failed (`verificationFailed.gate: spec-consistency` / `plan-critique`), does not enter this phase. When `state.research.md` exists, read it before writing code: what research closed, each answer with its source ([`features/research.md`]($HOME/.claude/multi-agent-refs/features/research.md)).
45
45
 
@@ -357,12 +357,24 @@ scenario indexes, localization keys, testing identifiers, design tokens. A gener
357
357
  file is regenerated on the next build, so an edit there is lost silently, and the
358
358
  matching hand-authored tree is the one that takes the change.
359
359
 
360
- Before writing into any path, check whether it is generated:
360
+ Before writing into any path, check whether it is generated or owned; the repo
361
+ profile answers both:
362
+
363
+ ```bash
364
+ node $HOME/.claude/scripts/repo-profile.mjs resolve generators --repo "$PROJECT_ROOT" --state "$STATE_FILE"
365
+ node $HOME/.claude/scripts/repo-profile.mjs resolve ownedPaths --repo "$PROJECT_ROOT" --state "$STATE_FILE"
366
+ ```
367
+
368
+ A target under a `generators[].output` goes to that entry's `input` and is
369
+ regenerated with its `command` (in the same commit when `sameCommit`). A target
370
+ under an `ownedPaths[].glob` is not edited at all: report the `owner`, and the
371
+ `bypass` label when there is one; exit Gate 0 enforces this. Without a profile,
372
+ look for the signals directly, case-insensitively:
361
373
 
362
374
  ```bash
363
375
  # a Generated/ segment, or a header saying so, is the signal
364
- find . -type d -name Generated -not -path './.*' | head
365
- grep -rl "DO NOT EDIT\|auto-generated\|Generated by" --include="*.swift" --include="*.kt" . | head
376
+ find . -type d -iname Generated -not -path './.*' | head
377
+ grep -rliE "do not edit|auto-?generated|generated by" --include="*.swift" --include="*.kt" . | head
366
378
  ```
367
379
 
368
380
  The pairing is usually `Generated/<x>` for output and `Custom<X>/` or
@@ -373,10 +385,6 @@ The pairing is usually `Generated/<x>` for output and `Custom<X>/` or
373
385
  | add a mock fixture / named scenario | a `Fixtures/` file under a generated tree | the repo's custom fixture tree, plus registering the scenario in the generated index the build reads |
374
386
  | add or change a service endpoint | the generated client method | the OpenAPI source the generator consumes, then regenerate |
375
387
 
376
- One run wrote a mock fixture into the generated fixtures tree; the fix commit moved it
377
- to the custom tree and registered the scenario in the generated index. Same content,
378
- wrong side of the generator, and the Debug menu never showed it.
379
-
380
388
  When the analysis doc has not recorded which trees are generated, that is a Phase 1
381
389
  gap - say so rather than guessing, since guessing wrong is invisible until the next
382
390
  regeneration.
@@ -385,16 +393,17 @@ regeneration.
385
393
 
386
394
  ## Exit gate: Verify (BLOCKING)
387
395
 
388
- These four deterministic gates ran at the top of Review until v19.0.0, which meant
389
- the build ran twice: once here at the end of Dev, once again as Review's Stage 1.
390
- They now run once, at this phase's exit, and Review inherits the logs instead of
391
- regenerating them. A failure here does not reach Review at all.
396
+ These deterministic gates run once, at this phase's exit, so the build runs once:
397
+ Review inherits the logs instead of regenerating them. A failure here does not
398
+ reach Review at all.
392
399
 
393
400
  #### Step 1 - Deterministic Gates (run BEFORE AI review)
394
401
 
395
402
  If any gate fails → fix first, don't waste AI tokens reviewing broken code.
396
403
 
397
404
  ```bash
405
+ # Gate 0: Owned paths (repo profile), before the build
406
+ node $HOME/.claude/scripts/owned-path-gate.mjs --repo "$WORKTREE" --base "origin/$BASE_BRANCH" --state "$STATE_FILE"
398
407
  # Gate 1: Build (xcodebuild/gradle assemble/tsc/py compile - stack-dependent; Xcode uses the build queue lock, see Phase 2) - tee output to a log
399
408
  <build-command> 2>&1 | tee "{worktreePath}/.build.log"
400
409
  # Gate 2: Lint (swiftlint/ktlint/ruff/eslint - stack-dependent)
@@ -404,6 +413,8 @@ If any gate fails → fix first, don't waste AI tokens reviewing broken code.
404
413
  bash $HOME/.claude/scripts/pre-commit-check.sh
405
414
  ```
406
415
 
416
+ **Gate 0** exit 1 blocks the phase: move each change where its `fix` points, revert the owned path, re-run. Exit 2 is a call error, never a pass: an unknown `--base` ref means fetch the base branch. Exit 0 with `skipped: true` gives its reason in `note`; log it. Contract: `features/repo-profile.md`, "Owned-path gate".
417
+
407
418
  **Default-FAIL evidence gate (required before recording any pass):** a green exit code is not enough - the captured log must actually show success. Before marking build/test passed, run the evidence gate against the tee'd log; it fails CLOSED when the log is missing, empty, or shows failure markers:
408
419
 
409
420
  ```bash
@@ -432,7 +443,7 @@ The subtraction never widens: match on identifier only, and when identifiers can
432
443
 
433
444
  - All pass (including the evidence gate) -> proceed to AI review
434
445
  - Any fail -> fix immediately, re-run gates (no AI review until clean)
435
- Log: "Phase 3: Gates - build:{pass/fail} lint:{pass/fail} test:{pass/fail/inherited-red} secrets:{clean/found} evidence:{ok/unverified}"
446
+ Log: "Phase 3: Gates - owned:{ok/fail} build:{pass/fail} lint:{pass/fail} test:{pass/fail/inherited-red} secrets:{clean/found} evidence:{ok/unverified}"
436
447
 
437
448
  ##### Gate 5 - Fortify SSC findings (runs when `state.contextLinks[]` contains a `fortify` entry, or when `prefs.global.fortify.alwaysCheck === true`)
438
449
 
@@ -135,6 +135,18 @@ TI_COUNT=$(jq -r '.count // 0' "$TI_FILE" 2>/dev/null || echo 0)
135
135
 
136
136
  The gate never blocks the phase; it *emits* blocking findings (unreadable input: zero). Feed it the FULL report: `--top` hides a shrinking test file below the cut. **No opt-out**: a run that can switch off its own anti-reward-hacking control cannot be trusted to report a pass.
137
137
 
138
+ #### Step 1.761 - Owned-path gate (BLOCKING findings)
139
+
140
+ Phase 2 Gate 0 again, same base:
141
+
142
+ ```bash
143
+ OP_FILE="$WORKTREE/.pipeline/owned-path.json"
144
+ node $HOME/.claude/scripts/owned-path-gate.mjs --repo "$WORKTREE" --base "origin/$BASE_BRANCH" \
145
+ --state "$STATE_FILE" --out "$OP_FILE"
146
+ ```
147
+
148
+ `--out` drops the old file; exit 2 writes none. A missing `$OP_FILE` is a gate error, never "no findings" (`review-decision-gate.mjs` exit 3 under the gates). Findings merge at Step 3.0. Contract: `features/repo-profile.md`.
149
+
138
150
  #### Step 1.77 - Reviewer scope (cost gate)
139
151
 
140
152
  On a trivial diff every reviewer agrees and the extra models plus triage are paid for nothing. Reviewer count comes from the same report, no LLM.
@@ -365,15 +377,15 @@ ANON=$(jq -n --argjson r "$REVIEWERS_JSON" --arg t "$TASK_ID" --argjson i "$ITER
365
377
 
366
378
  `$REVIEWERS_JSON` is `state.reviewIterations[i].reviewers`. Findings come back with `foundBy: "Source A|B|C"` and every identity key removed. Persist the map to `state.reviewIterations[i].anonymizationMap` for Phase 5 per-reviewer telemetry, and **never put the map in a prompt**.
367
379
 
368
- Then append the Step 1.76 test-integrity and Step 2.7 security-audit findings, so they are adjudicated rather than never seen:
380
+ Then append the Step 1.76 test-integrity, 1.761 owned-path and 2.7 security-audit findings, so they are adjudicated:
369
381
 
370
382
  ```bash
371
383
  SA_FILE="$WORKTREE/.pipeline/security-audit-$ITERATION.json"
372
- MERGED=$(printf '%s' "$ANON" | cat - "$TI_FILE" "$SA_FILE" 2>/dev/null \
384
+ MERGED=$(printf '%s' "$ANON" | cat - "$TI_FILE" "$OP_FILE" "$SA_FILE" 2>/dev/null \
373
385
  | jq -s '.[0] + ([.[1:][] | .findings // []] | add // [])')
374
386
  ```
375
387
 
376
- Deterministic findings keep `tag: test_integrity` and no `foundBy`: a reviewer finding may hallucinate, a gate finding is a fact. Security-audit findings carry their `security` envelope and `foundBy: "security-auditor"`; no audit file contributes nothing.
388
+ Deterministic findings keep their `tag` (`test_integrity`, `owned_path`) and no `foundBy`: a reviewer finding may hallucinate, a gate finding is a fact. Security-audit findings carry their `security` envelope and `foundBy: "security-auditor"`; no audit file contributes nothing.
377
389
 
378
390
  ##### 3.1 Short-circuit: no findings
379
391
 
@@ -538,7 +550,7 @@ A triage verdict is judgment; a failing repro test is proof. Runs only when `pre
538
550
 
539
551
  Compressed flow: dispatch ONE verifier agent (model `verifyByTest.model`, default `sonnet`) for up to `maxFindings` (default 3) accepted blocking findings. Per finding it writes ONE minimal repro test and runs ONLY that test (Phase 2 single-test invocation, build lock, log tee'd to `$WORKTREE/.pipeline/verify-<i>.test.log`). Outcomes: test FAILS as predicted -> `confirmed`, finding stays blocking and the test is KEPT in `redTests[]` as the Phase 2 rework RED test; test PASSES on every one of `verifyByTest.repeatCount` runs (default 3) -> `not-reproduced` ONLY if `evidence-gate.mjs --claim test --status passed` exits 0 on each log; a run that disagrees with the others -> `inconclusive` with `flaky: passed k/N`, finding moves to `deferred[]`, test deleted; compile error / timeout / not unit-testable -> `inconclusive`, judgment verdict stands. Stamp findings with `verification` (schema v3.2.0), persist `state.reviewIterations[-1].verifyByTest = {attempted, confirmed, downgraded, inconclusive, redTests[]}`, recompute `approved`, re-run `validate-triage.mjs` under the 3.2.1 gate. Whole step bounded by `stepTimeoutSec` (default 600); on breach or crash remaining findings keep judgment verdicts - never blocks. Telemetry per 3.4: `review.verify_by_test attempted= confirmed= downgraded= inconclusive= duration_ms=`.
540
552
 
541
- **Autopilot decision rule (every round, after 3.7):** run the call in `$HOME/.claude/multi-agent-refs/features/review-decision.md` (Wiring: `--integrity "$TI_FILE" --source "$SA_FILE"`), merge its JSON into `state.reviewIterations[-1].reviewDecision`, repeat the `cp`. A blocker backed by neither two reviewers nor a failing test becomes important; on exit 1 or 3 run `gate-ledger.mjs park --outcome verification-failed --gate review-decision`.
553
+ **Autopilot decision rule (every round, after 3.7):** run the call in `$HOME/.claude/multi-agent-refs/features/review-decision.md` (Wiring: `--integrity "$TI_FILE" --integrity "$OP_FILE" --source "$SA_FILE"`), merge its JSON into `state.reviewIterations[-1].reviewDecision`, repeat the `cp`. A blocker backed by neither two reviewers nor a failing test becomes important; on exit 1 or 3 run `gate-ledger.mjs park --outcome verification-failed --gate review-decision`.
542
554
 
543
555
  ##### 3.8 Cross-round delta + circuit-breaker trigger 2 (iteration >= 2)
544
556
 
@@ -109,7 +109,7 @@ Branch **deterministically**, no implicit fallback. Read `agent-state.json` and
109
109
  - Detect which submodules have changes: `git -C {worktree} diff --name-only | cut -d'/' -f1-3 | sort -u`
110
110
  - For EACH submodule with changes: stage, commit, push separately
111
111
  - Commit message uses same convention but scope reflects submodule: `{type}({submodule-scope}): {description} [{jiraId}]`
112
- 6. Commit with convention: `{type}({scope}): {description} [{jiraId}]` or `[#{shortId}]`. **Local-only repos** (state field `projects[i].provider == "local"`) carry the same conventional prefix but drop the `[{jiraId}]` suffix - there is no tracker to reference. The taskId (`LOCAL-...`) goes in the commit body footer instead: `Local-Task: {taskId}`.
112
+ 6. Commit with convention: `{type}({scope}): {description} [{jiraId}]` or `[#{shortId}]`. When `node $HOME/.claude/scripts/repo-profile.mjs resolve commit.format --repo "$PROJECT_ROOT"` returns `use: true`, its `ticketPosition` places the ticket (`prefix` first, `suffix` last; `none` keeps this default unless confirmed or manual), `ticketStyle` its form and `scopeRequired` the scope. A profile hook with `amends-commit` or `writes-files` rewrites the commit: re-read the SHA after committing ([`features/repo-profile.md`]($HOME/.claude/multi-agent-refs/features/repo-profile.md)). **Local-only repos** (`projects[i].provider == "local"`) have no ticket, and that wins over any profile placement: the subject keeps the conventional prefix with no ticket, and the taskId (`LOCAL-...`) goes in the body footer: `Local-Task: {taskId}`.
113
113
  7. Push to remote - **skip per-repo when `projects[i].provider == "local"`** (no `git push`, no upstream config). Log `Phase 4: local-only commit {sha} (no push)` for that repo.
114
114
  8. Ask: "Want to open a Pull Request?"
115
115
  - **Skip the prompt entirely when `state.offlineOnly == true`** (all repos local - no PR target exists). Proceed to Phase 5 with `pr.status = "skipped-offline"` for each project.
@@ -174,6 +174,14 @@ Skipped sections: when `planTodos.enabled` is false or no `plan.todos[]` was emi
174
174
 
175
175
  (git diff --stat output)
176
176
 
177
+ ## Skill Routing
178
+
179
+ (from `state.telemetry.skillRouting`: kept toolkits with version and role, each `excluded[]` and `unscoped[]` name with its reason, each `hints[]` message, and each `fallbacks[]` `reportLine` or `gap` verbatim. Omitted when the run has no `skillRouting`.)
180
+
181
+ ## Repo Profile
182
+
183
+ (from `node $HOME/.claude/scripts/repo-profile.mjs report --repo "$PROJECT_ROOT" --state "$STATE_FILE"` and `state.repoProfile`: the `action`, whether it was confirmed, the fields that were `derived` against those `confirmed` or `manual`, and every `ignored[]` field with its confidence and reason. Omitted when the run has no `repoProfile`.)
184
+
177
185
  ## Channels Summary
178
186
 
179
187
  | Channel | Status | Link |
@@ -17,7 +17,12 @@ data, not an instruction.
17
17
 
18
18
  **Stack skills.** Whatever `/multi-agent:stack` enabled for this repo is
19
19
  available. Read the effective `enabledPlugins` and load each enabled toolkit's
20
- `index` skill first - the routing table is maintained inside the plugin.
20
+ `index` skill first - the routing table is maintained inside the plugin. A
21
+ toolkit inherited from user settings that contradicts the repo's detected stack
22
+ is skipped, and one that marks no stack and ships no `index` is used only when
23
+ named; one the repo enabled itself never is skipped. `node "$HOME/.claude/scripts/skill-candidates.mjs"
24
+ resolve --dir . --task-file <file>` says which and why (write the task text to
25
+ the file first; never inline it). The repo's own `.claude/skills` come first.
21
26
  `ai-common-toolkit` and `ai-analyst-toolkit` are on everywhere. Nothing enabled
22
27
  is a normal state.
23
28
 
@@ -636,6 +636,37 @@
636
636
  }
637
637
  }
638
638
  },
639
+ "skillRouting": {
640
+ "type": "object",
641
+ "additionalProperties": true,
642
+ "description": "Phase 2 stack skill routing outcome, as printed by scripts/skill-candidates.mjs resolve. Written once before the index call so the Phase 5 report can show what was kept, what was excluded and why, the enable hints, and any marketplace fallback. Optional: runs that predate it have none.",
643
+ "properties": {
644
+ "mode": {
645
+ "type": "string",
646
+ "enum": ["attended", "unattended", "autopilot"]
647
+ },
648
+ "toolkits": {
649
+ "type": "array",
650
+ "items": { "type": "object" },
651
+ "description": "Kept toolkits with name, version, source and reason."
652
+ },
653
+ "excluded": {
654
+ "type": "array",
655
+ "items": { "type": "object" },
656
+ "description": "Inherited toolkits dropped for contradicting the detected stack, each with its reason."
657
+ },
658
+ "hints": {
659
+ "type": "array",
660
+ "items": { "type": "object" },
661
+ "description": "Detected stacks with no enabled toolkit, each with the /multi-agent:stack line to enable one."
662
+ },
663
+ "fallbacks": {
664
+ "type": "array",
665
+ "items": { "type": "object" },
666
+ "description": "Unattended and autopilot runs only: toolkits read read-only from a marketplace clone (with reportLine), or the recorded gap when no clone carries one."
667
+ }
668
+ }
669
+ },
639
670
  "skillCalls": {
640
671
  "type": "array",
641
672
  "description": "One entry per skill / plugin skill / stack guide consulted while writing code. Append at the moment of consultation, not retrospectively. This is a self-report and shares the known weakness of mcpCalls[]: an unrecorded consultation and no consultation are byte-identical here, so Step 1.78 treats the deterministic resolver as primary, defaults ledger.source to 'derived', and flags a declared skill it cannot bind to a changed file. Recording it still earns its keep - it is the only signal that distinguishes 'the dev phase applied X to the wrong files' from 'X was never opened'.",
@@ -653,7 +684,7 @@
653
684
  "type": "integer",
654
685
  "minimum": 0,
655
686
  "maximum": 5,
656
- "description": "Phase the consultation happened in. Normally 3."
687
+ "description": "Phase the consultation happened in. Normally 2 (Dev)."
657
688
  },
658
689
  "targetFiles": {
659
690
  "type": "array",
@@ -669,6 +700,19 @@
669
700
  "routedBy": {
670
701
  "type": "string",
671
702
  "description": "Set when a stack toolkit's own index skill chose this skill, as '<toolkit>:index@<version>'. Recorded so a finding can be traced to the index version that selected it - the skill set differs between plugin versions. Phase 3 Step 1.78 surfaces these separately in the manifest ledger; it does NOT grant them extra trust, because the resolver stays primary either way."
703
+ },
704
+ "source": {
705
+ "type": "string",
706
+ "enum": ["repo", "toolkit", "marketplace-fallback", "pipeline", "host"],
707
+ "description": "Where the loaded skill came from. 'repo' = the repo's own .claude/skills/<name>/SKILL.md (routedBy 'repo:.claude/skills'), which takes precedence over a toolkit skill of the same name; 'toolkit' = an enabled toolkit plugin; 'marketplace-fallback' = an unattended or autopilot run read a detected stack's toolkit read-only from a local marketplace clone because the repo does not enable it (toolkit and version are then required in practice); 'pipeline' = a skill the pipeline ships; 'host' = picked by the host's own description matching. Optional: entries written before this field existed carry no source."
708
+ },
709
+ "toolkit": {
710
+ "type": "string",
711
+ "description": "Toolkit plugin name the skill came from, set with source 'marketplace-fallback'."
712
+ },
713
+ "version": {
714
+ "type": "string",
715
+ "description": "That toolkit's version, from its plugin.json in the marketplace clone."
672
716
  }
673
717
  }
674
718
  }
@@ -783,6 +827,26 @@
783
827
  }
784
828
  }
785
829
  },
830
+ "repoProfile": {
831
+ "type": ["object", "null"],
832
+ "additionalProperties": false,
833
+ "description": "Outcome of `repo-profile.mjs ensure` at Phase 1: where the per-repo profile lives and how this run treated it. The profile itself stays at the path, outside the repo (multi-agent-refs/features/repo-profile.md).",
834
+ "properties": {
835
+ "path": { "type": "string" },
836
+ "action": { "type": "string", "enum": ["loaded", "derived", "rederived"] },
837
+ "mode": { "type": "string", "enum": ["attended", "unattended"] },
838
+ "confirmed": {
839
+ "type": "boolean",
840
+ "description": "The profile carried confirmedAt when the run used it."
841
+ },
842
+ "staleReasons": { "type": "array", "items": { "type": "string" } },
843
+ "ignored": {
844
+ "type": "array",
845
+ "description": "Fields the run did not act on, with the confidence policy's reason.",
846
+ "items": { "type": "object", "additionalProperties": true }
847
+ }
848
+ }
849
+ },
786
850
  "designCheck": {
787
851
  "type": ["object", "null"],
788
852
  "additionalProperties": true,
@@ -1328,7 +1392,7 @@
1328
1392
  "reviewDecision": {
1329
1393
  "type": "object",
1330
1394
  "additionalProperties": true,
1331
- "description": "review-decision-gate.mjs --json output for this iteration, merged by Phase 3 after Step 3.7 while the quality gates are active (features/review-decision.md). kept[] lists the blocking findings that stayed blocking with their basis (corroborated | failing-test | test-integrity); downgraded[] lists each blocking finding lowered to important with its fingerprint and reason. Absent on an attended run.",
1395
+ "description": "review-decision-gate.mjs --json output for this iteration, merged by Phase 3 after Step 3.7 while the quality gates are active (features/review-decision.md). kept[] lists the blocking findings that stayed blocking with their basis (corroborated | failing-test | test-integrity | owned-path); downgraded[] lists each blocking finding lowered to important with its fingerprint and reason. Absent on an attended run.",
1332
1396
  "properties": {
1333
1397
  "verdict": {
1334
1398
  "type": "string",
@@ -28,7 +28,7 @@
28
28
  "id": 1,
29
29
  "name": "Plan",
30
30
  "doc": "phase-1-plan.md",
31
- "maxTokens": 10000,
31
+ "maxTokens": 10250,
32
32
  "was": [1, 2]
33
33
  },
34
34
  {
@@ -49,18 +49,18 @@
49
49
  "id": 4,
50
50
  "name": "Commit",
51
51
  "doc": "phase-4-commit.md",
52
- "maxTokens": 6500,
52
+ "maxTokens": 6600,
53
53
  "was": [6]
54
54
  },
55
55
  {
56
56
  "id": 5,
57
57
  "name": "Report",
58
58
  "doc": "phase-5-report.md",
59
- "maxTokens": 5550,
59
+ "maxTokens": 5700,
60
60
  "was": [7]
61
61
  }
62
62
  ],
63
- "totalMaxTokens": 63300,
63
+ "totalMaxTokens": 63550,
64
64
  "modes": {
65
65
  "full": {
66
66
  "phases": [0, 1, 2, 3, 4, 5],