@mmerterden/multi-agent-pipeline 13.5.0 → 14.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/CHANGELOG.md +243 -0
  2. package/README.md +3 -3
  3. package/docs/features.md +1 -1
  4. package/install/_common.mjs +73 -0
  5. package/install/_mcp-register.mjs +70 -31
  6. package/install/_plugin-skills.mjs +73 -14
  7. package/install/claude.mjs +28 -4
  8. package/install/codex.mjs +33 -2
  9. package/install/copilot.mjs +145 -9
  10. package/install/index.mjs +10 -6
  11. package/install/templates/copilot-instructions.md +1 -1
  12. package/package.json +1 -1
  13. package/pipeline/agents/code-reviewer.md +58 -1
  14. package/pipeline/commands/multi-agent/SKILL.md +7 -5
  15. package/pipeline/commands/multi-agent/analysis/SKILL.md +7 -7
  16. package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +1 -1
  17. package/pipeline/commands/multi-agent/build-optimize/SKILL.md +7 -7
  18. package/pipeline/commands/multi-agent/channels/SKILL.md +5 -5
  19. package/pipeline/commands/multi-agent/dev/SKILL.md +23 -18
  20. package/pipeline/commands/multi-agent/dev-autopilot/SKILL.md +19 -13
  21. package/pipeline/commands/multi-agent/dev-local/SKILL.md +14 -12
  22. package/pipeline/commands/multi-agent/dev-local-autopilot/SKILL.md +17 -12
  23. package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
  24. package/pipeline/commands/multi-agent/help/SKILL.md +4 -4
  25. package/pipeline/commands/multi-agent/ios-coding-standard/SKILL.md +2 -2
  26. package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +4 -4
  27. package/pipeline/commands/multi-agent/resume/SKILL.md +1 -1
  28. package/pipeline/commands/multi-agent/review/SKILL.md +5 -5
  29. package/pipeline/commands/multi-agent/scan/SKILL.md +1 -1
  30. package/pipeline/commands/multi-agent/search/SKILL.md +1 -1
  31. package/pipeline/commands/multi-agent/setup/SKILL.md +6 -6
  32. package/pipeline/commands/multi-agent/{finish → ship}/SKILL.md +12 -12
  33. package/pipeline/commands/multi-agent/testflight-validation/SKILL.md +1 -1
  34. package/pipeline/commands/multi-agent/update/SKILL.md +5 -2
  35. package/pipeline/commands/sim-test.md +2 -2
  36. package/pipeline/lib/credential-store-resolver.sh +16 -0
  37. package/pipeline/lib/credential-store.sh +47 -4
  38. package/pipeline/lib/fetch-figma-annotations.sh +26 -28
  39. package/pipeline/lib/figma-screenshot.sh +28 -39
  40. package/pipeline/lib/figma-token.sh +63 -0
  41. package/pipeline/multi-agent-refs/analysis-template.md +1 -1
  42. package/pipeline/multi-agent-refs/android-guide.md +1 -1
  43. package/pipeline/multi-agent-refs/channels/issue-comment.md +1 -1
  44. package/pipeline/multi-agent-refs/component-dispatch.md +2 -2
  45. package/pipeline/multi-agent-refs/cross-cli-contract.md +4 -4
  46. package/pipeline/multi-agent-refs/features/dev-critic.md +2 -2
  47. package/pipeline/multi-agent-refs/features/model-fallback.md +35 -2
  48. package/pipeline/multi-agent-refs/features/plan-todos.md +1 -1
  49. package/pipeline/multi-agent-refs/features/repo-map.md +1 -1
  50. package/pipeline/multi-agent-refs/features/review-multi-repo.md +3 -3
  51. package/pipeline/multi-agent-refs/features/shadow-git.md +1 -1
  52. package/pipeline/multi-agent-refs/features/skill-conformance.md +116 -0
  53. package/pipeline/multi-agent-refs/features/verify-by-test.md +1 -1
  54. package/pipeline/multi-agent-refs/generate-issue.md +1 -1
  55. package/pipeline/multi-agent-refs/multi-repo-integration-build.md +1 -1
  56. package/pipeline/multi-agent-refs/phases/log-format.md +4 -4
  57. package/pipeline/multi-agent-refs/phases/modes.md +7 -7
  58. package/pipeline/multi-agent-refs/phases/phase-0-init.md +13 -11
  59. package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +17 -15
  60. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +7 -7
  61. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +28 -13
  62. package/pipeline/multi-agent-refs/phases/phase-4-review.md +90 -58
  63. package/pipeline/multi-agent-refs/phases/phase-5-test.md +7 -7
  64. package/pipeline/multi-agent-refs/phases/phase-6-commit.md +8 -8
  65. package/pipeline/multi-agent-refs/phases/phase-7-report.md +8 -8
  66. package/pipeline/multi-agent-refs/phases.md +13 -13
  67. package/pipeline/multi-agent-refs/progress-contract.md +2 -2
  68. package/pipeline/multi-agent-refs/rules.md +7 -5
  69. package/pipeline/multi-agent-refs/swiftui-guide.md +1 -1
  70. package/pipeline/multi-agent-refs/tracker-contract.md +16 -15
  71. package/pipeline/preferences-template.json +7 -1
  72. package/pipeline/rules/figma-pipeline.md +2 -2
  73. package/pipeline/schemas/agent-state.schema.json +333 -79
  74. package/pipeline/schemas/criteria-manifest.schema.json +228 -0
  75. package/pipeline/schemas/migrations/prefs-2.4.0-to-2.5.0.mjs +64 -0
  76. package/pipeline/schemas/prefs.schema.json +118 -262
  77. package/pipeline/schemas/reviewer-output.schema.json +48 -3
  78. package/pipeline/schemas/token-budget.json +34 -10
  79. package/pipeline/schemas/triage-output.schema.json +112 -27
  80. package/pipeline/scripts/cost-table.json +7 -4
  81. package/pipeline/scripts/gc-worktrees.sh +1 -1
  82. package/pipeline/scripts/gen-mode-dispatch.mjs +6 -6
  83. package/pipeline/scripts/match-skills.mjs +37 -4
  84. package/pipeline/scripts/migrate-prefs.mjs +88 -17
  85. package/pipeline/scripts/phase-tracker.sh +14 -3
  86. package/pipeline/scripts/pre-commit-check.sh +49 -2
  87. package/pipeline/scripts/skill-conformance.mjs +960 -0
  88. package/pipeline/scripts/smoke-schema-validation.sh +17 -4
  89. package/pipeline/scripts/uninstall.mjs +35 -9
  90. package/pipeline/scripts/validate-reviewer.mjs +108 -1
  91. package/pipeline/skills/.skill-manifest.json +1 -1
  92. package/pipeline/skills/.skills-index.json +36 -9
  93. package/pipeline/skills/shared/README.md +15 -12
  94. package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +1 -0
  95. package/pipeline/skills/shared/core/apple-archive-compliance/references/rules.yml +167 -0
  96. package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +1 -0
  97. package/pipeline/skills/shared/core/google-play-compliance/references/rules.yml +184 -0
  98. package/pipeline/skills/shared/core/multi-agent/SKILL.md +10 -10
  99. package/pipeline/skills/shared/core/multi-agent-analysis/SKILL.md +4 -4
  100. package/pipeline/skills/shared/core/multi-agent-analysis-resolve/SKILL.md +3 -3
  101. package/pipeline/skills/shared/core/multi-agent-build-optimize/SKILL.md +2 -2
  102. package/pipeline/skills/shared/core/multi-agent-create-jira/SKILL.md +1 -1
  103. package/pipeline/skills/shared/core/multi-agent-dev/SKILL.md +6 -5
  104. package/pipeline/skills/shared/core/multi-agent-dev-autopilot/SKILL.md +7 -6
  105. package/pipeline/skills/shared/core/multi-agent-dev-local/SKILL.md +4 -3
  106. package/pipeline/skills/shared/core/multi-agent-dev-local-autopilot/SKILL.md +2 -1
  107. package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +2 -2
  108. package/pipeline/skills/shared/core/multi-agent-ios-coding-standard/SKILL.md +1 -1
  109. package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +4 -4
  110. package/pipeline/skills/shared/core/multi-agent-review/SKILL.md +5 -5
  111. package/pipeline/skills/shared/core/multi-agent-scan/SKILL.md +1 -1
  112. package/pipeline/skills/shared/core/multi-agent-search/SKILL.md +1 -1
  113. package/pipeline/skills/shared/core/{multi-agent-finish → multi-agent-ship}/SKILL.md +8 -8
  114. package/pipeline/skills/shared/external/ios-coding-standard/SKILL.md +44 -5
  115. package/pipeline/skills/shared/external/ios-coding-standard/modules/_TEMPLATE.yml +82 -0
  116. package/pipeline/skills/shared/external/ios-coding-standard/references/STANDARD.md +169 -10
  117. package/pipeline/skills/shared/external/ios-coding-standard/references/lint-local.sh +13 -1
  118. package/pipeline/skills/shared/external/ios-coding-standard/references/rules.yml +335 -16
  119. package/pipeline/skills/skills-index.md +11 -8
@@ -1,9 +1,9 @@
1
1
  {
2
2
  "$schema": "https://json-schema.org/draft/2020-12/schema",
3
3
  "$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/reviewer-output.schema.json",
4
- "version": "1.0.0",
4
+ "version": "1.1.0",
5
5
  "title": "Multi-Agent Pipeline - Phase 4 reviewer output",
6
- "description": "Contract for a single code-reviewer subagent's JSON output in Phase 4 Step 2. Reviewer set is CLI-aware: Claude Code dispatches 2 parallel reviewers (Fable, Sonnet); Copilot CLI dispatches 3 (Opus, GPT-5.4, Sonnet); Codex CLI dispatches 3 (gpt-5.6 at xhigh, gpt-5.4, gpt-5.6 at medium). Every reviewer must return an object matching this shape before Opus triage merges them.",
6
+ "description": "Contract for a single code-reviewer subagent's JSON output in Phase 4 Step 2. Reviewer set is CLI-aware: Claude Code dispatches 2 parallel reviewers (Fable, Sonnet); Copilot CLI dispatches 3 (Opus, GPT-5.4, Sonnet); Codex CLI dispatches 3 (gpt-5.6 at xhigh, gpt-5.4, gpt-5.6 at medium). Every reviewer must return an object matching this shape before Opus triage merges them. v1.1.0 adds the rule-ID conformance checklist: when the orchestrator supplies a ${CRITERIA} block (Phase 4 Step 1.78), the reviewer must return one conformance row per selected rule ID. Findings alone cannot answer 'was this applied completely' - a reviewer that opened nothing returns the same empty findings array as one that checked everything.",
7
7
  "type": "object",
8
8
  "additionalProperties": false,
9
9
  "required": ["findings", "approved"],
@@ -11,7 +11,9 @@
11
11
  "findings": {
12
12
  "type": "array",
13
13
  "description": "Raw findings. May be empty. Duplicates and out-of-scope items are acceptable - Opus triage will deduplicate and filter.",
14
- "items": { "$ref": "#/$defs/finding" }
14
+ "items": {
15
+ "$ref": "#/$defs/finding"
16
+ }
15
17
  },
16
18
  "approved": {
17
19
  "type": "boolean",
@@ -20,6 +22,39 @@
20
22
  "reviewer": {
21
23
  "type": "string",
22
24
  "description": "Model label for this output (e.g. 'fable', 'opus', 'sonnet', 'gpt'). Present once the parallel reviewer outputs are merged into the Phase 4 array so triage/consensus can attribute each finding to its source. Optional on a single reviewer's raw pre-merge output."
25
+ },
26
+ "conformance": {
27
+ "type": "array",
28
+ "description": "One row per rule ID in the supplied ${CRITERIA} block, and none outside it. Required whenever a ${CRITERIA} block was supplied; omitted entirely when it was not. This is the denominator that makes completeness checkable: the set is fixed on disk before the reviewer runs, so it cannot be narrowed after seeing the diff.",
29
+ "items": {
30
+ "type": "object",
31
+ "additionalProperties": false,
32
+ "required": ["ruleId", "verdict"],
33
+ "properties": {
34
+ "ruleId": {
35
+ "type": "string",
36
+ "minLength": 1,
37
+ "description": "Must appear in the supplied criteria list."
38
+ },
39
+ "verdict": {
40
+ "type": "string",
41
+ "enum": ["conformant", "violated", "not-applicable"],
42
+ "description": "conformant = checked and honoured, requires file+line evidence. violated = checked and breached, requires a matching findings[] entry with the same ruleId. not-applicable = cannot bind here, requires a reason. There is deliberately no 'unchecked' value: an ID neither checked nor waived fails the stage rather than passing quietly."
43
+ },
44
+ "file": {
45
+ "type": "string",
46
+ "description": "Where the rule was checked. Required for conformant: a verdict with no evidence certifies completeness nobody established."
47
+ },
48
+ "line": {
49
+ "type": "integer",
50
+ "minimum": 0
51
+ },
52
+ "reason": {
53
+ "type": "string",
54
+ "description": "Why the rule does not apply. Required for not-applicable."
55
+ }
56
+ }
57
+ }
23
58
  }
24
59
  },
25
60
  "$defs": {
@@ -52,6 +87,16 @@
52
87
  "type": "string",
53
88
  "minLength": 4,
54
89
  "description": "Concrete remediation. Not 'refactor this' - actionable guidance."
90
+ },
91
+ "ruleId": {
92
+ "type": "string",
93
+ "minLength": 1,
94
+ "description": "Stable ID of the cited rule, e.g. SEC-01. Present when the finding comes from a rule in the ${CRITERIA} block, omitted otherwise. A cited ID carries evidence a reviewer opinion does not: the author can look it up and disagree with the rule rather than with the reviewer. Must be an ID from the supplied block - inventing one is a validator failure."
95
+ },
96
+ "criteriaSource": {
97
+ "type": "string",
98
+ "minLength": 1,
99
+ "description": "Which criteria source the rule came from: a registry name, a module-guide path, or 'exception-marker-audit'. Lets Phase 7 attribute findings to the standard that produced them."
55
100
  }
56
101
  }
57
102
  }
@@ -3,15 +3,39 @@
3
3
  "$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/token-budget.json",
4
4
  "description": "Per-phase token budget for lazy-loaded pipeline docs. Enforced by smoke-token-budget.sh.",
5
5
  "phases": {
6
- "phase-0-init": { "max_tokens": 12400, "warn_tokens": 10900 },
7
- "phase-1-analysis": { "max_tokens": 3750, "warn_tokens": 3300 },
8
- "phase-2-planning": { "max_tokens": 6500, "warn_tokens": 5750 },
9
- "phase-3-dev": { "max_tokens": 7900, "warn_tokens": 6950 },
10
- "phase-4-review": { "max_tokens": 13250, "warn_tokens": 11650 },
11
- "phase-5-test": { "max_tokens": 2550, "warn_tokens": 2250 },
12
- "phase-6-commit": { "max_tokens": 6150, "warn_tokens": 5400 },
13
- "phase-7-report": { "max_tokens": 5600, "warn_tokens": 4950 }
6
+ "phase-0-init": {
7
+ "max_tokens": 12400,
8
+ "warn_tokens": 11900
9
+ },
10
+ "phase-1-analysis": {
11
+ "max_tokens": 4600,
12
+ "warn_tokens": 4050
13
+ },
14
+ "phase-2-planning": {
15
+ "max_tokens": 6500,
16
+ "warn_tokens": 5650
17
+ },
18
+ "phase-3-dev": {
19
+ "max_tokens": 8950,
20
+ "warn_tokens": 7900
21
+ },
22
+ "phase-4-review": {
23
+ "max_tokens": 14750,
24
+ "warn_tokens": 12950
25
+ },
26
+ "phase-5-test": {
27
+ "max_tokens": 2900,
28
+ "warn_tokens": 2550
29
+ },
30
+ "phase-6-commit": {
31
+ "max_tokens": 6150,
32
+ "warn_tokens": 5450
33
+ },
34
+ "phase-7-report": {
35
+ "max_tokens": 6350,
36
+ "warn_tokens": 5600
37
+ }
14
38
  },
15
- "total_max_tokens": 51000,
16
- "note": "Token estimate = ceil(chars / 4). Per-phase budget rule: warn = current+10% (rounded to nearest 50), max = current+25%. Gives ~6 edit cycles of headroom before warn trips - intentionally quiet under normal maintenance, loud when a phase grows unusually. Only the active phase is loaded (lazy). Recalibrated at v10.0.0 after the validator/consistency/simplifier/lesson gate contracts landed in phases 1-4. Recalibrated again at v10.9.0 after the verify-by-test (Phase 4 Step 3.7), update-check (Phase 0 Step 0.6), immutable-test (Phase 3 GREEN) and redTests re-entry contracts landed - Step 3.7 prose was compressed to a pointer into refs/features/verify-by-test.md before the recalibration. Total bumped 50000 -> 51000 at v12.5.0 after the worktree residue/traversal-prune contract (Phase 0 + Phase 5 heal) and the Reflexion causal-diagnosis contract (Phase 4 lesson memory) landed; the prose was compressed first (161 tokens reclaimed) and every per-phase max still passes - only the aggregate needed room."
39
+ "total_max_tokens": 52200,
40
+ "note": "Token estimate = ceil(chars / 4). Per-phase budget rule: warn = current+10% (rounded to nearest 50), max = current+25%. Gives ~6 edit cycles of headroom before warn trips - intentionally quiet under normal maintenance, loud when a phase grows unusually. Only the active phase is loaded (lazy). Recalibrated at v10.0.0 after the validator/consistency/simplifier/lesson gate contracts landed in phases 1-4. Recalibrated again at v10.9.0 after the verify-by-test (Phase 4 Step 3.7), update-check (Phase 0 Step 0.6), immutable-test (Phase 3 GREEN) and redTests re-entry contracts landed - Step 3.7 prose was compressed to a pointer into refs/features/verify-by-test.md before the recalibration. Total bumped 50000 -> 51000 at v12.5.0 after the worktree residue/traversal-prune contract (Phase 0 + Phase 5 heal) and the Reflexion causal-diagnosis contract (Phase 4 lesson memory) landed; the prose was compressed first (161 tokens reclaimed) and every per-phase max still passes - only the aggregate needed room. Recalibrated again at v13.6.0 after the install-relative path correction: an instruction that names `pipeline/scripts/x` resolves only from a repo checkout, and a run happens in the user's worktree, so 157 references across these docs moved to `$HOME/.claude/...` at +5 bytes each - 196 tokens of pure correctness cost. Same discipline as before: prose was compressed FIRST (149 tokens reclaimed, by pointing Phase 1's Figma tier table at the Phase 0 probe that already resolved it and Phase 4's Codex constraints at the always-loaded AGENTS.md block), and only then were the budgets moved. Five warn lines had been permanently amber, which makes the amber tier useless as a signal, so every warn was reset to the documented current+10% and the four maxes that the new warn would have collided with were reset to current+25%. Aggregate 51000 -> 51500. Total bumped 51500 -> 52200 at v14.0.0 after Phase 4 Review entered the four --dev mode phase sets and the criteria-resolution contract (Step 1.78) landed. Same discipline as every prior bump: prose was compressed FIRST, 820 tokens reclaimed, before the number moved. Two of those compressions are structural rather than cosmetic - the hardcoded SwiftUI interaction list in Step 1.5 and the SwiftUI convention paragraph in Step 2.8 were transcriptions of rules that now live in a scoped registry, so keeping them here would have re-created the drift this release exists to remove, and the third moved the Step 1.78 full contract into refs/features/skill-conformance.md leaving a pointer. What remains is contract text that cannot be inferred: the manifest's four consumer-visible parts, the conformance checklist the reviewers must return, and the fail-closed semantics. Every per-phase max still passes (phase-4 12405/14750); only the aggregate needed room."
17
41
  }
@@ -1,9 +1,9 @@
1
1
  {
2
2
  "$schema": "https://json-schema.org/draft/2020-12/schema",
3
3
  "$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/triage-output.schema.json",
4
- "version": "3.2.0",
4
+ "version": "3.3.0",
5
5
  "title": "Multi-Agent Pipeline - Phase 4 triage output",
6
- "description": "Contract for the Opus triage agent's JSON output in Phase 4 Step 3. Triage consumes merged reviewer findings and splits them into accepted/deferred/rejected. Only `accepted` blocking/important items trigger Phase 3 rework. v3.1.0 adds the optional `consensus` block so triage can surface reviewer-agreement risk (false consensus among same-base-model reviewers) instead of silently merging. v3.2.0 adds the optional per-finding `verification` block written by Phase 4 Step 3.7 (verify-by-test): the empirical repro-test outcome for accepted blocking findings.",
6
+ "description": "Contract for the Opus triage agent's JSON output in Phase 4 Step 3. Triage consumes merged reviewer findings and splits them into accepted/deferred/rejected. Only `accepted` blocking/important items trigger Phase 3 rework. v3.1.0 adds the optional `consensus` block so triage can surface reviewer-agreement risk (false consensus among same-base-model reviewers) instead of silently merging. v3.2.0 adds the optional per-finding `verification` block written by Phase 4 Step 3.7 (verify-by-test): the empirical repro-test outcome for accepted blocking findings. v3.3.0 carries ruleId + criteriaSource through triage so a finding that cites a stable rule ID keeps that citation into Phase 6 and Phase 7, and the lesson loop can key durable learnings by rule.",
7
7
  "type": "object",
8
8
  "additionalProperties": false,
9
9
  "required": ["accepted", "deferred", "rejected", "approved"],
@@ -11,7 +11,9 @@
11
11
  "accepted": {
12
12
  "type": "array",
13
13
  "description": "Findings that are real, in-scope, and must be actioned in this iteration.",
14
- "items": { "$ref": "#/$defs/acceptedFinding" }
14
+ "items": {
15
+ "$ref": "#/$defs/acceptedFinding"
16
+ }
15
17
  },
16
18
  "deferred": {
17
19
  "type": "array",
@@ -21,7 +23,9 @@
21
23
  "additionalProperties": false,
22
24
  "required": ["finding", "reason"],
23
25
  "properties": {
24
- "finding": { "$ref": "#/$defs/rawFinding" },
26
+ "finding": {
27
+ "$ref": "#/$defs/rawFinding"
28
+ },
25
29
  "reason": {
26
30
  "type": "string",
27
31
  "minLength": 4,
@@ -38,7 +42,9 @@
38
42
  "additionalProperties": false,
39
43
  "required": ["finding", "reason"],
40
44
  "properties": {
41
- "finding": { "$ref": "#/$defs/rawFinding" },
45
+ "finding": {
46
+ "$ref": "#/$defs/rawFinding"
47
+ },
42
48
  "reason": {
43
49
  "type": "string",
44
50
  "minLength": 4,
@@ -51,21 +57,31 @@
51
57
  "type": "boolean",
52
58
  "description": "True if no accepted BLOCKING items remain. The pipeline uses this to gate Phase 5 / Phase 6 entry."
53
59
  },
54
- "consensus": { "$ref": "#/$defs/consensus" }
60
+ "consensus": {
61
+ "$ref": "#/$defs/consensus"
62
+ }
55
63
  },
56
64
  "if": {
57
65
  "properties": {
58
66
  "accepted": {
59
67
  "contains": {
60
68
  "type": "object",
61
- "properties": { "severity": { "const": "blocking" } },
69
+ "properties": {
70
+ "severity": {
71
+ "const": "blocking"
72
+ }
73
+ },
62
74
  "required": ["severity"]
63
75
  }
64
76
  }
65
77
  }
66
78
  },
67
79
  "then": {
68
- "properties": { "approved": { "const": false } }
80
+ "properties": {
81
+ "approved": {
82
+ "const": false
83
+ }
84
+ }
69
85
  },
70
86
  "$defs": {
71
87
  "severity": {
@@ -101,9 +117,18 @@
101
117
  "additionalProperties": false,
102
118
  "required": ["file", "issue", "note"],
103
119
  "properties": {
104
- "file": { "type": "string", "minLength": 1 },
105
- "line": { "type": "integer", "minimum": 0 },
106
- "issue": { "type": "string", "minLength": 4 },
120
+ "file": {
121
+ "type": "string",
122
+ "minLength": 1
123
+ },
124
+ "line": {
125
+ "type": "integer",
126
+ "minimum": 0
127
+ },
128
+ "issue": {
129
+ "type": "string",
130
+ "minLength": 4
131
+ },
107
132
  "note": {
108
133
  "type": "string",
109
134
  "minLength": 4,
@@ -134,10 +159,16 @@
134
159
  "minLength": 1,
135
160
  "description": "Path to the test-run log verified by evidence-gate.mjs, e.g. '.pipeline/verify-1.test.log'."
136
161
  },
137
- "note": { "type": "string" }
162
+ "note": {
163
+ "type": "string"
164
+ }
138
165
  },
139
166
  "if": {
140
- "properties": { "result": { "enum": ["confirmed", "not-reproduced"] } }
167
+ "properties": {
168
+ "result": {
169
+ "enum": ["confirmed", "not-reproduced"]
170
+ }
171
+ }
141
172
  },
142
173
  "then": {
143
174
  "required": ["result", "testRef", "evidencePath"]
@@ -148,34 +179,88 @@
148
179
  "additionalProperties": false,
149
180
  "required": ["severity", "file", "line", "issue"],
150
181
  "properties": {
151
- "severity": { "$ref": "#/$defs/severity" },
152
- "file": { "type": "string", "minLength": 1 },
153
- "line": { "type": "integer", "minimum": 0 },
154
- "issue": { "type": "string", "minLength": 4 },
155
- "fix": { "type": "string" },
156
- "reviewer": { "$ref": "#/$defs/reviewer" },
157
- "verification": { "$ref": "#/$defs/verification" }
182
+ "severity": {
183
+ "$ref": "#/$defs/severity"
184
+ },
185
+ "file": {
186
+ "type": "string",
187
+ "minLength": 1
188
+ },
189
+ "line": {
190
+ "type": "integer",
191
+ "minimum": 0
192
+ },
193
+ "issue": {
194
+ "type": "string",
195
+ "minLength": 4
196
+ },
197
+ "fix": {
198
+ "type": "string"
199
+ },
200
+ "reviewer": {
201
+ "$ref": "#/$defs/reviewer"
202
+ },
203
+ "verification": {
204
+ "$ref": "#/$defs/verification"
205
+ },
206
+ "ruleId": {
207
+ "type": "string",
208
+ "minLength": 1,
209
+ "description": "Stable ID of the cited rule (e.g. SEC-01), preserved from the reviewer output. A finding citing a registry rule carries evidence an opinion does not, so triage must not reject it as a matter of taste - it can still be deferred or rejected as out of scope for THIS task."
210
+ },
211
+ "criteriaSource": {
212
+ "type": "string",
213
+ "minLength": 1,
214
+ "description": "Which criteria source produced the rule: a registry name, a module-guide path, or a deterministic gate such as 'exception-marker-audit'."
215
+ }
158
216
  }
159
217
  },
160
218
  "acceptedFinding": {
161
219
  "allOf": [
162
- { "$ref": "#/$defs/rawFinding" },
220
+ {
221
+ "$ref": "#/$defs/rawFinding"
222
+ },
163
223
  {
164
224
  "type": "object",
165
225
  "additionalProperties": false,
166
226
  "required": ["severity", "file", "line", "issue", "reviewer", "fix"],
167
227
  "properties": {
168
- "severity": { "$ref": "#/$defs/severity" },
169
- "file": { "type": "string", "minLength": 1 },
170
- "line": { "type": "integer", "minimum": 0 },
171
- "issue": { "type": "string", "minLength": 4 },
172
- "reviewer": { "$ref": "#/$defs/reviewer" },
228
+ "severity": {
229
+ "$ref": "#/$defs/severity"
230
+ },
231
+ "file": {
232
+ "type": "string",
233
+ "minLength": 1
234
+ },
235
+ "line": {
236
+ "type": "integer",
237
+ "minimum": 0
238
+ },
239
+ "issue": {
240
+ "type": "string",
241
+ "minLength": 4
242
+ },
243
+ "reviewer": {
244
+ "$ref": "#/$defs/reviewer"
245
+ },
173
246
  "fix": {
174
247
  "type": "string",
175
248
  "minLength": 4,
176
249
  "description": "Concrete change the dev agent must make. Required for accepted items so Phase 3 re-entry has actionable direction."
177
250
  },
178
- "verification": { "$ref": "#/$defs/verification" }
251
+ "verification": {
252
+ "$ref": "#/$defs/verification"
253
+ },
254
+ "ruleId": {
255
+ "type": "string",
256
+ "minLength": 1,
257
+ "description": "Stable ID of the cited rule (e.g. SEC-01), preserved from the reviewer output. A finding citing a registry rule carries evidence an opinion does not, so triage must not reject it as a matter of taste - it can still be deferred or rejected as out of scope for THIS task."
258
+ },
259
+ "criteriaSource": {
260
+ "type": "string",
261
+ "minLength": 1,
262
+ "description": "Which criteria source produced the rule: a registry name, a module-guide path, or a deterministic gate such as 'exception-marker-audit'."
263
+ }
179
264
  }
180
265
  }
181
266
  ]
@@ -1,5 +1,5 @@
1
1
  {
2
- "_readme": "Per-model unit prices in USD per million tokens. Source: Anthropic public pricing (verified 2026-04-21). Update when Anthropic publishes new tiers. Unknown models render USD as ' - ' and emit a footnote - never block PR-body generation. cacheReadPerMtok is the discounted rate for prompt-cache hits (~10% of inPerMtok); the renderer prices a phase's tokens_cached at this rate when the tracker records it, so resume/cache reuse is visible in the ledger.",
2
+ "_readme": "Per-model unit prices in USD per million tokens. Source: Anthropic public pricing (Claude-family rows verified 2026-07-28 against the current model table; the gpt-* rows remain approximate). Update when Anthropic publishes new tiers. Rung names (fable / opus / sonnet / haiku) are the pipeline's stable identifiers and `modelId` is the wire value they currently resolve to - dispatch reads the rung, so a generation move is a `modelId` edit here plus the phase specs, never a rename of the rungs. Unknown models render USD as ' - ' and emit a footnote - never block PR-body generation. cacheReadPerMtok is the discounted rate for prompt-cache hits (~10% of inPerMtok); the renderer prices a phase's tokens_cached at this rate when the tracker records it, so resume/cache reuse is visible in the ledger.",
3
3
  "schemaVersion": "1.1.0",
4
4
  "prices": {
5
5
  "fable": {
@@ -13,19 +13,22 @@
13
13
  "inPerMtok": 5.0,
14
14
  "outPerMtok": 25.0,
15
15
  "cacheReadPerMtok": 0.5,
16
- "modelId": "claude-opus-4-8"
16
+ "modelId": "claude-opus-5",
17
+ "note": "Second tier - dev phase on --dev modes, Reviewer 1 and triage on Copilot CLI, and the opus rung of the fable -> opus -> sonnet fallback ladder. Same rate as the Opus 4.8 it replaces, so the ledger needed no reprice on the generation move. Claude Opus 5 draws on a rate-limit pool SEPARATE from the combined Opus 4.x pool - moving traffic here neither frees headroom on the old bucket nor inherits it."
17
18
  },
18
19
  "sonnet": {
19
20
  "inPerMtok": 3.0,
20
21
  "outPerMtok": 15.0,
21
22
  "cacheReadPerMtok": 0.3,
22
- "modelId": "claude-sonnet-4-6"
23
+ "modelId": "claude-sonnet-5",
24
+ "note": "Floor tier for Claude-family dispatch - Reviewer 3 on both hosts, and the terminal rung of the fallback ladder. Priced at the standard 3/15 rather than the 2/10 introductory rate that runs through 2026-08-31: over-reporting during the intro window is the safe direction for a cost ledger, and it needs no dated edit when the intro ends."
23
25
  },
24
26
  "haiku": {
25
27
  "inPerMtok": 1.0,
26
28
  "outPerMtok": 5.0,
27
29
  "cacheReadPerMtok": 0.1,
28
- "modelId": "claude-haiku-4-5-20251001"
30
+ "modelId": "claude-haiku-4-5",
31
+ "note": "Speed tier - task-clarifier and other latency-sensitive dispatches, plus the terminal rung of the fallback ladder. Named by its alias like every other rung here rather than by a dated snapshot, so the four rungs stay comparable at a glance."
29
32
  },
30
33
  "gpt-5.4": {
31
34
  "inPerMtok": 10.0,
@@ -16,7 +16,7 @@
16
16
  # .git/info/exclude so the gitlink class cannot re-occur.
17
17
  #
18
18
  # Registered, healthy worktrees are NEVER touched - finishing or killing a
19
- # task is /multi-agent:finish / /multi-agent:kill territory.
19
+ # task is /multi-agent:ship / /multi-agent:kill territory.
20
20
  #
21
21
  # SAFE BY DEFAULT: dry-run. Deletes nothing until you pass --yes.
22
22
  #
@@ -10,10 +10,10 @@
10
10
  * call, anti-pattern warning, and ref link are identical.
11
11
  *
12
12
  * Usage:
13
- * node pipeline/scripts/gen-mode-dispatch.mjs --mode=dev # 0/3/5/6/7 (Phase 5 User Test runs only in dev + full; autopilot/local variants drop it)
13
+ * node pipeline/scripts/gen-mode-dispatch.mjs --mode=dev # 0/3/4/5/6/7 (Phase 5 User Test runs only in dev + full; autopilot/local variants drop it)
14
14
  * node pipeline/scripts/gen-mode-dispatch.mjs --mode=full # 0..7
15
15
  * node pipeline/scripts/gen-mode-dispatch.mjs --mode=full-local # 0..7
16
- * node pipeline/scripts/gen-mode-dispatch.mjs --mode=dev-local # 0/3/6/7 + local-mode caveat
16
+ * node pipeline/scripts/gen-mode-dispatch.mjs --mode=dev-local # 0/3/4/6/7 + local-mode caveat
17
17
  *
18
18
  * Companion smoke `smoke-mode-dispatch-drift.sh` regenerates the section for
19
19
  * each mode file, diffs against the on-disk content, and fails on drift.
@@ -54,22 +54,22 @@ const PHASE_NAMES_NO_TEST = PHASE_NAMES.filter((p) => p !== "5:Test");
54
54
  */
55
55
  const MODES = {
56
56
  dev: {
57
- phases: ["0:Init", "3:Dev", "5:Test", "6:Commit", "7:Report"],
57
+ phases: ["0:Init", "3:Dev", "4:Review", "5:Test", "6:Commit", "7:Report"],
58
58
  local: false,
59
59
  autopilot: false,
60
60
  },
61
61
  "dev-autopilot": {
62
- phases: ["0:Init", "3:Dev", "6:Commit", "7:Report"],
62
+ phases: ["0:Init", "3:Dev", "4:Review", "6:Commit", "7:Report"],
63
63
  local: false,
64
64
  autopilot: true,
65
65
  },
66
66
  "dev-local": {
67
- phases: ["0:Init", "3:Dev", "6:Commit", "7:Report"],
67
+ phases: ["0:Init", "3:Dev", "4:Review", "6:Commit", "7:Report"],
68
68
  local: true,
69
69
  autopilot: false,
70
70
  },
71
71
  "dev-local-autopilot": {
72
- phases: ["0:Init", "3:Dev", "6:Commit", "7:Report"],
72
+ phases: ["0:Init", "3:Dev", "4:Review", "6:Commit", "7:Report"],
73
73
  local: true,
74
74
  autopilot: true,
75
75
  },
@@ -29,12 +29,42 @@
29
29
  //
30
30
  // Exit 0 on success; 1 on setup error.
31
31
 
32
- import { readFileSync } from "node:fs";
32
+ import { existsSync, readFileSync } from "node:fs";
33
33
  import { dirname, join, resolve } from "node:path";
34
34
  import { fileURLToPath } from "node:url";
35
35
 
36
36
  const here = dirname(fileURLToPath(import.meta.url));
37
- const repoRoot = resolve(here, "..", "..");
37
+
38
+ /**
39
+ * Candidate index locations, in the order they are tried.
40
+ *
41
+ * This script lives in two shapes: `<repo>/pipeline/scripts/` in a checkout, and
42
+ * `~/.claude/scripts/` (or `~/.copilot`, `~/.codex`) once installed. The default
43
+ * used to be `resolve(here, "..", "..") + "/pipeline/skills/.skills-index.json"`,
44
+ * which is right in a checkout and resolves to `$HOME/pipeline/skills/...` from an
45
+ * installed tree - a path that has never existed. So dynamic skill loading exited 1
46
+ * on every real install while passing its own smoke, which runs from the repo.
47
+ *
48
+ * Installed layout is tried first: that is where a user's run happens.
49
+ */
50
+ function defaultIndexCandidates() {
51
+ const installRoot = resolve(here, ".."); // ~/.claude when here is ~/.claude/scripts
52
+ const repoRoot = resolve(here, "..", ".."); // <repo> when here is <repo>/pipeline/scripts
53
+ return [
54
+ join(installRoot, "skills", ".skills-index.json"),
55
+ join(repoRoot, "pipeline", "skills", ".skills-index.json"),
56
+ ];
57
+ }
58
+
59
+ function resolveDefaultIndex() {
60
+ const candidates = defaultIndexCandidates();
61
+ for (const c of candidates) {
62
+ if (existsSync(c)) return c;
63
+ }
64
+ // Nothing found: return the first so the error message names a real intent
65
+ // rather than a path that only makes sense in a checkout.
66
+ return candidates[0];
67
+ }
38
68
 
39
69
  const args = process.argv.slice(2);
40
70
  if (args.length === 0 || args[0].startsWith("--")) {
@@ -49,7 +79,7 @@ const opts = {
49
79
  touchedFiles: [],
50
80
  stack: null,
51
81
  limit: 12,
52
- indexPath: join(repoRoot, "pipeline", "skills", ".skills-index.json"),
82
+ indexPath: resolveDefaultIndex(),
53
83
  json: false,
54
84
  };
55
85
  for (let i = 1; i < args.length; i++) {
@@ -69,7 +99,10 @@ try {
69
99
  index = JSON.parse(readFileSync(opts.indexPath, "utf-8"));
70
100
  } catch (e) {
71
101
  console.error(`match-skills: cannot read index ${opts.indexPath}: ${e.message}`);
72
- console.error("→ run: node pipeline/scripts/build-skills-index.mjs");
102
+ console.error(` tried: ${defaultIndexCandidates().join(", ")}`);
103
+ console.error(
104
+ "→ from a checkout: node pipeline/scripts/build-skills-index.mjs; from an install: re-run the installer, which ships the index",
105
+ );
73
106
  process.exit(1);
74
107
  }
75
108
 
@@ -18,7 +18,7 @@ const DEFAULT_FILE = path.join(os.homedir(), ".claude", "multi-agent-preferences
18
18
 
19
19
  // Single source of truth for the migration target version. Referenced by the
20
20
  // migrate() bump chain AND every user-facing message, so they cannot drift.
21
- const TARGET_VERSION = "2.4.0";
21
+ const TARGET_VERSION = "2.5.0";
22
22
 
23
23
  const USAGE = "Usage: migrate-prefs.mjs [--dry-run] [--file <path>]";
24
24
 
@@ -61,6 +61,40 @@ function defaultSettings() {
61
61
  };
62
62
  }
63
63
 
64
+ /**
65
+ * Versions this migrator accepts as input, read from the schema's own
66
+ * `schemaVersion` enum so the two lists cannot disagree.
67
+ *
68
+ * Resolved against both layouts: `<repo>/pipeline/schemas` in a checkout and
69
+ * `<install>/schemas` once installed - this script is copied into
70
+ * ~/.claude/scripts, ~/.copilot/scripts and ~/.codex/scripts without any path
71
+ * rewriting, so a checkout-relative path alone would resolve to nothing there.
72
+ *
73
+ * @param {string} target - TARGET_VERSION, excluded from the result
74
+ * @returns {Set<string>}
75
+ */
76
+ function readMigratableVersions(target) {
77
+ const here = path.dirname(new URL(import.meta.url).pathname);
78
+ const candidates = [
79
+ path.join(here, "..", "schemas", "prefs.schema.json"),
80
+ path.join(here, "..", "..", "pipeline", "schemas", "prefs.schema.json"),
81
+ ];
82
+ for (const c of candidates) {
83
+ try {
84
+ const schema = JSON.parse(fs.readFileSync(c, "utf-8"));
85
+ const versions = schema?.properties?.schemaVersion?.enum;
86
+ if (Array.isArray(versions) && versions.length > 0) {
87
+ return new Set(versions.filter((v) => v !== target));
88
+ }
89
+ } catch {
90
+ // try the next candidate
91
+ }
92
+ }
93
+ // Schema unreadable: refusing to migrate would be worse than migrating from a
94
+ // version this build cannot enumerate, so fall back to the known history.
95
+ return new Set(["2.0.0", "2.1.0", "2.2.0", "2.3.0", "2.4.0"]);
96
+ }
97
+
64
98
  function defaultKeychainMapping() {
65
99
  return {
66
100
  jira: null,
@@ -158,29 +192,30 @@ function migrate(prefs) {
158
192
  const changes = [];
159
193
  const out = JSON.parse(JSON.stringify(prefs)); // deep clone
160
194
 
161
- // 1. schemaVersion - detect whether a v2.0.0 → 2.1.0 → 2.2.0 → 2.3.0 bump chain is needed.
162
- // Sub-migrations run regardless (they're idempotent), so early-return is avoided.
163
- // v2.3.0 is the v7.4.0 target - adds optional `accounts[]` (per-account host
164
- // overrides) and `recentAccounts[]` (LRU for autopilot account picker).
165
- // Both default to empty arrays; old prefs continue to validate without them.
195
+ // 1. schemaVersion. Every known older version bumps STRAIGHT to TARGET: the
196
+ // sub-migrations below are cumulative and idempotent rather than pairwise, so
197
+ // there is no per-hop work to do.
198
+ //
199
+ // This is a SET rather than an if-else chain per version on purpose. The chain
200
+ // had to be extended by hand every time TARGET_VERSION moved, and forgetting
201
+ // made the migrator throw `unknown schemaVersion` on the very version it had
202
+ // just been released to migrate FROM.
203
+ //
204
+ // The set is READ FROM the schema's enum (minus the target) so the two cannot
205
+ // drift: adding a version to prefs.schema.json is what makes it migratable,
206
+ // and there is no second list to remember. Falls back to a literal set if the
207
+ // schema cannot be read, because refusing to migrate is worse than migrating
208
+ // from a version we cannot currently enumerate.
166
209
  const TARGET = TARGET_VERSION;
210
+ const MIGRATABLE_FROM = readMigratableVersions(TARGET);
167
211
  if (out.schemaVersion === TARGET) {
168
212
  // already on target - skip version bump, but still run sub-migrations below
169
- } else if (out.schemaVersion === "2.3.0") {
170
- out.schemaVersion = TARGET;
171
- changes.push(`schemaVersion: 2.3.0 → ${TARGET}`);
172
- } else if (out.schemaVersion === "2.2.0") {
173
- out.schemaVersion = TARGET;
174
- changes.push(`schemaVersion: 2.2.0 → ${TARGET}`);
175
- } else if (out.schemaVersion === "2.1.0") {
176
- out.schemaVersion = TARGET;
177
- changes.push(`schemaVersion: 2.1.0 → ${TARGET}`);
178
213
  } else if (!out.schemaVersion) {
179
214
  out.schemaVersion = TARGET;
180
215
  changes.push(`added schemaVersion: ${TARGET}`);
181
- } else if (out.schemaVersion === "2.0.0") {
216
+ } else if (MIGRATABLE_FROM.has(out.schemaVersion)) {
217
+ changes.push(`schemaVersion: ${out.schemaVersion} → ${TARGET}`);
182
218
  out.schemaVersion = TARGET;
183
- changes.push(`schemaVersion: 2.0.0 → ${TARGET}`);
184
219
  } else {
185
220
  throw new Error(`unknown schemaVersion: ${out.schemaVersion}`);
186
221
  }
@@ -253,6 +288,42 @@ function migrate(prefs) {
253
288
  out.global.routines = [];
254
289
  changes.push("added empty routines (v2.4.0 user-defined /multi-agent:save routines)");
255
290
  }
291
+ // v2.5.0 / v14.0.0: Phase 4 Review entered the --dev phase sets and Step 1.78
292
+ // (criteria resolution) landed. Default false because only Swift plus the two
293
+ // store-compliance catalogs ship a scoped registry today, so blocking on a
294
+ // coverage gap would stop every Kotlin/Python/Node/ObjC run from day one. The
295
+ // gap is reported either way; this only decides whether it halts.
296
+ if (!out.global.skillConformance || typeof out.global.skillConformance !== "object") {
297
+ out.global.skillConformance = {};
298
+ changes.push("added skillConformance (v2.5.0 Phase 4 criteria resolution)");
299
+ }
300
+ if (typeof out.global.skillConformance.blockOnCoverageGap !== "boolean") {
301
+ out.global.skillConformance.blockOnCoverageGap = false;
302
+ changes.push("added skillConformance.blockOnCoverageGap=false (report, do not halt)");
303
+ }
304
+ // v2.5.0: /multi-agent:finish became /multi-agent:ship, and its autoFix key -
305
+ // referenced by the command spec since it shipped but never declared in the
306
+ // schema - is declared here. Carry any value the user had rather than reset it.
307
+ if (!out.global.ship || typeof out.global.ship !== "object") {
308
+ out.global.ship = {};
309
+ changes.push("added ship (v2.5.0 - /multi-agent:finish renamed to :ship)");
310
+ }
311
+ if (typeof out.global.ship.autoFix !== "boolean") {
312
+ const legacyAutoFix = out.global.finish?.autoFix;
313
+ out.global.ship.autoFix = typeof legacyAutoFix === "boolean" ? legacyAutoFix : false;
314
+ changes.push(
315
+ typeof legacyAutoFix === "boolean"
316
+ ? "moved finish.autoFix -> ship.autoFix (value preserved)"
317
+ : "added ship.autoFix=false",
318
+ );
319
+ }
320
+ if (out.global.finish && typeof out.global.finish === "object") {
321
+ delete out.global.finish.autoFix;
322
+ if (Object.keys(out.global.finish).length === 0) {
323
+ delete out.global.finish;
324
+ changes.push("removed obsolete finish block (renamed to ship)");
325
+ }
326
+ }
256
327
  if (!out.global.serviceStatus) {
257
328
  out.global.serviceStatus = {};
258
329
  changes.push("added empty serviceStatus");