@mmerterden/multi-agent-pipeline 17.0.0 → 17.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/CHANGELOG.md +159 -0
  2. package/README.md +56 -4
  3. package/README.tr.md +57 -4
  4. package/docs/architecture.md +3 -3
  5. package/docs/ecosystem.md +5 -5
  6. package/docs/token-budget-history.md +22 -0
  7. package/install/_dev-only-files.mjs +1 -0
  8. package/install/codex.mjs +18 -1
  9. package/install/copilot.mjs +17 -1
  10. package/install/templates/multi-agent-autopilot.plist.template +79 -0
  11. package/package.json +1 -1
  12. package/pipeline/commands/multi-agent/autopilot-off/SKILL.md +64 -0
  13. package/pipeline/commands/multi-agent/autopilot-on/SKILL.md +181 -0
  14. package/pipeline/commands/multi-agent/autopilot-status/SKILL.md +74 -0
  15. package/pipeline/commands/multi-agent/channels/SKILL.md +41 -12
  16. package/pipeline/commands/multi-agent/help/SKILL.md +41 -35
  17. package/pipeline/commands/multi-agent/manual-test/SKILL.md +1 -1
  18. package/pipeline/commands/multi-agent/sync/SKILL.md +10 -9
  19. package/pipeline/commands/multi-agent/update/SKILL.md +1 -1
  20. package/pipeline/lib/autopilot-activation.sh +117 -0
  21. package/pipeline/lib/autopilot-state.sh +184 -0
  22. package/pipeline/lib/issue-fetcher.sh +18 -1
  23. package/pipeline/lib/plan-todos.sh +18 -0
  24. package/pipeline/multi-agent-refs/_dev-context.md +10 -0
  25. package/pipeline/multi-agent-refs/analysis/redesign.md +8 -0
  26. package/pipeline/multi-agent-refs/analysis/review.md +9 -0
  27. package/pipeline/multi-agent-refs/android-guide.md +14 -0
  28. package/pipeline/multi-agent-refs/audit-guide.md +12 -0
  29. package/pipeline/multi-agent-refs/backend-guide.md +10 -0
  30. package/pipeline/multi-agent-refs/channels/confluence.md +11 -0
  31. package/pipeline/multi-agent-refs/channels/issue-comment.md +12 -0
  32. package/pipeline/multi-agent-refs/channels/jira.md +90 -20
  33. package/pipeline/multi-agent-refs/channels/pr-review-actions.md +13 -0
  34. package/pipeline/multi-agent-refs/channels/pr.md +76 -19
  35. package/pipeline/multi-agent-refs/component-dispatch.md +11 -0
  36. package/pipeline/multi-agent-refs/component-generation.md +11 -0
  37. package/pipeline/multi-agent-refs/conventions-defaults.md +15 -0
  38. package/pipeline/multi-agent-refs/cross-cli-contract.md +49 -5
  39. package/pipeline/multi-agent-refs/features/analysis-jira.md +11 -0
  40. package/pipeline/multi-agent-refs/features/design-conformance.md +10 -0
  41. package/pipeline/multi-agent-refs/features/doctor.md +10 -0
  42. package/pipeline/multi-agent-refs/features/external-context-injection.md +7 -0
  43. package/pipeline/multi-agent-refs/features/jira-context.md +9 -0
  44. package/pipeline/multi-agent-refs/features/model-fallback.md +10 -0
  45. package/pipeline/multi-agent-refs/features/skill-conformance.md +13 -0
  46. package/pipeline/multi-agent-refs/features/url-enrichment.md +9 -0
  47. package/pipeline/multi-agent-refs/features/visual-evidence.md +61 -7
  48. package/pipeline/multi-agent-refs/generate-issue.md +7 -0
  49. package/pipeline/multi-agent-refs/issue-jira-triad.md +9 -0
  50. package/pipeline/multi-agent-refs/knowledge.md +6 -0
  51. package/pipeline/multi-agent-refs/multi-repo-integration-build.md +13 -0
  52. package/pipeline/multi-agent-refs/phases/modes.md +7 -0
  53. package/pipeline/multi-agent-refs/phases/operations.md +9 -0
  54. package/pipeline/multi-agent-refs/phases/phase-0-init.md +2 -2
  55. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +17 -15
  56. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +1 -1
  57. package/pipeline/multi-agent-refs/phases/phase-6-commit.md +1 -1
  58. package/pipeline/multi-agent-refs/phases.md +11 -0
  59. package/pipeline/multi-agent-refs/picker-contract.md +12 -0
  60. package/pipeline/multi-agent-refs/platform-parity.md +10 -0
  61. package/pipeline/multi-agent-refs/progress-contract.md +10 -0
  62. package/pipeline/multi-agent-refs/readiness-review.md +7 -1
  63. package/pipeline/multi-agent-refs/rules.md +3 -11
  64. package/pipeline/multi-agent-refs/setup/firebase.md +9 -0
  65. package/pipeline/multi-agent-refs/swiftui-guide.md +17 -0
  66. package/pipeline/multi-agent-refs/tracker-contract.md +44 -0
  67. package/pipeline/multi-agent-refs/web-guide.md +10 -0
  68. package/pipeline/multi-agent-refs/wiki-capture.md +11 -0
  69. package/pipeline/schemas/autopilot-config.schema.json +149 -0
  70. package/pipeline/schemas/prefs.schema.json +4 -0
  71. package/pipeline/schemas/token-budget.json +10 -19
  72. package/pipeline/scripts/autopilot-arming.mjs +147 -0
  73. package/pipeline/scripts/autopilot-intake.mjs +387 -0
  74. package/pipeline/scripts/autopilot-menubar.swift +361 -0
  75. package/pipeline/scripts/autopilot-runner.mjs +354 -0
  76. package/pipeline/scripts/autopilot-status.sh +213 -0
  77. package/pipeline/scripts/capture-evidence.sh +79 -11
  78. package/pipeline/scripts/gen-ref-toc.mjs +279 -0
  79. package/pipeline/scripts/jira-search.sh +70 -0
  80. package/pipeline/scripts/phase-tracker.sh +134 -12
  81. package/pipeline/scripts/probe-evidence-capability.sh +27 -3
  82. package/pipeline/scripts/run-ui-tests.sh +113 -4
  83. package/pipeline/skills/.skill-manifest.json +16 -4
  84. package/pipeline/skills/shared/core/multi-agent-autopilot-off/SKILL.md +67 -0
  85. package/pipeline/skills/shared/core/multi-agent-autopilot-on/SKILL.md +146 -0
  86. package/pipeline/skills/shared/core/multi-agent-autopilot-status/SKILL.md +64 -0
  87. package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +62 -11
  88. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +9 -8
package/CHANGELOG.md CHANGED
@@ -14,6 +14,165 @@ Internal file-layout changes that don't affect the slash-command surface are sti
14
14
 
15
15
  ---
16
16
 
17
+ ## [17.3.0] - 2026-09-14
18
+
19
+ A three-pass review of 17.2.0 found two things it had declared finished. Both
20
+ were the same shape: a chain fixed on one side and left broken on the other.
21
+
22
+ ### Fixed
23
+
24
+ - **Web was a platform the chain did not agree on.** 17.1.0 gave
25
+ `run-ui-tests.sh` a web arm and taught `probe-evidence-capability.sh` to open
26
+ tier 1 and tier 2 for web; `visualEvidence.platform` has carried `"web"` all
27
+ along. `capture-evidence.sh` never learned it and answered **exit 2**, a usage
28
+ error. So a UI change in a Playwright repo picked a tier the probe had opened
29
+ and got back a failure Phase 3 reads as a phase failure rather than the gap it
30
+ is. `capture-evidence.sh` now has a web arm on all three verbs.
31
+
32
+ Web does not share the iOS/Android recording model, and forcing it to would
33
+ record the same run twice: Playwright and Cypress write the video themselves.
34
+ `video start` writes a marker, `video stop` harvests the newest video under
35
+ `test-results/` or `cypress/videos/` written **after** it, and re-encodes webm
36
+ to h264 for the same reason the iOS arm re-encodes. The marker is what stops a
37
+ video from yesterday's run being attached as today's evidence, which is worse
38
+ than attaching nothing because a present artefact does not get re-checked.
39
+ `fit` gained `webm`, which had been falling through to the pass-through arm and
40
+ reaching the uploader at full size.
41
+ - **Continuous mode's node half still named one host.** 17.2.0 fixed the shell
42
+ side and closed the class. `autopilot-runner.mjs` was still resolving its three
43
+ siblings from `~/.claude/scripts`, and `autopilot-intake.mjs` looked for
44
+ `jira-search.sh` there behind an `existsSync` guard whose else-branch is
45
+ `return []` - on a Copilot-only or Codex-only host that is an **empty Jira
46
+ queue with no error**. Both resolve from their own directory now.
47
+ - **A sentence in `phase-0-init.md` contradicted the code three lines above it**
48
+ ("no simulator on web or backend, so the probe does not run", under a branch
49
+ that sets `EVIDENCE_PLATFORM=web` and runs the probe).
50
+
51
+ ### Added
52
+
53
+ - `visualEvidence.webBaseUrl` - the address a web capture points the browser at.
54
+ Web is the one evidence platform that needs one. Absent, the capture reports a
55
+ gap with that reason instead of guessing a port. Per-run override: `--url`.
56
+ - `features/visual-evidence.md` section 4.6, the web contract.
57
+ - `smoke-visual-evidence.sh` section 3d: for every platform the probe will open,
58
+ the capture script must answer with a reason or an artefact, never exit 2. Nine
59
+ assertions including a live harvest of a real webm and its h264 verdict; seven
60
+ of them fail against 17.2.0's code.
61
+ - **A gate for the rule the manifest could not see.** `dependencies: {}` was
62
+ already tested; a shipped script shelling out to a bare `npx <pkg>` is a
63
+ runtime dependency the manifest never shows, fetched from the network on the
64
+ user's machine. `test/project.test.mjs` now fails any `npx` invocation without
65
+ `--no-install`, allowing only our own `@mmerterden/multi-agent-*` packages.
66
+
67
+ ### Notes
68
+
69
+ - 17.2.0 was tagged but never published; npm goes 17.1.0 to 17.3.0.
70
+ - Five of this release's new assertions did not fire the first time they were
71
+ mutated. `grep -q installTemplates` was satisfied by `installTemplatesX`, the
72
+ host-resolver test only exercised the same-tree shortcut and never reached the
73
+ three-root loop, and the depth mutation targeted a file that is level 1 for
74
+ every reader. All fixed, all re-mutated, all red.
75
+
76
+ ## [17.2.0] - 2026-09-14
77
+
78
+ Three published rules we were measurably breaking, a budget tier that could not
79
+ fire, and continuous mode discovering that two other host trees exist.
80
+
81
+ ### Added
82
+
83
+ - **The 500-line SKILL.md rule.** Anthropic's guidance is a line count; ours was
84
+ a token count, and the difference is not cosmetic: all sixteen files over 500
85
+ lines were PASSING the token gate, because short lines and code blocks are
86
+ cheap at chars/4 and expensive to read (`api-security-best-practices` is 910
87
+ lines at 5,882 tokens against a 6,000 ceiling). `LINE_HARD_CAP` is a second
88
+ axis, not a tighter number on the first. The sixteen are frozen at their
89
+ current length in a closed grace list that only ratchets down, and
90
+ `test/skill-token-budget.test.mjs` is what makes "closed" enforceable.
91
+ - **`smoke-ref-depth.sh`.** Reference depth was spent every release and measured
92
+ never. The published rule asks for one level; the tree is three, and the
93
+ mechanism is written into our own gates - both the token-budget remedy and the
94
+ grace-list message told authors to move reasoning into a lazily loaded
95
+ reference and both said it as if the move were free. It is not: it spends a
96
+ level. The gate freezes the maximum at 3, lowers the ceiling when the tree
97
+ gets flatter, and fails any pointer naming a file that does not exist. Both
98
+ remedy texts now name the cost.
99
+ - **`gen-ref-toc.mjs` + `smoke-ref-toc.sh`.** A generated table of contents on
100
+ every reference over 100 lines (39 files, 297 anchors), refreshed rather than
101
+ hand-maintained, because a stale contents list is worse than none. Three sets
102
+ are excluded on measurement: a ToC puts 5 of 8 phase docs over their max and
103
+ the aggregate 1,533 tokens over, takes always-loaded `rules.md` from 59,908 to
104
+ 60,997 against a 60,000 ceiling, and the analysis refs 2,298 over theirs.
105
+ Those files are loaded whole, so the navigation buys nothing. The exclusion is
106
+ derived from the three budgets, never hand-listed.
107
+ - **`smoke-autopilot-hosts.sh`.** Continuous mode on a Copilot-only or
108
+ Codex-only machine.
109
+
110
+ ### Fixed
111
+
112
+ - **The warn tier that could not fire.** `warn_tokens` was a stored number
113
+ maintained by hand and it rotted twice the same way: reset at v13.6.0 with
114
+ five lines permanently amber, six of eight amber again by v17.1.0. It is gone
115
+ from the schema. The gate derives it per phase from that phase's own git
116
+ history as `median + 3*MAD`, anchored on HEAD rather than the working tree -
117
+ the first version anchored on the working tree, which made the threshold
118
+ follow the measurement so it could never fire, the exact defect being removed.
119
+ - **The ceiling that could only go up.** A `max_tokens` more than 25% above what
120
+ the document measures is now stale and fails, naming the number to lower it
121
+ to. Ceilings ratchet down, which the SKILL.md grace list always did and this
122
+ budget never had.
123
+ - **Continuous mode's hard-coded host paths.** `autopilot-on` read the launchd
124
+ plist template from `$HOME/.claude/templates/`, a tree only
125
+ `install/claude.mjs` creates and no path rewrite retargets, so on a
126
+ Copilot-only or Codex-only machine it rendered a launchd job from a file the
127
+ host does not have; `autopilot-status` lost its whole spend section the same
128
+ way. `ma_ap_asset` resolves across the three roots, and both installers now
129
+ ship `templates/`. The state root stays shared on purpose - two CLIs on one
130
+ machine must read ONE queue - and the gate asserts that too.
131
+ - **A leak scan that skipped the likeliest tree.** `smoke-autopilot-default-off`
132
+ asserted nothing outside the three autopilot commands references the runner,
133
+ but read only `commands/` and `multi-agent-refs/`, not `skills/`, where the
134
+ Copilot orchestrator is a 713-line inline file.
135
+
136
+ ### Changed
137
+
138
+ - The 23,769-character `note` in `token-budget.json` moved to
139
+ `docs/token-budget-history.md`. The schema is 1,668 characters and describes
140
+ the rules instead of the past.
141
+ - `cross-cli-contract.md` gains section 4.4: what continuous mode does on each
142
+ host, and why the state root is the one thing that is not per-host.
143
+
144
+ ## [17.1.0] - 2026-09-14
145
+
146
+ Continuous mode, and three things that were computed correctly and shown to
147
+ nobody: the plan's own steps, a web repo's UI tests, and the queue's status when
148
+ one field arrived as a number.
149
+
150
+ ### Added
151
+
152
+ - **Continuous mode.** `/multi-agent:autopilot-on` picks the repos ONE machine watches; labelled GitHub issues and assigned + labelled Jira items then run in a worktree and stop at an open PR. `:autopilot-status` is the single producer the terminal, the menu bar indicator and the session hook all render, `:autopilot-off` removes the schedule and keeps the selection. Nothing is on by default: installing writes no state and schedules nothing, and `smoke-autopilot-default-off.sh` fails the build if that changes.
153
+
154
+ The ordering is deterministic with no model call - explicit rank, repo grouping, priority, then OLDEST first, which is the opposite of the `jira` picker on purpose: a human wants what just landed, an unattended queue must not starve what has been waiting. Two preconditions are checked before any item is taken: a rolling 24-hour spend total, and a per-source credential gate, so a dead Jira token stops Jira items without stopping GitHub ones.
155
+
156
+ A menu bar indicator draws the same `status.json` in the top right when `swiftc` is present. ActivityKit is `@available(macOS, unavailable)` - there is no Live Activity on a Mac - so this is an `NSStatusItem`, built from source on demand rather than shipped as a binary that would need signing.
157
+
158
+ - **UI tests on web.** `run-ui-tests.sh` rejected every platform but `ios|android`, so a repo with a full Playwright suite reported "no UI test target" and the PR body said UI tests had not run - a statement true of the runner and false of the repo. Detection keys on the browser-driving import (`@playwright/test`, `cy.visit(`), not on a directory called `e2e`: the same distinction that keeps 475 iOS snapshot tests from being counted as UI tests.
159
+
160
+ ### Changed
161
+
162
+ - **The Jira comment and the PR body stopped being the same document.** Jira now carries Geliştirme Özeti, Test Senaryoları, Etki Analizi and Bağlantılar with the PR link on line 1 and no identifiers, file paths or diff hunks anywhere in it. The PR keeps all of that and gains Teknik Açıklama, Etki Analizi and Build - the build command, its result and the base sha it ran against. Given/When/Then is gone from the scenarios: it reads as translated English to the person running them, who wants a titled list they can follow with the app open.
163
+
164
+ ### Fixed
165
+
166
+ - **The menu bar indicator read "autopilot kapalı" with items in flight.** `phase` was declared `String?` while the producer passes the queue's own value through `jq` untouched, so a numeric phase made Swift's `Decodable` throw - and the throw did not lose one field, it failed the whole document. A silent blank is the worst failure a status indicator has, because it is indistinguishable from good news.
167
+
168
+ - **`ma_ap_boottime` returned the microseconds.** `.*sec = ` is greedy and walks past `sec` into `usec`, so the helper produced a six-digit number that looked plausible and never equalled the same fact read anywhere else. It surfaced as a live runner being declared dead.
169
+
170
+ - **Three phase documents said "MCP forbidden" without qualifying it**, while the gate that enforces it has always matched `figma` and nothing else. The prose therefore banned the screenshot, xcodebuild and UI-test tools that Phase 3 itself calls, which is one way a run reaches Phase 7 with no evidence.
171
+
172
+ - **`channels/jira.md` told Phase 7 to upload evidence Phase 6 had already uploaded**, so re-rendering a comment attached every file a second time.
173
+
174
+ ---
175
+
17
176
  ## [17.0.0] - 2026-09-14
18
177
 
19
178
  Five things in this release were not working, and four of them looked like they
package/README.md CHANGED
@@ -89,7 +89,7 @@ Depth, autopilot and `--local` are the only knobs on the run itself; everything
89
89
 
90
90
  ## Commands
91
91
 
92
- `/multi-agent` plus 53 sub-commands. `/multi-agent:help` renders the same catalog in your terminal, in your `outputLanguage`.
92
+ `/multi-agent` plus 56 sub-commands. `/multi-agent:help` renders the same catalog in your terminal, in your `outputLanguage`.
93
93
 
94
94
  ### Pipeline entries
95
95
 
@@ -195,6 +195,58 @@ Eight are installed alongside the commands and dispatched by the phases: `explor
195
195
 
196
196
  Two compliance skills install on every host and back the store gates: `apple-archive-compliance` (18-rule Apple review scan with ITMS code mapping) and `google-play-compliance` (21-rule Play policy catalog with Console error codes). Everything else stack-shaped - SwiftUI, Compose, backend, frontend - comes from the marketplace plugins described below.
197
197
 
198
+ ## Continuous mode
199
+
200
+ Everything above starts when you start it. Continuous mode is the same pipeline
201
+ picking work up on its own, on ONE machine you choose, from repos you choose.
202
+
203
+ | Command | What it does |
204
+ | --- | --- |
205
+ | `/multi-agent:autopilot-on` | Pick the repos this machine watches. Labelled GitHub issues and assigned + labelled Jira items then run in a worktree and stop at an open PR. Re-run to change the list |
206
+ | `/multi-agent:autopilot-status` | What is running and at which phase, what is queued, what is waiting for an answer, the PRs of the last day, and the rolling spend |
207
+ | `/multi-agent:autopilot-off` | Remove the schedule. Work already running finishes; the repo selection is kept |
208
+
209
+ **Nothing is on by default and nothing is added implicitly.** Installing the
210
+ package writes no state and schedules nothing; `smoke-autopilot-default-off.sh`
211
+ fails the build if that ever changes. A label is a filter, not a gate - anyone
212
+ who can open an issue in a repo you have push on could add one - so the gate is
213
+ the picker, and it is per machine.
214
+
215
+ The biggest win is not parallelism. Measured here, the median run is 44 minutes,
216
+ but an item that finishes at 14:00 waits until you sit down again: overnight that
217
+ is 16 hours against 40 minutes. Continuous mode removes the waiting, not the work.
218
+
219
+ **What it will not do.** It does not merge - the runner stops at an open PR and
220
+ the decision stays yours. It does not touch an attended run: per-repo concurrency
221
+ is always 1, so the queue steps around a repo you are working in rather than
222
+ competing for `.git/index.lock`. There is no cap on PRs; the bounds are
223
+ `costCeilingUsd` over a rolling 24 hours and what the machine can hold.
224
+
225
+ **A menu bar indicator**, when `swiftc` is present, draws the same `status.json`
226
+ in the top right and refreshes on its own: one row per item with its id, Full or
227
+ Short, the phase as a fraction, the elapsed time and the stack. A row disappears
228
+ the moment the item finishes and reappears under Reports with its PR. It only
229
+ draws - it cannot start, stop or change a run. ActivityKit is unavailable on
230
+ macOS, so this is an `NSStatusItem`, built from source on demand rather than
231
+ shipped as a binary that would need signing.
232
+
233
+ **It survives a restart with no command to run.** launchd loads the job at
234
+ **login**, not at boot, and that is correct rather than a limitation: the login
235
+ keychain is what unlocks the tokens, so a tick that fired before login could not
236
+ reach Jira or GitHub anyway. There is no `autopilot-resume` - a command you have
237
+ to remember is a queue that silently stops when you forget it. `doctor` reports
238
+ the real failure instead: configured, but launchd holds no job.
239
+
240
+ Sleep is held **only on AC**. A queue that flattens a laptop off the charger is a
241
+ bug; on battery the assertion is released and work resumes when you plug in.
242
+
243
+ **The three commands ship to all three CLIs and behave identically there**, because
244
+ what they control is not a CLI feature: it is a launchd user agent whose tick
245
+ spawns a `claude --bg` child regardless of which CLI you typed the command in.
246
+ Continuous mode therefore needs the `claude` binary on `PATH` everywhere. The
247
+ queue and the repo selection live in `~/.claude/autopilot/` on every host, on
248
+ purpose - two CLIs on one machine must read one queue, not two.
249
+
198
250
  ## Stacks
199
251
 
200
252
  Stack skills ship as versioned plugins in the [`mmerterden/multi-agent-plugins`](https://github.com/mmerterden/multi-agent-plugins) marketplace. Select a stack per-repo:
@@ -207,13 +259,13 @@ This enables the matching plugin (+ the shared `ai-common` plugin) in the repo's
207
259
 
208
260
  ## Tool support
209
261
 
210
- The pipeline runs natively on **Claude Code**, **Copilot CLI** and **Codex CLI** - all three install from the same `pipeline/` source and get the same 53 commands.
262
+ The pipeline runs natively on **Claude Code**, **Copilot CLI** and **Codex CLI** - all three install from the same `pipeline/` source and get the same 56 commands.
211
263
 
212
264
  | Tool | Flag | What it installs |
213
265
  | ----------- | -------------------- | ------------------------------------------------------------------------------------------------------ |
214
266
  | Claude Code | `--claude` (default) | slash commands + skills + agents + three `PreToolUse` hooks (secret scan, agent-guard, read-size gate) |
215
- | Copilot CLI | `--copilot` | instructions + 53 sub-command skills + scripts |
216
- | Codex CLI | `--codex` | one router skill + 53 specs as refs + 8 agent TOML + `AGENTS.md` block + `codex mcp add` |
267
+ | Copilot CLI | `--copilot` | instructions + 56 sub-command skills + scripts |
268
+ | Codex CLI | `--codex` | one router skill + 56 specs as refs + 8 agent TOML + `AGENTS.md` block + `codex mcp add` |
217
269
 
218
270
  Filter skills by stack with `--platform=ios\|android\|all`.
219
271
 
package/README.tr.md CHANGED
@@ -89,7 +89,7 @@ Koşunun kendisinde ayarlanabilen tek şey derinlik, autopilot ve `--local`; ger
89
89
 
90
90
  ## Komutlar
91
91
 
92
- `/multi-agent` ve 51 alt komut. `/multi-agent:help` aynı katalogu terminalde, `outputLanguage` ayarına göre gösterir.
92
+ `/multi-agent` ve 56 alt komut. `/multi-agent:help` aynı katalogu terminalde, `outputLanguage` ayarına göre gösterir.
93
93
 
94
94
  ### Pipeline girişleri
95
95
 
@@ -195,6 +195,59 @@ Komutlarla birlikte sekiz agent kurulur ve fazlar bunları çağırır: `explore
195
195
 
196
196
  İki uyumluluk skill'i her host'a kurulur ve store kapılarını besler: `apple-archive-compliance` (ITMS kod eşlemeli 18 kurallı Apple review taraması) ve `google-play-compliance` (Console hata kodlu 21 kurallı Play politika kataloğu). Stack'e bağlı geri kalan her şey - SwiftUI, Compose, backend, frontend - aşağıda anlatılan marketplace plugin'lerinden gelir.
197
197
 
198
+ ## Sürekli mod
199
+
200
+ Yukarıdaki her şey sen başlattığında başlar. Sürekli mod, aynı pipeline'ın
201
+ **senin seçtiğin tek makinede**, **senin seçtiğin repolardan** işi kendi başına
202
+ alması.
203
+
204
+ | Komut | Ne yapar |
205
+ | --- | --- |
206
+ | `/multi-agent:autopilot-on` | Bu makinenin izleyeceği repoları seçersin. Etiketli GitHub issue'ları ve sana atanmış + etiketli Jira maddeleri worktree'de koşar, açık PR'da durur. Listeyi değiştirmek için tekrar çalıştır |
207
+ | `/multi-agent:autopilot-status` | Ne koşuyor hangi fazda, sırada ne var, ne cevap bekliyor, son bir günün PR'ları ve yuvarlanan harcama |
208
+ | `/multi-agent:autopilot-off` | Zamanlamayı kaldırır. Koşan iş biter; repo seçimi saklanır |
209
+
210
+ **Varsayılan olarak hiçbir şey açık değil ve hiçbir repo kendiliğinden eklenmez.**
211
+ Paketi kurmak ne durum yazar ne zamanlama kurar; bu bir gün değişirse
212
+ `smoke-autopilot-default-off.sh` derlemeyi düşürür. Etiket bir filtredir, kapı
213
+ değil - push yetkin olan bir repoda issue açabilen herkes etiket ekleyebilir -
214
+ o yüzden kapı picker'dır ve makine başınadır.
215
+
216
+ En büyük kazanç paralellik değil. Ölçüm: medyan koşu 44 dakika, ama 14:00'te
217
+ biten bir madde sen masaya oturana kadar bekler; gece boyunca bu 40 dakikaya
218
+ karşı **16 saat**. Sürekli mod işi değil, beklemeyi kaldırıyor.
219
+
220
+ **Yapmayacakları.** Merge etmez - koşucu açık PR'da durur, karar sende kalır.
221
+ Senin elle koşturduğun bir işe dokunmaz: repo başına eşzamanlılık her zaman 1,
222
+ yani kuyruk senin çalıştığın repoyu `.git/index.lock` için yarışmak yerine atlar.
223
+ PR sayısına tavan yok; sınırlar 24 saatlik yuvarlanan `costCeilingUsd` ve
224
+ makinenin kapasitesi.
225
+
226
+ **`swiftc` varsa menü çubuğu göstergesi** aynı `status.json`'ı sağ üstte çizer ve
227
+ kendi kendine tazelenir: madde başına tek satır - id, Full ya da Short, kesir
228
+ olarak faz, geçen süre, stack. Madde biter bitmez satır kaybolur ve PR'ıyla
229
+ Raporlar altında görünür. Yalnızca **çizer**; bir koşuyu başlatamaz, durduramaz,
230
+ değiştiremez. ActivityKit macOS'ta yok, o yüzden bu bir `NSStatusItem` ve
231
+ imzalanması gerekecek bir ikili olarak değil, yerinde kaynaktan derleniyor.
232
+
233
+ **Yeniden başlatmadan sonra çalıştırman gereken komut yok.** launchd işi
234
+ **girişte** yükler, boot'ta değil - ve bu bir eksiklik değil doğrusu: token'ları
235
+ açan şey login keychain'in açılması, giriş öncesi ateşlenen bir tick zaten ne
236
+ Jira'ya ne GitHub'a bağlanabilirdi. `autopilot-resume` diye bir komut yok -
237
+ hatırlanması gereken komut, unutulduğunda sessizce duran kuyruk demektir.
238
+ Bunun yerine `doctor` gerçek arızayı söyler: kurulu, ama launchd'de iş yok.
239
+
240
+ Uyku **yalnızca prizdeyken** engellenir. Prizde olmayan bir laptop'u boşaltan
241
+ kuyruk hatadır; bataryada assertion bırakılır ve prize takınca kaldığı yerden
242
+ devam eder.
243
+
244
+ **Üç komut da üç CLI'a birden kurulur ve hepsinde aynı şekilde çalışır**, çünkü
245
+ kontrol ettikleri şey bir CLI özelliği değil: tick'i `claude --bg` çocuğu doğuran
246
+ bir launchd kullanıcı ajanı, ve komutu hangi CLI'da yazdığın bunu değiştirmiyor.
247
+ Dolayısıyla sürekli mod her hostta `claude` binary'sinin `PATH`'te olmasını
248
+ istiyor. Kuyruk ve repo seçimi her hostta `~/.claude/autopilot/` altında durur;
249
+ bu bilinçli - bir makinedeki iki CLI tek kuyruk okumalı, iki değil.
250
+
198
251
  ## Stack'ler
199
252
 
200
253
  Stack skill'leri [`mmerterden/multi-agent-plugins`](https://github.com/mmerterden/multi-agent-plugins) marketplace'inde versiyonlu plugin'ler olarak gönderilir. Repo başına bir stack seç:
@@ -207,13 +260,13 @@ Bu, ilgili plugin'i (+ ortak `ai-common` plugin'ini) repo'nun `.claude/settings.
207
260
 
208
261
  ## Araç desteği
209
262
 
210
- Pipeline **Claude Code**, **Copilot CLI** ve **Codex CLI** üzerinde native çalışır - üçü de aynı `pipeline/` kaynağından kurulur ve aynı 53 komutu alır.
263
+ Pipeline **Claude Code**, **Copilot CLI** ve **Codex CLI** üzerinde native çalışır - üçü de aynı `pipeline/` kaynağından kurulur ve aynı 56 komutu alır.
211
264
 
212
265
  | Araç | Bayrak | Ne kurar |
213
266
  | ----------- | ----------------------- | ---------------------------------------------------------------------------------------------------------------- |
214
267
  | Claude Code | `--claude` (varsayılan) | slash komutları + skill'ler + agent'lar + üç `PreToolUse` hook'u (secret scan, agent-guard, okuma-boyutu geçidi) |
215
- | Copilot CLI | `--copilot` | talimatlar + 53 alt-komut skill'i + script'ler |
216
- | Codex CLI | `--codex` | bir router skill + ref olarak 53 spec + 8 agent TOML + `AGENTS.md` bloğu + `codex mcp add` |
268
+ | Copilot CLI | `--copilot` | talimatlar + 56 alt-komut skill'i + script'ler |
269
+ | Codex CLI | `--codex` | bir router skill + ref olarak 56 spec + 8 agent TOML + `AGENTS.md` bloğu + `codex mcp add` |
217
270
 
218
271
  Skill'leri stack'e göre filtrele: `--platform=ios\|android\|all`.
219
272
 
@@ -117,7 +117,7 @@ graph TB
117
117
  end
118
118
 
119
119
  subgraph "Pipeline Specs"
120
- CMD[commands/<br/>53 command files]
120
+ CMD[commands/<br/>56 command files]
121
121
  AGT[agents/<br/>8 agent personas]
122
122
  RUL[rules/<br/>12 domain rules]
123
123
  PHS[multi-agent-refs/phases/<br/>phase specs + contracts]
@@ -169,8 +169,8 @@ revisions of this diagram - Codex CLI and the two independently-shipped repos
169
169
  ```mermaid
170
170
  graph TD
171
171
  CC["Claude Code<br/>(source of truth)"]
172
- COP["Copilot CLI<br/>(instructions + 53 skills)"]
173
- COD["Codex CLI<br/>(1 router skill + 53 refs)"]
172
+ COP["Copilot CLI<br/>(instructions + 56 skills)"]
173
+ COD["Codex CLI<br/>(1 router skill + 56 refs)"]
174
174
  REPO["Pipeline Repo<br/>(npm package)"]
175
175
  WEB["Website"]
176
176
  PLUGREPO["multi-agent-plugins<br/>(5 stack plugins, own repo)"]
package/docs/ecosystem.md CHANGED
@@ -5,7 +5,7 @@ separately, wired together at install time and at run time:
5
5
 
6
6
  | Repo | What it owns | Ships as |
7
7
  |---|---|---|
8
- | **`multi-agent-pipeline`** (this repo) | Orchestration: the 8-phase flow, the 53 slash commands, quality gates, review/triage, cross-CLI parity | npm package (`@mmerterden/multi-agent-pipeline`), installs itself onto Claude Code / Copilot CLI / Codex CLI |
8
+ | **`multi-agent-pipeline`** (this repo) | Orchestration: the 8-phase flow, the 56 slash commands, quality gates, review/triage, cross-CLI parity | npm package (`@mmerterden/multi-agent-pipeline`), installs itself onto Claude Code / Copilot CLI / Codex CLI |
9
9
  | **`multi-agent-plugins`** | Stack knowledge: per-platform component/lifecycle skills (iOS, Android, Frontend, Backend) + shared knowledge | Claude Code marketplace, 5 independently-versioned plugins |
10
10
  | **`multi-agent-toolkit-mcp`** | The pipeline's hands on devices and browsers: 80 MCP tools across 6 categories (simulator/emulator control, accessibility audit, store compliance, web automation, Figma-vs-mock design audit, an agent-DSL batch runner) | npm package, registered as a standard stdio MCP server on every host |
11
11
 
@@ -18,7 +18,7 @@ Either can be swapped or removed without touching the other two's source.
18
18
  graph LR
19
19
  subgraph PIPE ["multi-agent-pipeline (orchestrator)"]
20
20
  direction TB
21
- PHASES["8 phases · 53 commands"]
21
+ PHASES["8 phases · 56 commands"]
22
22
  GATES["deterministic gates + review triage"]
23
23
  end
24
24
 
@@ -64,8 +64,8 @@ only those:
64
64
  graph TD
65
65
  CC["Claude Code<br/>~/.claude/commands/multi-agent/<br/>(source of truth)"]
66
66
 
67
- CC -->|"Step 2: copy + reformat<br/>53 sub-command skills"| COP["Copilot CLI<br/>~/.copilot/skills/"]
68
- CC -->|"Step 2b: transform<br/>(install.js --codex)"| COD["Codex CLI<br/>1 router skill + 53 refs<br/>+ 8 agent TOML"]
67
+ CC -->|"Step 2: copy + reformat<br/>56 sub-command skills"| COP["Copilot CLI<br/>~/.copilot/skills/"]
68
+ CC -->|"Step 2b: transform<br/>(install.js --codex)"| COD["Codex CLI<br/>1 router skill + 56 refs<br/>+ 8 agent TOML"]
69
69
  CC -->|"Step 3: genericize<br/>(strip personal data)"| REPO["multi-agent-pipeline repo<br/>pipeline/"]
70
70
  CC -->|"Step 4: version + feature sync"| WEB["Website<br/>projects.ts / i18n.tsx"]
71
71
 
@@ -153,7 +153,7 @@ measurements behind this table):
153
153
 
154
154
  | | Claude Code | Copilot CLI | Codex CLI |
155
155
  |---|---|---|---|
156
- | **Pipeline commands** | 53 slash-command skills, native | 53 skills, `multi-agent-{cmd}` naming, copied in | 1 router skill (`multi-agent`) + 53 command specs as reference files - Codex silently truncates its skills block past a few dozen entries, so sub-commands are not peer skills here |
156
+ | **Pipeline commands** | 56 slash-command skills, native | 56 skills, `multi-agent-{cmd}` naming, copied in | 1 router skill (`multi-agent`) + 56 command specs as reference files - Codex silently truncates its skills block past a few dozen entries, so sub-commands are not peer skills here |
157
157
  | **Stack plugins** | Marketplace plugin, loaded natively, resolved by `.claude/settings.json` enabled-list | Enabled plugin's authored skills copied flat into `~/.copilot/skills/`; `knowledge/` **not** re-copied (already delivered via `shared/external`) | Copied as reference files under `~/.codex/multi-agent-refs/skills/`, plugin-prefixed on name clash (e.g. `architecture` → `ai-ios-toolkit-architecture`) |
158
158
  | **Component dispatch (Phase 3)** | Marketplace plugin's `create-component`/`create-screen` skill via the Skill tool | No plugin loader - the enabled stack plugin's authored skills (incl. `create-component`) are copied flat into `~/.copilot/skills/` at install time (the old frozen `figma-*` copies are pruned, they were never a fallback) | Not part of the enforced parity axis; classification + state-shape must match, skill *inventory* does not |
159
159
  | **multi-agent-toolkit-mcp** | `claude mcp add multi-agent-toolkit -- npx -y @mmerterden/multi-agent-toolkit-mcp` | `copilot mcp add multi-agent-toolkit -- npx -y @mmerterden/multi-agent-toolkit-mcp` | `codex mcp add multi-agent-toolkit -- npx -y @mmerterden/multi-agent-toolkit-mcp` (skipped with a warning if `codex` isn't on `PATH`) |
@@ -0,0 +1,22 @@
1
+ # Token budget history
2
+
3
+ Every ceiling change in `pipeline/schemas/token-budget.json`, with the reason
4
+ that was written at the time. It lives here rather than inside the schema for a
5
+ plain reason: it had grown to 23,769 characters, larger than four of the eight
6
+ documents it governs, inside a file whose job is to hold eight numbers.
7
+
8
+ It is kept rather than dropped because the reasons are the audit trail - they
9
+ are what separates a considered raise from threshold-chasing, and reading them
10
+ end to end is what made the pattern visible: almost every entry says "the
11
+ smallest step that clears it", which sets the ceiling to the measurement and is
12
+ why six of the eight phases ended up at 99-100% of their own limit.
13
+
14
+ As of v17.1.0 the rules changed. `warn` is no longer stored at all - the gate
15
+ computes it per phase from that phase's own git history. `max` stays a committed
16
+ constant, and a ceiling that sits more than 25% above the measurement now FAILS
17
+ as stale, so reclaimed space is banked instead of being left as room to drift
18
+ back into. The ceiling can still be raised; it can no longer be raised for free.
19
+
20
+ ---
21
+
22
+ Token estimate = ceil(chars / 4). Per-phase budget rule: warn = current+10% (rounded to nearest 50), max = current+25%. Gives ~6 edit cycles of headroom before warn trips - intentionally quiet under normal maintenance, loud when a phase grows unusually. Only the active phase is loaded (lazy). Recalibrated at v10.0.0 after the validator/consistency/simplifier/lesson gate contracts landed in phases 1-4. Recalibrated again at v10.9.0 after the verify-by-test (Phase 4 Step 3.7), update-check (Phase 0 Step 0.6), immutable-test (Phase 3 GREEN) and redTests re-entry contracts landed - Step 3.7 prose was compressed to a pointer into refs/features/verify-by-test.md before the recalibration. Total bumped 50000 -> 51000 at v12.5.0 after the worktree residue/traversal-prune contract (Phase 0 + Phase 5 heal) and the Reflexion causal-diagnosis contract (Phase 4 lesson memory) landed; the prose was compressed first (161 tokens reclaimed) and every per-phase max still passes - only the aggregate needed room. Recalibrated again at v13.6.0 after the install-relative path correction: an instruction that names `pipeline/scripts/x` resolves only from a repo checkout, and a run happens in the user's worktree, so 157 references across these docs moved to `$HOME/.claude/...` at +5 bytes each - 196 tokens of pure correctness cost. Same discipline as before: prose was compressed FIRST (149 tokens reclaimed, by pointing Phase 1's Figma tier table at the Phase 0 probe that already resolved it and Phase 4's Codex constraints at the always-loaded AGENTS.md block), and only then were the budgets moved. Five warn lines had been permanently amber, which makes the amber tier useless as a signal, so every warn was reset to the documented current+10% and the four maxes that the new warn would have collided with were reset to current+25%. Aggregate 51000 -> 51500. Total bumped 51500 -> 52200 at v14.0.0 after Phase 4 Review entered the four --dev mode phase sets and the criteria-resolution contract (Step 1.78) landed. Same discipline as every prior bump: prose was compressed FIRST, 820 tokens reclaimed, before the number moved. Two of those compressions are structural rather than cosmetic - the hardcoded SwiftUI interaction list in Step 1.5 and the SwiftUI convention paragraph in Step 2.8 were transcriptions of rules that now live in a scoped registry, so keeping them here would have re-created the drift this release exists to remove, and the third moved the Step 1.78 full contract into refs/features/skill-conformance.md leaving a pointer. What remains is contract text that cannot be inferred: the manifest's four consumer-visible parts, the conformance checklist the reviewers must return, and the fail-closed semantics. Every per-phase max still passes (phase-4 12405/14750); only the aggregate needed room. Total bumped 52200 -> 52700 at v14.1.0 after two more contracts landed: stack skill routing (Phase 3 pre-flight step 9) and worktree finalize (Phase 6 step 9). Compression came first, as always, and twice: 224 tokens out of Phase 3 by pointing its criteria-ledger and routing steps at their feature files instead of restating them, and 190 out of Phase 6 by moving the finalize contract into refs/features/worktree-finalize.md and leaving the invocation plus the exit-3 semantics. Both new contracts follow the pattern the earlier ones set: the phase doc carries the call and the decision, the feature file carries the reasoning, and the feature files are outside this budget because it loops only the eight phase-N-* keys. Every per-phase max still passes (phase-3 7677/8950, phase-6 5223/6150 and both under warn); only the aggregate needed room. Total bumped 52700 -> 52750 for the Phase 0 Step 3 branch-persistence correction: the step wrote the legacy `projects[].branches` while the TTL filter two sections below read `global.recentBranches`, and both spots named a `{name, lastUsed}` shape the schema rejects (`branch` required, `additionalProperties: false`), so the recent-branch picker option could never populate and a literal implementation would have failed prefs validation. Naming the right target, the right key and the legacy field to avoid costs 41 tokens over the one line it replaces. Compression came first and was applied three times to the replacement text itself, from 120 tokens down to 66, by moving the rationale out of the phase doc entirely: the reasoning now lives where it is enforced, in the migrate-prefs carry-forward comment and the smoke-pref-migration f7 block, leaving the phase doc with only the instruction. 50 was the smallest step that clears it; phase-0-init sits at 10893/12400, far under its own max, so this is purely an aggregate ceiling. v15.0.0: total 52750 -> 53100, the stack-skill tables in phase-1/2/4 now carry plugin-namespaced names (ai-<stack>-toolkit:<skill>) - functional prefixes, ~170 tokens. v15.10.0: total 53350 -> 53950 for the memory-recall + context-offload contracts (Phase 1 two-block durable-knowledge injection and its telemetry, Phase 3 build-log offload pipe, Phase 4 ranked prior art, offload pipe and recall telemetry). Compression came first and twice, taking the new prose from 1168 tokens to 580: the reasoning behind the two blocks lives in multi-agent-refs/prompt-assembly.md and the reasoning behind the offload filter lives in the offload-ref.sh header, both outside this budget, so the phase docs carry only the call, the pref that gates it and the one fact an agent cannot infer - that the evidence gate still reads the whole build log, so offloading changes what is read, never what counts as a verified pass. Every per-phase max still passes (phase-3 7985/8950, phase-4 12997/14750); phase-3 and phase-4 crossed their warn lines and are left amber on purpose, because that is the signal that those two docs are the next ones needing structural compression rather than another bump. v15.13.0: total 53950 -> 54050 for the prefs-to-flag bridges. Five settings had shipped declared-but-inert: contextOffload.minLines and .tailLines (fixed in 15.11.0), learningsLedger.maxBriefEntries, and testGap.scanTree and .promoteSeverity - the last two declared in the schema AND implemented as flags in the scanner, with nothing in between reading the pref and passing the flag. Wiring three of them costs the phase docs 94 tokens, which is the wiring itself and not prose: two `--max` substitutions and a three-line GAP_FLAGS block. Compression came first and twice, as always: the rationale that would have sat in phase-5 now lives in the header of smoke-prefs-consumed.sh, the gate that makes this class fail a build instead of shipping, and a `--severity-promote` table row was dropped because the invocation above it now shows the flag and names the pref that triggers it, which the row did not. 100 was the smallest step that clears it. Every per-phase max still passes; phase-3 and phase-4 remain amber on purpose. v15.14.0: total 54050 -> 54400 for the supported-version gate. Phase 0 Step 0.6 stopped being purely advisory: a release can now publish an npm dist-tag `required` that names the oldest runnable version, and below it the run halts instead of nagging. What the phase doc has to carry is the part an agent cannot infer - the third stdout field, that the halt is identical in autopilot, and that the run must NOT continue on the freshly updated install because its docs were already loaded from the old version. Compression came first, as always, and took the new prose from 469 tokens to 337: the rationale for the floor, the exemption list, the fail-open rules and the `npm dist-tag add` recipe all moved to multi-agent-refs/rules.md "Supported Version Gate" (loaded by 25 commands, outside this budget) and to the header of require-supported-version.sh, leaving the phase doc with the call, the decision table and the halt. 350 was the smallest step that clears it. Every per-phase max still passes (phase-0-init 11230/12400); phase-3 and phase-4 remain amber on purpose. v15.17.0: total 54400 -> 54900 for the Phase 1 analysis-document step. Phase 2 and Phase 3 pre-flights had BLOCKED on `analysis/<feature>-<platform>.md` since v9.0.0 while nothing produced it, so a full run either aborted at Phase 2 or the model ignored its own BLOCKING contract; Step 4 is the producer. What the phase doc carries is only what cannot be inferred: the when-table (taskType x Figma reference), the four refs in load order, the two artefacts, and that the doc validator fails closed. Compression came first and took the step from 745 tokens to 497: the history of why the gap existed moved to the CHANGELOG, the per-ref one-line descriptions moved into the refs' own headers, and the autopilot carve-out collapsed to one clause. The 17.4k-token analysis engine itself is NOT in this budget - it moved out of commands/ into multi-agent-refs/analysis/{locked,evidence,synthesis,render}.md, loaded on demand, which also took analysis/SKILL.md from 18081 to 5974 tokens and retired its lint grace entry. 500 was the smallest step that clears it; phase-1-analysis sits at 4338/4600 and is amber on purpose, like phase-3 and phase-4. v15.18.0: total 54900 -> 55250 for analysis mode. Three phase docs gained a mode branch that cannot be inferred: Phase 4 reviews a document instead of a diff (validator, the one question reviewers answer, the open-question walk), and Phase 6 publishes instead of committing. Compression came first and was applied twice to the new prose and once to old: the Phase 4 branch went from 320 tokens to 180 and the Phase 6 branch from 190 to 120 by pointing at multi-agent-refs/analysis/{resolve,render}.md, which now hold the walks themselves, and the front-matter parse contract stopped being spelled out in both pre-flights. The analysis engine keeps leaving this budget rather than entering it: intake joined locked/evidence/synthesis/render/resolve in multi-agent-refs/analysis/, which is what let analysis/SKILL.md drop under the 6000 hard cap after its grace entry was retired. 350 was the smallest step that clears it; phase-4 and phase-6 are amber on purpose, as phase-1 and phase-3 already were. v15.20.0: total 55250 -> 55500 for the TDD bridge. Phase 3 pre-flight read the analysis doc's concept table and even said test method names come from it, while nothing read Section 15 - so the RED step invented tests and the analysis test matrix never reached development. Phase 3 step 5b now loads it into state.dev.testPlan[] and Phase 4 step 1.45 cross-checks every planned row against a real test, which is what turns "analysis quality is output quality" from a slogan into a finding. Compression came first on both blocks, 300 tokens down to 175, by dropping the enumerated failure modes to one line each and the rationale to one clause; the reasoning lives in the CHANGELOG. 250 was the smallest step that clears it. v15.21.0: total 55500 -> 55800 for the post-analysis confirmation. Phase 2 gained Step 0.9, the last human checkpoint before Phase 3: derived values are shown for confirmation and only Section 20 rows are asked, through the resolve engine that already exists in refs. It belongs here rather than Phase 4 because Phase 4 runs after development, where an answer arrives too late to change anything. Compression came first and twice, 430 tokens down to 250, by collapsing the derived-vs-asked explanation to one sentence each and moving the walk itself to multi-agent-refs/analysis/resolve.md, which Phase 4 and analysis-resolve already mount. 300 was the smallest step that clears it. v15.22.0: total 55800 -> 55900 for the analyst-toolkit hooks. Phase 1 Step 4 now names the two prefs that decide whether a document is produced at all and how deep it goes (forceFull, mode) - the first of those had shipped declared-but-inert and smoke-prefs-consumed caught it - and Phase 4 triage gained one clause: a finding that blames a third-party library asks evidence-github whether it is already open upstream, which turns it into a deferred item with a citation instead of Phase 3 rework on code that is not ours. Compression came first and three times, taking the new prose from 220 tokens to 110, and the Phase 1d evidence contract itself never entered this budget - it lives in multi-agent-refs/analysis/evidence.md beside the phases it belongs to. 100 was the smallest step that clears it, leaving 34 tokens of headroom. phase-4 stays amber and the debt named at v15.10.0 stands: it is the doc that needs structural compression rather than another bump, and the two candidates are the inline triage JSON shape and the 3.4 telemetry block, both of which restate something already authoritative elsewhere. v16.0.0: total 55900 -> 56350 for the depth picker. `--dev` and the four dev-* commands are gone; depth is Phase 0 Step 7.5, which costs phase-0-init a step it did not have. Compression came first and three times, taking the step from 530 tokens to 300: the question wording, the per-taskType recommendation and the mode tables all live in phases/modes.md (outside this budget), so the phase doc carries only what an agent cannot infer - that the step runs after Step 7 and why, who is exempt, that ASK_CHOICE_DEFAULT must be passed explicitly because ask-choice.sh takes the FIRST option on a non-TTY, and that Short flips the Phase 1/2 tiles late rather than pre-marking them. The phase-4 telemetry block named as compression debt at v15.22.0 was collapsed to an emit() helper (-27) and the four dev-* mode files left the tree entirely, but neither offsets a genuinely new phase step. 450 was the smallest step that clears it, leaving 119 tokens of headroom. phase-4 remains amber and its other named candidate, the inline triage JSON shape, was left alone on purpose: it is the prompt the triage agent is handed, not a restatement for readers. v16.2.0: total 56350 -> 56600 for the spec-freshness and reuse-tag contracts. Phase 3 step 3 had compared `state.run.lastAnalysisDigest` since it was written, against a key nothing ever set and that the state schema did not declare, so the staleness branch was unreachable and every run reported fresh by default. Phase 1 now persists the digest and a `base_commit` anchor, and step 3 gained the repo-drift half the digest cannot see: a reused document keeps a matching digest precisely because its evidence inputs did not change, while the code underneath it moved. The second contract is the Section 14 tag reaching development: Phase 2 carries it onto the todo as `sourceTag` and Phase 3 treats it as an instruction, which is what stops a Reuse row from being re-implemented. Compression came first and took the four additions from 380 tokens to 214, by moving every rationale clause out of the phase docs: why the commit anchor exists rather than a digest recomputation lives in this note and the CHANGELOG, and the schema descriptions carry the field semantics. The baseline had 9 tokens of headroom, so no addition of any size could have fit without a bump. 250 was the smallest step that clears it, leaving 45 tokens. phase-3 and phase-4 remain amber. v16.13.0: total 57600 -> 57700 for the code-graph injection and the fable-rung switch. Phase 1 gained Step 2.6 (query the graph, hand Explore a ranked starting set), Phase 7 gained the post-branch graph refresh, and Phase 0 Step 0 gained one line: a prefs switch that resolves every preferredModel: fable persona to opus for the run, which also collapses the Phase 4 Claude Code panel from three reviewers to two. Compression came first and mostly structurally: of roughly 1,630 tokens of new contract text, 1,310 never entered this budget at all - the whole code-graph contract lives in multi-agent-refs/features/code-graph.md (604) and the fable switch's scope table, per-host effects and cost-accounting consequence live in features/model-fallback.md (+707), leaving the phase docs with the call, the pref that gates it and the one fact an agent cannot infer. Phase 4 was compressed on top of that: its TLDR restated the reviewer matrix 270 lines below it, so 36 tokens came back and the doc nets +6 despite carrying two new clauses. One of those clauses is a correction rather than a feature - the consensus rule still said reviewerCount is 2 on Claude Code, which stopped being true when the third reviewer landed in 16.12.0, and the cross-CLI smoke never caught it because it reads the matrix line instead. 100 was the smallest step that clears it, leaving 54 tokens. phase-3 and phase-4 remain amber. v16.17.0: total 57700 -> 57850 for the platform-parity cross-check. Phase 4 gained Step 1.8: when dev-context carries a counterpart app repo, the review compares the change against the other platform on four axes. Compression came first and structurally, as always - of roughly 1,610 tokens of new contract text, 1,490 never entered this budget at all, because the four axes, the file cap, the graph-query recipe, the read-only prohibitions and the rule that an extractor miss may not be reported as an absence all live in multi-agent-refs/platform-parity.md. The step itself was then cut from ~200 tokens to 120 by deleting everything the ref already owns, leaving the trigger, the pointer and the two facts an agent must not infer: the counterpart repo is read-only, and parity findings are never blocking. The baseline had 13 tokens of headroom, so no addition of any size could have fit without a bump. 150 was the smallest step that clears it, leaving 35 tokens. phase-3 and phase-4 remain amber, and phase-4's structural-compression debt still stands. v16.20.0: phase-4-review max 14750 -> 15150 and total 58250 -> 60250 for the cross-round review delta, the scope self-check handoff and the circuit-breaker wiring. Compression came first and structurally: of roughly 3,900 tokens of new contract text, 2,700 never entered this budget at all - the previous-round-findings block, the scope-self-check block, the Step 3.8 state merge, telemetry and picker wording live in multi-agent-refs/features/review-delta.md, and the scope-check record rules and consumers in features/scope-check.md - so the phase docs carry the call, the pref that gates it and the exit table. The Phase 3 stability rule and the trigger-3 write were cut twice more before the bump; phase-3 stays under its max (8692/8950). Phase 4 is the first per-phase max raised since v10.9.0: the doc gained three steps that cannot be inferred (a per-round triage file, a prefix block that changes what reviewers report, and a halt condition), and its structural-compression debt (the inline triage JSON shape, named at v15.10.0) still stands and is the next candidate. 15150 and 60250 were the smallest steps that clear it, leaving 25 and 45 tokens. v16.23.0: phase-0-init max 12400 -> 12500 and total 60250 -> 60500 for the widget-registration call and the accounting gate. Phase 0 gained the `tiles` call and the exit-3 rule, Phase 7 gained the run report; together they are contract an agent cannot infer - which call registers this host's widget, and that a completion is refused without recorded spend. Compression came first and twice, taking the new prose from 472 tokens to 255: the per-host call list moved into tracker-contract.md "The card is not the widget" and the record-then-rerun recovery into "Accounting is a gate", both outside this budget, leaving the phase docs with the call and the one fact that cannot be looked up. 100 and 250 were the smallest steps that clear it, leaving 74 and 40 tokens. phase-3 and phase-4 remain amber. v16.24.0: total 60500 -> 60750 for visual evidence. Four phase docs gained one instruction each that cannot be inferred: Phase 0 keeps the issue's own images as the pre-fix evidence, Phase 3 captures the fixed state (there and not Phase 5, because every autopilot and --local entry drops Phase 5), Phase 5 hosts the flow recording when it runs, and Phase 6 blocks on a required artefact that is neither attached nor explained. Compression came first and twice, 42 tokens back, and the contract itself never entered this budget: the trigger matrix, the three video tiers, the size-degradation ladder and both render shapes live in multi-agent-refs/features/visual-evidence.md. phase-0-init cleared its own max without a bump. 250 was the smallest step that clears the aggregate. phase-3 and phase-4 remain amber. v17.0.0: phase-0-init max 13000 -> 13100 and total 62400 -> 62500 for the evidence-verdict writer. Phase 0 Step 7.7 probed with `--platform "$PLATFORM"`, a variable no phase document ever assigned, and it was gated on `visualEvidence.required`, which no phase document ever wrote - five readers, zero writers - so the step, Phase 3's capture and Phase 6's blocker were all unreachable and the pipeline reported nothing wrong. The step now writes the verdict and derives the platform from the stack, skipping the probe with a recorded reason when there is no device platform rather than passing the empty string the probe refuses with exit 2. Compression came first and three times, 30 tokens back from the Step 7.7 index rule that restated Step 7.5 verbatim and 55 from the new block itself; the reasoning never entered this budget, because who writes the verdict and how the platform is derived live in features/visual-evidence.md sections 1a and 1b. phase-3-dev max 9250 -> 9300 in the same change: it reads the platform back from state and re-decides the provisional verdict before capturing, which is the half of the fix that makes Phase 3 honest rather than merely reachable. Compressed three times first, 29 tokens back, by pointing its Phase-5 rationale and its tier mapping at visual-evidence.md sections 3 and 4.3 where both already live. 100, 50 and 150 were the smallest steps that clear it, leaving 24, 15 and 26 tokens. phase-3 and phase-4 remain amber. v17.1.0: total 62550 -> 62600 for the Figma/toolkit MCP distinction and web as an evidence platform. Three phase docs said "MCP forbidden" without qualifying it, while the gate that enforces it (smoke-no-mcp-in-dev-phases.sh) has always matched `figma` and nothing else - so the prose banned the screenshot, xcodebuild and UI-test tools that Phase 3 Steps 3.4 and 3.55 actually call, which is one way a run reaches Phase 7 with no evidence. Phase 0 gained one `web` arm in the platform derivation, now that run-ui-tests.sh has a web arm to derive it for. Compression came first and three times, taking the new prose from 175 tokens to 30: the reasoning moved to rules.md, whose own seven-row Figma phase matrix was deleted in the same pass because it duplicated rules/figma-pipeline.md "Phase access matrix" two paragraphs below this file's own instruction not to duplicate that rule file - 74 tokens back there, which is why phase-3-dev cleared its max without a bump (9298/9300). 50 was the smallest step that clears the aggregate, leaving 37 tokens. phase-3 and phase-4 remain amber. v17.1.0 (2): total 62600 -> 62700 for the plan reaching the task widget. Phase 2 computed tasks[], their order and their dependsOn[] edges, stored them, and used them to drive Phase 3's ready-task picker - and none of it was visible on the surface the user actually watches; the card had drawn sub-phases for releases, the widget never had. Phase 2 gains one call. Compression came first and twice: the tasks[]-to-todos[] jq blob left the phase doc for plan-todos.sh `set`, which now accepts a planning-output document directly (-28, and it removes a mapping two files defined, of which this was the untested copy), and the new step's own prose was cut from 116 tokens to 61. The parsing of the plan itself never entered this budget - it lives in phase-tracker.sh `plan`, which owns the sub-phase structure it writes. 100 was the smallest step that clears it, leaving 75 tokens. phase-3 and phase-4 remain amber.
@@ -62,6 +62,7 @@ const DEV_ONLY_TOOLING = Object.freeze([
62
62
  "lint-personas.mjs", // imports install/_codex-agents.mjs, which never ships to scripts/
63
63
  "lint-mcp-refs.mjs",
64
64
  "check-md-links.mjs",
65
+ "gen-ref-toc.mjs", // regenerates the repo's own reference tree; an install has no pipeline/
65
66
  "validate-schemas.mjs", // validates the repo's own schema files, needs ajv
66
67
  "sync-parity-check.sh",
67
68
  "benchmark-phase-0.sh",
package/install/codex.mjs CHANGED
@@ -18,7 +18,7 @@
18
18
  */
19
19
 
20
20
  import { existsSync, readFileSync, readdirSync, rmSync } from "fs";
21
- import { join } from "path";
21
+ import { dirname, join } from "path";
22
22
 
23
23
  import {
24
24
  copyDir,
@@ -206,6 +206,7 @@ export function installCodex(ctx) {
206
206
  const CODEX_SCRIPTS = join(CODEX_DIR, "scripts");
207
207
  const CODEX_LIB = join(CODEX_DIR, "lib");
208
208
  const CODEX_SCHEMAS = join(CODEX_DIR, "schemas");
209
+ const CODEX_TEMPLATES = join(CODEX_DIR, "templates");
209
210
  const CODEX_RULES = join(CODEX_DIR, "rules");
210
211
  const CODEX_SKILL_REFS = join(CODEX_MA_REFS, "skills");
211
212
 
@@ -221,6 +222,7 @@ export function installCodex(ctx) {
221
222
  installTree("lib", pipelineSrc, CODEX_LIB, useSymlinks);
222
223
  installTree("schemas", pipelineSrc, CODEX_SCHEMAS, useSymlinks);
223
224
  installTree("rules", pipelineSrc, CODEX_RULES, useSymlinks);
225
+ installTemplates(pipelineSrc, CODEX_TEMPLATES, useSymlinks);
224
226
  writeAgentsMd(join(CODEX_DIR, "AGENTS.md"));
225
227
  ensureSharedPreferences(home, pipelineSrc);
226
228
  registerMcpServer("codex", "Codex CLI");
@@ -501,6 +503,21 @@ function installScripts(pipelineSrc, dest, useSymlinks) {
501
503
  // path from the caller's side, so they stay byte-identical.
502
504
  const REWRITTEN_TREES = new Set(["rules"]);
503
505
 
506
+ function installTemplates(pipelineSrc, dest, useSymlinks) {
507
+ // Lives under install/, not pipeline/, so installTree cannot reach it. Only
508
+ // install/claude.mjs laid it down before, which left autopilot-on reading the
509
+ // launchd plist template from a directory a Codex-only host never had.
510
+ const src = join(dirname(pipelineSrc), "install", "templates");
511
+ if (!existsSync(src)) return;
512
+ if (!useSymlinks) {
513
+ ensureRealDir(dest);
514
+ ensureDir(dest);
515
+ }
516
+ wipeDir(dest);
517
+ copyDir(src, dest, { useSymlinks });
518
+ console.log(` -> ${countFiles(src)} template file(s) copied to ${dest}`);
519
+ }
520
+
504
521
  function installTree(name, pipelineSrc, dest, useSymlinks) {
505
522
  const src = join(pipelineSrc, name);
506
523
  if (!existsSync(src)) return;
@@ -9,7 +9,7 @@
9
9
  */
10
10
 
11
11
  import { existsSync, mkdirSync, readFileSync, readdirSync, rmSync } from "fs";
12
- import { join } from "path";
12
+ import { dirname, join } from "path";
13
13
 
14
14
  import {
15
15
  copyDir,
@@ -61,6 +61,7 @@ export function installCopilot(ctx) {
61
61
  const COPILOT_SCHEMAS = join(COPILOT_DIR, "schemas");
62
62
  const COPILOT_LIB = join(COPILOT_DIR, "lib");
63
63
  const COPILOT_RULES = join(COPILOT_DIR, "rules");
64
+ const COPILOT_TEMPLATES = join(COPILOT_DIR, "templates");
64
65
 
65
66
  console.log(" [Copilot CLI] Installing pipeline instructions...");
66
67
  ensureDir(COPILOT_DIR);
@@ -71,6 +72,7 @@ export function installCopilot(ctx) {
71
72
  installSchemas(pipelineSrc, COPILOT_SCHEMAS, useSymlinks);
72
73
  installLib(pipelineSrc, COPILOT_LIB, useSymlinks);
73
74
  installRules(pipelineSrc, COPILOT_RULES, useSymlinks);
75
+ installTemplates(pipelineSrc, COPILOT_TEMPLATES, useSymlinks);
74
76
  installSkills({ home, pipelineSrc, dest: COPILOT_SKILLS, indexOnly, useSymlinks, platformFlag });
75
77
 
76
78
  // Copilot never registered the companion MCP server, while Codex did from the day its
@@ -258,6 +260,20 @@ function installSchemas(pipelineSrc, dest, useSymlinks) {
258
260
  console.log(` -> ${countFiles(schemasSrc)} files copied to ${dest}`);
259
261
  }
260
262
 
263
+ function installTemplates(pipelineSrc, dest, useSymlinks) {
264
+ // Only install/claude.mjs used to lay this tree down, so a Copilot-only or
265
+ // Codex-only machine had no ~/.claude/templates - and autopilot-on reads the
266
+ // launchd plist template out of it. The mode would have written a job for a
267
+ // file the host does not have.
268
+ console.log(" [Copilot CLI] Installing templates...");
269
+ const templatesSrc = join(dirname(pipelineSrc), "install", "templates");
270
+ if (!existsSync(templatesSrc)) return;
271
+ if (!useSymlinks) ensureRealDir(dest);
272
+ wipeDir(dest);
273
+ copyDir(templatesSrc, dest, { useSymlinks });
274
+ console.log(` -> ${countFiles(templatesSrc)} files copied to ${dest}`);
275
+ }
276
+
261
277
  function installLib(pipelineSrc, dest, useSymlinks) {
262
278
  console.log(" [Copilot CLI] Installing shell libraries...");
263
279
  const libSrc = join(pipelineSrc, "lib");