@mmerterden/multi-agent-pipeline 14.1.0 → 14.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/CHANGELOG.md +177 -1
  2. package/README.md +4 -4
  3. package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -1
  4. package/package.json +1 -1
  5. package/pipeline/commands/deploy.md +4 -1
  6. package/pipeline/commands/multi-agent/SKILL.md +6 -3
  7. package/pipeline/commands/multi-agent/dev/SKILL.md +5 -1
  8. package/pipeline/commands/multi-agent/help/SKILL.md +49 -11
  9. package/pipeline/commands/multi-agent/setup/SKILL.md +1 -1
  10. package/pipeline/commands/multi-agent/store-ready/SKILL.md +340 -0
  11. package/pipeline/commands/multi-agent/sync/SKILL.md +11 -5
  12. package/pipeline/commands/multi-agent/test/SKILL.md +18 -8
  13. package/pipeline/commands/multi-agent/test-accessibility/SKILL.md +33 -0
  14. package/pipeline/commands/multi-agent/test-dark-mode/SKILL.md +33 -0
  15. package/pipeline/commands/multi-agent/test-dynamic-type/SKILL.md +33 -0
  16. package/pipeline/commands/multi-agent/test-screenshots/SKILL.md +41 -0
  17. package/pipeline/commands/multi-agent/testflight-validation/SKILL.md +28 -201
  18. package/pipeline/commands/sim-test.md +45 -36
  19. package/pipeline/multi-agent-refs/cross-cli-contract.md +3 -2
  20. package/pipeline/multi-agent-refs/knowledge.md +1 -1
  21. package/pipeline/multi-agent-refs/phases/phase-0-init.md +7 -4
  22. package/pipeline/schemas/prefs.schema.json +1 -1
  23. package/pipeline/schemas/token-budget.json +2 -2
  24. package/pipeline/scripts/build-stack-plugins.mjs +21 -0
  25. package/pipeline/scripts/migrate-prefs.mjs +30 -0
  26. package/pipeline/skills/.skills-index.json +57 -12
  27. package/pipeline/skills/shared/README.md +11 -6
  28. package/pipeline/skills/shared/core/multi-agent/SKILL.md +13 -17
  29. package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +50 -12
  30. package/pipeline/skills/shared/core/multi-agent-purge/SKILL.md +18 -3
  31. package/pipeline/skills/shared/core/multi-agent-store-ready/SKILL.md +50 -0
  32. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +4 -3
  33. package/pipeline/skills/shared/core/multi-agent-test/SKILL.md +18 -8
  34. package/pipeline/skills/shared/core/multi-agent-test-accessibility/SKILL.md +37 -0
  35. package/pipeline/skills/shared/core/multi-agent-test-dark-mode/SKILL.md +37 -0
  36. package/pipeline/skills/shared/core/multi-agent-test-dynamic-type/SKILL.md +37 -0
  37. package/pipeline/skills/shared/core/multi-agent-test-screenshots/SKILL.md +44 -0
  38. package/pipeline/skills/shared/core/multi-agent-testflight-validation/SKILL.md +29 -101
  39. package/pipeline/skills/shared/external/firebase/SKILL.md +1 -1
  40. package/pipeline/skills/skills-index.md +9 -4
@@ -1,219 +1,46 @@
1
1
  ---
2
- description: "Pre-submission validation for a TestFlight / App Store build (iOS, local-only). Three gates: static archive audit, Apple's own `altool --validate-app`, and a Review-Guidelines check. ITMS codes are mapped to the rule each implies. Validates only, never uploads. Use when a build is about to go to TestFlight, or a submission was rejected and you need why."
3
- description-tr: "TestFlight / App Store yüklemesi öncesi doğrulama (iOS, yalnızca lokal). Repo + branch seç, sonra ya build'i sen ver ya da koşu archive alsın; üç kapıyı geç: statik 18-kurallı archive denetimi, Apple'ın kendi `altool --validate-app`'i, ve App Store Review Guidelines'a karşı guideline incelemesi. Her kapı için ayrı verdict, ITMS kodları ilgili kurala eşlenmiş. Sadece doğrular - asla yüklemez."
2
+ description: "iOS-pinned alias of store-ready: same three gates on an iOS build, one implementation. Validates a TestFlight / App Store package, never uploads. Use when an iOS build is about to go to TestFlight, or a submission was rejected and you need why."
3
+ description-tr: "store-ready komutunun iOS'a sabitlenmiş alias'ı: iOS build'inde aynı üç kapı, tek implementasyon. TestFlight / App Store paketini doğrular, asla yüklemez."
4
4
  argument-hint: "[repo] - empty = pick from prefs; repo name or path; --ipa=<path>; --archive=<path>; --resume"
5
5
  allowed-tools: Agent, Bash, Read, Write, Edit, Glob, Grep, TaskCreate, TaskUpdate, AskUserQuestion, Skill, mcp__dev-toolkit__ios_app_store_audit, mcp__dev-toolkit__ios_export_ipa, mcp__dev-toolkit__ios_testflight_validate, mcp__dev-toolkit__ios_xcodebuild, mcp__dev-toolkit__ios_xcresult
6
6
  ---
7
7
 
8
- # /multi-agent:testflight-validation - pre-submission validation
8
+ # /multi-agent:testflight-validation - iOS alias for store-ready
9
9
 
10
- Catch, before you upload, what App Store Connect would send back after you do.
10
+ This command is an alias. The implementation it used to carry was merged into
11
+ `/multi-agent:store-ready`, which runs the same three gates on iOS and adds the
12
+ Android side, so there is one flow to maintain instead of two that had already
13
+ started to drift.
11
14
 
12
- **Local-only.** No commits, no push, no PR, no channels. The worktree exists only
13
- to archive without touching your working tree.
15
+ The name is kept because it is the one people reach for when the target is
16
+ TestFlight, and removing a command is a breaking change to the slash-command
17
+ surface.
14
18
 
15
- **It never uploads.** Only `--validate-app` is ever invoked, never `--upload-app`.
16
- A validation run must not be able to ship a build by accident.
19
+ ## Dispatcher
17
20
 
18
- ## Why three gates and not one
19
-
20
- Each gate sees something the others structurally cannot. Reporting one of them as
21
- "the check" is how a build passes locally and gets rejected anyway.
22
-
23
- | Gate | What runs | Needs | Sees | Blind to |
24
- |---|---|---|---|---|
25
- | **1. Static** | `ios_app_store_audit` (18 rules, real ITMS codes) | an `.xcarchive` | privacy manifest, required-reason API, Info.plist, code signing, entitlements, embedded SDK, IPv6, debug-tool leak, binary size | anything that depends on the App Store Connect account |
26
- | **2. Authoritative** | `ios_testflight_validate` → `altool --validate-app` | an `.ipa` + credentials | unregistered bundle ID, profile that does not match the app record, **a version+build pair already used**, entitlements not provisioned for the App ID | the Review Guidelines - Apple's validator does not read them |
27
- | **3. Guideline** | `app-store-review` skill + repo evidence | repo checkout | ATT flow, privacy policy, account deletion, IAP rules, purpose-string wording, permission justification | anything not visible in source |
28
-
29
- Gate 2 is the only one that asks Apple, and Gate 3 is the only one that covers the
30
- rejections a human reviewer writes. Most "we passed validation and still got
31
- rejected" cases are Gate 3 findings.
32
-
33
- ## Step 0 - parse input
34
-
35
- | Input | Meaning |
36
- |---|---|
37
- | (empty) | ask which repo (Step 1) |
38
- | `my-ios-app` or a path | that repo |
39
- | `--ipa=<path>` | Mode B with an `.ipa`; Gate 1 cannot run (see Step 3) |
40
- | `--archive=<path>` | Mode B with an `.xcarchive`; all three gates run |
41
- | `--resume` | continue the last run from its state file |
42
-
43
- State lives at `$HOME/.claude/logs/multi-agent/<task_id>/agent-state.json` with
44
- `taskId = TFV-<repo>-<yyyymmddHHMM>`. Register phases with the tracker
45
- (`$HOME/.claude/multi-agent-refs/tracker-contract.md`) so `:resume` and `:status`
46
- work like any other run.
47
-
48
- ## Step 1 - pickers (native, always)
49
-
50
- Use `AskUserQuestion` for every step - never a numbered text menu. Questions and
51
- descriptions render in `prefs.global.outputLanguage`; `label` and `header` stay
52
- English, per `$HOME/.claude/multi-agent-refs/picker-contract.md`. Print the
53
- `Step <i>/<n>: <what this decides>` breadcrumb for each.
54
-
55
- 1. **Repo** - from `prefs.projects` where the stack is iOS. A single match
56
- auto-resolves (say so in the breadcrumb, do not silently skip the step).
57
- 2. **Branch** - the branch to validate. Resolution order:
58
- - `git fetch --prune` first, capturing **stderr**. **If the fetch fails, do not
59
- silently fall back to a cached ref**, and **classify before naming a cause** -
60
- the same rule as the `/multi-agent:dev` remote gate:
61
-
62
- | stderr contains | Cause | Remedy |
63
- |---|---|---|
64
- | `could not read Password`, `Authentication failed`, `403` | credential | store the PAT in the credential helper or switch the remote to SSH. **A VPN cannot fix this**, and the base ref being stale is unrelated to what broke - do not offer the cached-ref fallback. |
65
- | `Could not resolve host`, `Operation timed out`, `Connection refused` | network | retry / continue on the cached ref with an explicit warning / switch remote / abort, per `$HOME/.claude/multi-agent-refs/rules.md` |
66
- | `Repository not found`, `404` | wrong remote | show `git remote -v` and ask |
67
-
68
- Always print the observed stderr line next to the classification. Asserting
69
- `unreachable (VPN/DNS)` for a missing-credential error that returns in under a
70
- second sends the user to fix something that was never broken.
71
- - Offer the current branch, the default branch, and any `release/*` /
72
- `tkdevelop/*` heads.
73
- 3. **Mode** - how the build is obtained:
74
- - `Supply a build` (Mode B, default) - fastest, no signing needed in-run.
75
- - `Archive from this branch` (Mode A) - needs a distribution certificate and
76
- profile in the keychain, and takes as long as a release archive.
77
-
78
- ## Step 2 - pre-flight, before anything expensive
79
-
80
- Report every line; a missing prerequisite is a halt, not a warning.
81
-
82
- ```bash
83
- xcrun --find altool >/dev/null 2>&1 || echo "MISSING: altool (install Xcode)"
84
- xcodebuild -version | head -1
85
- ```
86
-
87
- Then resolve credentials, and **state which tier is active in the report**:
88
-
89
- | Tier | Source | Effect |
90
- |---|---|---|
91
- | 1 | ASC API key - key id + issuer id from the keychain via `prefs.global.keychainMapping`, `.p8` at `~/.appstoreconnect/private_keys/AuthKey_<keyId>.p8` | Gate 2 runs |
92
- | 2 | Apple ID + app-specific password, referenced as a keychain item | Gate 2 runs |
93
- | 3 | neither | **Gate 2 reports `SKIPPED`, and the run says so in the verdict line** |
94
-
95
- Credentials come from `/multi-agent:setup`; never prompt for a secret value in
96
- chat. If nothing is configured, tell the user which of the two tiers they can set
97
- up and that tier 2 needs no elevated App Store Connect role.
98
-
99
- Multi-provider accounts need `--provider-public-id`. When it is not in prefs, run
100
- `ios_testflight_validate({list_providers: true})` once and ask which provider.
101
-
102
- ## Step 3 - obtain the build
103
-
104
- ### Mode B - a build you supply
105
-
106
- - `.xcarchive` → Gate 1 runs on it. To reach Gate 2 the archive must be exported,
107
- so run `ios_export_ipa` (see Mode A step 3 for the signing inputs).
108
- - `.ipa` only → **Gate 1 is reported `SKIPPED (needs .xcarchive)`.** The static
109
- audit reads archive structure that an `.ipa` does not carry. Do not present a
110
- two-gate run as a full pass; say which gate did not run and why, and offer to
111
- re-run with the archive.
112
-
113
- ### Mode A - archive from the branch
114
-
115
- 1. Worktree at `{projectRoot}/{worktreeBasePath}/{taskId}` on the chosen branch.
116
- **Never under `$HOME`**, never a direct checkout of the main working tree.
117
- 2. Resolve the scheme and workspace/project from prefs; ask if ambiguous.
118
- 3. Archive:
119
- `ios_xcodebuild({workspace|project, scheme, action: "archive", configuration: "Release", destination: "generic/platform=iOS"})`
120
- Note the destination: the simulator default would produce an archive that
121
- cannot be exported for distribution.
122
- 4. Export:
123
- `ios_export_ipa({archive_path, output_dir, method: "app-store-connect", team_id, provisioning_profiles?, signing_style?})`
124
- Leave `allow_provisioning_updates` off unless the user asks: it lets xcodebuild
125
- create or modify profiles in the developer account, which a validation run has
126
- no business doing.
127
- 5. A failed export halts with the parsed errors. The usual causes are a missing
128
- distribution certificate, a profile that does not match the bundle ID, or
129
- `signing_style: "manual"` with no `provisioning_profiles` map.
130
-
131
- ## Step 4 - Gate 1, static audit
132
-
133
- `ios_app_store_audit({archive_path, rules: "all"})`.
134
-
135
- `error` findings are blocking; `warning` is advisory. Group the output by severity
136
- and keep each finding's ITMS code - Gate 2 may return the same code, and seeing
137
- it in both places tells the user it is real rather than a heuristic.
138
-
139
- ## Step 5 - Gate 2, Apple's own validation
140
-
141
- `ios_testflight_validate({ipa_path, platform: "ios", <credential args>})`.
142
-
143
- Render the verdict exactly as returned:
144
-
145
- - `PASS` - Apple accepted the binary for delivery.
146
- - `FAIL` - list each issue with its ITMS code, the mapped guideline, and the hint.
147
- - `SKIPPED` - print the reason. **Never render this as a pass.** The verdict line
148
- for the whole run must read `2 of 3 gates cleared, 1 skipped`, not `passed`.
149
-
150
- Gate 2 is the only gate that catches a build number already used - the most
151
- common wasted upload - so when it fails on that, say so plainly and name the next
152
- free build number.
153
-
154
- ## Step 6 - Gate 3, guideline review
155
-
156
- Load the `app-store-review` skill and review the repo against it. This is the gate
157
- that catches what a human reviewer rejects, so it reads source, not the binary:
158
-
159
- | Area | Evidence to gather |
160
- |---|---|
161
- | Purpose strings | every `NS*UsageDescription` in Info.plist - present, specific, user-facing, and matching what the code actually does with the data |
162
- | Privacy manifest | `PrivacyInfo.xcprivacy` exists, declares required-reason APIs, and matches the SDKs actually linked |
163
- | Tracking | if any tracking API or SDK is present, an ATT prompt exists and runs before collection |
164
- | Account deletion | if the app creates accounts, an in-app deletion path exists (guideline 5.1.1(v)) |
165
- | Privacy policy | reachable in-app and in the metadata |
166
- | IAP | anything unlocking features goes through StoreKit, with no external purchase path |
167
- | Sign in with Apple | present when a third-party social login is offered |
168
-
169
- For each: `pass` / `fail` / `not-applicable` with the evidence path that justifies
170
- it. `not-applicable` needs a reason - an unexamined area is not a pass.
171
-
172
- ## Step 7 - report
173
-
174
- Write to `~/TestFlightChecks/<repo>-<branch>-<timestamp>/report.md` and print a
175
- summary. Structure:
21
+ Read this file and apply its instructions:
176
22
 
177
23
  ```
178
- Verdict: <N> of 3 gates cleared[, <M> skipped]
179
- Build: <ipa or archive path> · <bundle id> <version> (<build>)
180
- Auth: tier <1|2|none> · <method>
181
-
182
- Gate 1 static audit PASS | FAIL (<n> blocking, <n> advisory) | SKIPPED (<reason>)
183
- Gate 2 Apple validation PASS | FAIL (<n> issues) | SKIPPED (<reason>)
184
- Gate 3 guideline review PASS | FAIL (<n> findings) | <n> not-applicable
185
-
186
- Blocking - fix before uploading
187
- [ITMS-90683] Info.plist: NSCameraUsageDescription missing
188
- guideline 5.1.1 Data Collection and Storage
189
- <hint>
190
- <file:line>
191
-
192
- Advisory
193
- ...
194
-
195
- Not run
196
- Gate 1: needs an .xcarchive; only an .ipa was supplied
24
+ $HOME/.claude/commands/multi-agent/store-ready/SKILL.md
197
25
  ```
198
26
 
199
- Rules for the report:
27
+ Run it with **platform pinned to `ios`** and skip its platform-detection step. Pass
28
+ `$ARGUMENTS` through verbatim: `--ipa=`, `--archive=`, a repo name or path, and
29
+ `--resume` all mean exactly what they mean there.
200
30
 
201
- - A skipped gate is never folded into the pass count. The verdict line states the
202
- skip.
203
- - Every blocking finding carries a file path or an ITMS code. A finding the user
204
- cannot act on is noise.
205
- - No AI or assistant attribution anywhere, per
206
- `$HOME/.claude/rules/git-conventions.md`.
207
- - Real newlines, no HTML entities, per
208
- `$HOME/.claude/multi-agent-refs/rules.md "External System Outputs"`.
31
+ `--aab=` / `--apk=` are Android inputs and are not valid here. If one is supplied,
32
+ do not silently switch platform - say the Android inputs belong to
33
+ `/multi-agent:store-ready`, and stop.
209
34
 
210
- ## Step 8 - offer the next action, do not take it
35
+ ## What you get
211
36
 
212
- Print, and stop:
37
+ Unchanged from before the merge, on the iOS path:
213
38
 
214
- - the exact `xcrun altool --upload-app` command for when the gates are clear, so
215
- uploading stays an explicit human act
216
- - `/multi-agent:testflight-validation --resume` to re-run after fixes
217
- - `/multi-agent:fix-bug` when Gate 3 produced code-level findings
39
+ | Gate | What runs | Needs |
40
+ |---|---|---|
41
+ | **1. Static** | `ios_app_store_audit` (18 rules, real ITMS codes) | an `.xcarchive` |
42
+ | **2. Authoritative** | `ios_testflight_validate` → `altool --validate-app` | an `.ipa` + credentials |
43
+ | **3. Policy** | `app-store-review` skill vs repo source | repo checkout |
218
44
 
219
- Never upload, never bump the build number, never commit.
45
+ A skipped gate is never folded into the pass count, and the run never uploads. The
46
+ report lands in `~/StoreChecks/ios-<repo>-<branch>-<timestamp>/report.md`.
@@ -35,12 +35,36 @@ For Android: use `mcp__dev-toolkit__android_*` tools
35
35
  /multi-agent test "accessibility" -> accessibility audit (visual + MCP audit tool)
36
36
  /multi-agent test "dynamic type" -> large text test
37
37
  /multi-agent test "screenshot tr" -> locale screenshots
38
- /multi-agent test "store-ready" -> full audit: visual + accessibility + archive/APK compliance
39
- /multi-agent test "biometric" -> Face ID / Touch ID flow test
40
- /multi-agent test "performance" -> launch time + scroll performance
38
+ /multi-agent test "store-ready" -> hands off to /multi-agent:store-ready (package validation)
41
39
  /sim-test -> standalone (same thing)
42
40
  ```
43
41
 
42
+ Not offered, deliberately - `"biometric"` and `"performance"` were listed here with
43
+ no implementation section, so reaching one fell through to the general sweep and got
44
+ reported as the scenario asked for. Neither can be implemented symmetrically today:
45
+ biometric has an iOS tool (`ios_biometric`) and no Android counterpart, and launch
46
+ timing has `android_launch_time` and no iOS counterpart. Platform is auto-detected,
47
+ so either one would work on one platform and silently do nothing on the other. They
48
+ come back when the missing side exists, not before - do not re-add the rows to make
49
+ the list look complete.
50
+
51
+ Four scenarios also have a fixed-scenario command that pins the tag, so it does not
52
+ have to be typed or quoted. They arrive here with the scenario already resolved -
53
+ treat them as identical to the quoted form:
54
+
55
+ ```
56
+ /multi-agent:test-dark-mode -> scenario "dark mode"
57
+ /multi-agent:test-accessibility -> scenario "accessibility"
58
+ /multi-agent:test-dynamic-type -> scenario "dynamic type"
59
+ /multi-agent:test-screenshots [locale] -> scenario "screenshot <locale>", locale defaults to tr
60
+ ```
61
+
62
+ `store-ready` is not one of them, and does not belong to this file at all: it
63
+ validates a built package on either platform and lives at
64
+ `/multi-agent:store-ready`. The `"store-ready"` tag is kept as a hand-off so an
65
+ existing invocation still lands somewhere correct. `/multi-agent:testflight-validation`
66
+ is the iOS-pinned alias of that same command.
67
+
44
68
  ## Flow
45
69
 
46
70
  ### Step 1 - Device & App Discovery
@@ -121,43 +145,28 @@ Call: ios_set_locale(language: "<lang>", bundle_id: "...")
121
145
 
122
146
  **"store-ready":**
123
147
 
124
- Runs visual + accessibility pass AND dispatches the platform-matching compliance skill so the build is pre-validated against Apple / Google store requirements before submission.
148
+ Not implemented here. `store-ready` validates a built **package**, which is a
149
+ different job from driving a running app, and it is owned by one command on both
150
+ platforms:
125
151
 
126
152
  ```
127
- Platform detection:
128
- cwd contains .xcodeproj OR Package.swift -> iOS
129
- cwd contains build.gradle OR build.gradle.kts -> Android
130
- Otherwise -> error: "store-ready needs an iOS or Android project"
131
-
132
- Step A - run the standard visual + accessibility sweep (light + dark + large text).
133
-
134
- Step B - dispatch compliance skill:
135
- iOS:
136
- Load $HOME/.claude/skills/apple-archive-compliance/SKILL.md
137
- Prereq: @mmerterden/dev-toolkit-mcp ≥ v2.9.0 (provides ios_app_store_audit MCP tool)
138
- - installable via npm; standalone ArchiveGuard binary deprecated in v8.4.0
139
- Artifact: pick newest .xcarchive under ~/Library/Developer/Xcode/Archives/**
140
- OR accept explicit path argument after "store-ready"
141
- Invoke: mcp__dev-toolkit__ios_app_store_audit({ archive_path: <path>, rules: "all" })
142
- OR (Node fallback): node -e "import('@mmerterden/dev-toolkit-mcp/tools/ios-app-store-audit/index.js').then(m => m.runAudit({ archivePath: '<path>', rules: 'all' }).then(r => console.log(JSON.stringify(r))))"
143
- Humanize via --lang en (promptLanguage is locked to "en"; pass --lang=tr explicitly to opt into Turkish)
144
- Android:
145
- Load $HOME/.claude/skills/google-play-compliance/SKILL.md
146
- Prereq: bundletool, aapt2, apksigner in PATH (Android SDK build-tools)
147
- Artifact: newest .aab under **/build/outputs/bundle/**/*.aab
148
- OR explicit path argument
149
- Invoke: bundletool validate + dump manifest, aapt2 dump badging,
150
- apksigner verify, ABI/native scan
151
- Merge findings with UI-hunter findings -> single report, severity-grouped.
152
-
153
- Step C - humanize via `--lang en` by default (promptLanguage is locked to "en"; pass --lang=tr explicitly to opt into Turkish), group by severity
154
- (error = blocker, warning = risk, info = hygiene), link each finding to the
155
- Apple ITMS / Play policy reference supplied by the skill catalog.
156
-
157
- Step D - offer `/multi-agent:channels` follow-up so findings can land in
158
- Jira / Confluence / Wiki / PR body via the normal Phase 7 machinery.
153
+ $HOME/.claude/commands/multi-agent/store-ready/SKILL.md
159
154
  ```
160
155
 
156
+ Read that file and follow it. Pass through any artifact path given after
157
+ `store-ready` as its `--archive=` / `--ipa=` / `--aab=` / `--apk=` input.
158
+
159
+ Its Step A is the visual + accessibility sweep in this file - it calls back here
160
+ for the running-app half, then runs three gates per platform that nothing in this
161
+ file can do: the static package audit, the store's own validator, and a policy
162
+ review against repo source. It merges both halves into one severity-grouped report
163
+ and offers the `/multi-agent:channels` follow-up.
164
+
165
+ The archive-compliance audit used to be duplicated here, invoking
166
+ `ios_app_store_audit` with exactly the arguments the store-ready command's Gate 1
167
+ uses. Two copies of one call is how the iOS path grew a second door with no Gate 2,
168
+ no Gate 3 and no Android parity, so the copy is gone rather than kept in sync.
169
+
161
170
  **No argument (full test):**
162
171
 
163
172
  - Run light mode -> all screens
@@ -6,14 +6,15 @@
6
6
 
7
7
  ---
8
8
 
9
- ## 1. Command Inventory (44 commands)
9
+ ## 1. Command Inventory (49 commands)
10
10
 
11
11
  ```
12
12
  analysis, analysis-resolve, autopilot, build-optimize, channels, create-jira, design-check, dev,
13
13
  dev-autopilot, dev-local, dev-local-autopilot, diff-explain, forget, garbage-collect,
14
14
  help, ios-coding-standard, issue, jira, kill, language, local,
15
15
  local-autopilot, log, manual-test, prune-logs, purge, refactor, resume, review, review-issue, review-jira,
16
- routines, save, scan, search, setup, ship, stack, status, sync, test, testflight-validation, uninstall, update
16
+ routines, save, scan, search, setup, ship, stack, status, store-ready, sync, test, test-accessibility,
17
+ test-dark-mode, test-dynamic-type, test-screenshots, testflight-validation, uninstall, update
17
18
  ```
18
19
 
19
20
  Categories:
@@ -56,7 +56,7 @@ Knowledge files grow over time. Maintenance rules:
56
56
  - 90-day-old entries are considered "stale" - not used without verification
57
57
  - Stale check: orchestrator checks file mtime during Phase 1 knowledge injection
58
58
  - Stale entries are added to the prompt with a "STALE - verify before relying" tag
59
- - `clear-logs` does not touch knowledge - only deletes logs and state
59
+ - `prune-logs` does not touch knowledge - only deletes logs and state
60
60
  - `purge` does not touch knowledge either - separate command: `/multi-agent clear-knowledge {project}`
61
61
 
62
62
  ---
@@ -73,7 +73,7 @@ Used for: input parsing, branch naming, commit messages.
73
73
  |---|---|---|---|
74
74
  | Project picked (Step 2) | `global.recentProjects` | `[{path, label, count, lastUsed}]` | 20 |
75
75
  | Multi-repo group picked or saved (Step 2) | `global.recentGroups` | `[{label?, repos[], count, lastUsed}]` | 10 |
76
- | Branch picked (Step 3) | `global.recentBranches[{projectKey}]` | `[{name, lastUsed}]` (TTL `settings.branchTtlDays`, default 15d) | implicit (TTL prunes) |
76
+ | Branch picked (Step 3) | `global.recentBranches[{projectKey}]` | `[{branch, lastUsed, count?}]` (TTL `settings.branchTtlDays`, default 15d) | 10 (TTL also prunes) |
77
77
  | Service ping (any external API call) | `global.serviceStatus[{service}]` | `{ok, checkedAt, reason?}` (TTL `settings.serviceStatusCacheSeconds`, default 300s) | n/a |
78
78
  | Git identity routed (Step 6a) | `projects[{name}].lastIdentity` | int (index into `global.identities`) | n/a |
79
79
 
@@ -261,7 +261,10 @@ Scan `$HOME` (maxdepth 2) for project markers (`.xcodeproj`, `Package.swift`, `b
261
261
  header: "Base branch"
262
262
  options: origin/develop (Recommended, reused from last run) | origin/main | release/8.4.0 | Other
263
263
  ```
264
- 7. User picks → store as `baseBranch`. Save to `prefs.projects[{project}].branches` (dedup, max 10).
264
+ 7. User picks → store as `baseBranch`, and append `{branch, lastUsed, count?}` to
265
+ `prefs.global.recentBranches[{projectKey}]` (dedup by `branch`, cap 10) - what the TTL
266
+ filter below reads. Key is `branch`, not `name`. Never the legacy
267
+ `projects[{project}].branches`.
265
268
 
266
269
  **MUST: this step is not skippable (BLOCKING).** The only legitimate skip is rule 4
267
270
  above - `baseBranch` already supplied in the input. Everything else asks. A run once
@@ -277,7 +280,7 @@ which still writes the fields - it does not leave them unset.
277
280
 
278
281
  **TTL filter for recent branches**:
279
282
 
280
- - `prefs.global.recentBranches[{projectKey}][]` carries `{name, lastUsed}`. Filter to those whose `lastUsed` is within `settings.branchTtlDays` (default 15).
283
+ - `prefs.global.recentBranches[{projectKey}][]` carries `{branch, lastUsed, count?}`. Filter to those whose `lastUsed` is within `settings.branchTtlDays` (default 15).
281
284
  - Stale entries (>TTL) are pruned in-place during the read - keeps the picker uncluttered without a separate cleanup pass.
282
285
  - The filtered "Recent" list precedes the fresh `git branch -r` list; cap at 5 visible recent entries.
283
286
 
@@ -403,7 +406,7 @@ git -C $PROJECT_ROOT config user.email "{identity.email}"
403
406
 
404
407
  `worktreePath` = `$PROJECT_ROOT`, `localMode` = `true`.
405
408
 
406
- **If normal mode** (worktree - default): 2. Worktree path: Jira → `.worktrees/{jiraId}/`, GitHub → `.worktrees/GH{issueNo}/`, free-text → `.worktrees/task-{shortId}/` 3. **Heal stale admin state first** (see "Worktree stale-lock heal" below) and **apply the residue guard** (see "Worktree residue guard" below), then `git -C $PROJECT_ROOT worktree add {path} -b {branch} origin/{baseBranch}` (if exists: enter, pull) 4. Set identity: `git -C {worktree-path} config user.name/email` 5. Create log dir + `agent-log.md` + `agent-state.json`:
409
+ **If normal mode** (worktree - default): 2. Worktree path: Jira → `.worktrees/{jiraId}/`, GitHub → `.worktrees/GH{issueNo}/`, free-text → `.worktrees/task-{shortId}/` 3. **Heal stale admin state first** (see "Worktree stale-lock heal" below) and **apply the residue guard** (see "Worktree residue guard" below), then `git -C $PROJECT_ROOT worktree add {path} -b {branch} origin/{baseBranch}` (if exists: enter, pull) 4. Set identity: `git -C {worktree-path} config user.name/email` 5. Create log dir + `agent-log.md` + `agent-state.json` at `$HOME/.claude/logs/multi-agent/{project}/{task-id}/`, never inside the worktree:
407
410
 
408
411
  **Worktree stale-lock heal (required before every `worktree add`):** a run killed mid-`worktree add` (OOM, SIGTERM, disk full) leaves a locked or broken admin entry under `.git/worktrees/{id}/`, so the retry fails with `fatal: '<path>' already exists`. Always run the heal first - it is a no-op on a clean repo:
409
412
 
@@ -184,7 +184,7 @@
184
184
  },
185
185
  "appstore_connect_key_id": {
186
186
  "type": ["string", "null"],
187
- "description": "App Store Connect API key ID. Tier 1 of the App Store Connect access chain, used by /multi-agent:testflight-validation Gate 2. An identifier rather than a secret; mapped anyway so every credential is read through the same layer. Creating an API key needs an Admin or App Manager role, which is why Tier 2 exists."
187
+ "description": "App Store Connect API key ID. Tier 1 of the App Store Connect access chain, used by /multi-agent:store-ready Gate 2 (and its iOS alias /multi-agent:testflight-validation). An identifier rather than a secret; mapped anyway so every credential is read through the same layer. Creating an API key needs an Admin or App Manager role, which is why Tier 2 exists."
188
188
  },
189
189
  "appstore_connect_issuer_id": {
190
190
  "type": ["string", "null"],
@@ -36,6 +36,6 @@
36
36
  "warn_tokens": 5600
37
37
  }
38
38
  },
39
- "total_max_tokens": 52700,
40
- "note": "Token estimate = ceil(chars / 4). Per-phase budget rule: warn = current+10% (rounded to nearest 50), max = current+25%. Gives ~6 edit cycles of headroom before warn trips - intentionally quiet under normal maintenance, loud when a phase grows unusually. Only the active phase is loaded (lazy). Recalibrated at v10.0.0 after the validator/consistency/simplifier/lesson gate contracts landed in phases 1-4. Recalibrated again at v10.9.0 after the verify-by-test (Phase 4 Step 3.7), update-check (Phase 0 Step 0.6), immutable-test (Phase 3 GREEN) and redTests re-entry contracts landed - Step 3.7 prose was compressed to a pointer into refs/features/verify-by-test.md before the recalibration. Total bumped 50000 -> 51000 at v12.5.0 after the worktree residue/traversal-prune contract (Phase 0 + Phase 5 heal) and the Reflexion causal-diagnosis contract (Phase 4 lesson memory) landed; the prose was compressed first (161 tokens reclaimed) and every per-phase max still passes - only the aggregate needed room. Recalibrated again at v13.6.0 after the install-relative path correction: an instruction that names `pipeline/scripts/x` resolves only from a repo checkout, and a run happens in the user's worktree, so 157 references across these docs moved to `$HOME/.claude/...` at +5 bytes each - 196 tokens of pure correctness cost. Same discipline as before: prose was compressed FIRST (149 tokens reclaimed, by pointing Phase 1's Figma tier table at the Phase 0 probe that already resolved it and Phase 4's Codex constraints at the always-loaded AGENTS.md block), and only then were the budgets moved. Five warn lines had been permanently amber, which makes the amber tier useless as a signal, so every warn was reset to the documented current+10% and the four maxes that the new warn would have collided with were reset to current+25%. Aggregate 51000 -> 51500. Total bumped 51500 -> 52200 at v14.0.0 after Phase 4 Review entered the four --dev mode phase sets and the criteria-resolution contract (Step 1.78) landed. Same discipline as every prior bump: prose was compressed FIRST, 820 tokens reclaimed, before the number moved. Two of those compressions are structural rather than cosmetic - the hardcoded SwiftUI interaction list in Step 1.5 and the SwiftUI convention paragraph in Step 2.8 were transcriptions of rules that now live in a scoped registry, so keeping them here would have re-created the drift this release exists to remove, and the third moved the Step 1.78 full contract into refs/features/skill-conformance.md leaving a pointer. What remains is contract text that cannot be inferred: the manifest's four consumer-visible parts, the conformance checklist the reviewers must return, and the fail-closed semantics. Every per-phase max still passes (phase-4 12405/14750); only the aggregate needed room. Total bumped 52200 -> 52700 at v14.1.0 after two more contracts landed: stack skill routing (Phase 3 pre-flight step 9) and worktree finalize (Phase 6 step 9). Compression came first, as always, and twice: 224 tokens out of Phase 3 by pointing its criteria-ledger and routing steps at their feature files instead of restating them, and 190 out of Phase 6 by moving the finalize contract into refs/features/worktree-finalize.md and leaving the invocation plus the exit-3 semantics. Both new contracts follow the pattern the earlier ones set: the phase doc carries the call and the decision, the feature file carries the reasoning, and the feature files are outside this budget because it loops only the eight phase-N-* keys. Every per-phase max still passes (phase-3 7677/8950, phase-6 5223/6150 and both under warn); only the aggregate needed room."
39
+ "total_max_tokens": 52750,
40
+ "note": "Token estimate = ceil(chars / 4). Per-phase budget rule: warn = current+10% (rounded to nearest 50), max = current+25%. Gives ~6 edit cycles of headroom before warn trips - intentionally quiet under normal maintenance, loud when a phase grows unusually. Only the active phase is loaded (lazy). Recalibrated at v10.0.0 after the validator/consistency/simplifier/lesson gate contracts landed in phases 1-4. Recalibrated again at v10.9.0 after the verify-by-test (Phase 4 Step 3.7), update-check (Phase 0 Step 0.6), immutable-test (Phase 3 GREEN) and redTests re-entry contracts landed - Step 3.7 prose was compressed to a pointer into refs/features/verify-by-test.md before the recalibration. Total bumped 50000 -> 51000 at v12.5.0 after the worktree residue/traversal-prune contract (Phase 0 + Phase 5 heal) and the Reflexion causal-diagnosis contract (Phase 4 lesson memory) landed; the prose was compressed first (161 tokens reclaimed) and every per-phase max still passes - only the aggregate needed room. Recalibrated again at v13.6.0 after the install-relative path correction: an instruction that names `pipeline/scripts/x` resolves only from a repo checkout, and a run happens in the user's worktree, so 157 references across these docs moved to `$HOME/.claude/...` at +5 bytes each - 196 tokens of pure correctness cost. Same discipline as before: prose was compressed FIRST (149 tokens reclaimed, by pointing Phase 1's Figma tier table at the Phase 0 probe that already resolved it and Phase 4's Codex constraints at the always-loaded AGENTS.md block), and only then were the budgets moved. Five warn lines had been permanently amber, which makes the amber tier useless as a signal, so every warn was reset to the documented current+10% and the four maxes that the new warn would have collided with were reset to current+25%. Aggregate 51000 -> 51500. Total bumped 51500 -> 52200 at v14.0.0 after Phase 4 Review entered the four --dev mode phase sets and the criteria-resolution contract (Step 1.78) landed. Same discipline as every prior bump: prose was compressed FIRST, 820 tokens reclaimed, before the number moved. Two of those compressions are structural rather than cosmetic - the hardcoded SwiftUI interaction list in Step 1.5 and the SwiftUI convention paragraph in Step 2.8 were transcriptions of rules that now live in a scoped registry, so keeping them here would have re-created the drift this release exists to remove, and the third moved the Step 1.78 full contract into refs/features/skill-conformance.md leaving a pointer. What remains is contract text that cannot be inferred: the manifest's four consumer-visible parts, the conformance checklist the reviewers must return, and the fail-closed semantics. Every per-phase max still passes (phase-4 12405/14750); only the aggregate needed room. Total bumped 52200 -> 52700 at v14.1.0 after two more contracts landed: stack skill routing (Phase 3 pre-flight step 9) and worktree finalize (Phase 6 step 9). Compression came first, as always, and twice: 224 tokens out of Phase 3 by pointing its criteria-ledger and routing steps at their feature files instead of restating them, and 190 out of Phase 6 by moving the finalize contract into refs/features/worktree-finalize.md and leaving the invocation plus the exit-3 semantics. Both new contracts follow the pattern the earlier ones set: the phase doc carries the call and the decision, the feature file carries the reasoning, and the feature files are outside this budget because it loops only the eight phase-N-* keys. Every per-phase max still passes (phase-3 7677/8950, phase-6 5223/6150 and both under warn); only the aggregate needed room. Total bumped 52700 -> 52750 for the Phase 0 Step 3 branch-persistence correction: the step wrote the legacy `projects[].branches` while the TTL filter two sections below read `global.recentBranches`, and both spots named a `{name, lastUsed}` shape the schema rejects (`branch` required, `additionalProperties: false`), so the recent-branch picker option could never populate and a literal implementation would have failed prefs validation. Naming the right target, the right key and the legacy field to avoid costs 41 tokens over the one line it replaces. Compression came first and was applied three times to the replacement text itself, from 120 tokens down to 66, by moving the rationale out of the phase doc entirely: the reasoning now lives where it is enforced, in the migrate-prefs carry-forward comment and the smoke-pref-migration f7 block, leaving the phase doc with only the instruction. 50 was the smallest step that clears it; phase-0-init sits at 10893/12400, far under its own max, so this is purely an aggregate ceiling."
41
41
  }
@@ -33,6 +33,7 @@ import {
33
33
  statSync,
34
34
  } from "node:fs";
35
35
  import { join } from "node:path";
36
+ import { spawnSync } from "node:child_process";
36
37
 
37
38
  const args = process.argv.slice(2);
38
39
  // Accepts both `--key value` and `--key=value`. The `=` form used to fall through to
@@ -251,6 +252,26 @@ for (const [plugin, want] of Object.entries(desired)) {
251
252
  pj.skills = newSkills;
252
253
  pj.version = newV;
253
254
  writeFileSync(pjPath, `${JSON.stringify(pj, null, 2)}\n`);
255
+
256
+ // The marketplace README carries a per-plugin table of version + skill count,
257
+ // and that table is what a human reads to know what is published. Bumping
258
+ // plugin.json without it drifted five of the five rows, silently, because the
259
+ // drift is only visible if someone runs the checker. `tools/bump.py` owns the
260
+ // table's format, so call it rather than reproducing the row layout here - a
261
+ // second formatter would be the one that rots.
262
+ const bumpTool = join(PLUGINS_REPO, "tools", "bump.py");
263
+ if (existsSync(bumpTool)) {
264
+ const r = spawnSync("python3", [bumpTool, plugin, "--set", newV], {
265
+ cwd: PLUGINS_REPO,
266
+ encoding: "utf8",
267
+ });
268
+ if (r.status !== 0) {
269
+ console.warn(
270
+ ` WARN ${plugin}: plugin.json is ${newV} but the README table was not synced ` +
271
+ `(bump.py exit ${r.status ?? "n/a"}). Run \`tools/bump.py --check\` in ${PLUGINS_REPO}.`,
272
+ );
273
+ }
274
+ }
254
275
  }
255
276
  report.push({
256
277
  plugin,
@@ -270,6 +270,36 @@ function migrate(prefs) {
270
270
  out.global.recentBranches = {};
271
271
  changes.push("added empty recentBranches");
272
272
  }
273
+ // Carry the legacy per-project branch list into the canonical LRU. Phase 0 Step 3
274
+ // read global.recentBranches while its own step 7 wrote projects[].branches, so
275
+ // every branch a user ever picked landed in the field nothing reads and the
276
+ // "reused from last run" picker option stayed empty. Seeding here makes that
277
+ // history usable instead of stranding it. Idempotent: an existing entry for the
278
+ // same branch is never duplicated or overwritten.
279
+ //
280
+ // `lastUsed` is stamped with the migration time, not the real pick time, which the
281
+ // legacy field never recorded. Stamping the true unknown (epoch) would put every
282
+ // seeded entry outside settings.branchTtlDays and the TTL filter would prune it on
283
+ // the first read, making this carry-forward a no-op. `count: 0` marks the entry as
284
+ // seeded rather than observed, and the normal TTL retires anything the user does not
285
+ // actually pick again within the window.
286
+ const seededAt = new Date().toISOString();
287
+ for (const [projectKey, project] of Object.entries(out.projects ?? {})) {
288
+ if (!Array.isArray(project?.branches) || project.branches.length === 0) continue;
289
+ const lru = (out.global.recentBranches[projectKey] ??= []);
290
+ const known = new Set(lru.map((e) => e?.branch));
291
+ let seeded = 0;
292
+ for (const branch of project.branches) {
293
+ if (typeof branch !== "string" || !branch || known.has(branch)) continue;
294
+ if (lru.length >= 10) break;
295
+ lru.push({ branch, lastUsed: seededAt, count: 0 });
296
+ known.add(branch);
297
+ seeded += 1;
298
+ }
299
+ if (seeded > 0) {
300
+ changes.push(`seeded ${seeded} recentBranches entr${seeded === 1 ? "y" : "ies"} for ${projectKey} from legacy projects[].branches`);
301
+ }
302
+ }
273
303
  if (!Array.isArray(out.global.recentGroups)) {
274
304
  out.global.recentGroups = [];
275
305
  changes.push("added empty recentGroups");