@mmerterden/multi-agent-pipeline 16.18.0 → 16.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (238) hide show
  1. package/CHANGELOG.md +68 -0
  2. package/README.md +5 -5
  3. package/README.tr.md +2 -2
  4. package/docs/FIGMA_PIPELINE.md +1 -1
  5. package/docs/adr/0001-three-model-triage.md +4 -2
  6. package/docs/architecture.md +2 -2
  7. package/docs/ecosystem.md +13 -11
  8. package/docs/features.md +2 -2
  9. package/index.js +1 -1
  10. package/install/_codex-agents.mjs +2 -2
  11. package/install/_common.mjs +25 -1
  12. package/install/_dev-only-files.mjs +3 -2
  13. package/install/_mcp-register.mjs +4 -3
  14. package/install/_plugin-skills.mjs +1 -3
  15. package/install/copilot.mjs +18 -9
  16. package/install/index.mjs +2 -4
  17. package/install/templates/copilot-instructions.md +7 -7
  18. package/package.json +4 -3
  19. package/pipeline/agents/android-architect.md +1 -0
  20. package/pipeline/agents/backend-architect.md +1 -0
  21. package/pipeline/agents/code-reviewer.md +1 -0
  22. package/pipeline/agents/dev-critic.md +2 -1
  23. package/pipeline/agents/explorer.md +1 -0
  24. package/pipeline/agents/ios-architect.md +1 -0
  25. package/pipeline/agents/security-auditor.md +1 -0
  26. package/pipeline/agents/task-clarifier.md +1 -0
  27. package/pipeline/claude-md-template.md +2 -2
  28. package/pipeline/commands/multi-agent/SKILL.md +5 -5
  29. package/pipeline/commands/multi-agent/complaint-analysis/SKILL.md +1 -1
  30. package/pipeline/commands/multi-agent/design-check/SKILL.md +1 -1
  31. package/pipeline/commands/multi-agent/help/SKILL.md +2 -2
  32. package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +1 -1
  33. package/pipeline/commands/multi-agent/manual-test/SKILL.md +5 -1
  34. package/pipeline/commands/multi-agent/resume/SKILL.md +1 -0
  35. package/pipeline/commands/multi-agent/review/SKILL.md +27 -14
  36. package/pipeline/commands/multi-agent/review-analysis/SKILL.md +1 -1
  37. package/pipeline/commands/multi-agent/store-ready/SKILL.md +24 -2
  38. package/pipeline/commands/multi-agent/sync/SKILL.md +2 -2
  39. package/pipeline/commands/sim-test.md +64 -20
  40. package/pipeline/lib/credential-inventory.sh +15 -2
  41. package/pipeline/lib/credential-store-resolver.sh +14 -4
  42. package/pipeline/lib/credential-store.sh +8 -2
  43. package/pipeline/lib/extract-conventions.sh +1 -14
  44. package/pipeline/lib/fetch-confluence.sh +12 -4
  45. package/pipeline/lib/fetch-crashlytics.sh +11 -8
  46. package/pipeline/lib/fetch-document.sh +1 -1
  47. package/pipeline/lib/fetch-figma-annotations.sh +7 -5
  48. package/pipeline/lib/fetch-fortify.sh +5 -3
  49. package/pipeline/lib/fetch-graylog.sh +5 -3
  50. package/pipeline/lib/figma-mcp-refresh.sh +1 -1
  51. package/pipeline/lib/figma-screenshot.sh +27 -24
  52. package/pipeline/lib/figma-token.sh +8 -4
  53. package/pipeline/lib/issue-fetcher.sh +0 -1
  54. package/pipeline/lib/jira-publish.sh +7 -5
  55. package/pipeline/lib/md2confluence-v3.py +13 -7
  56. package/pipeline/lib/multi-repo-pipeline.sh +18 -8
  57. package/pipeline/lib/plan-todos.sh +11 -0
  58. package/pipeline/lib/post-pr-review.sh +9 -2
  59. package/pipeline/lib/repo-cache.sh +18 -10
  60. package/pipeline/lib/review-watch.sh +60 -14
  61. package/pipeline/lib/shadow-git.sh +8 -4
  62. package/pipeline/lib/vercel-deploy.sh +2 -2
  63. package/pipeline/multi-agent-refs/_dev-context.md +5 -2
  64. package/pipeline/multi-agent-refs/analysis/locked.md +4 -4
  65. package/pipeline/multi-agent-refs/analysis/render.md +1 -1
  66. package/pipeline/multi-agent-refs/channels/pr.md +22 -4
  67. package/pipeline/multi-agent-refs/cross-cli-contract.md +1 -1
  68. package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +15 -2
  69. package/pipeline/multi-agent-refs/features/review-delta.md +89 -0
  70. package/pipeline/multi-agent-refs/features/scope-check.md +41 -0
  71. package/pipeline/multi-agent-refs/features/verify-by-test.md +6 -5
  72. package/pipeline/multi-agent-refs/features/worktree-finalize.md +1 -1
  73. package/pipeline/multi-agent-refs/outside-the-pipeline.md +6 -6
  74. package/pipeline/multi-agent-refs/payload-contracts.md +1 -1
  75. package/pipeline/multi-agent-refs/phases/log-format.md +1 -1
  76. package/pipeline/multi-agent-refs/phases/modes.md +1 -1
  77. package/pipeline/multi-agent-refs/phases/phase-0-init.md +4 -2
  78. package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +3 -3
  79. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +5 -5
  80. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +17 -2
  81. package/pipeline/multi-agent-refs/phases/phase-4-review.md +67 -24
  82. package/pipeline/multi-agent-refs/phases/phase-5-test.md +10 -0
  83. package/pipeline/multi-agent-refs/phases/phase-6-commit.md +2 -0
  84. package/pipeline/multi-agent-refs/phases/phase-7-report.md +4 -2
  85. package/pipeline/multi-agent-refs/rules.md +2 -2
  86. package/pipeline/multi-agent-refs/tracker-contract.md +1 -1
  87. package/pipeline/rules/figma-pipeline.md +18 -72
  88. package/pipeline/rules/outside-the-pipeline.md +4 -3
  89. package/pipeline/schemas/agent-state.schema.json +130 -1
  90. package/pipeline/schemas/dev-critic-output.schema.json +5 -0
  91. package/pipeline/schemas/prefs.schema.json +47 -0
  92. package/pipeline/schemas/reviewer-output.schema.json +7 -2
  93. package/pipeline/schemas/scope-check.schema.json +55 -0
  94. package/pipeline/schemas/token-budget.json +3 -3
  95. package/pipeline/schemas/triage-output.schema.json +12 -2
  96. package/pipeline/scripts/README.md +3 -2
  97. package/pipeline/scripts/_fingerprint.mjs +173 -0
  98. package/pipeline/scripts/_stack-routing.mjs +1 -1
  99. package/pipeline/scripts/agent-guard.py +102 -21
  100. package/pipeline/scripts/anonymize-findings.mjs +7 -6
  101. package/pipeline/scripts/build-skills-index.mjs +14 -3
  102. package/pipeline/scripts/cost-budget-check.mjs +5 -3
  103. package/pipeline/scripts/cost-lib.sh +0 -15
  104. package/pipeline/scripts/diff-explain.mjs +22 -12
  105. package/pipeline/scripts/evidence-gate.mjs +73 -5
  106. package/pipeline/scripts/finding-fingerprint.mjs +101 -0
  107. package/pipeline/scripts/gc-refs.sh +6 -2
  108. package/pipeline/scripts/gc-tmp.sh +1 -1
  109. package/pipeline/scripts/gc-worktrees.sh +1 -1
  110. package/pipeline/scripts/gen-mode-dispatch.mjs +3 -3
  111. package/pipeline/scripts/github-ssh-setup.sh +7 -2
  112. package/pipeline/scripts/graph-build.mjs +2 -2
  113. package/pipeline/scripts/jira-wiki-escape.mjs +2 -1
  114. package/pipeline/scripts/keychain.py +12 -11
  115. package/pipeline/scripts/learning-curve.mjs +1 -1
  116. package/pipeline/scripts/migrate-prefs.mjs +1 -1
  117. package/pipeline/scripts/output-quality-check.sh +3 -1
  118. package/pipeline/scripts/phase-tracker.sh +1 -1
  119. package/pipeline/scripts/phase0-exit-gate.mjs +2 -1
  120. package/pipeline/scripts/plan-coverage-gate.mjs +2 -1
  121. package/pipeline/scripts/pre-commit-check.sh +23 -13
  122. package/pipeline/scripts/prune-logs.sh +1 -1
  123. package/pipeline/scripts/render-agent-log-cost.sh +3 -1
  124. package/pipeline/scripts/render-cost-summary.sh +4 -2
  125. package/pipeline/scripts/render-work-summary.sh +5 -3
  126. package/pipeline/scripts/repo-map.mjs +3 -2
  127. package/pipeline/scripts/review-delta.mjs +217 -0
  128. package/pipeline/scripts/run-metrics.mjs +20 -0
  129. package/pipeline/scripts/scan-skills.sh +6 -2
  130. package/pipeline/scripts/scope-check-gate.mjs +90 -0
  131. package/pipeline/scripts/search-logs.sh +8 -6
  132. package/pipeline/scripts/sign-skills.sh +3 -1
  133. package/pipeline/scripts/smoke-cross-cli-behavior.sh +12 -5
  134. package/pipeline/scripts/triage-memory.mjs +25 -4
  135. package/pipeline/scripts/uninstall.mjs +20 -12
  136. package/pipeline/scripts/update-check.sh +2 -2
  137. package/pipeline/scripts/update-issue-progress.sh +6 -5
  138. package/pipeline/scripts/validate-analysis-doc.mjs +6 -6
  139. package/pipeline/scripts/validate-reviewer.mjs +6 -0
  140. package/pipeline/scripts/validate-triage.mjs +20 -0
  141. package/pipeline/scripts/verify-skills.sh +3 -1
  142. package/pipeline/scripts/worktree-finalize.sh +22 -10
  143. package/pipeline/skills/.skill-manifest.json +81 -57
  144. package/pipeline/skills/.skills-index.json +19 -19
  145. package/pipeline/skills/shared/README.md +8 -8
  146. package/pipeline/skills/shared/core/multi-agent/SKILL.md +7 -7
  147. package/pipeline/skills/shared/core/multi-agent-design-check/SKILL.md +1 -1
  148. package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +2 -2
  149. package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +1 -1
  150. package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +2 -2
  151. package/pipeline/skills/shared/core/multi-agent-review/SKILL.md +23 -14
  152. package/pipeline/skills/shared/core/multi-agent-store-ready/SKILL.md +5 -0
  153. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +2 -2
  154. package/pipeline/skills/shared/external/NOTICE-dimillian-skills.md +56 -0
  155. package/pipeline/skills/shared/external/accessibility-compliance-accessibility-audit/SKILL.md +0 -4
  156. package/pipeline/skills/shared/external/api-patterns/SKILL.md +12 -24
  157. package/pipeline/skills/shared/external/app-store-changelog/references/release-notes-guidelines.md +34 -0
  158. package/pipeline/skills/shared/external/app-store-changelog/scripts/collect_release_changes.sh +33 -0
  159. package/pipeline/skills/shared/external/architecture/SKILL.md +7 -9
  160. package/pipeline/skills/shared/external/debugging-strategies/SKILL.md +0 -4
  161. package/pipeline/skills/shared/external/fastapi-pro/SKILL.md +0 -1
  162. package/pipeline/skills/shared/external/github-actions-templates/SKILL.md +0 -14
  163. package/pipeline/skills/shared/external/hig-components-content/SKILL.md +13 -13
  164. package/pipeline/skills/shared/external/hig-components-layout/SKILL.md +16 -16
  165. package/pipeline/skills/shared/external/hig-components-status/SKILL.md +6 -6
  166. package/pipeline/skills/shared/external/hig-components-system/SKILL.md +13 -13
  167. package/pipeline/skills/shared/external/hig-foundations/SKILL.md +23 -23
  168. package/pipeline/skills/shared/external/hig-inputs/SKILL.md +18 -18
  169. package/pipeline/skills/shared/external/hig-patterns/SKILL.md +30 -30
  170. package/pipeline/skills/shared/external/hig-platforms/SKILL.md +11 -11
  171. package/pipeline/skills/shared/external/hig-technologies/SKILL.md +33 -33
  172. package/pipeline/skills/shared/external/ios-coding-standard/references/STANDARD.md +52 -52
  173. package/pipeline/skills/shared/external/ios-coding-standard/references/lint-local.sh +1 -1
  174. package/pipeline/skills/shared/external/ios-coding-standard/references/rules.yml +11 -11
  175. package/pipeline/skills/shared/external/ios-developer/SKILL.md +0 -1
  176. package/pipeline/skills/shared/external/ios-module-structure/SKILL.md +7 -3
  177. package/pipeline/skills/shared/external/localization-reuse-map/SKILL.md +9 -15
  178. package/pipeline/skills/shared/external/macos-spm-app-packaging/SKILL.md +0 -5
  179. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/Package.swift +17 -0
  180. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/Sources/MyApp/Resources/.keep +0 -0
  181. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/Sources/MyApp/main.swift +11 -0
  182. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/version.env +2 -0
  183. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/build_icon.sh +49 -0
  184. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/compile_and_run.sh +63 -0
  185. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/launch.sh +28 -0
  186. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/make_appcast.sh +82 -0
  187. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/package_app.sh +206 -0
  188. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/setup_dev_signing.sh +52 -0
  189. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/sign-and-notarize.sh +52 -0
  190. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/version.env +2 -0
  191. package/pipeline/skills/shared/external/macos-spm-app-packaging/references/packaging.md +17 -0
  192. package/pipeline/skills/shared/external/macos-spm-app-packaging/references/release.md +32 -0
  193. package/pipeline/skills/shared/external/macos-spm-app-packaging/references/scaffold.md +79 -0
  194. package/pipeline/skills/shared/external/monorepo-architect/SKILL.md +0 -1
  195. package/pipeline/skills/shared/external/nodejs-backend-patterns/SKILL.md +0 -4
  196. package/pipeline/skills/shared/external/swift-concurrency-expert/references/approachable-concurrency.md +63 -0
  197. package/pipeline/skills/shared/external/swift-concurrency-expert/references/swift-6-2-concurrency.md +272 -0
  198. package/pipeline/skills/shared/external/swift-concurrency-expert/references/swiftui-concurrency-tour-wwdc.md +33 -0
  199. package/pipeline/skills/shared/external/swiftui-performance-audit/references/code-smells.md +150 -0
  200. package/pipeline/skills/shared/external/swiftui-performance-audit/references/demystify-swiftui-performance-wwdc23.md +46 -0
  201. package/pipeline/skills/shared/external/swiftui-performance-audit/references/optimizing-swiftui-performance-instruments.md +29 -0
  202. package/pipeline/skills/shared/external/swiftui-performance-audit/references/profiling-intake.md +44 -0
  203. package/pipeline/skills/shared/external/swiftui-performance-audit/references/report-template.md +47 -0
  204. package/pipeline/skills/shared/external/swiftui-performance-audit/references/understanding-hangs-in-your-app.md +33 -0
  205. package/pipeline/skills/shared/external/swiftui-performance-audit/references/understanding-improving-swiftui-performance.md +52 -0
  206. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/app-wiring.md +201 -0
  207. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/async-state.md +96 -0
  208. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/components-index.md +46 -0
  209. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/controls.md +57 -0
  210. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/deeplinks.md +66 -0
  211. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/focus.md +90 -0
  212. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/form.md +97 -0
  213. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/grids.md +71 -0
  214. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/haptics.md +71 -0
  215. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/input-toolbar.md +51 -0
  216. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/lightweight-clients.md +93 -0
  217. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/list.md +86 -0
  218. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/loading-placeholders.md +38 -0
  219. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/macos-settings.md +71 -0
  220. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/matched-transitions.md +59 -0
  221. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/media.md +73 -0
  222. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/menu-bar.md +101 -0
  223. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/navigationstack.md +159 -0
  224. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/overlay.md +45 -0
  225. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/performance.md +62 -0
  226. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/previews.md +48 -0
  227. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/scroll-reveal.md +133 -0
  228. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/scrollview.md +87 -0
  229. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/searchable.md +71 -0
  230. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/sheets.md +155 -0
  231. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/split-views.md +72 -0
  232. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/tabview.md +114 -0
  233. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/theming.md +71 -0
  234. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/title-menus.md +93 -0
  235. package/pipeline/skills/shared/external/swiftui-ui-patterns/references/top-bar.md +49 -0
  236. package/pipeline/skills/shared/external/swiftui-view-refactor/references/mv-patterns.md +161 -0
  237. package/pipeline/skills/skills-index.md +8 -8
  238. package/pipeline/skills/shared/external/help-skills/SKILL.md +0 -166
@@ -1,5 +1,5 @@
1
1
  ---
2
- description: "Task orchestrator - full pipeline via Jira ID + branch or GitHub Issue URL: analysis, plan, TDD development, parallel review + Fable triage (CLI-aware: 2-model on Claude Code, 3-model on Copilot CLI), commit, log. Use when given a Jira ID, a GitHub issue or a free-text task and the whole pipeline should run."
2
+ description: "Task orchestrator - full pipeline via Jira ID + branch or GitHub Issue URL: analysis, plan, TDD development, parallel review + Fable triage (3 reviewers per host: Fable + Opus + Sonnet on Claude Code, GPT-5.4 + Opus + Sonnet on Copilot CLI), commit, log. Use when given a Jira ID, a GitHub issue or a free-text task and the whole pipeline should run."
3
3
  description-tr: "Görev orkestratörü - Jira ID + branch veya GitHub Issue URL ile tam pipeline: analiz, plan, TDD geliştirme, paralel review + Fable triyajı (CLI'ya göre: Claude Code'da 3, Copilot CLI'da 3 model), commit, log"
4
4
  allowed-tools: Agent, Bash, Read, Write, Edit, Glob, Grep, TaskCreate, TaskUpdate, TaskList, TaskGet, AskUserQuestion, WebFetch, WebSearch, NotebookEdit, Skill
5
5
  ---
@@ -48,7 +48,7 @@ Classification schema lives in `$HOME/.claude/multi-agent-refs/_input-parser.md`
48
48
  | 7 | `issue` | full picker | account → repo (multi) → issue → maturity → dev-context |
49
49
  | 8 | Free-text | `freetext` | account → repo (single) → dev-context (maturity skip) |
50
50
 
51
- **Rule**: Whatever the type, **account is always asked** (autopilot picks a default). After issue fetch, **maturity check is mandatory** - blockers halt the pipeline. Picker `label` + `header` render in English (`promptLanguage` is locked to `en`); the `question` and each option's `description` render in `outputLanguage`, per the canonical matrix in `multi-agent-refs/rules.md`.
51
+ **Rule**: Whatever the type, **account is always asked** (autopilot picks a default). After issue fetch, **maturity check is mandatory** - blockers halt the pipeline. Picker `header` renders in English (the 12-char chip); the `question`, each option's `label` and each option's `description` render in `outputLanguage`, per the canonical matrix in `multi-agent-refs/rules.md`.
52
52
 
53
53
  Lib scripts (`~/.claude/lib/`):
54
54
  - `account-resolver.sh` - keychain account inventory
@@ -102,8 +102,8 @@ Lib scripts (`~/.claude/lib/`):
102
102
  This command uses lazy loading for token efficiency. Read the relevant sub-file based on the routed action:
103
103
 
104
104
  **File layout:**
105
- - `commands/multi-agent/*.md` → **invocable actions** (each gets its own `/multi-agent:<name>` slash command)
106
- - `$HOME/.claude/multi-agent-refs/**` → **internal references**, read by the main command, never invoked directly (surface as `/multi-agent:refs:...` in autocomplete - the prefix signals "not an action")
105
+ - `commands/multi-agent/{cmd}/SKILL.md` → **invocable actions** (each gets its own `/multi-agent:<name>` slash command)
106
+ - `$HOME/.claude/multi-agent-refs/**` → **internal references**, read by the main command, never invoked directly. They live outside `commands/` precisely so they never appear in slash-command autocomplete.
107
107
 
108
108
  | Route | File to Read |
109
109
  |-------|-------------|
@@ -262,7 +262,7 @@ When called with `review`:
262
262
  1. Detect current branch and project from cwd (or ask)
263
263
  2. Get diff: `git diff HEAD` (unstaged + staged)
264
264
  3. If no diff, get diff against base branch: `git diff origin/{baseBranch}...HEAD`
265
- 4. Launch Phase 4 review (parallel + Fable triage - 2-model on Claude Code, 3-model on Copilot CLI) on the diff
265
+ 4. Launch Phase 4 review (parallel + Fable triage - 3 reviewers on every host: Fable + Opus + Sonnet on Claude Code, GPT-5.4 + Opus + Sonnet on Copilot CLI) on the diff
266
266
  5. No worktree, no state file - lightweight one-shot review
267
267
  6. Print findings to terminal
268
268
 
@@ -127,7 +127,7 @@ Set `phase: "drafting"`.
127
127
  ### Phase 4 - Draft, humanize, buffer
128
128
 
129
129
  1. Render the report per `$HOME/.claude/multi-agent-refs/complaint-analysis-template.md` (8 fixed sections; single-language body in `outputLanguage`; verdict tokens English per Locked 9) to `/tmp/complaint-analysis-<run-slug>-<UTC-iso8601>/report.md`. Store `outputs.draftDir`.
130
- 2. **Humanizer pass (MANDATORY: actually invoke the `ai-common-toolkit:humanizer` skill; the punctuation grep alone does NOT satisfy this)** with `language: <tr|en>`, `tone: technical-explanatory`, `stripFancyPunctuation: true`. Diacritics preserved (Locked 8).
130
+ 2. **Humanizer pass (required: actually invoke the `ai-common-toolkit:humanizer` skill; the punctuation grep alone does NOT satisfy this)** with `language: <tr|en>`, `tone: technical-explanatory`, `stripFancyPunctuation: true`. Diacritics preserved (Locked 8).
131
131
  3. Punctuation gate: `node $HOME/.claude/scripts/validate-complaint-doc.mjs <draft>` reports no banned-punctuation error. It checks the policy in Node, so the same result holds on macOS, Linux and Windows; `grep -P` is absent from BSD grep and would never run there.
132
132
  4. Show the draft path + size to the user. Set `phase: "awaiting_output_decision"`.
133
133
 
@@ -133,7 +133,7 @@ for l in sys.stdin:
133
133
  Print the resolved set grouped by screen **with its relaunch cost**: `<n> targets · <relaunchCount> relaunches`. Read the cost from `plan`, not from the target count - one relaunch serves every in-app target on that screen, so a 50-target module is typically a dozen relaunches, not fifty. **Whole-module is the intended default**; only ask for confirmation when `relaunchCount` exceeds `config coverage.confirmAbove` (default 25), and phrase it as a cost estimate, not as an invitation to shrink the audit. Never propose a smaller scope as the easy path - a scoped run is for resuming or for a focused re-check, not for avoiding work.
134
134
  7. **`--resume`** - read the most recent `~/DesignChecks/{repo}__{module}/*/run-state.json`; the scope becomes that run's targets minus its `covered` and minus its `skipped` entries. Skips carry a reason and are honoured, so a resume covers the genuinely unaudited remainder. No previous run → tell the user and fall back to whole-module scope after confirmation.
135
135
  8. **Report dir** - create `~/DesignChecks/{repo}__{module}/{UTC-timestamp}/` (and `assets/` inside it) now, and persist it as `state.designCheck.reportDir`. Phase 3 writes captures, comparison images, and `run-state.json` into it, so it must exist before driving starts - not at export time. This is report output, not a worktree; $HOME is intended here.
136
- 9. **Worktree** - build the Debug app in an isolated worktree so the user's tree is untouched. Follow `phase-0-init.md` Step 8 convention exactly: `{projectRoot}/{worktreeBasePath}/{taskId}` (default `.worktrees/DC-<shortId>`), **never under $HOME** (`feedback_worktree_path_convention`). Stale-lock heal (`git worktree prune`) + residue guard (`.worktrees/` in `.git/info/exclude`) first. Local mode is not offered - the audit always uses a worktree checkout of the current branch's HEAD (no fetch/push).
136
+ 9. **Worktree** - build the Debug app in an isolated worktree so the user's tree is untouched. Follow the `phase-0-init.md` Step 6 worktree location convention exactly: `{projectRoot}/.worktrees/{taskId}` with `taskId` = `DC-<shortId>`, **never under $HOME** (the `.worktrees` segment is fixed, not a preference). Stale-lock heal (`git worktree prune`) + residue guard (`.worktrees/` in `.git/info/exclude`) first. Local mode is not offered - the audit always uses a worktree checkout of the current branch's HEAD (no fetch/push).
137
137
 
138
138
  Persist `agent-state.json` with `taskId`, `mode: "design-check"`, `platform`, `projectRoot`, `worktreePath`, `module`, `designCheck`.
139
139
 
@@ -53,7 +53,7 @@ How It Works (Phase 0 - Interactive Flow):
53
53
  Pipeline (after Phase 0) - shown as visual cards in terminal:
54
54
 
55
55
  Phase 0: Init -> The 8 steps above
56
- Phase 1: Analysis -> Stack detection + codebase scan (Fable)
56
+ Phase 1: Analysis -> Stack detection + codebase scan (Sonnet)
57
57
  Phase 2: Planning -> Task breakdown + architecture review + Plan Approval Gate
58
58
  (clarification max 2 rounds + approval loop - Full + interactive
59
59
  only; a Short run has no plan, autopilot may not ask)
@@ -331,7 +331,7 @@ Nasıl Çalışır (Phase 0 - İnteraktif Akış):
331
331
  Pipeline (Phase 0'dan sonra) - terminalde görsel kart olarak görünür:
332
332
 
333
333
  Phase 0: Init -> Yukarıdaki 8 adım
334
- Phase 1: Analysis -> Stack tespiti + codebase taraması (Fable)
334
+ Phase 1: Analysis -> Stack tespiti + codebase taraması (Sonnet)
335
335
  Phase 2: Planning -> Task kırılımı + mimari inceleme + Plan Onay Kapısı
336
336
  (clarification max 2 tur + onay döngüsü - sadece Tam +
337
337
  etkileşimli; Kısa'da plan yok, autopilot soru soramaz)
@@ -67,7 +67,7 @@ Depth is a separate axis, asked at Phase 0 Step 7.5 rather than encoded in the c
67
67
 
68
68
  ## Delegation
69
69
 
70
- Orchestrator routing: the routing table in `$HOME/.claude/commands/multi-agent/SKILL.md` resolves `local-autopilot` as the union of the `dev-local` + `autopilot` mode mixins. Contract details: `$HOME/.claude/multi-agent-refs/phases/phase-0-init.md` Step 8 (local branch) + `$HOME/.claude/multi-agent-refs/phases/phase-2-planning.md` Step 5 (autopilot gate skip + safety classifier).
70
+ Orchestrator routing: the routing table in `$HOME/.claude/commands/multi-agent/SKILL.md` resolves `local-autopilot` as the union of the `dev-local` + `autopilot` mode mixins. Contract details: `$HOME/.claude/multi-agent-refs/phases/phase-0-init.md` Step 6 (local branch) + `$HOME/.claude/multi-agent-refs/phases/phase-2-planning.md` Step 5 (autopilot gate skip + safety classifier).
71
71
  ## Required: outward-facing payload contracts
72
72
 
73
73
  Before writing anything outward-facing - PR body, Jira comment, Confluence page, closing report - load `$HOME/.claude/multi-agent-refs/payload-contracts.md`. It names the canonical section set for each payload, the markup dialect per surface (PR body is Markdown, Jira is wiki markup - mixing them is a defect), and the token/duration numbers the closing report must carry. Improvising a payload shape from memory is the most common failure of the short modes.
@@ -47,5 +47,9 @@ Lets you switch to the task branch for manual testing in Xcode before the PR is
47
47
  5. **Wait for the user's reply**
48
48
 
49
49
  6. **Branch on the answer**:
50
- - **OK** → `phase-tracker.sh update 5 completed` + `phase-tracker.sh meta 5 Result "local test passed (user)"`, recreate the worktree, continue to Phase 6
50
+ - **OK** → first write `$WORKTREE/.pipeline/manual-test.json` (one entry per acceptance criterion from the analysis doc test plan, the plan tasks, or the user's own words):
51
+ ```json
52
+ {"criteria":[{"spec":"<quote>","source":"analysis 15.2 | plan task 3 | user","observed":"<what was seen>","verdict":"pass|fail|not-tested","reason":"<required when not-tested>","screenshot":"<path or null>"}],"verdict":"passed|failed"}
53
+ ```
54
+ then run `node $HOME/.claude/scripts/evidence-gate.mjs --claim manual --status passed --evidence "$WORKTREE/.pipeline/manual-test.json"`. Exit 1 means the "ok" is not accepted: name the criterion that is missing evidence and wait for the next reply. Exit 0 → `phase-tracker.sh update 5 completed` + `phase-tracker.sh meta 5 Result "local test passed (user)"`, recreate the worktree, continue to Phase 6. Full contract: `$HOME/.claude/multi-agent-refs/phases/phase-5-test.md` step 5.
51
55
  - **Fix needed** → `phase-tracker.sh now 5 "applying fix: <summary>"`, recreate the worktree, apply the fix
@@ -23,6 +23,7 @@ Resume a paused or failed task from the last successful phase.
23
23
  - `currentPhase` - last completed phase
24
24
  - `status` - `paused` | `failed` | `in_progress`
25
25
  - `haltReason` - if set, show it so the user knows why the run stopped; clear it on successful re-entry
26
+ - `circuitBreaker` - if `tripped`, show `trigger` + `detail`, then set `tripped: false` and keep `counters`; if the same trigger fires again at the next checkpoint the breaker re-trips (no silent bypass)
26
27
  - `autopilot` - preserve the mode
27
28
 
28
29
  3. **Load context** - rebuild working context from durable artifacts, never from conversation memory:
@@ -117,12 +117,12 @@ jq --arg sha "$HEAD_SHA" '.review.headCommitSha = $sha' "$AGENT_STATE" > "$tmp_s
117
117
  **pr mode - bitbucket-server:**
118
118
 
119
119
  ```bash
120
- # Resolve credentials via prefs.keychainMapping (never hardcode key names).
120
+ # Resolve credentials via prefs.global.keychainMapping (never hardcode key names).
121
121
  . "$HOME/.claude/lib/credential-store-resolver.sh" \
122
122
  || . "$HOME/.copilot/lib/credential-store-resolver.sh"
123
123
  resolve_credential_store
124
- USER_KEY=$(jq -r '.keychainMapping.bitbucket_user' "$HOME/.claude/multi-agent-preferences.json")
125
- TOKEN_KEY=$(jq -r '.keychainMapping.bitbucket_token' "$HOME/.claude/multi-agent-preferences.json")
124
+ USER_KEY=$(jq -r '.global.keychainMapping.bitbucket_user' "$HOME/.claude/multi-agent-preferences.json")
125
+ TOKEN_KEY=$(jq -r '.global.keychainMapping.bitbucket_token' "$HOME/.claude/multi-agent-preferences.json")
126
126
  BB_USER=$("$CRED_STORE" get "$USER_KEY")
127
127
  BB_TOKEN=$("$CRED_STORE" get "$TOKEN_KEY")
128
128
 
@@ -175,15 +175,25 @@ Scope note: a guide governs only files under its own directory - a guide found
175
175
 
176
176
  ### 3. Launch parallel reviewers - host-CLI dependent
177
177
 
178
+ Every host runs three reviewers; only the second slot differs, because GPT-5.4 exists on Copilot and Codex but not on Claude Code, where Opus fills it.
179
+
178
180
  **Claude Code (3 in parallel):**
179
181
  - Agent 1: `claude-fable-5` → security + architecture
180
- - Agent 2: `claude-sonnet-5` → general quality
182
+ - Agent 2: `claude-opus-5` → edge cases, alternate perspective
183
+ - Agent 3: `claude-sonnet-5` → general quality
181
184
 
182
185
  **Copilot CLI (3 in parallel):**
183
186
  - Agent 1: `claude-opus-5` → security + architecture (Fable 5 is not offered on Copilot CLI)
184
187
  - Agent 2: `gpt-5.4` → edge cases, alternate perspective
185
188
  - Agent 3: `claude-sonnet-5` → general quality
186
189
 
190
+ **Codex CLI (3 in parallel, single vendor):**
191
+ - Agent 1: `gpt-5.6` @ `xhigh` → security + architecture
192
+ - Agent 2: `gpt-5.4` @ `high` → edge cases, alternate perspective
193
+ - Agent 3: `gpt-5.6` @ `medium` → general quality
194
+
195
+ With the `fable` rung disabled by prefs the Claude Code panel is two reviewers (Opus + Sonnet); see `$HOME/.claude/multi-agent-refs/features/model-fallback.md`.
196
+
187
197
  Each reviewer receives the diff, the module review guides from Step 2b (when any were found), plus the standard reviewer system prompt (see `$HOME/.claude/multi-agent-refs/phases/phase-4-review.md` for the prompt contract). Output: structured `findings[]` per reviewer.
188
198
 
189
199
  ### 4. Store-compliance cross-reference
@@ -203,16 +213,18 @@ Catalog-only - does NOT invoke binaries. For a full scan, use `/multi-agent:te
203
213
 
204
214
  ### 4b. Platform parity cross-check (advisory, read-only)
205
215
 
206
- When `state.siblings[]` holds a checked-out repo whose `stack` is the other
207
- mobile platform and the diff touches a screen, a service, a request model or a
208
- localization file, compare the change against that repo on four axes: endpoints
209
- called, parameters sent, business rules around the call, localization keys used.
216
+ When a counterpart repo resolves whose `stack` is the other mobile platform and
217
+ the diff touches a screen, a service, a request model or a localization file,
218
+ compare the change against that repo on four axes: endpoints called, parameters
219
+ sent, business rules around the call, localization keys used.
210
220
 
211
- The counterpart is resolved automatically, in both directions (ios ↔ android):
212
- a remembered `prefs.projects[<slug>].counterpartRoots[]` first, then the primary
213
- checkout's sibling directories, then `state.siblings[]`. One match is used and
214
- remembered, so the next review of the same project asks nothing. `--with
215
- <path|owner/repo>` overrides all of it for a one-off and is remembered too.
221
+ The counterpart is resolved automatically, in both directions (ios ↔ android),
222
+ from four sources in this order, first hit wins: `--with <path|owner/repo>` (a
223
+ one-off override, remembered like any other resolution), then a remembered
224
+ `prefs.projects[<slug>].counterpartRoots[]`, then the primary checkout's sibling
225
+ directories, then `state.siblings[]` from the Phase 0 dev-context picker. None of
226
+ them is a precondition for the others. One match is used and remembered, so the
227
+ next review of the same project asks nothing.
216
228
 
217
229
  Several candidates → interactive runs ask once; autopilot and non-interactive
218
230
  runs skip silently, because an unattended run must not block on a picker. No
@@ -237,12 +249,13 @@ Triage also marks each finding as `accepted` (real issue), `deferred` (real but
237
249
  | Model | Verdict | Blocking | Important | Suggestion |
238
250
  |----------|-----------|----------|-----------|------------|
239
251
  | Fable | approved | 0 | 1 | 3 |
252
+ | Opus | approved | 0 | 2 | 2 |
240
253
  | Sonnet | rejected | 1 | 2 | 5 |
241
254
 
242
255
  Consensus: ⚠ DISAGREEMENT - see Fable triage
243
256
  ```
244
257
 
245
- This summary ALWAYS prints, regardless of input mode. The chat is the live conversation; on the PR side, the durable artifacts are inline comments + the review state (Step 7).
258
+ One row per reviewer that ran: Fable + Opus + Sonnet on Claude Code, Opus + GPT-5.4 + Sonnet on Copilot CLI, the three GPT rows on Codex CLI. This summary ALWAYS prints, regardless of input mode. The chat is the live conversation; on the PR side, the durable artifacts are inline comments + the review state (Step 7).
246
259
 
247
260
  ### 7. Post to PR - only when input.kind === "pr"
248
261
 
@@ -23,7 +23,7 @@ Read `$HOME/.claude/multi-agent-refs/analysis/review.md` and execute it:
23
23
 
24
24
  ## Why it cites Locked decisions
25
25
 
26
- The analysis flow already declares 35 Locked decisions and two deterministic validators. A reviewer that says "I would have written this differently" gives the author nothing to act on; one that says "Locked 34: the Confluence page is in the evidence record but not in Section 21" gives them a fix and a reason. Findings that map to no rule are still allowed, but they are marked as judgement, not dressed up as a violation.
26
+ The analysis flow already declares 36 Locked decisions and two deterministic validators. A reviewer that says "I would have written this differently" gives the author nothing to act on; one that says "Locked 34: the Confluence page is in the evidence record but not in Section 21" gives them a fix and a reason. Findings that map to no rule are still allowed, but they are marked as judgement, not dressed up as a violation.
27
27
 
28
28
  ## What it never does
29
29
 
@@ -183,7 +183,7 @@ is a guess the user must be able to correct.
183
183
 
184
184
  ### Mode A - build from the branch
185
185
 
186
- 1. Worktree at `{projectRoot}/{worktreeBasePath}/{taskId}` on the chosen branch.
186
+ 1. Worktree at `{projectRoot}/.worktrees/{taskId}` on the chosen branch (phase-0-init.md Step 6 convention; the `.worktrees` segment is fixed, not a preference).
187
187
  **Never under `$HOME`**, never a direct checkout of the main working tree.
188
188
  2. Resolve the scheme / module and workspace / project from prefs; ask if ambiguous.
189
189
  3. Build:
@@ -218,10 +218,22 @@ is a guess the user must be able to correct.
218
218
  Use the fallback only when the tool is genuinely absent, and say in the report
219
219
  which path ran - a rule set that silently differed between two invocations is
220
220
  worse than a missing gate.
221
+
222
+ Then, whichever path ran, count `<archive>/dSYMs/*.dSYM`. Zero is a blocking
223
+ finding: `[SYMBOLS] archive carries no dSYM; crash reports will not symbolicate`,
224
+ with the hint `DEBUG_INFORMATION_FORMAT = dwarf-with-dsym` for the Release
225
+ configuration. This check runs from the package alone and does not need the
226
+ MCP tool.
221
227
  - **Android**: `android_apk_audit` on the artifact, plus the
222
228
  `google-play-compliance` skill's 21 rules - `bundletool validate` and manifest
223
229
  dump, `aapt2 dump badging`, `apksigner verify`, ABI / native scan.
224
230
 
231
+ Then, when the module has `minifyEnabled true` and the bundle build produced no
232
+ `mapping.txt`, raise the same class of blocking finding:
233
+ `[SYMBOLS] minified bundle carries no mapping.txt; crash reports will not
234
+ deobfuscate`. This check reads the module config and the build output alone
235
+ and does not need the MCP tool.
236
+
225
237
  `error` findings are blocking; `warning` is advisory. Group by severity and keep
226
238
  each finding's ITMS / Play policy code - Gate 2 may return the same code on iOS,
227
239
  and seeing it in both places tells the user it is real rather than a heuristic.
@@ -259,6 +271,7 @@ that catches what a human reviewer rejects, so it reads source, not the binary.
259
271
  | Privacy policy | reachable in-app and in the metadata |
260
272
  | IAP | anything unlocking features goes through StoreKit, with no external purchase path |
261
273
  | Sign in with Apple | present when a third-party social login is offered |
274
+ | Crash symbolication | a dSYM upload step exists: an Xcode run-script calling `upload-symbols`, or a Crashlytics / Sentry / Datadog upload in CI |
262
275
 
263
276
  **Android** - `ai-android-toolkit:play-store-review`:
264
277
 
@@ -272,6 +285,7 @@ that catches what a human reviewer rejects, so it reads source, not the binary.
272
285
  | Account deletion | if the app creates accounts, an in-app deletion path exists, plus the web deletion URL Play requires |
273
286
  | Content rating | the questionnaire answers match the app's actual content |
274
287
  | Signing | Play App Signing configured, upload key distinct from the app signing key |
288
+ | Crash symbolication | a `mapping.txt` upload exists: the Firebase Crashlytics Gradle plugin, or the Play App Bundle deobfuscation file |
275
289
 
276
290
  For each: `pass` / `fail` / `not-applicable` with the evidence path that justifies
277
291
  it. `not-applicable` needs a reason - an unexamined area is not a pass.
@@ -303,6 +317,12 @@ Advisory
303
317
 
304
318
  Not run
305
319
  Gate 2: no local Play validator - authoritative check is server-side only
320
+
321
+ Before rollout (human inputs, not verified)
322
+ Phased rollout: <1% -> 10% -> 50% -> 100% | full>
323
+ Halt thresholds: crash-free < 99.5% (iOS, Android) or ANR > 0.47% (Android) => pause the rollout
324
+ Rollback / forward-fix owner: <name>
325
+ Forward-fix plan: <one line>
306
326
  ```
307
327
 
308
328
  Rules for the report:
@@ -311,6 +331,8 @@ Rules for the report:
311
331
  skip. An Android run therefore reads `2 of 3 gates cleared, 1 skipped` at best.
312
332
  - Every blocking finding carries a file path or a store code. A finding the user
313
333
  cannot act on is noise.
334
+ - The `Before rollout` block is filled by a human, never inferred. When it is
335
+ left unfilled the verdict line gains the suffix `, rollout plan missing`.
314
336
  - Humanize via `--lang en` by default (`promptLanguage` is locked to `"en"`); pass
315
337
  `--lang=tr` explicitly to opt into Turkish.
316
338
  - No AI or assistant attribution anywhere, per
@@ -329,7 +351,7 @@ Print, and stop:
329
351
  declares - never run it
330
352
  - `/multi-agent:store-ready --resume` to re-run after fixes
331
353
  - `/multi-agent:channels` to land the findings in Jira / Confluence / Wiki / PR
332
- - `/multi-agent:fix-bug` when Gate 3 produced code-level findings
354
+ - the enabled stack plugin's `fix-bug` skill (`ai-ios-toolkit:fix-bug` or the stack equivalent) when Gate 3 produced code-level findings
333
355
 
334
356
  Never upload, never commit a Play edit, never bump the build number, never commit.
335
357
 
@@ -60,7 +60,7 @@ Run every step automatically:
60
60
  Step 1: PLATFORM Detect macOS / Linux / Windows (Git Bash / WSL); export PLATFORM env
61
61
  Step 1.5: DETECT Compare timestamps, find stale targets
62
62
  Step 2: COPILOT Claude Code -> Copilot CLI (instructions + 55 sub-command skills)
63
- Step 2b: CODEX Claude Code -> Codex CLI (1 router skill + 51 specs as refs + 8 agent TOML)
63
+ Step 2b: CODEX Claude Code -> Codex CLI (1 router skill + 55 specs as refs + 8 agent TOML)
64
64
  Step 3: REPO Claude Code -> pipeline repo (genericized, personal data scrub, bash -n on all sh)
65
65
  Step 3c: PLUGINS pipeline shared/external -> multi-agent-plugins marketplace (rebuild knowledge/,
66
66
  bump changed plugins' patch version, commit + push the plugins repo)
@@ -117,7 +117,7 @@ If nothing is stale → report "All targets up to date" and stop.
117
117
  - `~/.claude/CLAUDE.md`, `~/.claude/rules/`, `~/.claude/knowledge/`
118
118
  - `~/.claude/scripts/` - EXCEPT `pre-commit-check.sh`, `agent-guard.sh`, `agent-guard.py`, and `build-stack-plugins.mjs` (generic, synced)
119
119
  - `~/.claude/settings.json`
120
- - **Any `~/.claude/commands/multi-agent/*.md` whose frontmatter has `local-only: true`** - these are user/repo-specific alias wrappers that delegate to a private marketplace plugin (e.g. corporate `ai-ios-toolkit` skills exposed as `multi-agent:<name>`). Syncing them would leak the private plugin/skill names into the public pipeline. Filter before copy: skip every source file containing `local-only: true`, and after copy assert none reached `pipeline/commands/`.
120
+ - **Any `~/.claude/commands/multi-agent/*/SKILL.md` whose frontmatter has `local-only: true`** - these are user/repo-specific alias wrappers that delegate to a private marketplace plugin's skills exposed as `multi-agent:<name>`. Syncing them would leak the private plugin/skill names into the public pipeline. Filter before copy: skip every source file containing `local-only: true`, and after copy assert none reached `pipeline/commands/`.
121
121
  ```bash
122
122
  # backstop: no local-only wrapper may exist in the synced target
123
123
  grep -rl "^local-only: true" ~/multi-agent-pipeline/pipeline/commands/ 2>/dev/null \
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  description: "Mobile UI Bug Hunter - iOS Simulator (simctl) + Android Emulator (adb) - auto-detects platform"
3
- allowed-tools: Agent, Bash, Read, Write, Edit, TaskCreate, TaskUpdate, TaskList, TaskGet, mcp__multi-agent-toolkit__ios_list_devices, mcp__multi-agent-toolkit__ios_boot_device, mcp__multi-agent-toolkit__ios_screenshot, mcp__multi-agent-toolkit__ios_tap, mcp__multi-agent-toolkit__ios_swipe, mcp__multi-agent-toolkit__ios_type_text, mcp__multi-agent-toolkit__ios_launch_app, mcp__multi-agent-toolkit__ios_terminate_app, mcp__multi-agent-toolkit__ios_list_apps, mcp__multi-agent-toolkit__ios_go_home, mcp__multi-agent-toolkit__ios_set_appearance, mcp__multi-agent-toolkit__ios_set_content_size, mcp__multi-agent-toolkit__ios_set_locale, mcp__multi-agent-toolkit__ios_open_url, mcp__multi-agent-toolkit__ios_status_bar, mcp__multi-agent-toolkit__ios_push_notification, mcp__multi-agent-toolkit__ios_grant_permission, mcp__multi-agent-toolkit__ios_revoke_permission, mcp__multi-agent-toolkit__ios_reset_permissions, mcp__multi-agent-toolkit__ios_set_location, mcp__multi-agent-toolkit__ios_clear_location, mcp__multi-agent-toolkit__ios_set_increase_contrast, mcp__multi-agent-toolkit__ios_record_video, mcp__multi-agent-toolkit__ios_add_media, mcp__multi-agent-toolkit__ios_keychain_reset, mcp__multi-agent-toolkit__ios_get_app_container, mcp__multi-agent-toolkit__ios_erase_device, mcp__multi-agent-toolkit__ios_get_ui_tree, mcp__multi-agent-toolkit__android_list_devices, mcp__multi-agent-toolkit__android_screenshot, mcp__multi-agent-toolkit__android_tap, mcp__multi-agent-toolkit__android_swipe, mcp__multi-agent-toolkit__android_type_text, mcp__multi-agent-toolkit__android_key_event, mcp__multi-agent-toolkit__android_launch_app, mcp__multi-agent-toolkit__android_stop_app, mcp__multi-agent-toolkit__android_list_packages, mcp__multi-agent-toolkit__android_go_home, mcp__multi-agent-toolkit__android_go_back, mcp__multi-agent-toolkit__android_get_ui_tree, mcp__multi-agent-toolkit__android_set_dark_mode, mcp__multi-agent-toolkit__android_set_font_scale, mcp__multi-agent-toolkit__android_set_locale, mcp__multi-agent-toolkit__android_set_location, mcp__multi-agent-toolkit__android_grant_permission, mcp__multi-agent-toolkit__android_revoke_permission, mcp__multi-agent-toolkit__android_record_screen, mcp__multi-agent-toolkit__android_install_apk, mcp__multi-agent-toolkit__android_uninstall_app, mcp__multi-agent-toolkit__android_logcat, mcp__multi-agent-toolkit__android_get_screen_size, mcp__multi-agent-toolkit__android_open_url, mcp__multi-agent-toolkit__android_clear_app_data
3
+ allowed-tools: Agent, Bash, Read, Write, Edit, TaskCreate, TaskUpdate, TaskList, TaskGet, mcp__multi-agent-toolkit__ios_list_devices, mcp__multi-agent-toolkit__ios_boot_device, mcp__multi-agent-toolkit__ios_screenshot, mcp__multi-agent-toolkit__ios_tap, mcp__multi-agent-toolkit__ios_swipe, mcp__multi-agent-toolkit__ios_type_text, mcp__multi-agent-toolkit__ios_launch_app, mcp__multi-agent-toolkit__ios_terminate_app, mcp__multi-agent-toolkit__ios_list_apps, mcp__multi-agent-toolkit__ios_go_home, mcp__multi-agent-toolkit__ios_set_appearance, mcp__multi-agent-toolkit__ios_set_content_size, mcp__multi-agent-toolkit__ios_set_locale, mcp__multi-agent-toolkit__ios_open_url, mcp__multi-agent-toolkit__ios_status_bar, mcp__multi-agent-toolkit__ios_push_notification, mcp__multi-agent-toolkit__ios_grant_permission, mcp__multi-agent-toolkit__ios_revoke_permission, mcp__multi-agent-toolkit__ios_reset_permissions, mcp__multi-agent-toolkit__ios_set_location, mcp__multi-agent-toolkit__ios_clear_location, mcp__multi-agent-toolkit__ios_set_increase_contrast, mcp__multi-agent-toolkit__ios_record_video, mcp__multi-agent-toolkit__ios_add_media, mcp__multi-agent-toolkit__ios_keychain_reset, mcp__multi-agent-toolkit__ios_get_app_container, mcp__multi-agent-toolkit__ios_erase_device, mcp__multi-agent-toolkit__ios_get_ui_tree, mcp__multi-agent-toolkit__ios_accessibility_audit, mcp__multi-agent-toolkit__ios_accessibility_audit_deep, mcp__multi-agent-toolkit__ios_list_crashes, mcp__multi-agent-toolkit__agent_run_steps, mcp__multi-agent-toolkit__android_list_devices, mcp__multi-agent-toolkit__android_screenshot, mcp__multi-agent-toolkit__android_tap, mcp__multi-agent-toolkit__android_swipe, mcp__multi-agent-toolkit__android_type_text, mcp__multi-agent-toolkit__android_key_event, mcp__multi-agent-toolkit__android_launch_app, mcp__multi-agent-toolkit__android_stop_app, mcp__multi-agent-toolkit__android_list_packages, mcp__multi-agent-toolkit__android_go_home, mcp__multi-agent-toolkit__android_go_back, mcp__multi-agent-toolkit__android_get_ui_tree, mcp__multi-agent-toolkit__android_set_dark_mode, mcp__multi-agent-toolkit__android_set_font_scale, mcp__multi-agent-toolkit__android_set_locale, mcp__multi-agent-toolkit__android_set_location, mcp__multi-agent-toolkit__android_grant_permission, mcp__multi-agent-toolkit__android_revoke_permission, mcp__multi-agent-toolkit__android_record_screen, mcp__multi-agent-toolkit__android_install_apk, mcp__multi-agent-toolkit__android_uninstall_app, mcp__multi-agent-toolkit__android_logcat, mcp__multi-agent-toolkit__android_get_screen_size, mcp__multi-agent-toolkit__android_open_url, mcp__multi-agent-toolkit__android_clear_app_data, mcp__multi-agent-toolkit__android_accessibility_audit, mcp__multi-agent-toolkit__android_list_crashes
4
4
  ---
5
5
 
6
6
  # Mobile UI Bug Hunter
@@ -25,7 +25,7 @@ Auto-detects platform: iOS (simctl) or Android (adb). No external apps needed.
25
25
  For iOS: use `mcp__multi-agent-toolkit__ios_*` tools
26
26
  For Android: use `mcp__multi-agent-toolkit__android_*` tools
27
27
 
28
- **Android extras**: `get_ui_tree` returns XML with bounds/text/resource-id (uiautomator dump), `logcat` for crash detection, `go_back` button.
28
+ **Android extras**: `get_ui_tree` returns XML with bounds/text/resource-id (uiautomator dump), `logcat` for live output, `list_crashes` for the crash buffer, `go_back` button.
29
29
 
30
30
  ## Activation
31
31
 
@@ -90,23 +90,34 @@ Call: ios_status_bar (time: "09:41", battery_level: 100)
90
90
 
91
91
  ### Step 3 - Systematic Screen Exploration
92
92
 
93
- For each screen in the app:
93
+ For each screen in the app, read before you look, and batch what you can:
94
94
 
95
95
  ```
96
- Call: ios_screenshot
97
- -> Claude analyzes the image for bugs (see Bug Detection below)
96
+ Call: ios_get_ui_tree (text: labels, frames, traits; ~0.1 s, no image)
97
+ -> pick tap targets from element frames, never by guessing pixels from a picture
98
+
99
+ Call: agent_run_steps (one round trip for a scripted sequence)
100
+ steps: [ {tool: "ios_tap", args: {x, y}}, {wait_ms: 400},
101
+ {tool: "ios_screenshot", args: {path: "<dir>/<screen>.png"}},
102
+ {tool: "ios_get_ui_tree", args: {}} ]
98
103
 
99
- Call: ios_tap (on each tab bar item, button, navigation link)
100
- -> After each tap: ios_screenshot -> analyze
101
- -> Navigate back: ios_swipe (left edge swipe) or tap back button coordinates
104
+ Call: ios_screenshot (inline, only when a visual judgment is needed)
105
+ -> Claude analyzes the image for bugs (see Bug Detection below)
102
106
  ```
103
107
 
108
+ Every inline `ios_screenshot` / `android_screenshot` is an 800 px JPEG (about 60 KB) by
109
+ default; pass `format: "png"` or a larger `max_width` only when a pixel-level look is the
110
+ point. Bulk captures (dark mode, dynamic type, locale sweeps) always go through `path`,
111
+ so a run of 100 screens does not push 100 images through the model: look at the ones
112
+ whose tree or diff changed. A screen whose tree is unchanged after a tap is the
113
+ "button did not respond" finding without a second image.
114
+
104
115
  **Navigation strategy:**
105
116
 
106
- 1. Screenshot initial screen -> identify tab bar (usually bottom ~680-700y area)
107
- 2. Tap each tab position -> screenshot each
108
- 3. On each screen: tap interactive elements -> screenshot results
109
- 4. Scroll: `ios_swipe(200, 600, 200, 200)` to scroll down -> screenshot
117
+ 1. Tree of the initial screen -> tab bar items are the elements with a tab-bar trait or the bottom row of buttons; use their frames
118
+ 2. Tap each tab (batched via `agent_run_steps`) -> tree + file capture each
119
+ 3. On each screen: tap interactive elements from the tree -> tree after; inline screenshot only where the tree cannot tell (layout, contrast, clipping)
120
+ 4. Scroll: `ios_swipe(200, 600, 200, 200)` to scroll down -> tree, capture when new content appeared
110
121
  5. Type in text fields: `ios_type_text("test input")`
111
122
 
112
123
  ### Step 4 - Variant Testing (based on input argument)
@@ -131,9 +142,24 @@ Call: ios_set_content_size("medium") // reset
131
142
 
132
143
  **"accessibility":**
133
144
 
134
- - Check each screenshot for: small tap targets, missing labels, poor contrast
135
- - Verify minimum 44x44pt touch areas
136
- - Check text readability at default + large sizes
145
+ Two halves per screen, and the report keeps them apart: what the tool measured and what the screenshot suggests.
146
+
147
+ ```
148
+ On each screen, after ios_screenshot:
149
+ Call: ios_accessibility_audit (Android: android_accessibility_audit)
150
+ -> findings[]: missing labels / contentDescription, controls a screen reader cannot
151
+ name, tap targets under 44pt (iOS) / 48dp (Android), missing identifiers,
152
+ reading order that does not follow the layout. Each finding becomes a BUG task
153
+ with the element identifier from the tree, not a guess from the pixels.
154
+ -> measurable:false (with a reason) means the tree could not be read (on iOS,
155
+ Simulator.app itself must be running, a booted device is not enough). Record
156
+ the screen as "not audited: <reason>" in the report. Never read it as clean.
157
+ -> Pass `scope: "<prefix>"` to narrow to one screen's identifiers when the tree is large.
158
+ ```
159
+
160
+ The screenshot pass still runs on every screen for what the tree cannot see: poor contrast, text readability at default + large sizes, and visual crowding. Report those as screenshot findings, separately from the audit findings.
161
+
162
+ Optional deep pass (iOS only, minutes rather than seconds): when the project has an XCUITest target whose test calls `performAccessibilityAudit()`, run `ios_accessibility_audit_deep` (`scheme`, `project` or `workspace`, optional `test_identifier`) once at the end. It reaches contrast, Dynamic Type, clipped text, traits and hit regions that a tree dump cannot. The result says whether the named test ran; a test that never called the audit reports as "did not run", never as clean. Skip it, and say so in the report, when no such test exists.
137
163
 
138
164
  **"screenshot <lang>":**
139
165
 
@@ -172,6 +198,17 @@ no Gate 3 and no Android parity, so the copy is gone rather than kept in sync.
172
198
  - Run light mode -> all screens
173
199
  - Run dark mode -> all screens
174
200
  - Run large text -> all screens
201
+ - Crash sweep, after the last screen and before the report:
202
+
203
+ ```
204
+ Call: ios_list_crashes (app: <process name>, since_min: <minutes since Step 1 launch>, limit: 20)
205
+ Android: android_list_crashes (lines: 200)
206
+ -> Every report / stack newer than the launch becomes a BUG task with severity Critical,
207
+ the crashed process and the top frame in the description.
208
+ -> Empty result -> "Crashes: none during this run" in the report. Only the bounded window
209
+ counts; older reports on the host are not this run's.
210
+ ```
211
+
175
212
  - Compile complete report
176
213
 
177
214
  ### Step 5 - Bug Detection (on EVERY screenshot)
@@ -216,16 +253,18 @@ Then output full report:
216
253
  - **Steps**: 1. Open app -> 2. Tap {X} -> 3. Observe {issue}
217
254
  - **Expected**: {correct behavior}
218
255
  - **Actual**: {what's wrong}
256
+ - **Spec**: "<quoted acceptance criterion from the analysis doc Section 15 / 20, or: no spec, model expectation>" (<source>)
257
+ - **Evidence**: before=<png path> after=<png path>
219
258
 
220
259
  ### BUG-2: ...
221
260
 
222
261
  ## Screens Visited ({N})
223
262
 
224
- | # | Screen | Light | Dark | Large Text | Bugs |
225
- | --- | -------- | ----- | ----- | ---------- | ---- |
226
- | 1 | Home | ok | BUG-1 | ok | 1 |
227
- | 2 | Login | ok | ok | BUG-2 | 1 |
228
- | 3 | Settings | ok | ok | ok | 0 |
263
+ | # | Screen | Light | Dark | Large Text | Bugs | Evidence |
264
+ | --- | -------- | ----- | ----- | ---------- | ---- | -------- |
265
+ | 1 | Home | ok | BUG-1 | ok | 1 | 3 |
266
+ | 2 | Login | ok | ok | BUG-2 | 1 | 4 |
267
+ | 3 | Settings | ok | ok | ok | 0 | 2 |
229
268
 
230
269
  ## Summary
231
270
 
@@ -233,8 +272,13 @@ Then output full report:
233
272
  - Clean: {N}
234
273
  - With bugs: {N}
235
274
  - Critical: {N} | Major: {N} | Minor: {N}
275
+ - Accessibility audit (accessibility scenario): {N} audited, {N} not audited (reason per screen), deep pass: {ran / skipped: reason}
276
+ - Crashes (full scenario): {N} during this run
277
+ - Production readiness: FAILED | NEEDS WORK | READY
236
278
  ```
237
279
 
280
+ `Evidence` in a bug block is the captures already written to files in Step 4: a tap-driven finding needs both `before` and `after`, a static finding (layout, contrast, dark mode, large text) needs `after` only. The `Evidence` column in Screens Visited is the count of screenshot files written for that screen. Production readiness defaults to FAILED; it is NEEDS WORK when only Minor bugs remain, and READY only when there are zero Critical / Major bugs and every planned screen was visited.
281
+
238
282
  Save to: `$HOME/.claude/logs/sim-test/{bundle_id}/{timestamp}.md`
239
283
 
240
284
  ### Step 7 - Fix Offer
@@ -46,7 +46,8 @@
46
46
  # not to enable, not a problem to report.
47
47
  #
48
48
  # Exit codes: 0 = inventory produced (or the queried key is present), 1 = queried key
49
- # absent, 3 = usage error.
49
+ # absent, 2 = no credential helper on this host (nothing could be probed), 3 = usage
50
+ # error.
50
51
 
51
52
  set -uo pipefail
52
53
 
@@ -290,11 +291,23 @@ if [ -n "$QUERY" ]; then
290
291
  echo "$QUERY: present - can $(capability_of "$QUERY")"
291
292
  fi
292
293
  exit 0 ;;
293
- 2) echo "$QUERY: no credential helper on this host" >&2; exit 1 ;;
294
+ 2) echo "$QUERY: no credential helper on this host - run the pipeline installer (credential-store.sh missing)" >&2; exit 2 ;;
294
295
  *) echo "$QUERY: NOT AVAILABLE - onboard it via /multi-agent:setup before relying on it" >&2; exit 1 ;;
295
296
  esac
296
297
  fi
297
298
 
299
+ # Without a helper nothing can be probed: every mapped key would read as
300
+ # "mapped-but-missing" and the inventory would send the user to re-onboard
301
+ # tokens that are sitting in the store.
302
+ if [ -z "$STORE" ]; then
303
+ if [ "$MODE" = "json" ]; then
304
+ echo '{"status":"no-backend","reason":"credential-store.sh not found on this host","credentials":[]}'
305
+ else
306
+ echo "no credential helper on this host - run the pipeline installer (credential-store.sh missing)" >&2
307
+ fi
308
+ exit 2
309
+ fi
310
+
298
311
  ROWS=""
299
312
  while IFS=$'\t' read -r key mapped; do
300
313
  [ -z "$key" ] && continue
@@ -81,12 +81,22 @@ MSG
81
81
  return 1
82
82
  }
83
83
 
84
- # When sourced (BASH_SOURCE != $0), auto-resolve into the caller's environment.
85
- # When run directly (./credential-store-resolver.sh), print the resolved path
86
- # or a non-zero exit code with the message above.
87
- if [ "${BASH_SOURCE[0]:-$0}" = "${0}" ]; then
84
+ # When sourced, auto-resolve into the caller's environment. When run directly
85
+ # (./credential-store-resolver.sh), print the resolved path or a non-zero exit
86
+ # code with the message above. zsh sources this too (the pipeline's Bash tool
87
+ # is zsh), where BASH_SOURCE is unset and $0 is the sourced file's own name, so
88
+ # the bash test alone always looked like a direct run there.
89
+ _csr_sourced=0
90
+ if [ -n "${ZSH_VERSION:-}" ]; then
91
+ case "${ZSH_EVAL_CONTEXT:-}" in *:file|*:file:*) _csr_sourced=1 ;; esac
92
+ elif [ -n "${BASH_SOURCE[0]:-}" ] && [ "${BASH_SOURCE[0]}" != "$0" ]; then
93
+ _csr_sourced=1
94
+ fi
95
+ if [ "$_csr_sourced" -eq 0 ]; then
96
+ unset _csr_sourced
88
97
  resolve_credential_store && echo "$CRED_STORE"
89
98
  else
99
+ unset _csr_sourced
90
100
  # `|| :` matters, and it is not cosmetic.
91
101
  #
92
102
  # A sourced file runs in the caller's shell, so a bare failing command at this top
@@ -164,6 +164,10 @@ do_get() {
164
164
  return 0
165
165
  fi
166
166
  audit_lookup "$logical" false
167
+ case "$rc" in
168
+ 2) echo "ERR: credential backend unavailable on $PLATFORM" >&2; return 2 ;;
169
+ 4) echo "ERR: credential backend error while reading '$logical'" >&2; return 4 ;;
170
+ esac
167
171
  return 1
168
172
  fi
169
173
  local val=""
@@ -204,8 +208,10 @@ do_set() {
204
208
  if [ "$val" = "-" ]; then
205
209
  val=$(cat)
206
210
  fi
207
- if delegate_to_python; then
208
- # Pass the value via stdin to keep it out of process listings / shell history.
211
+ # macOS writes go through `security -i` below: the secret travels on stdin
212
+ # and never lands on any argv, which is the property the delegate cannot
213
+ # promise from here.
214
+ if [ "$PLATFORM" != "macos" ] && delegate_to_python; then
209
215
  printf '%s' "$val" | python3 "$KEYCHAIN_PY" set "$key" -
210
216
  return $?
211
217
  fi
@@ -2,7 +2,7 @@
2
2
  # extract-conventions.sh
3
3
  #
4
4
  # Phase 1c helper for /multi-agent:analysis.
5
- # Scans a repo and emits 12 convention buckets as a single JSON object on stdout.
5
+ # Scans a repo and emits 13 convention buckets as a single JSON object on stdout.
6
6
  #
7
7
  # Usage:
8
8
  # extract-conventions.sh <repo-path> <platform>
@@ -202,19 +202,6 @@ files_to_json() {
202
202
  printf '%s\n' "$input" | head -5 | jq -R . | jq -s .
203
203
  }
204
204
 
205
- # Top-N suffix frequency from a stream of file basenames.
206
- # stdin: basenames (one per line)
207
- # arg1: regex (POSIX ERE) capturing the suffix in group 1
208
- # stdout: lines "count<TAB>suffix" sorted desc
209
- suffix_freq() {
210
- local re="$1"
211
- grep -Eo "$re" 2>/dev/null \
212
- | sort \
213
- | uniq -c \
214
- | sort -rn \
215
- | sed -E 's/^ *([0-9]+) +/\1\t/'
216
- }
217
-
218
205
  # Run a bucket function with a soft timeout. We wrap the function in a subshell
219
206
  # and use a watchdog so we stay compatible with macOS bash 3.2.
220
207
  run_bucket_with_timeout() {
@@ -48,10 +48,12 @@ HOST_OVERRIDE="${CONFLUENCE_HOST_OVERRIDE:-}"
48
48
  PAGE_ID=""
49
49
  TIMEOUT="${CONFLUENCE_TIMEOUT_SECONDS:-20}"
50
50
 
51
+ need_value() { [ $# -ge 2 ] || { echo "ERR: $1 needs a value" >&2; exit 4; }; }
52
+
51
53
  while [ $# -gt 0 ]; do
52
54
  case "$1" in
53
- --host) HOST_OVERRIDE="$2"; shift 2 ;;
54
- --page-id) PAGE_ID="$2"; shift 2 ;;
55
+ --host) need_value "$@"; HOST_OVERRIDE="$2"; shift 2 ;;
56
+ --page-id) need_value "$@"; PAGE_ID="$2"; shift 2 ;;
55
57
  -h|--help)
56
58
  echo "usage: $0 <page-url> | $0 --host <host> --page-id <id>" >&2
57
59
  exit 4 ;;
@@ -220,11 +222,17 @@ PAGE_ID_FINAL="$PAGE_ID_RESOLVED" \
220
222
  SPACE_KEY_FINAL="$SPACE_KEY" \
221
223
  PAGE_JSON_RAW="$PAGE_JSON" \
222
224
  python3 - <<'PY'
223
- import json, os, re, datetime, html
225
+ import json, os, re, sys, datetime, html
224
226
  from html.parser import HTMLParser
225
227
 
226
228
  raw = os.environ["PAGE_JSON_RAW"]
227
- data = json.loads(raw)
229
+ try:
230
+ data = json.loads(raw)
231
+ except ValueError:
232
+ # A 200 with an HTML body is an SSO login page or a proxy, not the page.
233
+ sys.stderr.write("ERR: Confluence GET for pageId=%s returned a non-JSON body (login page or proxy?)\n"
234
+ % os.environ.get("PAGE_ID_FINAL", ""))
235
+ sys.exit(3)
228
236
 
229
237
  storage = ((data.get("body") or {}).get("storage") or {}).get("value") or ""
230
238
  title = data.get("title") or "<untitled>"