@mmerterden/multi-agent-pipeline 16.18.0 → 16.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +68 -0
- package/README.md +5 -5
- package/README.tr.md +2 -2
- package/docs/FIGMA_PIPELINE.md +1 -1
- package/docs/adr/0001-three-model-triage.md +4 -2
- package/docs/architecture.md +2 -2
- package/docs/ecosystem.md +13 -11
- package/docs/features.md +2 -2
- package/index.js +1 -1
- package/install/_codex-agents.mjs +2 -2
- package/install/_common.mjs +25 -1
- package/install/_dev-only-files.mjs +3 -2
- package/install/_mcp-register.mjs +4 -3
- package/install/_plugin-skills.mjs +1 -3
- package/install/copilot.mjs +18 -9
- package/install/index.mjs +2 -4
- package/install/templates/copilot-instructions.md +7 -7
- package/package.json +4 -3
- package/pipeline/agents/android-architect.md +1 -0
- package/pipeline/agents/backend-architect.md +1 -0
- package/pipeline/agents/code-reviewer.md +1 -0
- package/pipeline/agents/dev-critic.md +2 -1
- package/pipeline/agents/explorer.md +1 -0
- package/pipeline/agents/ios-architect.md +1 -0
- package/pipeline/agents/security-auditor.md +1 -0
- package/pipeline/agents/task-clarifier.md +1 -0
- package/pipeline/claude-md-template.md +2 -2
- package/pipeline/commands/multi-agent/SKILL.md +5 -5
- package/pipeline/commands/multi-agent/complaint-analysis/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/design-check/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/help/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/manual-test/SKILL.md +5 -1
- package/pipeline/commands/multi-agent/resume/SKILL.md +1 -0
- package/pipeline/commands/multi-agent/review/SKILL.md +27 -14
- package/pipeline/commands/multi-agent/review-analysis/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/store-ready/SKILL.md +24 -2
- package/pipeline/commands/multi-agent/sync/SKILL.md +2 -2
- package/pipeline/commands/sim-test.md +64 -20
- package/pipeline/lib/credential-inventory.sh +15 -2
- package/pipeline/lib/credential-store-resolver.sh +14 -4
- package/pipeline/lib/credential-store.sh +8 -2
- package/pipeline/lib/extract-conventions.sh +1 -14
- package/pipeline/lib/fetch-confluence.sh +12 -4
- package/pipeline/lib/fetch-crashlytics.sh +11 -8
- package/pipeline/lib/fetch-document.sh +1 -1
- package/pipeline/lib/fetch-figma-annotations.sh +7 -5
- package/pipeline/lib/fetch-fortify.sh +5 -3
- package/pipeline/lib/fetch-graylog.sh +5 -3
- package/pipeline/lib/figma-mcp-refresh.sh +1 -1
- package/pipeline/lib/figma-screenshot.sh +27 -24
- package/pipeline/lib/figma-token.sh +8 -4
- package/pipeline/lib/issue-fetcher.sh +0 -1
- package/pipeline/lib/jira-publish.sh +7 -5
- package/pipeline/lib/md2confluence-v3.py +13 -7
- package/pipeline/lib/multi-repo-pipeline.sh +18 -8
- package/pipeline/lib/plan-todos.sh +11 -0
- package/pipeline/lib/post-pr-review.sh +9 -2
- package/pipeline/lib/repo-cache.sh +18 -10
- package/pipeline/lib/review-watch.sh +60 -14
- package/pipeline/lib/shadow-git.sh +8 -4
- package/pipeline/lib/vercel-deploy.sh +2 -2
- package/pipeline/multi-agent-refs/_dev-context.md +5 -2
- package/pipeline/multi-agent-refs/analysis/locked.md +4 -4
- package/pipeline/multi-agent-refs/analysis/render.md +1 -1
- package/pipeline/multi-agent-refs/channels/pr.md +22 -4
- package/pipeline/multi-agent-refs/cross-cli-contract.md +1 -1
- package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +15 -2
- package/pipeline/multi-agent-refs/features/review-delta.md +89 -0
- package/pipeline/multi-agent-refs/features/scope-check.md +41 -0
- package/pipeline/multi-agent-refs/features/verify-by-test.md +6 -5
- package/pipeline/multi-agent-refs/features/worktree-finalize.md +1 -1
- package/pipeline/multi-agent-refs/outside-the-pipeline.md +6 -6
- package/pipeline/multi-agent-refs/payload-contracts.md +1 -1
- package/pipeline/multi-agent-refs/phases/log-format.md +1 -1
- package/pipeline/multi-agent-refs/phases/modes.md +1 -1
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +4 -2
- package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +3 -3
- package/pipeline/multi-agent-refs/phases/phase-2-planning.md +5 -5
- package/pipeline/multi-agent-refs/phases/phase-3-dev.md +17 -2
- package/pipeline/multi-agent-refs/phases/phase-4-review.md +67 -24
- package/pipeline/multi-agent-refs/phases/phase-5-test.md +10 -0
- package/pipeline/multi-agent-refs/phases/phase-6-commit.md +2 -0
- package/pipeline/multi-agent-refs/phases/phase-7-report.md +4 -2
- package/pipeline/multi-agent-refs/rules.md +2 -2
- package/pipeline/multi-agent-refs/tracker-contract.md +1 -1
- package/pipeline/rules/figma-pipeline.md +18 -72
- package/pipeline/rules/outside-the-pipeline.md +4 -3
- package/pipeline/schemas/agent-state.schema.json +130 -1
- package/pipeline/schemas/dev-critic-output.schema.json +5 -0
- package/pipeline/schemas/prefs.schema.json +47 -0
- package/pipeline/schemas/reviewer-output.schema.json +7 -2
- package/pipeline/schemas/scope-check.schema.json +55 -0
- package/pipeline/schemas/token-budget.json +3 -3
- package/pipeline/schemas/triage-output.schema.json +12 -2
- package/pipeline/scripts/README.md +3 -2
- package/pipeline/scripts/_fingerprint.mjs +173 -0
- package/pipeline/scripts/_stack-routing.mjs +1 -1
- package/pipeline/scripts/agent-guard.py +102 -21
- package/pipeline/scripts/anonymize-findings.mjs +7 -6
- package/pipeline/scripts/build-skills-index.mjs +14 -3
- package/pipeline/scripts/cost-budget-check.mjs +5 -3
- package/pipeline/scripts/cost-lib.sh +0 -15
- package/pipeline/scripts/diff-explain.mjs +22 -12
- package/pipeline/scripts/evidence-gate.mjs +73 -5
- package/pipeline/scripts/finding-fingerprint.mjs +101 -0
- package/pipeline/scripts/gc-refs.sh +6 -2
- package/pipeline/scripts/gc-tmp.sh +1 -1
- package/pipeline/scripts/gc-worktrees.sh +1 -1
- package/pipeline/scripts/gen-mode-dispatch.mjs +3 -3
- package/pipeline/scripts/github-ssh-setup.sh +7 -2
- package/pipeline/scripts/graph-build.mjs +2 -2
- package/pipeline/scripts/jira-wiki-escape.mjs +2 -1
- package/pipeline/scripts/keychain.py +12 -11
- package/pipeline/scripts/learning-curve.mjs +1 -1
- package/pipeline/scripts/migrate-prefs.mjs +1 -1
- package/pipeline/scripts/output-quality-check.sh +3 -1
- package/pipeline/scripts/phase-tracker.sh +1 -1
- package/pipeline/scripts/phase0-exit-gate.mjs +2 -1
- package/pipeline/scripts/plan-coverage-gate.mjs +2 -1
- package/pipeline/scripts/pre-commit-check.sh +23 -13
- package/pipeline/scripts/prune-logs.sh +1 -1
- package/pipeline/scripts/render-agent-log-cost.sh +3 -1
- package/pipeline/scripts/render-cost-summary.sh +4 -2
- package/pipeline/scripts/render-work-summary.sh +5 -3
- package/pipeline/scripts/repo-map.mjs +3 -2
- package/pipeline/scripts/review-delta.mjs +217 -0
- package/pipeline/scripts/run-metrics.mjs +20 -0
- package/pipeline/scripts/scan-skills.sh +6 -2
- package/pipeline/scripts/scope-check-gate.mjs +90 -0
- package/pipeline/scripts/search-logs.sh +8 -6
- package/pipeline/scripts/sign-skills.sh +3 -1
- package/pipeline/scripts/smoke-cross-cli-behavior.sh +12 -5
- package/pipeline/scripts/triage-memory.mjs +25 -4
- package/pipeline/scripts/uninstall.mjs +20 -12
- package/pipeline/scripts/update-check.sh +2 -2
- package/pipeline/scripts/update-issue-progress.sh +6 -5
- package/pipeline/scripts/validate-analysis-doc.mjs +6 -6
- package/pipeline/scripts/validate-reviewer.mjs +6 -0
- package/pipeline/scripts/validate-triage.mjs +20 -0
- package/pipeline/scripts/verify-skills.sh +3 -1
- package/pipeline/scripts/worktree-finalize.sh +22 -10
- package/pipeline/skills/.skill-manifest.json +81 -57
- package/pipeline/skills/.skills-index.json +19 -19
- package/pipeline/skills/shared/README.md +8 -8
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +7 -7
- package/pipeline/skills/shared/core/multi-agent-design-check/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-review/SKILL.md +23 -14
- package/pipeline/skills/shared/core/multi-agent-store-ready/SKILL.md +5 -0
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +2 -2
- package/pipeline/skills/shared/external/NOTICE-dimillian-skills.md +56 -0
- package/pipeline/skills/shared/external/accessibility-compliance-accessibility-audit/SKILL.md +0 -4
- package/pipeline/skills/shared/external/api-patterns/SKILL.md +12 -24
- package/pipeline/skills/shared/external/app-store-changelog/references/release-notes-guidelines.md +34 -0
- package/pipeline/skills/shared/external/app-store-changelog/scripts/collect_release_changes.sh +33 -0
- package/pipeline/skills/shared/external/architecture/SKILL.md +7 -9
- package/pipeline/skills/shared/external/debugging-strategies/SKILL.md +0 -4
- package/pipeline/skills/shared/external/fastapi-pro/SKILL.md +0 -1
- package/pipeline/skills/shared/external/github-actions-templates/SKILL.md +0 -14
- package/pipeline/skills/shared/external/hig-components-content/SKILL.md +13 -13
- package/pipeline/skills/shared/external/hig-components-layout/SKILL.md +16 -16
- package/pipeline/skills/shared/external/hig-components-status/SKILL.md +6 -6
- package/pipeline/skills/shared/external/hig-components-system/SKILL.md +13 -13
- package/pipeline/skills/shared/external/hig-foundations/SKILL.md +23 -23
- package/pipeline/skills/shared/external/hig-inputs/SKILL.md +18 -18
- package/pipeline/skills/shared/external/hig-patterns/SKILL.md +30 -30
- package/pipeline/skills/shared/external/hig-platforms/SKILL.md +11 -11
- package/pipeline/skills/shared/external/hig-technologies/SKILL.md +33 -33
- package/pipeline/skills/shared/external/ios-coding-standard/references/STANDARD.md +52 -52
- package/pipeline/skills/shared/external/ios-coding-standard/references/lint-local.sh +1 -1
- package/pipeline/skills/shared/external/ios-coding-standard/references/rules.yml +11 -11
- package/pipeline/skills/shared/external/ios-developer/SKILL.md +0 -1
- package/pipeline/skills/shared/external/ios-module-structure/SKILL.md +7 -3
- package/pipeline/skills/shared/external/localization-reuse-map/SKILL.md +9 -15
- package/pipeline/skills/shared/external/macos-spm-app-packaging/SKILL.md +0 -5
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/Package.swift +17 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/Sources/MyApp/Resources/.keep +0 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/Sources/MyApp/main.swift +11 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/bootstrap/version.env +2 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/build_icon.sh +49 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/compile_and_run.sh +63 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/launch.sh +28 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/make_appcast.sh +82 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/package_app.sh +206 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/setup_dev_signing.sh +52 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/sign-and-notarize.sh +52 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/version.env +2 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/references/packaging.md +17 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/references/release.md +32 -0
- package/pipeline/skills/shared/external/macos-spm-app-packaging/references/scaffold.md +79 -0
- package/pipeline/skills/shared/external/monorepo-architect/SKILL.md +0 -1
- package/pipeline/skills/shared/external/nodejs-backend-patterns/SKILL.md +0 -4
- package/pipeline/skills/shared/external/swift-concurrency-expert/references/approachable-concurrency.md +63 -0
- package/pipeline/skills/shared/external/swift-concurrency-expert/references/swift-6-2-concurrency.md +272 -0
- package/pipeline/skills/shared/external/swift-concurrency-expert/references/swiftui-concurrency-tour-wwdc.md +33 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/code-smells.md +150 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/demystify-swiftui-performance-wwdc23.md +46 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/optimizing-swiftui-performance-instruments.md +29 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/profiling-intake.md +44 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/report-template.md +47 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/understanding-hangs-in-your-app.md +33 -0
- package/pipeline/skills/shared/external/swiftui-performance-audit/references/understanding-improving-swiftui-performance.md +52 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/app-wiring.md +201 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/async-state.md +96 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/components-index.md +46 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/controls.md +57 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/deeplinks.md +66 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/focus.md +90 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/form.md +97 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/grids.md +71 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/haptics.md +71 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/input-toolbar.md +51 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/lightweight-clients.md +93 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/list.md +86 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/loading-placeholders.md +38 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/macos-settings.md +71 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/matched-transitions.md +59 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/media.md +73 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/menu-bar.md +101 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/navigationstack.md +159 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/overlay.md +45 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/performance.md +62 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/previews.md +48 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/scroll-reveal.md +133 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/scrollview.md +87 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/searchable.md +71 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/sheets.md +155 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/split-views.md +72 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/tabview.md +114 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/theming.md +71 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/title-menus.md +93 -0
- package/pipeline/skills/shared/external/swiftui-ui-patterns/references/top-bar.md +49 -0
- package/pipeline/skills/shared/external/swiftui-view-refactor/references/mv-patterns.md +161 -0
- package/pipeline/skills/skills-index.md +8 -8
- package/pipeline/skills/shared/external/help-skills/SKILL.md +0 -166
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
---
|
|
2
|
-
description: "Task orchestrator - full pipeline via Jira ID + branch or GitHub Issue URL: analysis, plan, TDD development, parallel review + Fable triage (
|
|
2
|
+
description: "Task orchestrator - full pipeline via Jira ID + branch or GitHub Issue URL: analysis, plan, TDD development, parallel review + Fable triage (3 reviewers per host: Fable + Opus + Sonnet on Claude Code, GPT-5.4 + Opus + Sonnet on Copilot CLI), commit, log. Use when given a Jira ID, a GitHub issue or a free-text task and the whole pipeline should run."
|
|
3
3
|
description-tr: "Görev orkestratörü - Jira ID + branch veya GitHub Issue URL ile tam pipeline: analiz, plan, TDD geliştirme, paralel review + Fable triyajı (CLI'ya göre: Claude Code'da 3, Copilot CLI'da 3 model), commit, log"
|
|
4
4
|
allowed-tools: Agent, Bash, Read, Write, Edit, Glob, Grep, TaskCreate, TaskUpdate, TaskList, TaskGet, AskUserQuestion, WebFetch, WebSearch, NotebookEdit, Skill
|
|
5
5
|
---
|
|
@@ -48,7 +48,7 @@ Classification schema lives in `$HOME/.claude/multi-agent-refs/_input-parser.md`
|
|
|
48
48
|
| 7 | `issue` | full picker | account → repo (multi) → issue → maturity → dev-context |
|
|
49
49
|
| 8 | Free-text | `freetext` | account → repo (single) → dev-context (maturity skip) |
|
|
50
50
|
|
|
51
|
-
**Rule**: Whatever the type, **account is always asked** (autopilot picks a default). After issue fetch, **maturity check is mandatory** - blockers halt the pipeline. Picker `
|
|
51
|
+
**Rule**: Whatever the type, **account is always asked** (autopilot picks a default). After issue fetch, **maturity check is mandatory** - blockers halt the pipeline. Picker `header` renders in English (the 12-char chip); the `question`, each option's `label` and each option's `description` render in `outputLanguage`, per the canonical matrix in `multi-agent-refs/rules.md`.
|
|
52
52
|
|
|
53
53
|
Lib scripts (`~/.claude/lib/`):
|
|
54
54
|
- `account-resolver.sh` - keychain account inventory
|
|
@@ -102,8 +102,8 @@ Lib scripts (`~/.claude/lib/`):
|
|
|
102
102
|
This command uses lazy loading for token efficiency. Read the relevant sub-file based on the routed action:
|
|
103
103
|
|
|
104
104
|
**File layout:**
|
|
105
|
-
- `commands/multi-agent
|
|
106
|
-
- `$HOME/.claude/multi-agent-refs/**` → **internal references**, read by the main command, never invoked directly
|
|
105
|
+
- `commands/multi-agent/{cmd}/SKILL.md` → **invocable actions** (each gets its own `/multi-agent:<name>` slash command)
|
|
106
|
+
- `$HOME/.claude/multi-agent-refs/**` → **internal references**, read by the main command, never invoked directly. They live outside `commands/` precisely so they never appear in slash-command autocomplete.
|
|
107
107
|
|
|
108
108
|
| Route | File to Read |
|
|
109
109
|
|-------|-------------|
|
|
@@ -262,7 +262,7 @@ When called with `review`:
|
|
|
262
262
|
1. Detect current branch and project from cwd (or ask)
|
|
263
263
|
2. Get diff: `git diff HEAD` (unstaged + staged)
|
|
264
264
|
3. If no diff, get diff against base branch: `git diff origin/{baseBranch}...HEAD`
|
|
265
|
-
4. Launch Phase 4 review (parallel + Fable triage -
|
|
265
|
+
4. Launch Phase 4 review (parallel + Fable triage - 3 reviewers on every host: Fable + Opus + Sonnet on Claude Code, GPT-5.4 + Opus + Sonnet on Copilot CLI) on the diff
|
|
266
266
|
5. No worktree, no state file - lightweight one-shot review
|
|
267
267
|
6. Print findings to terminal
|
|
268
268
|
|
|
@@ -127,7 +127,7 @@ Set `phase: "drafting"`.
|
|
|
127
127
|
### Phase 4 - Draft, humanize, buffer
|
|
128
128
|
|
|
129
129
|
1. Render the report per `$HOME/.claude/multi-agent-refs/complaint-analysis-template.md` (8 fixed sections; single-language body in `outputLanguage`; verdict tokens English per Locked 9) to `/tmp/complaint-analysis-<run-slug>-<UTC-iso8601>/report.md`. Store `outputs.draftDir`.
|
|
130
|
-
2. **Humanizer pass (
|
|
130
|
+
2. **Humanizer pass (required: actually invoke the `ai-common-toolkit:humanizer` skill; the punctuation grep alone does NOT satisfy this)** with `language: <tr|en>`, `tone: technical-explanatory`, `stripFancyPunctuation: true`. Diacritics preserved (Locked 8).
|
|
131
131
|
3. Punctuation gate: `node $HOME/.claude/scripts/validate-complaint-doc.mjs <draft>` reports no banned-punctuation error. It checks the policy in Node, so the same result holds on macOS, Linux and Windows; `grep -P` is absent from BSD grep and would never run there.
|
|
132
132
|
4. Show the draft path + size to the user. Set `phase: "awaiting_output_decision"`.
|
|
133
133
|
|
|
@@ -133,7 +133,7 @@ for l in sys.stdin:
|
|
|
133
133
|
Print the resolved set grouped by screen **with its relaunch cost**: `<n> targets · <relaunchCount> relaunches`. Read the cost from `plan`, not from the target count - one relaunch serves every in-app target on that screen, so a 50-target module is typically a dozen relaunches, not fifty. **Whole-module is the intended default**; only ask for confirmation when `relaunchCount` exceeds `config coverage.confirmAbove` (default 25), and phrase it as a cost estimate, not as an invitation to shrink the audit. Never propose a smaller scope as the easy path - a scoped run is for resuming or for a focused re-check, not for avoiding work.
|
|
134
134
|
7. **`--resume`** - read the most recent `~/DesignChecks/{repo}__{module}/*/run-state.json`; the scope becomes that run's targets minus its `covered` and minus its `skipped` entries. Skips carry a reason and are honoured, so a resume covers the genuinely unaudited remainder. No previous run → tell the user and fall back to whole-module scope after confirmation.
|
|
135
135
|
8. **Report dir** - create `~/DesignChecks/{repo}__{module}/{UTC-timestamp}/` (and `assets/` inside it) now, and persist it as `state.designCheck.reportDir`. Phase 3 writes captures, comparison images, and `run-state.json` into it, so it must exist before driving starts - not at export time. This is report output, not a worktree; $HOME is intended here.
|
|
136
|
-
9. **Worktree** - build the Debug app in an isolated worktree so the user's tree is untouched. Follow `phase-0-init.md` Step
|
|
136
|
+
9. **Worktree** - build the Debug app in an isolated worktree so the user's tree is untouched. Follow the `phase-0-init.md` Step 6 worktree location convention exactly: `{projectRoot}/.worktrees/{taskId}` with `taskId` = `DC-<shortId>`, **never under $HOME** (the `.worktrees` segment is fixed, not a preference). Stale-lock heal (`git worktree prune`) + residue guard (`.worktrees/` in `.git/info/exclude`) first. Local mode is not offered - the audit always uses a worktree checkout of the current branch's HEAD (no fetch/push).
|
|
137
137
|
|
|
138
138
|
Persist `agent-state.json` with `taskId`, `mode: "design-check"`, `platform`, `projectRoot`, `worktreePath`, `module`, `designCheck`.
|
|
139
139
|
|
|
@@ -53,7 +53,7 @@ How It Works (Phase 0 - Interactive Flow):
|
|
|
53
53
|
Pipeline (after Phase 0) - shown as visual cards in terminal:
|
|
54
54
|
|
|
55
55
|
Phase 0: Init -> The 8 steps above
|
|
56
|
-
Phase 1: Analysis -> Stack detection + codebase scan (
|
|
56
|
+
Phase 1: Analysis -> Stack detection + codebase scan (Sonnet)
|
|
57
57
|
Phase 2: Planning -> Task breakdown + architecture review + Plan Approval Gate
|
|
58
58
|
(clarification max 2 rounds + approval loop - Full + interactive
|
|
59
59
|
only; a Short run has no plan, autopilot may not ask)
|
|
@@ -331,7 +331,7 @@ Nasıl Çalışır (Phase 0 - İnteraktif Akış):
|
|
|
331
331
|
Pipeline (Phase 0'dan sonra) - terminalde görsel kart olarak görünür:
|
|
332
332
|
|
|
333
333
|
Phase 0: Init -> Yukarıdaki 8 adım
|
|
334
|
-
Phase 1: Analysis -> Stack tespiti + codebase taraması (
|
|
334
|
+
Phase 1: Analysis -> Stack tespiti + codebase taraması (Sonnet)
|
|
335
335
|
Phase 2: Planning -> Task kırılımı + mimari inceleme + Plan Onay Kapısı
|
|
336
336
|
(clarification max 2 tur + onay döngüsü - sadece Tam +
|
|
337
337
|
etkileşimli; Kısa'da plan yok, autopilot soru soramaz)
|
|
@@ -67,7 +67,7 @@ Depth is a separate axis, asked at Phase 0 Step 7.5 rather than encoded in the c
|
|
|
67
67
|
|
|
68
68
|
## Delegation
|
|
69
69
|
|
|
70
|
-
Orchestrator routing: the routing table in `$HOME/.claude/commands/multi-agent/SKILL.md` resolves `local-autopilot` as the union of the `dev-local` + `autopilot` mode mixins. Contract details: `$HOME/.claude/multi-agent-refs/phases/phase-0-init.md` Step
|
|
70
|
+
Orchestrator routing: the routing table in `$HOME/.claude/commands/multi-agent/SKILL.md` resolves `local-autopilot` as the union of the `dev-local` + `autopilot` mode mixins. Contract details: `$HOME/.claude/multi-agent-refs/phases/phase-0-init.md` Step 6 (local branch) + `$HOME/.claude/multi-agent-refs/phases/phase-2-planning.md` Step 5 (autopilot gate skip + safety classifier).
|
|
71
71
|
## Required: outward-facing payload contracts
|
|
72
72
|
|
|
73
73
|
Before writing anything outward-facing - PR body, Jira comment, Confluence page, closing report - load `$HOME/.claude/multi-agent-refs/payload-contracts.md`. It names the canonical section set for each payload, the markup dialect per surface (PR body is Markdown, Jira is wiki markup - mixing them is a defect), and the token/duration numbers the closing report must carry. Improvising a payload shape from memory is the most common failure of the short modes.
|
|
@@ -47,5 +47,9 @@ Lets you switch to the task branch for manual testing in Xcode before the PR is
|
|
|
47
47
|
5. **Wait for the user's reply**
|
|
48
48
|
|
|
49
49
|
6. **Branch on the answer**:
|
|
50
|
-
- **OK** →
|
|
50
|
+
- **OK** → first write `$WORKTREE/.pipeline/manual-test.json` (one entry per acceptance criterion from the analysis doc test plan, the plan tasks, or the user's own words):
|
|
51
|
+
```json
|
|
52
|
+
{"criteria":[{"spec":"<quote>","source":"analysis 15.2 | plan task 3 | user","observed":"<what was seen>","verdict":"pass|fail|not-tested","reason":"<required when not-tested>","screenshot":"<path or null>"}],"verdict":"passed|failed"}
|
|
53
|
+
```
|
|
54
|
+
then run `node $HOME/.claude/scripts/evidence-gate.mjs --claim manual --status passed --evidence "$WORKTREE/.pipeline/manual-test.json"`. Exit 1 means the "ok" is not accepted: name the criterion that is missing evidence and wait for the next reply. Exit 0 → `phase-tracker.sh update 5 completed` + `phase-tracker.sh meta 5 Result "local test passed (user)"`, recreate the worktree, continue to Phase 6. Full contract: `$HOME/.claude/multi-agent-refs/phases/phase-5-test.md` step 5.
|
|
51
55
|
- **Fix needed** → `phase-tracker.sh now 5 "applying fix: <summary>"`, recreate the worktree, apply the fix
|
|
@@ -23,6 +23,7 @@ Resume a paused or failed task from the last successful phase.
|
|
|
23
23
|
- `currentPhase` - last completed phase
|
|
24
24
|
- `status` - `paused` | `failed` | `in_progress`
|
|
25
25
|
- `haltReason` - if set, show it so the user knows why the run stopped; clear it on successful re-entry
|
|
26
|
+
- `circuitBreaker` - if `tripped`, show `trigger` + `detail`, then set `tripped: false` and keep `counters`; if the same trigger fires again at the next checkpoint the breaker re-trips (no silent bypass)
|
|
26
27
|
- `autopilot` - preserve the mode
|
|
27
28
|
|
|
28
29
|
3. **Load context** - rebuild working context from durable artifacts, never from conversation memory:
|
|
@@ -117,12 +117,12 @@ jq --arg sha "$HEAD_SHA" '.review.headCommitSha = $sha' "$AGENT_STATE" > "$tmp_s
|
|
|
117
117
|
**pr mode - bitbucket-server:**
|
|
118
118
|
|
|
119
119
|
```bash
|
|
120
|
-
# Resolve credentials via prefs.keychainMapping (never hardcode key names).
|
|
120
|
+
# Resolve credentials via prefs.global.keychainMapping (never hardcode key names).
|
|
121
121
|
. "$HOME/.claude/lib/credential-store-resolver.sh" \
|
|
122
122
|
|| . "$HOME/.copilot/lib/credential-store-resolver.sh"
|
|
123
123
|
resolve_credential_store
|
|
124
|
-
USER_KEY=$(jq -r '.keychainMapping.bitbucket_user' "$HOME/.claude/multi-agent-preferences.json")
|
|
125
|
-
TOKEN_KEY=$(jq -r '.keychainMapping.bitbucket_token' "$HOME/.claude/multi-agent-preferences.json")
|
|
124
|
+
USER_KEY=$(jq -r '.global.keychainMapping.bitbucket_user' "$HOME/.claude/multi-agent-preferences.json")
|
|
125
|
+
TOKEN_KEY=$(jq -r '.global.keychainMapping.bitbucket_token' "$HOME/.claude/multi-agent-preferences.json")
|
|
126
126
|
BB_USER=$("$CRED_STORE" get "$USER_KEY")
|
|
127
127
|
BB_TOKEN=$("$CRED_STORE" get "$TOKEN_KEY")
|
|
128
128
|
|
|
@@ -175,15 +175,25 @@ Scope note: a guide governs only files under its own directory - a guide found
|
|
|
175
175
|
|
|
176
176
|
### 3. Launch parallel reviewers - host-CLI dependent
|
|
177
177
|
|
|
178
|
+
Every host runs three reviewers; only the second slot differs, because GPT-5.4 exists on Copilot and Codex but not on Claude Code, where Opus fills it.
|
|
179
|
+
|
|
178
180
|
**Claude Code (3 in parallel):**
|
|
179
181
|
- Agent 1: `claude-fable-5` → security + architecture
|
|
180
|
-
- Agent 2: `claude-
|
|
182
|
+
- Agent 2: `claude-opus-5` → edge cases, alternate perspective
|
|
183
|
+
- Agent 3: `claude-sonnet-5` → general quality
|
|
181
184
|
|
|
182
185
|
**Copilot CLI (3 in parallel):**
|
|
183
186
|
- Agent 1: `claude-opus-5` → security + architecture (Fable 5 is not offered on Copilot CLI)
|
|
184
187
|
- Agent 2: `gpt-5.4` → edge cases, alternate perspective
|
|
185
188
|
- Agent 3: `claude-sonnet-5` → general quality
|
|
186
189
|
|
|
190
|
+
**Codex CLI (3 in parallel, single vendor):**
|
|
191
|
+
- Agent 1: `gpt-5.6` @ `xhigh` → security + architecture
|
|
192
|
+
- Agent 2: `gpt-5.4` @ `high` → edge cases, alternate perspective
|
|
193
|
+
- Agent 3: `gpt-5.6` @ `medium` → general quality
|
|
194
|
+
|
|
195
|
+
With the `fable` rung disabled by prefs the Claude Code panel is two reviewers (Opus + Sonnet); see `$HOME/.claude/multi-agent-refs/features/model-fallback.md`.
|
|
196
|
+
|
|
187
197
|
Each reviewer receives the diff, the module review guides from Step 2b (when any were found), plus the standard reviewer system prompt (see `$HOME/.claude/multi-agent-refs/phases/phase-4-review.md` for the prompt contract). Output: structured `findings[]` per reviewer.
|
|
188
198
|
|
|
189
199
|
### 4. Store-compliance cross-reference
|
|
@@ -203,16 +213,18 @@ Catalog-only - does NOT invoke binaries. For a full scan, use `/multi-agent:te
|
|
|
203
213
|
|
|
204
214
|
### 4b. Platform parity cross-check (advisory, read-only)
|
|
205
215
|
|
|
206
|
-
When
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
216
|
+
When a counterpart repo resolves whose `stack` is the other mobile platform and
|
|
217
|
+
the diff touches a screen, a service, a request model or a localization file,
|
|
218
|
+
compare the change against that repo on four axes: endpoints called, parameters
|
|
219
|
+
sent, business rules around the call, localization keys used.
|
|
210
220
|
|
|
211
|
-
The counterpart is resolved automatically, in both directions (ios ↔ android)
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
221
|
+
The counterpart is resolved automatically, in both directions (ios ↔ android),
|
|
222
|
+
from four sources in this order, first hit wins: `--with <path|owner/repo>` (a
|
|
223
|
+
one-off override, remembered like any other resolution), then a remembered
|
|
224
|
+
`prefs.projects[<slug>].counterpartRoots[]`, then the primary checkout's sibling
|
|
225
|
+
directories, then `state.siblings[]` from the Phase 0 dev-context picker. None of
|
|
226
|
+
them is a precondition for the others. One match is used and remembered, so the
|
|
227
|
+
next review of the same project asks nothing.
|
|
216
228
|
|
|
217
229
|
Several candidates → interactive runs ask once; autopilot and non-interactive
|
|
218
230
|
runs skip silently, because an unattended run must not block on a picker. No
|
|
@@ -237,12 +249,13 @@ Triage also marks each finding as `accepted` (real issue), `deferred` (real but
|
|
|
237
249
|
| Model | Verdict | Blocking | Important | Suggestion |
|
|
238
250
|
|----------|-----------|----------|-----------|------------|
|
|
239
251
|
| Fable | approved | 0 | 1 | 3 |
|
|
252
|
+
| Opus | approved | 0 | 2 | 2 |
|
|
240
253
|
| Sonnet | rejected | 1 | 2 | 5 |
|
|
241
254
|
|
|
242
255
|
Consensus: ⚠ DISAGREEMENT - see Fable triage
|
|
243
256
|
```
|
|
244
257
|
|
|
245
|
-
This summary ALWAYS prints, regardless of input mode. The chat is the live conversation; on the PR side, the durable artifacts are inline comments + the review state (Step 7).
|
|
258
|
+
One row per reviewer that ran: Fable + Opus + Sonnet on Claude Code, Opus + GPT-5.4 + Sonnet on Copilot CLI, the three GPT rows on Codex CLI. This summary ALWAYS prints, regardless of input mode. The chat is the live conversation; on the PR side, the durable artifacts are inline comments + the review state (Step 7).
|
|
246
259
|
|
|
247
260
|
### 7. Post to PR - only when input.kind === "pr"
|
|
248
261
|
|
|
@@ -23,7 +23,7 @@ Read `$HOME/.claude/multi-agent-refs/analysis/review.md` and execute it:
|
|
|
23
23
|
|
|
24
24
|
## Why it cites Locked decisions
|
|
25
25
|
|
|
26
|
-
The analysis flow already declares
|
|
26
|
+
The analysis flow already declares 36 Locked decisions and two deterministic validators. A reviewer that says "I would have written this differently" gives the author nothing to act on; one that says "Locked 34: the Confluence page is in the evidence record but not in Section 21" gives them a fix and a reason. Findings that map to no rule are still allowed, but they are marked as judgement, not dressed up as a violation.
|
|
27
27
|
|
|
28
28
|
## What it never does
|
|
29
29
|
|
|
@@ -183,7 +183,7 @@ is a guess the user must be able to correct.
|
|
|
183
183
|
|
|
184
184
|
### Mode A - build from the branch
|
|
185
185
|
|
|
186
|
-
1. Worktree at `{projectRoot}/{
|
|
186
|
+
1. Worktree at `{projectRoot}/.worktrees/{taskId}` on the chosen branch (phase-0-init.md Step 6 convention; the `.worktrees` segment is fixed, not a preference).
|
|
187
187
|
**Never under `$HOME`**, never a direct checkout of the main working tree.
|
|
188
188
|
2. Resolve the scheme / module and workspace / project from prefs; ask if ambiguous.
|
|
189
189
|
3. Build:
|
|
@@ -218,10 +218,22 @@ is a guess the user must be able to correct.
|
|
|
218
218
|
Use the fallback only when the tool is genuinely absent, and say in the report
|
|
219
219
|
which path ran - a rule set that silently differed between two invocations is
|
|
220
220
|
worse than a missing gate.
|
|
221
|
+
|
|
222
|
+
Then, whichever path ran, count `<archive>/dSYMs/*.dSYM`. Zero is a blocking
|
|
223
|
+
finding: `[SYMBOLS] archive carries no dSYM; crash reports will not symbolicate`,
|
|
224
|
+
with the hint `DEBUG_INFORMATION_FORMAT = dwarf-with-dsym` for the Release
|
|
225
|
+
configuration. This check runs from the package alone and does not need the
|
|
226
|
+
MCP tool.
|
|
221
227
|
- **Android**: `android_apk_audit` on the artifact, plus the
|
|
222
228
|
`google-play-compliance` skill's 21 rules - `bundletool validate` and manifest
|
|
223
229
|
dump, `aapt2 dump badging`, `apksigner verify`, ABI / native scan.
|
|
224
230
|
|
|
231
|
+
Then, when the module has `minifyEnabled true` and the bundle build produced no
|
|
232
|
+
`mapping.txt`, raise the same class of blocking finding:
|
|
233
|
+
`[SYMBOLS] minified bundle carries no mapping.txt; crash reports will not
|
|
234
|
+
deobfuscate`. This check reads the module config and the build output alone
|
|
235
|
+
and does not need the MCP tool.
|
|
236
|
+
|
|
225
237
|
`error` findings are blocking; `warning` is advisory. Group by severity and keep
|
|
226
238
|
each finding's ITMS / Play policy code - Gate 2 may return the same code on iOS,
|
|
227
239
|
and seeing it in both places tells the user it is real rather than a heuristic.
|
|
@@ -259,6 +271,7 @@ that catches what a human reviewer rejects, so it reads source, not the binary.
|
|
|
259
271
|
| Privacy policy | reachable in-app and in the metadata |
|
|
260
272
|
| IAP | anything unlocking features goes through StoreKit, with no external purchase path |
|
|
261
273
|
| Sign in with Apple | present when a third-party social login is offered |
|
|
274
|
+
| Crash symbolication | a dSYM upload step exists: an Xcode run-script calling `upload-symbols`, or a Crashlytics / Sentry / Datadog upload in CI |
|
|
262
275
|
|
|
263
276
|
**Android** - `ai-android-toolkit:play-store-review`:
|
|
264
277
|
|
|
@@ -272,6 +285,7 @@ that catches what a human reviewer rejects, so it reads source, not the binary.
|
|
|
272
285
|
| Account deletion | if the app creates accounts, an in-app deletion path exists, plus the web deletion URL Play requires |
|
|
273
286
|
| Content rating | the questionnaire answers match the app's actual content |
|
|
274
287
|
| Signing | Play App Signing configured, upload key distinct from the app signing key |
|
|
288
|
+
| Crash symbolication | a `mapping.txt` upload exists: the Firebase Crashlytics Gradle plugin, or the Play App Bundle deobfuscation file |
|
|
275
289
|
|
|
276
290
|
For each: `pass` / `fail` / `not-applicable` with the evidence path that justifies
|
|
277
291
|
it. `not-applicable` needs a reason - an unexamined area is not a pass.
|
|
@@ -303,6 +317,12 @@ Advisory
|
|
|
303
317
|
|
|
304
318
|
Not run
|
|
305
319
|
Gate 2: no local Play validator - authoritative check is server-side only
|
|
320
|
+
|
|
321
|
+
Before rollout (human inputs, not verified)
|
|
322
|
+
Phased rollout: <1% -> 10% -> 50% -> 100% | full>
|
|
323
|
+
Halt thresholds: crash-free < 99.5% (iOS, Android) or ANR > 0.47% (Android) => pause the rollout
|
|
324
|
+
Rollback / forward-fix owner: <name>
|
|
325
|
+
Forward-fix plan: <one line>
|
|
306
326
|
```
|
|
307
327
|
|
|
308
328
|
Rules for the report:
|
|
@@ -311,6 +331,8 @@ Rules for the report:
|
|
|
311
331
|
skip. An Android run therefore reads `2 of 3 gates cleared, 1 skipped` at best.
|
|
312
332
|
- Every blocking finding carries a file path or a store code. A finding the user
|
|
313
333
|
cannot act on is noise.
|
|
334
|
+
- The `Before rollout` block is filled by a human, never inferred. When it is
|
|
335
|
+
left unfilled the verdict line gains the suffix `, rollout plan missing`.
|
|
314
336
|
- Humanize via `--lang en` by default (`promptLanguage` is locked to `"en"`); pass
|
|
315
337
|
`--lang=tr` explicitly to opt into Turkish.
|
|
316
338
|
- No AI or assistant attribution anywhere, per
|
|
@@ -329,7 +351,7 @@ Print, and stop:
|
|
|
329
351
|
declares - never run it
|
|
330
352
|
- `/multi-agent:store-ready --resume` to re-run after fixes
|
|
331
353
|
- `/multi-agent:channels` to land the findings in Jira / Confluence / Wiki / PR
|
|
332
|
-
-
|
|
354
|
+
- the enabled stack plugin's `fix-bug` skill (`ai-ios-toolkit:fix-bug` or the stack equivalent) when Gate 3 produced code-level findings
|
|
333
355
|
|
|
334
356
|
Never upload, never commit a Play edit, never bump the build number, never commit.
|
|
335
357
|
|
|
@@ -60,7 +60,7 @@ Run every step automatically:
|
|
|
60
60
|
Step 1: PLATFORM Detect macOS / Linux / Windows (Git Bash / WSL); export PLATFORM env
|
|
61
61
|
Step 1.5: DETECT Compare timestamps, find stale targets
|
|
62
62
|
Step 2: COPILOT Claude Code -> Copilot CLI (instructions + 55 sub-command skills)
|
|
63
|
-
Step 2b: CODEX Claude Code -> Codex CLI (1 router skill +
|
|
63
|
+
Step 2b: CODEX Claude Code -> Codex CLI (1 router skill + 55 specs as refs + 8 agent TOML)
|
|
64
64
|
Step 3: REPO Claude Code -> pipeline repo (genericized, personal data scrub, bash -n on all sh)
|
|
65
65
|
Step 3c: PLUGINS pipeline shared/external -> multi-agent-plugins marketplace (rebuild knowledge/,
|
|
66
66
|
bump changed plugins' patch version, commit + push the plugins repo)
|
|
@@ -117,7 +117,7 @@ If nothing is stale → report "All targets up to date" and stop.
|
|
|
117
117
|
- `~/.claude/CLAUDE.md`, `~/.claude/rules/`, `~/.claude/knowledge/`
|
|
118
118
|
- `~/.claude/scripts/` - EXCEPT `pre-commit-check.sh`, `agent-guard.sh`, `agent-guard.py`, and `build-stack-plugins.mjs` (generic, synced)
|
|
119
119
|
- `~/.claude/settings.json`
|
|
120
|
-
- **Any `~/.claude/commands/multi-agent
|
|
120
|
+
- **Any `~/.claude/commands/multi-agent/*/SKILL.md` whose frontmatter has `local-only: true`** - these are user/repo-specific alias wrappers that delegate to a private marketplace plugin's skills exposed as `multi-agent:<name>`. Syncing them would leak the private plugin/skill names into the public pipeline. Filter before copy: skip every source file containing `local-only: true`, and after copy assert none reached `pipeline/commands/`.
|
|
121
121
|
```bash
|
|
122
122
|
# backstop: no local-only wrapper may exist in the synced target
|
|
123
123
|
grep -rl "^local-only: true" ~/multi-agent-pipeline/pipeline/commands/ 2>/dev/null \
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
description: "Mobile UI Bug Hunter - iOS Simulator (simctl) + Android Emulator (adb) - auto-detects platform"
|
|
3
|
-
allowed-tools: Agent, Bash, Read, Write, Edit, TaskCreate, TaskUpdate, TaskList, TaskGet, mcp__multi-agent-toolkit__ios_list_devices, mcp__multi-agent-toolkit__ios_boot_device, mcp__multi-agent-toolkit__ios_screenshot, mcp__multi-agent-toolkit__ios_tap, mcp__multi-agent-toolkit__ios_swipe, mcp__multi-agent-toolkit__ios_type_text, mcp__multi-agent-toolkit__ios_launch_app, mcp__multi-agent-toolkit__ios_terminate_app, mcp__multi-agent-toolkit__ios_list_apps, mcp__multi-agent-toolkit__ios_go_home, mcp__multi-agent-toolkit__ios_set_appearance, mcp__multi-agent-toolkit__ios_set_content_size, mcp__multi-agent-toolkit__ios_set_locale, mcp__multi-agent-toolkit__ios_open_url, mcp__multi-agent-toolkit__ios_status_bar, mcp__multi-agent-toolkit__ios_push_notification, mcp__multi-agent-toolkit__ios_grant_permission, mcp__multi-agent-toolkit__ios_revoke_permission, mcp__multi-agent-toolkit__ios_reset_permissions, mcp__multi-agent-toolkit__ios_set_location, mcp__multi-agent-toolkit__ios_clear_location, mcp__multi-agent-toolkit__ios_set_increase_contrast, mcp__multi-agent-toolkit__ios_record_video, mcp__multi-agent-toolkit__ios_add_media, mcp__multi-agent-toolkit__ios_keychain_reset, mcp__multi-agent-toolkit__ios_get_app_container, mcp__multi-agent-toolkit__ios_erase_device, mcp__multi-agent-toolkit__ios_get_ui_tree, mcp__multi-agent-toolkit__android_list_devices, mcp__multi-agent-toolkit__android_screenshot, mcp__multi-agent-toolkit__android_tap, mcp__multi-agent-toolkit__android_swipe, mcp__multi-agent-toolkit__android_type_text, mcp__multi-agent-toolkit__android_key_event, mcp__multi-agent-toolkit__android_launch_app, mcp__multi-agent-toolkit__android_stop_app, mcp__multi-agent-toolkit__android_list_packages, mcp__multi-agent-toolkit__android_go_home, mcp__multi-agent-toolkit__android_go_back, mcp__multi-agent-toolkit__android_get_ui_tree, mcp__multi-agent-toolkit__android_set_dark_mode, mcp__multi-agent-toolkit__android_set_font_scale, mcp__multi-agent-toolkit__android_set_locale, mcp__multi-agent-toolkit__android_set_location, mcp__multi-agent-toolkit__android_grant_permission, mcp__multi-agent-toolkit__android_revoke_permission, mcp__multi-agent-toolkit__android_record_screen, mcp__multi-agent-toolkit__android_install_apk, mcp__multi-agent-toolkit__android_uninstall_app, mcp__multi-agent-toolkit__android_logcat, mcp__multi-agent-toolkit__android_get_screen_size, mcp__multi-agent-toolkit__android_open_url, mcp__multi-agent-toolkit__android_clear_app_data
|
|
3
|
+
allowed-tools: Agent, Bash, Read, Write, Edit, TaskCreate, TaskUpdate, TaskList, TaskGet, mcp__multi-agent-toolkit__ios_list_devices, mcp__multi-agent-toolkit__ios_boot_device, mcp__multi-agent-toolkit__ios_screenshot, mcp__multi-agent-toolkit__ios_tap, mcp__multi-agent-toolkit__ios_swipe, mcp__multi-agent-toolkit__ios_type_text, mcp__multi-agent-toolkit__ios_launch_app, mcp__multi-agent-toolkit__ios_terminate_app, mcp__multi-agent-toolkit__ios_list_apps, mcp__multi-agent-toolkit__ios_go_home, mcp__multi-agent-toolkit__ios_set_appearance, mcp__multi-agent-toolkit__ios_set_content_size, mcp__multi-agent-toolkit__ios_set_locale, mcp__multi-agent-toolkit__ios_open_url, mcp__multi-agent-toolkit__ios_status_bar, mcp__multi-agent-toolkit__ios_push_notification, mcp__multi-agent-toolkit__ios_grant_permission, mcp__multi-agent-toolkit__ios_revoke_permission, mcp__multi-agent-toolkit__ios_reset_permissions, mcp__multi-agent-toolkit__ios_set_location, mcp__multi-agent-toolkit__ios_clear_location, mcp__multi-agent-toolkit__ios_set_increase_contrast, mcp__multi-agent-toolkit__ios_record_video, mcp__multi-agent-toolkit__ios_add_media, mcp__multi-agent-toolkit__ios_keychain_reset, mcp__multi-agent-toolkit__ios_get_app_container, mcp__multi-agent-toolkit__ios_erase_device, mcp__multi-agent-toolkit__ios_get_ui_tree, mcp__multi-agent-toolkit__ios_accessibility_audit, mcp__multi-agent-toolkit__ios_accessibility_audit_deep, mcp__multi-agent-toolkit__ios_list_crashes, mcp__multi-agent-toolkit__agent_run_steps, mcp__multi-agent-toolkit__android_list_devices, mcp__multi-agent-toolkit__android_screenshot, mcp__multi-agent-toolkit__android_tap, mcp__multi-agent-toolkit__android_swipe, mcp__multi-agent-toolkit__android_type_text, mcp__multi-agent-toolkit__android_key_event, mcp__multi-agent-toolkit__android_launch_app, mcp__multi-agent-toolkit__android_stop_app, mcp__multi-agent-toolkit__android_list_packages, mcp__multi-agent-toolkit__android_go_home, mcp__multi-agent-toolkit__android_go_back, mcp__multi-agent-toolkit__android_get_ui_tree, mcp__multi-agent-toolkit__android_set_dark_mode, mcp__multi-agent-toolkit__android_set_font_scale, mcp__multi-agent-toolkit__android_set_locale, mcp__multi-agent-toolkit__android_set_location, mcp__multi-agent-toolkit__android_grant_permission, mcp__multi-agent-toolkit__android_revoke_permission, mcp__multi-agent-toolkit__android_record_screen, mcp__multi-agent-toolkit__android_install_apk, mcp__multi-agent-toolkit__android_uninstall_app, mcp__multi-agent-toolkit__android_logcat, mcp__multi-agent-toolkit__android_get_screen_size, mcp__multi-agent-toolkit__android_open_url, mcp__multi-agent-toolkit__android_clear_app_data, mcp__multi-agent-toolkit__android_accessibility_audit, mcp__multi-agent-toolkit__android_list_crashes
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Mobile UI Bug Hunter
|
|
@@ -25,7 +25,7 @@ Auto-detects platform: iOS (simctl) or Android (adb). No external apps needed.
|
|
|
25
25
|
For iOS: use `mcp__multi-agent-toolkit__ios_*` tools
|
|
26
26
|
For Android: use `mcp__multi-agent-toolkit__android_*` tools
|
|
27
27
|
|
|
28
|
-
**Android extras**: `get_ui_tree` returns XML with bounds/text/resource-id (uiautomator dump), `logcat` for crash
|
|
28
|
+
**Android extras**: `get_ui_tree` returns XML with bounds/text/resource-id (uiautomator dump), `logcat` for live output, `list_crashes` for the crash buffer, `go_back` button.
|
|
29
29
|
|
|
30
30
|
## Activation
|
|
31
31
|
|
|
@@ -90,23 +90,34 @@ Call: ios_status_bar (time: "09:41", battery_level: 100)
|
|
|
90
90
|
|
|
91
91
|
### Step 3 - Systematic Screen Exploration
|
|
92
92
|
|
|
93
|
-
For each screen in the app:
|
|
93
|
+
For each screen in the app, read before you look, and batch what you can:
|
|
94
94
|
|
|
95
95
|
```
|
|
96
|
-
Call:
|
|
97
|
-
->
|
|
96
|
+
Call: ios_get_ui_tree (text: labels, frames, traits; ~0.1 s, no image)
|
|
97
|
+
-> pick tap targets from element frames, never by guessing pixels from a picture
|
|
98
|
+
|
|
99
|
+
Call: agent_run_steps (one round trip for a scripted sequence)
|
|
100
|
+
steps: [ {tool: "ios_tap", args: {x, y}}, {wait_ms: 400},
|
|
101
|
+
{tool: "ios_screenshot", args: {path: "<dir>/<screen>.png"}},
|
|
102
|
+
{tool: "ios_get_ui_tree", args: {}} ]
|
|
98
103
|
|
|
99
|
-
Call:
|
|
100
|
-
->
|
|
101
|
-
-> Navigate back: ios_swipe (left edge swipe) or tap back button coordinates
|
|
104
|
+
Call: ios_screenshot (inline, only when a visual judgment is needed)
|
|
105
|
+
-> Claude analyzes the image for bugs (see Bug Detection below)
|
|
102
106
|
```
|
|
103
107
|
|
|
108
|
+
Every inline `ios_screenshot` / `android_screenshot` is an 800 px JPEG (about 60 KB) by
|
|
109
|
+
default; pass `format: "png"` or a larger `max_width` only when a pixel-level look is the
|
|
110
|
+
point. Bulk captures (dark mode, dynamic type, locale sweeps) always go through `path`,
|
|
111
|
+
so a run of 100 screens does not push 100 images through the model: look at the ones
|
|
112
|
+
whose tree or diff changed. A screen whose tree is unchanged after a tap is the
|
|
113
|
+
"button did not respond" finding without a second image.
|
|
114
|
+
|
|
104
115
|
**Navigation strategy:**
|
|
105
116
|
|
|
106
|
-
1.
|
|
107
|
-
2. Tap each tab
|
|
108
|
-
3. On each screen: tap interactive elements -> screenshot
|
|
109
|
-
4. Scroll: `ios_swipe(200, 600, 200, 200)` to scroll down ->
|
|
117
|
+
1. Tree of the initial screen -> tab bar items are the elements with a tab-bar trait or the bottom row of buttons; use their frames
|
|
118
|
+
2. Tap each tab (batched via `agent_run_steps`) -> tree + file capture each
|
|
119
|
+
3. On each screen: tap interactive elements from the tree -> tree after; inline screenshot only where the tree cannot tell (layout, contrast, clipping)
|
|
120
|
+
4. Scroll: `ios_swipe(200, 600, 200, 200)` to scroll down -> tree, capture when new content appeared
|
|
110
121
|
5. Type in text fields: `ios_type_text("test input")`
|
|
111
122
|
|
|
112
123
|
### Step 4 - Variant Testing (based on input argument)
|
|
@@ -131,9 +142,24 @@ Call: ios_set_content_size("medium") // reset
|
|
|
131
142
|
|
|
132
143
|
**"accessibility":**
|
|
133
144
|
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
145
|
+
Two halves per screen, and the report keeps them apart: what the tool measured and what the screenshot suggests.
|
|
146
|
+
|
|
147
|
+
```
|
|
148
|
+
On each screen, after ios_screenshot:
|
|
149
|
+
Call: ios_accessibility_audit (Android: android_accessibility_audit)
|
|
150
|
+
-> findings[]: missing labels / contentDescription, controls a screen reader cannot
|
|
151
|
+
name, tap targets under 44pt (iOS) / 48dp (Android), missing identifiers,
|
|
152
|
+
reading order that does not follow the layout. Each finding becomes a BUG task
|
|
153
|
+
with the element identifier from the tree, not a guess from the pixels.
|
|
154
|
+
-> measurable:false (with a reason) means the tree could not be read (on iOS,
|
|
155
|
+
Simulator.app itself must be running, a booted device is not enough). Record
|
|
156
|
+
the screen as "not audited: <reason>" in the report. Never read it as clean.
|
|
157
|
+
-> Pass `scope: "<prefix>"` to narrow to one screen's identifiers when the tree is large.
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
The screenshot pass still runs on every screen for what the tree cannot see: poor contrast, text readability at default + large sizes, and visual crowding. Report those as screenshot findings, separately from the audit findings.
|
|
161
|
+
|
|
162
|
+
Optional deep pass (iOS only, minutes rather than seconds): when the project has an XCUITest target whose test calls `performAccessibilityAudit()`, run `ios_accessibility_audit_deep` (`scheme`, `project` or `workspace`, optional `test_identifier`) once at the end. It reaches contrast, Dynamic Type, clipped text, traits and hit regions that a tree dump cannot. The result says whether the named test ran; a test that never called the audit reports as "did not run", never as clean. Skip it, and say so in the report, when no such test exists.
|
|
137
163
|
|
|
138
164
|
**"screenshot <lang>":**
|
|
139
165
|
|
|
@@ -172,6 +198,17 @@ no Gate 3 and no Android parity, so the copy is gone rather than kept in sync.
|
|
|
172
198
|
- Run light mode -> all screens
|
|
173
199
|
- Run dark mode -> all screens
|
|
174
200
|
- Run large text -> all screens
|
|
201
|
+
- Crash sweep, after the last screen and before the report:
|
|
202
|
+
|
|
203
|
+
```
|
|
204
|
+
Call: ios_list_crashes (app: <process name>, since_min: <minutes since Step 1 launch>, limit: 20)
|
|
205
|
+
Android: android_list_crashes (lines: 200)
|
|
206
|
+
-> Every report / stack newer than the launch becomes a BUG task with severity Critical,
|
|
207
|
+
the crashed process and the top frame in the description.
|
|
208
|
+
-> Empty result -> "Crashes: none during this run" in the report. Only the bounded window
|
|
209
|
+
counts; older reports on the host are not this run's.
|
|
210
|
+
```
|
|
211
|
+
|
|
175
212
|
- Compile complete report
|
|
176
213
|
|
|
177
214
|
### Step 5 - Bug Detection (on EVERY screenshot)
|
|
@@ -216,16 +253,18 @@ Then output full report:
|
|
|
216
253
|
- **Steps**: 1. Open app -> 2. Tap {X} -> 3. Observe {issue}
|
|
217
254
|
- **Expected**: {correct behavior}
|
|
218
255
|
- **Actual**: {what's wrong}
|
|
256
|
+
- **Spec**: "<quoted acceptance criterion from the analysis doc Section 15 / 20, or: no spec, model expectation>" (<source>)
|
|
257
|
+
- **Evidence**: before=<png path> after=<png path>
|
|
219
258
|
|
|
220
259
|
### BUG-2: ...
|
|
221
260
|
|
|
222
261
|
## Screens Visited ({N})
|
|
223
262
|
|
|
224
|
-
| # | Screen | Light | Dark | Large Text | Bugs |
|
|
225
|
-
| --- | -------- | ----- | ----- | ---------- | ---- |
|
|
226
|
-
| 1 | Home | ok | BUG-1 | ok | 1 |
|
|
227
|
-
| 2 | Login | ok | ok | BUG-2 | 1 |
|
|
228
|
-
| 3 | Settings | ok | ok | ok | 0 |
|
|
263
|
+
| # | Screen | Light | Dark | Large Text | Bugs | Evidence |
|
|
264
|
+
| --- | -------- | ----- | ----- | ---------- | ---- | -------- |
|
|
265
|
+
| 1 | Home | ok | BUG-1 | ok | 1 | 3 |
|
|
266
|
+
| 2 | Login | ok | ok | BUG-2 | 1 | 4 |
|
|
267
|
+
| 3 | Settings | ok | ok | ok | 0 | 2 |
|
|
229
268
|
|
|
230
269
|
## Summary
|
|
231
270
|
|
|
@@ -233,8 +272,13 @@ Then output full report:
|
|
|
233
272
|
- Clean: {N}
|
|
234
273
|
- With bugs: {N}
|
|
235
274
|
- Critical: {N} | Major: {N} | Minor: {N}
|
|
275
|
+
- Accessibility audit (accessibility scenario): {N} audited, {N} not audited (reason per screen), deep pass: {ran / skipped: reason}
|
|
276
|
+
- Crashes (full scenario): {N} during this run
|
|
277
|
+
- Production readiness: FAILED | NEEDS WORK | READY
|
|
236
278
|
```
|
|
237
279
|
|
|
280
|
+
`Evidence` in a bug block is the captures already written to files in Step 4: a tap-driven finding needs both `before` and `after`, a static finding (layout, contrast, dark mode, large text) needs `after` only. The `Evidence` column in Screens Visited is the count of screenshot files written for that screen. Production readiness defaults to FAILED; it is NEEDS WORK when only Minor bugs remain, and READY only when there are zero Critical / Major bugs and every planned screen was visited.
|
|
281
|
+
|
|
238
282
|
Save to: `$HOME/.claude/logs/sim-test/{bundle_id}/{timestamp}.md`
|
|
239
283
|
|
|
240
284
|
### Step 7 - Fix Offer
|
|
@@ -46,7 +46,8 @@
|
|
|
46
46
|
# not to enable, not a problem to report.
|
|
47
47
|
#
|
|
48
48
|
# Exit codes: 0 = inventory produced (or the queried key is present), 1 = queried key
|
|
49
|
-
# absent, 3 = usage
|
|
49
|
+
# absent, 2 = no credential helper on this host (nothing could be probed), 3 = usage
|
|
50
|
+
# error.
|
|
50
51
|
|
|
51
52
|
set -uo pipefail
|
|
52
53
|
|
|
@@ -290,11 +291,23 @@ if [ -n "$QUERY" ]; then
|
|
|
290
291
|
echo "$QUERY: present - can $(capability_of "$QUERY")"
|
|
291
292
|
fi
|
|
292
293
|
exit 0 ;;
|
|
293
|
-
2) echo "$QUERY: no credential helper on this host" >&2; exit
|
|
294
|
+
2) echo "$QUERY: no credential helper on this host - run the pipeline installer (credential-store.sh missing)" >&2; exit 2 ;;
|
|
294
295
|
*) echo "$QUERY: NOT AVAILABLE - onboard it via /multi-agent:setup before relying on it" >&2; exit 1 ;;
|
|
295
296
|
esac
|
|
296
297
|
fi
|
|
297
298
|
|
|
299
|
+
# Without a helper nothing can be probed: every mapped key would read as
|
|
300
|
+
# "mapped-but-missing" and the inventory would send the user to re-onboard
|
|
301
|
+
# tokens that are sitting in the store.
|
|
302
|
+
if [ -z "$STORE" ]; then
|
|
303
|
+
if [ "$MODE" = "json" ]; then
|
|
304
|
+
echo '{"status":"no-backend","reason":"credential-store.sh not found on this host","credentials":[]}'
|
|
305
|
+
else
|
|
306
|
+
echo "no credential helper on this host - run the pipeline installer (credential-store.sh missing)" >&2
|
|
307
|
+
fi
|
|
308
|
+
exit 2
|
|
309
|
+
fi
|
|
310
|
+
|
|
298
311
|
ROWS=""
|
|
299
312
|
while IFS=$'\t' read -r key mapped; do
|
|
300
313
|
[ -z "$key" ] && continue
|
|
@@ -81,12 +81,22 @@ MSG
|
|
|
81
81
|
return 1
|
|
82
82
|
}
|
|
83
83
|
|
|
84
|
-
# When sourced
|
|
85
|
-
#
|
|
86
|
-
#
|
|
87
|
-
|
|
84
|
+
# When sourced, auto-resolve into the caller's environment. When run directly
|
|
85
|
+
# (./credential-store-resolver.sh), print the resolved path or a non-zero exit
|
|
86
|
+
# code with the message above. zsh sources this too (the pipeline's Bash tool
|
|
87
|
+
# is zsh), where BASH_SOURCE is unset and $0 is the sourced file's own name, so
|
|
88
|
+
# the bash test alone always looked like a direct run there.
|
|
89
|
+
_csr_sourced=0
|
|
90
|
+
if [ -n "${ZSH_VERSION:-}" ]; then
|
|
91
|
+
case "${ZSH_EVAL_CONTEXT:-}" in *:file|*:file:*) _csr_sourced=1 ;; esac
|
|
92
|
+
elif [ -n "${BASH_SOURCE[0]:-}" ] && [ "${BASH_SOURCE[0]}" != "$0" ]; then
|
|
93
|
+
_csr_sourced=1
|
|
94
|
+
fi
|
|
95
|
+
if [ "$_csr_sourced" -eq 0 ]; then
|
|
96
|
+
unset _csr_sourced
|
|
88
97
|
resolve_credential_store && echo "$CRED_STORE"
|
|
89
98
|
else
|
|
99
|
+
unset _csr_sourced
|
|
90
100
|
# `|| :` matters, and it is not cosmetic.
|
|
91
101
|
#
|
|
92
102
|
# A sourced file runs in the caller's shell, so a bare failing command at this top
|
|
@@ -164,6 +164,10 @@ do_get() {
|
|
|
164
164
|
return 0
|
|
165
165
|
fi
|
|
166
166
|
audit_lookup "$logical" false
|
|
167
|
+
case "$rc" in
|
|
168
|
+
2) echo "ERR: credential backend unavailable on $PLATFORM" >&2; return 2 ;;
|
|
169
|
+
4) echo "ERR: credential backend error while reading '$logical'" >&2; return 4 ;;
|
|
170
|
+
esac
|
|
167
171
|
return 1
|
|
168
172
|
fi
|
|
169
173
|
local val=""
|
|
@@ -204,8 +208,10 @@ do_set() {
|
|
|
204
208
|
if [ "$val" = "-" ]; then
|
|
205
209
|
val=$(cat)
|
|
206
210
|
fi
|
|
207
|
-
|
|
208
|
-
|
|
211
|
+
# macOS writes go through `security -i` below: the secret travels on stdin
|
|
212
|
+
# and never lands on any argv, which is the property the delegate cannot
|
|
213
|
+
# promise from here.
|
|
214
|
+
if [ "$PLATFORM" != "macos" ] && delegate_to_python; then
|
|
209
215
|
printf '%s' "$val" | python3 "$KEYCHAIN_PY" set "$key" -
|
|
210
216
|
return $?
|
|
211
217
|
fi
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
# extract-conventions.sh
|
|
3
3
|
#
|
|
4
4
|
# Phase 1c helper for /multi-agent:analysis.
|
|
5
|
-
# Scans a repo and emits
|
|
5
|
+
# Scans a repo and emits 13 convention buckets as a single JSON object on stdout.
|
|
6
6
|
#
|
|
7
7
|
# Usage:
|
|
8
8
|
# extract-conventions.sh <repo-path> <platform>
|
|
@@ -202,19 +202,6 @@ files_to_json() {
|
|
|
202
202
|
printf '%s\n' "$input" | head -5 | jq -R . | jq -s .
|
|
203
203
|
}
|
|
204
204
|
|
|
205
|
-
# Top-N suffix frequency from a stream of file basenames.
|
|
206
|
-
# stdin: basenames (one per line)
|
|
207
|
-
# arg1: regex (POSIX ERE) capturing the suffix in group 1
|
|
208
|
-
# stdout: lines "count<TAB>suffix" sorted desc
|
|
209
|
-
suffix_freq() {
|
|
210
|
-
local re="$1"
|
|
211
|
-
grep -Eo "$re" 2>/dev/null \
|
|
212
|
-
| sort \
|
|
213
|
-
| uniq -c \
|
|
214
|
-
| sort -rn \
|
|
215
|
-
| sed -E 's/^ *([0-9]+) +/\1\t/'
|
|
216
|
-
}
|
|
217
|
-
|
|
218
205
|
# Run a bucket function with a soft timeout. We wrap the function in a subshell
|
|
219
206
|
# and use a watchdog so we stay compatible with macOS bash 3.2.
|
|
220
207
|
run_bucket_with_timeout() {
|
|
@@ -48,10 +48,12 @@ HOST_OVERRIDE="${CONFLUENCE_HOST_OVERRIDE:-}"
|
|
|
48
48
|
PAGE_ID=""
|
|
49
49
|
TIMEOUT="${CONFLUENCE_TIMEOUT_SECONDS:-20}"
|
|
50
50
|
|
|
51
|
+
need_value() { [ $# -ge 2 ] || { echo "ERR: $1 needs a value" >&2; exit 4; }; }
|
|
52
|
+
|
|
51
53
|
while [ $# -gt 0 ]; do
|
|
52
54
|
case "$1" in
|
|
53
|
-
--host) HOST_OVERRIDE="$2"; shift 2 ;;
|
|
54
|
-
--page-id) PAGE_ID="$2"; shift 2 ;;
|
|
55
|
+
--host) need_value "$@"; HOST_OVERRIDE="$2"; shift 2 ;;
|
|
56
|
+
--page-id) need_value "$@"; PAGE_ID="$2"; shift 2 ;;
|
|
55
57
|
-h|--help)
|
|
56
58
|
echo "usage: $0 <page-url> | $0 --host <host> --page-id <id>" >&2
|
|
57
59
|
exit 4 ;;
|
|
@@ -220,11 +222,17 @@ PAGE_ID_FINAL="$PAGE_ID_RESOLVED" \
|
|
|
220
222
|
SPACE_KEY_FINAL="$SPACE_KEY" \
|
|
221
223
|
PAGE_JSON_RAW="$PAGE_JSON" \
|
|
222
224
|
python3 - <<'PY'
|
|
223
|
-
import json, os, re, datetime, html
|
|
225
|
+
import json, os, re, sys, datetime, html
|
|
224
226
|
from html.parser import HTMLParser
|
|
225
227
|
|
|
226
228
|
raw = os.environ["PAGE_JSON_RAW"]
|
|
227
|
-
|
|
229
|
+
try:
|
|
230
|
+
data = json.loads(raw)
|
|
231
|
+
except ValueError:
|
|
232
|
+
# A 200 with an HTML body is an SSO login page or a proxy, not the page.
|
|
233
|
+
sys.stderr.write("ERR: Confluence GET for pageId=%s returned a non-JSON body (login page or proxy?)\n"
|
|
234
|
+
% os.environ.get("PAGE_ID_FINAL", ""))
|
|
235
|
+
sys.exit(3)
|
|
228
236
|
|
|
229
237
|
storage = ((data.get("body") or {}).get("storage") or {}).get("value") or ""
|
|
230
238
|
title = data.get("title") or "<untitled>"
|