@mmerterden/multi-agent-pipeline 18.0.0 → 19.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +287 -0
- package/README.md +36 -20
- package/README.tr.md +14 -16
- package/docs/adr/0002-instruction-driven-flag.md +1 -0
- package/docs/adr/0005-lazy-phase-docs.md +11 -1
- package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
- package/docs/adr/0010-own-code-graph.md +1 -0
- package/docs/adr/0014-six-phase-consolidation.md +134 -0
- package/docs/adr/README.md +2 -1
- package/docs/architecture.md +37 -38
- package/docs/best-practices.md +1 -1
- package/docs/ecosystem.md +46 -27
- package/docs/engineering.md +1 -1
- package/docs/facts.json +61 -0
- package/docs/features.md +55 -54
- package/docs/performance.md +5 -5
- package/docs/recovery-guide.md +17 -17
- package/docs/token-budget-history.md +3 -1
- package/index.js +2 -2
- package/install/_codex-agents.mjs +1 -1
- package/install/templates/claude-hooks.json +1 -1
- package/install/templates/codex-instructions.md +1 -1
- package/install/templates/copilot-instructions.md +28 -28
- package/manifest.json +234 -216
- package/package.json +2 -2
- package/pipeline/agents/dev-critic.md +7 -7
- package/pipeline/commands/figma-to-swiftui.md +1 -1
- package/pipeline/commands/multi-agent/SKILL.md +9 -9
- package/pipeline/commands/multi-agent/analysis/SKILL.md +15 -15
- package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/autopilot/SKILL.md +7 -7
- package/pipeline/commands/multi-agent/channels/SKILL.md +15 -15
- package/pipeline/commands/multi-agent/diff-explain/SKILL.md +6 -6
- package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/graph/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/help/SKILL.md +62 -62
- package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/local/SKILL.md +11 -11
- package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +13 -13
- package/pipeline/commands/multi-agent/log/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/manual-test/SKILL.md +9 -9
- package/pipeline/commands/multi-agent/model/SKILL.md +69 -0
- package/pipeline/commands/multi-agent/refactor/SKILL.md +3 -3
- package/pipeline/commands/multi-agent/resume/SKILL.md +4 -4
- package/pipeline/commands/multi-agent/resume-local/SKILL.md +19 -17
- package/pipeline/commands/multi-agent/review/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/review-analysis/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/route-off/SKILL.md +36 -0
- package/pipeline/commands/multi-agent/route-on/SKILL.md +74 -0
- package/pipeline/commands/multi-agent/route-status/SKILL.md +56 -0
- package/pipeline/commands/multi-agent/setup/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/status/SKILL.md +5 -5
- package/pipeline/commands/multi-agent/steer/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/sync/SKILL.md +12 -13
- package/pipeline/commands/multi-agent/test/SKILL.md +1 -1
- package/pipeline/lib/credential-inventory.sh +1 -1
- package/pipeline/lib/fetch-fortify.sh +1 -1
- package/pipeline/lib/model-dispatch.sh +140 -0
- package/pipeline/lib/model-rung.sh +142 -0
- package/pipeline/lib/outbound-gate.mjs +14 -0
- package/pipeline/lib/phase-schema.mjs +88 -0
- package/pipeline/lib/plan-todos.sh +5 -5
- package/pipeline/lib/route-state.sh +161 -0
- package/pipeline/lib/run-paths.sh +2 -2
- package/pipeline/multi-agent-refs/_account-picker.md +1 -1
- package/pipeline/multi-agent-refs/_dev-context.md +6 -6
- package/pipeline/multi-agent-refs/_input-parser.md +1 -1
- package/pipeline/multi-agent-refs/analysis/evidence.md +2 -11
- package/pipeline/multi-agent-refs/analysis/intake.md +7 -7
- package/pipeline/multi-agent-refs/analysis/locked.md +48 -22
- package/pipeline/multi-agent-refs/analysis/redesign.md +1 -1
- package/pipeline/multi-agent-refs/analysis/render.md +10 -10
- package/pipeline/multi-agent-refs/analysis/resolve.md +1 -1
- package/pipeline/multi-agent-refs/analysis/review.md +2 -2
- package/pipeline/multi-agent-refs/analysis/synthesis.md +13 -7
- package/pipeline/multi-agent-refs/analysis-template-corporate.md +9 -9
- package/pipeline/multi-agent-refs/analysis-template.md +19 -19
- package/pipeline/multi-agent-refs/android-guide.md +1 -1
- package/pipeline/multi-agent-refs/audit-guide.md +13 -13
- package/pipeline/multi-agent-refs/channels/issue-comment.md +2 -2
- package/pipeline/multi-agent-refs/channels/jira.md +3 -3
- package/pipeline/multi-agent-refs/channels/pr.md +4 -4
- package/pipeline/multi-agent-refs/channels/wiki.md +1 -1
- package/pipeline/multi-agent-refs/component-dispatch.md +8 -8
- package/pipeline/multi-agent-refs/conventions-defaults.md +2 -2
- package/pipeline/multi-agent-refs/cross-cli-contract.md +31 -6
- package/pipeline/multi-agent-refs/features/analysis-jira.md +1 -1
- package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +4 -4
- package/pipeline/multi-agent-refs/features/code-graph.md +5 -5
- package/pipeline/multi-agent-refs/features/design-conformance.md +1 -1
- package/pipeline/multi-agent-refs/features/dev-critic.md +3 -3
- package/pipeline/multi-agent-refs/features/doctor.md +3 -3
- package/pipeline/multi-agent-refs/features/external-context-injection.md +3 -3
- package/pipeline/multi-agent-refs/features/maturity-followup.md +3 -3
- package/pipeline/multi-agent-refs/features/model-fallback.md +41 -5
- package/pipeline/multi-agent-refs/features/plan-todos.md +1 -1
- package/pipeline/multi-agent-refs/features/repo-map.md +1 -1
- package/pipeline/multi-agent-refs/features/review-delta.md +3 -3
- package/pipeline/multi-agent-refs/features/review-multi-repo.md +2 -2
- package/pipeline/multi-agent-refs/features/scope-check.md +4 -4
- package/pipeline/multi-agent-refs/features/skill-conformance.md +2 -2
- package/pipeline/multi-agent-refs/features/stack-skill-routing.md +1 -1
- package/pipeline/multi-agent-refs/features/url-enrichment.md +1 -1
- package/pipeline/multi-agent-refs/features/verify-by-test.md +4 -4
- package/pipeline/multi-agent-refs/features/visual-evidence.md +19 -19
- package/pipeline/multi-agent-refs/features/worktree-finalize.md +6 -6
- package/pipeline/multi-agent-refs/issue-jira-triad.md +10 -10
- package/pipeline/multi-agent-refs/knowledge.md +11 -11
- package/pipeline/multi-agent-refs/multi-repo-integration-build.md +13 -13
- package/pipeline/multi-agent-refs/payload-contracts.md +8 -8
- package/pipeline/multi-agent-refs/phases/log-format.md +10 -10
- package/pipeline/multi-agent-refs/phases/modes.md +30 -30
- package/pipeline/multi-agent-refs/phases/operations.md +8 -8
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +24 -24
- package/pipeline/multi-agent-refs/phases/phase-1-plan.md +599 -0
- package/pipeline/multi-agent-refs/phases/{phase-3-dev.md → phase-2-dev.md} +129 -49
- package/pipeline/multi-agent-refs/phases/{phase-4-review.md → phase-3-review.md} +225 -107
- package/pipeline/multi-agent-refs/phases/{phase-6-commit.md → phase-4-commit.md} +23 -23
- package/pipeline/multi-agent-refs/phases/{phase-7-report.md → phase-5-report.md} +29 -29
- package/pipeline/multi-agent-refs/phases.md +44 -48
- package/pipeline/multi-agent-refs/picker-contract.md +1 -1
- package/pipeline/multi-agent-refs/progress-contract.md +6 -6
- package/pipeline/multi-agent-refs/readiness-review.md +1 -1
- package/pipeline/multi-agent-refs/rules.md +7 -7
- package/pipeline/multi-agent-refs/swiftui-guide.md +2 -2
- package/pipeline/multi-agent-refs/tracker-contract.md +31 -32
- package/pipeline/multi-agent-refs/wiki-capture.md +14 -14
- package/pipeline/preferences-template.json +9 -1
- package/pipeline/rules/figma-pipeline.md +8 -8
- package/pipeline/rules/outside-the-pipeline.md +1 -1
- package/pipeline/schemas/agent-state.schema.json +50 -50
- package/pipeline/schemas/analysis-output.schema.json +3 -3
- package/pipeline/schemas/analysis-spec.schema.json +2 -2
- package/pipeline/schemas/autopilot-config.schema.json +1 -1
- package/pipeline/schemas/code-graph.schema.json +1 -1
- package/pipeline/schemas/criteria-manifest.schema.json +1 -1
- package/pipeline/schemas/dev-critic-output.schema.json +1 -1
- package/pipeline/schemas/diff-risk.schema.json +1 -1
- package/pipeline/schemas/figma-project-config.schema.json +1 -1
- package/pipeline/schemas/migrations/prefs-2.4.0-to-2.5.0.mjs +2 -2
- package/pipeline/schemas/migrations/prefs-2.6.0-to-2.7.0.mjs +31 -0
- package/pipeline/schemas/migrations/state-2.1.0-to-2.2.0.mjs +129 -0
- package/pipeline/schemas/phases.json +105 -0
- package/pipeline/schemas/plan-todos.schema.json +5 -5
- package/pipeline/schemas/planning-output.schema.json +1 -1
- package/pipeline/schemas/prefs.schema.json +102 -58
- package/pipeline/schemas/reviewer-output.schema.json +3 -3
- package/pipeline/schemas/route-config.schema.json +74 -0
- package/pipeline/schemas/scope-check.schema.json +1 -1
- package/pipeline/schemas/secret-patterns.json +124 -0
- package/pipeline/schemas/test-gap.schema.json +1 -1
- package/pipeline/schemas/token-budget.json +12 -18
- package/pipeline/schemas/triage-output.schema.json +6 -6
- package/pipeline/scripts/README.md +3 -3
- package/pipeline/scripts/_code-graph.mjs +2 -2
- package/pipeline/scripts/_run-paths.mjs +2 -2
- package/pipeline/scripts/_smoke-root.sh +1 -1
- package/pipeline/scripts/aggregate-metrics.mjs +1 -1
- package/pipeline/scripts/build-references.mjs +2 -2
- package/pipeline/scripts/bulk-read.sh +10 -1
- package/pipeline/scripts/capture-flush.sh +8 -8
- package/pipeline/scripts/capture-resume.sh +3 -3
- package/pipeline/scripts/classify-plan-safety.mjs +1 -1
- package/pipeline/scripts/cost-table.json +8 -1
- package/pipeline/scripts/diff-explain.mjs +1 -1
- package/pipeline/scripts/doctor.mjs +3 -3
- package/pipeline/scripts/gc-abandoned.sh +3 -3
- package/pipeline/scripts/gc-tmp.sh +1 -1
- package/pipeline/scripts/gc-worktrees.sh +1 -1
- package/pipeline/scripts/gen-facts.mjs +280 -0
- package/pipeline/scripts/gen-mode-dispatch.mjs +32 -37
- package/pipeline/scripts/gen-ref-toc.mjs +1 -1
- package/pipeline/scripts/graph-report.mjs +1 -1
- package/pipeline/scripts/jira-attach.sh +1 -1
- package/pipeline/scripts/learn-from-transcripts.mjs +1 -1
- package/pipeline/scripts/learning-curve.mjs +2 -2
- package/pipeline/scripts/log-metric.sh +17 -4
- package/pipeline/scripts/memory-save.sh +1 -1
- package/pipeline/scripts/migrate-prefs.mjs +22 -5
- package/pipeline/scripts/phase-banner.sh +20 -20
- package/pipeline/scripts/phase-tracker.sh +12 -12
- package/pipeline/scripts/plan-coverage-gate.mjs +2 -2
- package/pipeline/scripts/pre-commit-check.sh +30 -1
- package/pipeline/scripts/render-agent-log-cost.sh +1 -1
- package/pipeline/scripts/render-work-summary.sh +3 -3
- package/pipeline/scripts/review-file-filter.mjs +1 -1
- package/pipeline/scripts/run-aggregator.mjs +13 -6
- package/pipeline/scripts/run-metrics.mjs +1 -1
- package/pipeline/scripts/runs-index.mjs +11 -1
- package/pipeline/scripts/scan-skills.sh +26 -0
- package/pipeline/scripts/smoke-cross-cli-behavior.sh +6 -6
- package/pipeline/scripts/smoke-schema-validation.sh +26 -7
- package/pipeline/scripts/token-budget-report.mjs +13 -2
- package/pipeline/scripts/triage-memory.mjs +2 -2
- package/pipeline/scripts/validate-analysis-doc.mjs +274 -43
- package/pipeline/scripts/validate-planning.mjs +1 -1
- package/pipeline/scripts/validate-reviewer.mjs +1 -1
- package/pipeline/scripts/validate-state.mjs +45 -5
- package/pipeline/scripts/validate-triage.mjs +3 -3
- package/pipeline/scripts/verify-citations.mjs +1 -1
- package/pipeline/scripts/worktree-finalize.sh +5 -5
- package/pipeline/scripts/write-state.mjs +32 -0
- package/pipeline/skills/.skill-manifest.json +38 -22
- package/pipeline/skills/.skills-index.json +49 -5
- package/pipeline/skills/shared/README.md +10 -6
- package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +8 -8
- package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +8 -8
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +81 -82
- package/pipeline/skills/shared/core/multi-agent-autopilot/SKILL.md +3 -3
- package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +14 -14
- package/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md +5 -5
- package/pipeline/skills/shared/core/multi-agent-graph/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +25 -23
- package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +8 -8
- package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +6 -6
- package/pipeline/skills/shared/core/multi-agent-model/SKILL.md +71 -0
- package/pipeline/skills/shared/core/multi-agent-refactor/SKILL.md +3 -3
- package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +7 -7
- package/pipeline/skills/shared/core/multi-agent-route-off/SKILL.md +39 -0
- package/pipeline/skills/shared/core/multi-agent-route-on/SKILL.md +76 -0
- package/pipeline/skills/shared/core/multi-agent-route-status/SKILL.md +59 -0
- package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +5 -5
- package/pipeline/skills/shared/core/multi-agent-steer/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +6 -5
- package/pipeline/skills/shared/external/NOTICE-swift-ios-skills.md +1 -1
- package/pipeline/skills/shared/external/signal-community/SKILL.md +8 -1
- package/pipeline/skills/skills-index.md +8 -4
- package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +0 -263
- package/pipeline/multi-agent-refs/phases/phase-2-planning.md +0 -344
- package/pipeline/multi-agent-refs/phases/phase-5-test.md +0 -182
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
|
-
description: "Continue already-done LOCAL work through the pipeline tail: Review
|
|
3
|
-
description-tr: "Halihazırda bitmiş LOKAL işi pipeline kuyruğundan geçirir: Review
|
|
2
|
+
description: "Continue already-done LOCAL work through the pipeline tail: Review (with its build gate) → Commit/PR → Report (technical analysis + Jira test-scenario comment). No dev phase. Use when local work is already done and only review, build, commit and reporting remain."
|
|
3
|
+
description-tr: "Halihazırda bitmiş LOKAL işi pipeline kuyruğundan geçirir: Review (build kapısıyla) → Commit/PR → Report (teknik analiz + Jira test-senaryosu yorumu). Dev fazı yok."
|
|
4
4
|
allowed-tools: Agent, Bash, Read, Write, Edit, Glob, Grep, TaskCreate, TaskUpdate, TaskList, TaskGet, AskUserQuestion, WebFetch, WebSearch, Skill
|
|
5
5
|
---
|
|
6
6
|
|
|
@@ -25,7 +25,7 @@ You already did the work locally - wrote code on the current branch and maybe
|
|
|
25
25
|
|
|
26
26
|
```bash
|
|
27
27
|
/multi-agent:resume-local # current branch vs its base; resolve Jira id from branch name
|
|
28
|
-
/multi-agent:resume-local PROJ-12345 # bind to an explicit Jira id for the Phase
|
|
28
|
+
/multi-agent:resume-local PROJ-12345 # bind to an explicit Jira id for the Phase 5 comment
|
|
29
29
|
/multi-agent:resume-local --base develop # override the base branch for the diff
|
|
30
30
|
/multi-agent:resume-local autopilot # no gate prompts: auto-fix blocking findings, auto-PR, auto-comment
|
|
31
31
|
```
|
|
@@ -34,13 +34,15 @@ You already did the work locally - wrote code on the current branch and maybe
|
|
|
34
34
|
|
|
35
35
|
```
|
|
36
36
|
Phase 0: Init → project/branch detect, resolve base + diff (work-already-done), Jira id, state (NO worktree)
|
|
37
|
-
Phase
|
|
38
|
-
|
|
39
|
-
Phase
|
|
40
|
-
Phase
|
|
37
|
+
Phase 3: Review → the Verify gate first (stack-aware build + existing tests; SUCCESS required), then
|
|
38
|
+
parallel review (Fable + Opus + Sonnet) + Fable triage
|
|
39
|
+
Phase 4: Commit → commit remaining local changes + push + open PR if none exists
|
|
40
|
+
Phase 5: Report → technical analysis + Jira comment with test scenarios (channels: Jira / PR / Confluence / Wiki)
|
|
41
41
|
```
|
|
42
42
|
|
|
43
|
-
Phases 1-
|
|
43
|
+
Phases 1-2 (Plan / Dev) are skipped by design - `ship` treats the current branch's local changes as the Phase 2 output.
|
|
44
|
+
|
|
45
|
+
**Why Review runs the build gate here.** Verify is Phase 2 Dev's exit gate since v19.0.0, and this mode has no Dev: the work arrived already written. The gate still has to run, so Review runs it before dispatching reviewers, which is what Review Stage 1 did before the consolidation. Reviewing a branch whose build was never checked is the failure this ordering prevents.
|
|
44
46
|
|
|
45
47
|
## Phase 0 - Context resolution (finish-specific)
|
|
46
48
|
|
|
@@ -52,12 +54,12 @@ Phases 1-3 (Analysis / Planning / Dev) are skipped by design - `ship` treats t
|
|
|
52
54
|
|
|
53
55
|
## Phase execution (reuse the existing phase contracts)
|
|
54
56
|
|
|
55
|
-
- **Phase
|
|
57
|
+
- **Phase 3 Review** - run per `$HOME/.claude/multi-agent-refs/phases/phase-3-review.md` against the resolved diff: deterministic gates (Step 1.x), stack-specific parallel reviewers (Fable + Opus + Sonnet on Claude Code; GPT + Opus + Sonnet on Copilot CLI), Fable triage → `triage.accepted`. Blocking/important accepted findings:
|
|
56
58
|
- interactive: present them and ask (`AskUserQuestion`) whether to fix now (loop back through a minimal Phase-3-style TDD fix) or proceed;
|
|
57
59
|
- `autopilot` (or `prefs.global.resumeLocal.autoFix == true`): auto-fix accepted blocking/important findings, then re-review the fix, before advancing.
|
|
58
|
-
- **Phase
|
|
59
|
-
- **Phase
|
|
60
|
-
- **Phase
|
|
60
|
+
- **Phase 3 Verify gate** - the **automated success gate** (this is what "build+test success" means here; the interactive device user-test is `/multi-agent:manual-test`). Stack-aware: build via `figma-config.build` (iOS scheme / Android gradle / detected backend/web build) and run the existing test suite if present (`swift test` / `xcodebuild test` / `./gradlew test` / `pytest` / `npm test` / `vitest`). Require success to advance; on failure, surface logs and (interactive) stop or (autopilot) attempt a bounded fix loop. **If the repo has no tests, report "no tests present" - never fabricate test results.**
|
|
61
|
+
- **Phase 4 Commit/PR** - per `$HOME/.claude/multi-agent-refs/phases/phase-4-commit.md`: stage + commit any remaining local changes with a conventional message (`{type}(scope): desc [{JIRA_KEY}-{id}]`), push, and open a PR **only if one does not already exist** for the branch. PR body per `$HOME/.claude/multi-agent-refs/rules.md "External System Outputs"` and `$HOME/.claude/rules/git-conventions.md` - `Ref: #N` / `Related: #N`, never `Closes/Fixes/Resolves`; NO AI/bot attribution anywhere.
|
|
62
|
+
- **Phase 5 Report** - per `$HOME/.claude/multi-agent-refs/phases/phase-5-report.md` + `channels.md`: produce the **technical analysis** and **test scenarios**, then post to the configured channels. Default content for `ship`: a Jira **comment** carrying the technical analysis + the test scenarios (and, when the PR was opened, the PR description). Every body runs through the humanizer; bot/tool/AI signatures are FORBIDDEN in comments.
|
|
61
63
|
|
|
62
64
|
## Modes
|
|
63
65
|
|
|
@@ -70,17 +72,17 @@ Before writing anything outward-facing - PR body, Jira comment, Confluence pag
|
|
|
70
72
|
|
|
71
73
|
## Required: Phase Tracker Contract
|
|
72
74
|
|
|
73
|
-
**The phase tracker is required.** Full spec: [`$HOME/.claude/multi-agent-refs/tracker-contract.md`]($HOME/.claude/multi-agent-refs/tracker-contract.md). `ship` registers its active phase set - `0:Init
|
|
75
|
+
**The phase tracker is required.** Full spec: [`$HOME/.claude/multi-agent-refs/tracker-contract.md`]($HOME/.claude/multi-agent-refs/tracker-contract.md). `ship` registers its active phase set - `0:Init 3:Review 4:Commit 5:Report` - but NEVER resets a tracker that an earlier run already built (load-or-continue; contract section "Continuation runs"):
|
|
74
76
|
|
|
75
77
|
```bash
|
|
76
78
|
# Phase 0, first shell call (every CLI). init ONLY when no prior state exists -
|
|
77
|
-
# a task handed off from Phase
|
|
78
|
-
# phase 0-
|
|
79
|
+
# a task handed off from the user test inside Phase 3 ("awaiting local test")
|
|
80
|
+
# keeps its full phase 0-2 history (elapsed, tokens, USD).
|
|
79
81
|
STATE="$HOME/.claude/logs/multi-agent/${TASK_ID}/tracker-state.json"
|
|
80
82
|
if [ ! -f "$STATE" ]; then
|
|
81
83
|
bash $HOME/.claude/scripts/phase-tracker.sh init "$TASK_ID"
|
|
82
84
|
fi
|
|
83
|
-
for p in "0:Init" "
|
|
85
|
+
for p in "0:Init" "3:Review" "4:Commit" "5:Report"; do
|
|
84
86
|
bash $HOME/.claude/scripts/phase-tracker.sh add "${p%%:*}" "${p#*:}" # idempotent: existing phases keep their history
|
|
85
87
|
done
|
|
86
88
|
bash $HOME/.claude/scripts/phase-tracker.sh update 0 in_progress
|
|
@@ -91,7 +93,7 @@ bash $HOME/.claude/scripts/phase-tracker.sh update <N> in_progress|completed|fai
|
|
|
91
93
|
bash $HOME/.claude/scripts/phase-tracker.sh tokens <N> <in> <out> [cached]
|
|
92
94
|
```
|
|
93
95
|
|
|
94
|
-
**Continuation path (prior state existed):** (1) if Phase
|
|
96
|
+
**Continuation path (prior state existed):** (1) if Phase 3 was left `in_progress` with `Now: awaiting local test (user)`, mark it `update 3 completed` + `meta 3 Result "local test done (user)"` before finish's own work (finish re-opens it with `update 3 in_progress` when its Verify gate runs; elapsed keeps the original `started_at`, which is expected); (2) print ONE line in `outputLanguage` summarizing the inherited history, e.g. `Continuing PROJ-12345: phases 0-3 finished earlier (12m, 38.4k tok, ~$0.74)` (USD via `phase-tracker.sh cost total`); (3) `render`.
|
|
95
97
|
|
|
96
98
|
### Visual channel - Claude Code (native TaskList widget, required)
|
|
97
99
|
|
|
@@ -194,7 +194,7 @@ Every host runs three reviewers; only the second slot differs, because GPT-5.4 e
|
|
|
194
194
|
|
|
195
195
|
With the `fable` rung disabled by prefs the Claude Code panel is two reviewers (Opus + Sonnet); see `$HOME/.claude/multi-agent-refs/features/model-fallback.md`.
|
|
196
196
|
|
|
197
|
-
Each reviewer receives the diff, the module review guides from Step 2b (when any were found), plus the standard reviewer system prompt (see `$HOME/.claude/multi-agent-refs/phases/phase-
|
|
197
|
+
Each reviewer receives the diff, the module review guides from Step 2b (when any were found), plus the standard reviewer system prompt (see `$HOME/.claude/multi-agent-refs/phases/phase-3-review.md` for the prompt contract). Output: structured `findings[]` per reviewer.
|
|
198
198
|
|
|
199
199
|
### 4. Store-compliance cross-reference
|
|
200
200
|
|
|
@@ -312,7 +312,7 @@ Per the `$HOME/.claude/multi-agent-refs/tracker-contract.md` tokens contract, af
|
|
|
312
312
|
bash $HOME/.claude/scripts/phase-tracker.sh tokens 4 <input_count> <output_count>
|
|
313
313
|
```
|
|
314
314
|
|
|
315
|
-
Standalone `/multi-agent:review` runs use phase id
|
|
315
|
+
Standalone `/multi-agent:review` runs use phase id 3 (matches Phase 3 Review in the full pipeline) so cost aggregation stays consistent.
|
|
316
316
|
|
|
317
317
|
## Input matrix
|
|
318
318
|
|
|
@@ -23,7 +23,7 @@ Read `$HOME/.claude/multi-agent-refs/analysis/review.md` and execute it:
|
|
|
23
23
|
|
|
24
24
|
## Why it cites Locked decisions
|
|
25
25
|
|
|
26
|
-
The analysis flow already declares 36 Locked decisions and two deterministic validators. A reviewer that says "I would have written this differently" gives the author nothing to act on; one that says "Locked
|
|
26
|
+
The analysis flow already declares 36 Locked decisions and two deterministic validators. A reviewer that says "I would have written this differently" gives the author nothing to act on; one that says "Locked 33: the Confluence page is in the evidence record but not in Section 21" gives them a fix and a reason. Findings that map to no rule are still allowed, but they are marked as judgement, not dressed up as a violation.
|
|
27
27
|
|
|
28
28
|
## What it never does
|
|
29
29
|
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: "Disable model routing and keep its rules, so turning it back on does not re-ask for the same configuration. Use when asked to turn model routing off."
|
|
3
|
+
description-tr: "Model yönlendirmesini kapatır ve kurallarını saklar; tekrar açıldığında aynı yapılandırma yeniden sorulmaz."
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# multi-agent route-off - disarm routing, keep the configuration
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
bash "$HOME/.claude/lib/route-state.sh" off
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
Sets `prefs.global.modelRouting.enabled` to `false`. Dispatch returns to the
|
|
13
|
+
plain ladder immediately: every persona takes its `preferredModel`, and the
|
|
14
|
+
`modelFallback` rules are the only thing that can move it.
|
|
15
|
+
|
|
16
|
+
## The rules are not deleted
|
|
17
|
+
|
|
18
|
+
`strategy`, `scope`, `rules[]` and `budgetCeilingUsd` all survive. `route-on`
|
|
19
|
+
brings back exactly what was configured, without asking again.
|
|
20
|
+
|
|
21
|
+
This is the same contract `autopilot-off` follows with its repo selection, for
|
|
22
|
+
the same reason: a command named after a toggle that quietly discards
|
|
23
|
+
configuration is a destructive action in disguise. To actually remove the rules,
|
|
24
|
+
replace them with an empty array:
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
echo '[]' > /tmp/none.json
|
|
28
|
+
bash "$HOME/.claude/lib/route-state.sh" set-rules /tmp/none.json
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## What stays behind
|
|
32
|
+
|
|
33
|
+
Routing decisions already written to the cost ledger stay there. They are a
|
|
34
|
+
record of what happened on past runs, and deleting them would make a run's cost
|
|
35
|
+
unexplainable after the fact - which is the one thing the ledger exists to
|
|
36
|
+
prevent.
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: "Enable policy-driven model routing: pick a strategy, a scope, and the rules that say which rung a call lands on. Use when asked to turn model routing on."
|
|
3
|
+
description-tr: "Politika güdümlü model yönlendirmesini açar: strateji, kapsam ve hangi çağrının hangi basamağa düşeceğini söyleyen kuralları sorar."
|
|
4
|
+
argument-hint: "[--strategy=manual|task-fit|cost-ceiling] [--scope=subagent,bulk-read,research]"
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
# multi-agent route-on - arm model routing
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
bash "$HOME/.claude/lib/route-state.sh" on ${ARGUMENTS}
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
Routing ships **off**. This is the command that turns it on, and it writes to
|
|
14
|
+
`prefs.global.modelRouting`, validated against the repo's `schemas/route-config.schema.json`.
|
|
15
|
+
|
|
16
|
+
## What a rule is
|
|
17
|
+
|
|
18
|
+
```json
|
|
19
|
+
{ "when": { "persona": "code-reviewer" }, "prefer": ["opus", "sonnet"] }
|
|
20
|
+
{ "when": { "phase": 2 }, "prefer": ["sonnet", "haiku"] }
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Ordered, first match wins. `when` matches on `persona`, `phase` (0..5) or
|
|
24
|
+
`taskKind`; `prefer` lists rungs in descending preference.
|
|
25
|
+
|
|
26
|
+
**Rung names are the contract; model ids are not.** `opus` is a rung here and a
|
|
27
|
+
model id in `cost-table.json`, and the second can change without anyone editing
|
|
28
|
+
a rule. Writing `claude-opus-5` into a rule pins a decision to a string that will
|
|
29
|
+
go stale.
|
|
30
|
+
|
|
31
|
+
Set rules with:
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
bash "$HOME/.claude/lib/route-state.sh" set-rules path/to/rules.json
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Strategies
|
|
38
|
+
|
|
39
|
+
| Strategy | What it does |
|
|
40
|
+
|---|---|
|
|
41
|
+
| `manual` | only the explicit rules apply, nothing is inferred. The default, because a router that guesses is a router nobody can predict |
|
|
42
|
+
| `task-fit` | a rule may match on `taskKind`, and the cheapest rung clearing it is chosen |
|
|
43
|
+
| `cost-ceiling` | rungs downgrade as the run approaches `budgetCeilingUsd` |
|
|
44
|
+
|
|
45
|
+
## Scope, and the value that is not in it
|
|
46
|
+
|
|
47
|
+
`scope` names the call sites routing may act on: `subagent`, `bulk-read`,
|
|
48
|
+
`research`. Every one of them is a call **this pipeline makes itself**.
|
|
49
|
+
|
|
50
|
+
There is no `host-session` value, and that absence is enforced by that schema
|
|
51
|
+
rather than written as advice. Routing a host session means rewriting the CLI's
|
|
52
|
+
base URL to point at a local gateway - which sends the user's *entire* session
|
|
53
|
+
through a third layer, including work that has nothing to do with this pipeline,
|
|
54
|
+
breaks the subscription's auth model, and silently changes which model answered.
|
|
55
|
+
Passing `--scope=host-session` is refused with that reason, not ignored.
|
|
56
|
+
|
|
57
|
+
## The limit this command prints every time
|
|
58
|
+
|
|
59
|
+
On Claude Code a subagent cannot be dispatched to a non-Anthropic model: subagent
|
|
60
|
+
dispatch belongs to the host, not to us. So Phase 1, 2 and 3 personas stay inside
|
|
61
|
+
the Anthropic ladder no matter what the rules say, and external providers apply
|
|
62
|
+
only at `bulk-read` and `research`, where the pipeline makes the HTTP call.
|
|
63
|
+
|
|
64
|
+
This is printed by `route-status` on every invocation instead of living in a doc,
|
|
65
|
+
because the question it answers - "routing is on, why is the reviewer still on
|
|
66
|
+
Opus" - otherwise arrives days later as a bug report.
|
|
67
|
+
|
|
68
|
+
## Related
|
|
69
|
+
|
|
70
|
+
- `/multi-agent:route-off` - disables routing and **keeps** the rules
|
|
71
|
+
- `/multi-agent:route-status` - what is active, and what it costs
|
|
72
|
+
- `/multi-agent:model` - whether the top rung exists at all. That is a different
|
|
73
|
+
question: this command decides which rung a call picks, that one decides
|
|
74
|
+
whether the top one is in play
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: "Report model routing: whether it is armed, which rule applies where, which rung the last dispatches took, and what this run has cost. Use when asked which model is being used or why."
|
|
3
|
+
description-tr: "Model yönlendirmesinin durumunu raporlar: açık mı, hangi kural nerede geçerli, son çağrılar hangi basamağa gitti, bu koşu ne tuttu."
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# multi-agent route-status - what is actually routing
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
bash "$HOME/.claude/lib/route-state.sh" status
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
Reports the stored policy, then the part that matters more: **what it can and
|
|
13
|
+
cannot reach.**
|
|
14
|
+
|
|
15
|
+
## Three states, told apart
|
|
16
|
+
|
|
17
|
+
| Output | Meaning |
|
|
18
|
+
|---|---|
|
|
19
|
+
| `routing: false`, rules present | configured and disarmed. `route-on` restores it as-is; nothing is lost |
|
|
20
|
+
| `routing: true`, `rules: 0` | armed with nothing to match. Not an error - a configuration state, and the most common "I turned it on and nothing changed" |
|
|
21
|
+
| `routing: true` with rules listed | live. Each rule is printed as `when <key>=<value> -> rung > rung` |
|
|
22
|
+
|
|
23
|
+
A disabled router with rules is deliberately not reported as "off" alone: that
|
|
24
|
+
reads as "unconfigured" and sends the user through `route-on`'s questions a
|
|
25
|
+
second time.
|
|
26
|
+
|
|
27
|
+
## The limit, printed every time
|
|
28
|
+
|
|
29
|
+
On Claude Code a subagent cannot be sent to a non-Anthropic model. Subagent
|
|
30
|
+
dispatch belongs to the host; the pipeline asks for a persona and the host
|
|
31
|
+
decides what answers. So Phase 1, 2 and 3 personas stay inside the Anthropic
|
|
32
|
+
ladder whatever the rules say, and an external provider is only reachable where
|
|
33
|
+
the pipeline makes the HTTP call itself - `bulk-read.sh` and `research_ask`.
|
|
34
|
+
|
|
35
|
+
This paragraph is output, not documentation, because the alternative is the
|
|
36
|
+
question arriving later as "routing is on but the reviewer is still on Opus, is
|
|
37
|
+
it broken". It is not broken; it is the seam.
|
|
38
|
+
|
|
39
|
+
## Where decisions are recorded
|
|
40
|
+
|
|
41
|
+
With `recordDecisions: true` (the default) every routing decision is written to
|
|
42
|
+
the cost ledger: which rule matched, which rung it chose, and why. That is what
|
|
43
|
+
makes a run's cost explainable after it finished - a router whose choices are not
|
|
44
|
+
recorded cannot be audited, and the cost question always arrives after the run,
|
|
45
|
+
never during it.
|
|
46
|
+
|
|
47
|
+
Per-run cost and the model breakdown come from the same ledger:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
node "$HOME/.claude/scripts/token-budget-report.mjs" --json
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
## Related
|
|
54
|
+
|
|
55
|
+
- `/multi-agent:route-on` / `/multi-agent:route-off` - arm and disarm
|
|
56
|
+
- `/multi-agent:model` - whether the top rung exists at all
|
|
@@ -558,7 +558,7 @@ Re-run scan from Step 1. Show final status:
|
|
|
558
558
|
All tokens present. Pipeline ready to use.
|
|
559
559
|
|
|
560
560
|
Optional per-project features (configured on first use, nothing to do now):
|
|
561
|
-
• Phase
|
|
561
|
+
• Phase 5 Report Step 2 Wiki - auto-generates component wiki pages + Figma
|
|
562
562
|
screenshots. Activates when (a) task is a component AND (b) a Figma token
|
|
563
563
|
is in Keychain. Four adapters supported: submodule / in-repo / github-wiki
|
|
564
564
|
/ separate-repo. First run asks: use auto-detected path, use a custom
|
|
@@ -803,7 +803,7 @@ All tokens are optional in the sense that every service can be answered with Ski
|
|
|
803
803
|
Offer to merge `$HOME/.claude/templates/claude-hooks.json`: three `PreToolUse` gates that block on a non-zero exit (secret scan, agent-guard, read-size) plus three capture hooks that block nothing (`PreCompact`, `SessionEnd`, `SessionStart`). What each does: `$HOME/.claude/multi-agent-refs/picker-contract.md`.
|
|
804
804
|
|
|
805
805
|
- Ask (picker): "Install the pipeline's hooks into `~/.claude/settings.json`?" Default Yes.
|
|
806
|
-
- On Yes, deep-merge EVERY event in the template's `hooks` object, not `PreToolUse` alone - merging one event silently drops the capture hooks, and a run killed before Phase
|
|
806
|
+
- On Yes, deep-merge EVERY event in the template's `hooks` object, not `PreToolUse` alone - merging one event silently drops the capture hooks, and a run killed before Phase 5 then loses its findings exactly as it did before they existed. Preserve existing hooks; never duplicate a matcher already calling the same script.
|
|
807
807
|
- Say what the merge does NOT cover: only the three gates need no run-specific arguments, so only they are hookable; the rest are phase-enforced.
|
|
808
808
|
- Say what it does not turn on: the read-size gate is inert until `prefs.global.bulkRead.mode` is set. Recommend `observe` first. Why, and the Phase 3 exemption: `$HOME/.claude/multi-agent-refs/picker-contract.md`.
|
|
809
809
|
|
|
@@ -19,7 +19,7 @@ Show every active and completed task as a table.
|
|
|
19
19
|
|
|
20
20
|
`runs-index.mjs` resolves through `lib/run-paths.sh` / `scripts/_run-paths.mjs`,
|
|
21
21
|
so it sees both directory layouts (`<project>/<id>/` and the flat `<id>/`),
|
|
22
|
-
the salvaged `artifacts/` copy Phase
|
|
22
|
+
the salvaged `artifacts/` copy Phase 4 leaves behind, and every spelling of a
|
|
23
23
|
task id - and it counts a run that exists in both layouts once. Do NOT
|
|
24
24
|
re-scan the tree by hand: the earlier instruction here listed three
|
|
25
25
|
hard-coded `.worktrees/` paths and a single `find` depth, and on a real
|
|
@@ -33,7 +33,7 @@ Show every active and completed task as a table.
|
|
|
33
33
|
2. **Fields per run** (already in the output): `taskId`, `project`, `branch`,
|
|
34
34
|
`currentPhase`, `status`, `startedAt`, `worktreePath`, `prUrl`, `autopilot`,
|
|
35
35
|
`phases[]`, `tokens`, `estUsd`, `group`, plus `duplicateOf` when the run also
|
|
36
|
-
exists in the other layout and `salvaged` when its state is the Phase
|
|
36
|
+
exists in the other layout and `salvaged` when its state is the Phase 4 copy.
|
|
37
37
|
|
|
38
38
|
3. **Groups are computed, not judged.** `runs-index.mjs` assigns `group` by the
|
|
39
39
|
table below; report what it returns rather than re-deriving it. `in_progress`
|
|
@@ -42,7 +42,7 @@ Show every active and completed task as a table.
|
|
|
42
42
|
|
|
43
43
|
| Group | Test | Action offered |
|
|
44
44
|
|---|---|---|
|
|
45
|
-
| `waiting` - Waiting on you | `status == "awaiting_input"`, or a `pr.url`, or phase >=
|
|
45
|
+
| `waiting` - Waiting on you | `status == "awaiting_input"`, or a `pr.url`, or phase >= 4 | `resume #N` - the work landed, it needs your answer |
|
|
46
46
|
| `stopped` - Stopped mid-development | anything else past phase 0 | `resume #N` or `kill #N` |
|
|
47
47
|
| `question` - Left at a question | phase 0 | `garbage-collect --abandoned` - nothing was built |
|
|
48
48
|
| `unknown` - Status not recorded | no `status` field | say so; offer nothing |
|
|
@@ -57,8 +57,8 @@ Show every active and completed task as a table.
|
|
|
57
57
|
|
|
58
58
|
| ID | Jira/Task | Branch | Phase | Status | Duration |
|
|
59
59
|
|----|-----------|--------|-------|--------|----------|
|
|
60
|
-
| #1 | {JIRA-KEY}-12345 | feature/{JIRA-KEY}-12345-... |
|
|
61
|
-
| #3 | {JIRA-KEY}-12345 | feature/{JIRA-KEY}-12345-... |
|
|
60
|
+
| #1 | {JIRA-KEY}-12345 | feature/{JIRA-KEY}-12345-... | 5/5 DONE | ✅ Complete | 12m |
|
|
61
|
+
| #3 | {JIRA-KEY}-12345 | feature/{JIRA-KEY}-12345-... | 5/5 DONE | ✅ Complete | 8m |
|
|
62
62
|
|
|
63
63
|
💡 log #1 | resume #N | kill #N
|
|
64
64
|
```
|
|
@@ -41,7 +41,7 @@ is being replaced.
|
|
|
41
41
|
| `paused` / `failed` | Say the task is not running, and that `/multi-agent:resume #N` will re-enter with the instruction applied at that phase's entry. Queue it. |
|
|
42
42
|
| `complete` | Refuse. Nothing will read it. Point at `/multi-agent` for a follow-up run. |
|
|
43
43
|
|
|
44
|
-
`currentPhase` is 7 and status is `in_progress` → warn that Phase
|
|
44
|
+
`currentPhase` is 7 and status is `in_progress` → warn that Phase 5 is the
|
|
45
45
|
last one, so an instruction queued now may never be consumed.
|
|
46
46
|
|
|
47
47
|
3. **Read the instruction** - from the argument, or ask for it when the
|
|
@@ -52,7 +52,7 @@ is being replaced.
|
|
|
52
52
|
4. **Show what will be queued, and ask**:
|
|
53
53
|
|
|
54
54
|
```
|
|
55
|
-
Steer #3 ({JIRA-KEY}-12345, Phase
|
|
55
|
+
Steer #3 ({JIRA-KEY}-12345, Phase 2 Dev, in_progress)
|
|
56
56
|
|
|
57
57
|
"the field is called web, not frontend"
|
|
58
58
|
|
|
@@ -65,8 +65,8 @@ Run every step automatically:
|
|
|
65
65
|
```
|
|
66
66
|
Step 0: DOCTOR doctor.mjs - exit 2 or 4 stops the sync
|
|
67
67
|
Step 1.5: DETECT Compare timestamps, find stale targets
|
|
68
|
-
Step 2: COPILOT Claude Code -> Copilot CLI (instructions +
|
|
69
|
-
Step 2b: CODEX Claude Code -> Codex CLI (1 router skill +
|
|
68
|
+
Step 2: COPILOT Claude Code -> Copilot CLI (instructions + 60 sub-command skills)
|
|
69
|
+
Step 2b: CODEX Claude Code -> Codex CLI (1 router skill + 60 specs as refs + 8 agent TOML)
|
|
70
70
|
Step 3: REPO Claude Code -> pipeline repo (genericized, personal data scrub, bash -n on all sh)
|
|
71
71
|
Step 3c: PLUGINS pipeline shared/external -> multi-agent-plugins marketplace (rebuild knowledge/,
|
|
72
72
|
bump changed plugins' patch version, commit + push the plugins repo)
|
|
@@ -494,19 +494,18 @@ same 56 specs as reference files rather than as peer skills, via Step 2b - see
|
|
|
494
494
|
|-------------|-------------|
|
|
495
495
|
| `~/.claude/commands/multi-agent/{cmd}/SKILL.md` | `~/.copilot/skills/multi-agent-{cmd}/SKILL.md` |
|
|
496
496
|
|
|
497
|
-
**
|
|
497
|
+
**60 commands are synced** (canonical inventory - must match `cross-cli-contract.md` section 1; drift = contract violation):
|
|
498
498
|
|
|
499
499
|
```
|
|
500
|
-
analysis, analysis-jira, analysis-resolve, autopilot, autopilot-off,
|
|
501
|
-
autopilot-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
uninstall, update
|
|
500
|
+
analysis, analysis-jira, analysis-resolve, autopilot, autopilot-off, autopilot-on,
|
|
501
|
+
autopilot-status, build-optimize, channels, complaint-analysis, create-jira,
|
|
502
|
+
design-check, diff-explain, doctor, feedback, forget, garbage-collect, graph, help,
|
|
503
|
+
ios-coding-standard, issue, jira, kill, language, local, local-autopilot, log,
|
|
504
|
+
manual-test, model, prune-logs, prune-prompts, purge, refactor, resume, resume-local,
|
|
505
|
+
review, review-analysis, review-issue, review-jira, route-off, route-on, route-status,
|
|
506
|
+
routines, save, scan, search, setup, stack, status, steer, store-ready, sync, test,
|
|
507
|
+
test-accessibility, test-dark-mode, test-dynamic-type, test-screenshots,
|
|
508
|
+
testflight-validation, uninstall, update
|
|
510
509
|
```
|
|
511
510
|
|
|
512
511
|
**NOT synced**: `$HOME/.claude/multi-agent-refs/*` - lazy-load references, Claude Code specific
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
description: "UI Bug Hunter - iOS Simulator (simctl) + Android Emulator (adb). Auto-detects platform. Screenshot + tap + analyze on the booted device. The Phase 5 Manual Test flow lives at /multi-agent:manual-test. Use when a running app should be driven on a simulator or emulator to hunt UI bugs."
|
|
3
|
-
description-tr: "UI Bug Hunter - iOS Simulator (simctl) + Android Emulator (adb). Platformu otomatik algılar. Açık cihazda screenshot + tap + analiz. Faz
|
|
3
|
+
description-tr: "UI Bug Hunter - iOS Simulator (simctl) + Android Emulator (adb). Platformu otomatik algılar. Açık cihazda screenshot + tap + analiz. Faz 3 Manuel Test akışı /multi-agent:manual-test'te."
|
|
4
4
|
argument-hint: '[scenario] - e.g. "dark mode" | "accessibility" | "dynamic type" | "screenshot tr" | "store-ready" | (empty = full sweep)'
|
|
5
5
|
---
|
|
6
6
|
|
|
@@ -345,7 +345,7 @@ capability_of() {
|
|
|
345
345
|
esac
|
|
346
346
|
fi
|
|
347
347
|
case "$1" in
|
|
348
|
-
jira) echo "read the ticket, its comments and its linked issues; post the Phase
|
|
348
|
+
jira) echo "read the ticket, its comments and its linked issues; post the Phase 5 comment" ;;
|
|
349
349
|
bitbucket_token) echo "read the repo, open and update pull requests" ;;
|
|
350
350
|
bitbucket_user) echo "identify the PR author (paired with bitbucket_token)" ;;
|
|
351
351
|
github) echo "read issues, open pull requests, read Actions runs" ;;
|
|
@@ -49,7 +49,7 @@
|
|
|
49
49
|
# 6 not configured (prefs.global.hosts.fortify empty, or an instance-id-only
|
|
50
50
|
# lookup with no prefs.global.fortify.versionIds to search)
|
|
51
51
|
#
|
|
52
|
-
# Phase
|
|
52
|
+
# Phase 3 review gate contract:
|
|
53
53
|
# Critical > 0 → blocking=true, reason="critical-findings"
|
|
54
54
|
# High > 0 → blocking=false, reason="high-findings-warning"
|
|
55
55
|
# else → blocking=false, reason="clean"
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# model-dispatch.sh - which rung answers this call.
|
|
3
|
+
#
|
|
4
|
+
# The one place `prefs.global.modelRouting` turns into a rung. Every call site
|
|
5
|
+
# that may be routed asks here, so the policy has a single implementation and
|
|
6
|
+
# `/multi-agent:route-status` describes something real.
|
|
7
|
+
#
|
|
8
|
+
# Usage:
|
|
9
|
+
# model-dispatch.sh <call-site> [--persona P] [--phase N] [--task-kind K]
|
|
10
|
+
# [--default RUNG] [--task-id ID]
|
|
11
|
+
#
|
|
12
|
+
# <call-site> is one of the `scope` members: subagent | bulk-read | research.
|
|
13
|
+
# stdout is a single rung NAME. Rung names are the contract; the model id behind
|
|
14
|
+
# one lives in cost-table.json and moves without a config edit.
|
|
15
|
+
#
|
|
16
|
+
# Exit is always 0 and stdout is always a usable rung. A router that can fail
|
|
17
|
+
# turns every call site into a place the run can die, for a feature that ships
|
|
18
|
+
# disabled; when anything is missing, unparseable or out of scope, the caller's
|
|
19
|
+
# default comes back unchanged.
|
|
20
|
+
#
|
|
21
|
+
# Two layers, and the difference is the safety argument the schema spells out:
|
|
22
|
+
#
|
|
23
|
+
# Layer 1 an Anthropic rung. Reachable from every call site, because it
|
|
24
|
+
# writes to a seam that already exists and opens no new network path.
|
|
25
|
+
# Layer 2 a non-Anthropic rung. Reachable ONLY from a call site the pipeline
|
|
26
|
+
# makes itself. A subagent is dispatched by the HOST, so a rule that
|
|
27
|
+
# prefers an external rung for a subagent cannot be honoured - and
|
|
28
|
+
# this script says so on stderr rather than pretending it was.
|
|
29
|
+
|
|
30
|
+
set -uo pipefail
|
|
31
|
+
|
|
32
|
+
CALL_SITE="${1:-}"
|
|
33
|
+
shift || true
|
|
34
|
+
|
|
35
|
+
PERSONA=""; PHASE=""; TASK_KIND=""; DEFAULT_RUNG=""; TASK_ID="${MULTI_AGENT_TASK_ID:-unknown}"
|
|
36
|
+
while [ $# -gt 0 ]; do
|
|
37
|
+
case "$1" in
|
|
38
|
+
--persona) PERSONA="${2:-}"; shift 2 ;;
|
|
39
|
+
--phase) PHASE="${2:-}"; shift 2 ;;
|
|
40
|
+
--task-kind) TASK_KIND="${2:-}"; shift 2 ;;
|
|
41
|
+
--default) DEFAULT_RUNG="${2:-}"; shift 2 ;;
|
|
42
|
+
--task-id) TASK_ID="${2:-}"; shift 2 ;;
|
|
43
|
+
*) shift ;;
|
|
44
|
+
esac
|
|
45
|
+
done
|
|
46
|
+
|
|
47
|
+
emit() { printf '%s\n' "$1"; exit 0; }
|
|
48
|
+
|
|
49
|
+
case "$CALL_SITE" in
|
|
50
|
+
subagent|bulk-read|research) ;;
|
|
51
|
+
*) emit "$DEFAULT_RUNG" ;;
|
|
52
|
+
esac
|
|
53
|
+
|
|
54
|
+
[ -n "$DEFAULT_RUNG" ] || DEFAULT_RUNG="sonnet"
|
|
55
|
+
|
|
56
|
+
command -v jq >/dev/null 2>&1 || emit "$DEFAULT_RUNG"
|
|
57
|
+
|
|
58
|
+
PREFS=""
|
|
59
|
+
for candidate in \
|
|
60
|
+
"${MULTI_AGENT_PREFS:-}" \
|
|
61
|
+
"$HOME/.claude/multi-agent-preferences.json" \
|
|
62
|
+
"$HOME/.config/multi-agent-pipeline/multi-agent-preferences.json"
|
|
63
|
+
do
|
|
64
|
+
[ -n "$candidate" ] && [ -f "$candidate" ] && { PREFS="$candidate"; break; }
|
|
65
|
+
done
|
|
66
|
+
[ -n "$PREFS" ] || emit "$DEFAULT_RUNG"
|
|
67
|
+
|
|
68
|
+
ENABLED=$(jq -r '.global.modelRouting.enabled // false' "$PREFS" 2>/dev/null) || emit "$DEFAULT_RUNG"
|
|
69
|
+
[ "$ENABLED" = "true" ] || emit "$DEFAULT_RUNG"
|
|
70
|
+
|
|
71
|
+
# Out of scope is not a failure and not a warning. The user named the call sites
|
|
72
|
+
# routing may touch; the ones they left out keep their existing behaviour, which
|
|
73
|
+
# is the point of naming them.
|
|
74
|
+
IN_SCOPE=$(jq -r --arg cs "$CALL_SITE" \
|
|
75
|
+
'((.global.modelRouting.scope // ["subagent"]) | index($cs)) != null' "$PREFS" 2>/dev/null)
|
|
76
|
+
[ "$IN_SCOPE" = "true" ] || emit "$DEFAULT_RUNG"
|
|
77
|
+
|
|
78
|
+
# First rule whose every stated condition matches. A `when` with three keys has
|
|
79
|
+
# to match on all three - the schema requires at least one, so an always-matching
|
|
80
|
+
# rule cannot be written by accident.
|
|
81
|
+
MATCH=$(jq -r \
|
|
82
|
+
--arg persona "$PERSONA" --arg phase "$PHASE" --arg kind "$TASK_KIND" \
|
|
83
|
+
'[ (.global.modelRouting.rules // [])[]
|
|
84
|
+
| select(
|
|
85
|
+
((.when.persona // null) as $p | $p == null or $p == $persona)
|
|
86
|
+
and ((.when.phase // null) as $h | $h == null or ($phase != "" and ($h | tostring) == $phase))
|
|
87
|
+
and ((.when.taskKind // null) as $k | $k == null or $k == $kind)
|
|
88
|
+
)
|
|
89
|
+
] | first | (.prefer // []) | join(" ")' "$PREFS" 2>/dev/null) || emit "$DEFAULT_RUNG"
|
|
90
|
+
[ -n "$MATCH" ] && [ "$MATCH" != "null" ] || emit "$DEFAULT_RUNG"
|
|
91
|
+
|
|
92
|
+
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
93
|
+
COST_TABLE="$HERE/../scripts/cost-table.json"
|
|
94
|
+
|
|
95
|
+
# The fable rung answers only when the fable switch is on. Routing is a policy
|
|
96
|
+
# over what is available; it is not a second way to turn a rung on, or
|
|
97
|
+
# `/multi-agent:model off` would stop meaning anything the moment a rule named
|
|
98
|
+
# fable.
|
|
99
|
+
FABLE_ON=$(jq -r '.global.modelFallback.fableEnabled // false' "$PREFS" 2>/dev/null)
|
|
100
|
+
|
|
101
|
+
record() {
|
|
102
|
+
[ -x "$HERE/../scripts/log-metric.sh" ] || return 0
|
|
103
|
+
local rec
|
|
104
|
+
rec=$(jq -r '.global.modelRouting.recordDecisions // true' "$PREFS" 2>/dev/null)
|
|
105
|
+
[ "$rec" = "false" ] && return 0
|
|
106
|
+
"$HERE/../scripts/log-metric.sh" "$TASK_ID" "${PHASE:-0}" model_routing.decision \
|
|
107
|
+
call_site="$CALL_SITE" rung="$1" reason="$2" >/dev/null 2>&1 || true
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
for rung in $MATCH; do
|
|
111
|
+
if [ ! -f "$COST_TABLE" ]; then
|
|
112
|
+
record "$rung" "no cost table; rung taken on trust"
|
|
113
|
+
emit "$rung"
|
|
114
|
+
fi
|
|
115
|
+
PROVIDER=$(jq -r --arg r "$rung" '.prices[$r].provider // ""' "$COST_TABLE" 2>/dev/null)
|
|
116
|
+
# A rung nobody priced is a typo in a config far more often than it is a new
|
|
117
|
+
# model. Skipping it silently would route to the next one and leave the user
|
|
118
|
+
# certain their rule applied.
|
|
119
|
+
if [ -z "$PROVIDER" ]; then
|
|
120
|
+
echo "model-dispatch: rung '$rung' is not in cost-table.json; skipping it" >&2
|
|
121
|
+
continue
|
|
122
|
+
fi
|
|
123
|
+
if [ "$rung" = "fable" ] && [ "$FABLE_ON" != "true" ]; then
|
|
124
|
+
echo "model-dispatch: rule prefers 'fable' but modelFallback.fableEnabled is false; skipping it" >&2
|
|
125
|
+
continue
|
|
126
|
+
fi
|
|
127
|
+
if [ "$PROVIDER" != "anthropic" ] && [ "$CALL_SITE" = "subagent" ]; then
|
|
128
|
+
# The honest limit, said out loud at the moment it bites. Subagent dispatch
|
|
129
|
+
# belongs to the host; we cannot send one to another provider, and a router
|
|
130
|
+
# that quietly downgraded to an Anthropic rung here would leave the user
|
|
131
|
+
# believing a rule worked that never could.
|
|
132
|
+
echo "model-dispatch: rung '$rung' is $PROVIDER and subagent dispatch belongs to the host; skipping it" >&2
|
|
133
|
+
continue
|
|
134
|
+
fi
|
|
135
|
+
record "$rung" "rule matched"
|
|
136
|
+
emit "$rung"
|
|
137
|
+
done
|
|
138
|
+
|
|
139
|
+
record "$DEFAULT_RUNG" "every preferred rung was unavailable"
|
|
140
|
+
emit "$DEFAULT_RUNG"
|