@mmerterden/multi-agent-pipeline 12.11.0 → 13.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +120 -0
- package/README.md +24 -7
- package/index.js +5 -2
- package/install/_codex-agents.mjs +211 -0
- package/install/_codex-instructions.mjs +33 -0
- package/install/_managed-block.mjs +99 -0
- package/install/codex.mjs +478 -0
- package/install/copilot.mjs +34 -80
- package/install/index.mjs +25 -9
- package/install/templates/codex-instructions.md +45 -0
- package/package.json +5 -3
- package/pipeline/claude-md-template.md +1 -0
- package/pipeline/commands/multi-agent/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/dev/SKILL.md +31 -6
- package/pipeline/commands/multi-agent/finish/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/setup/SKILL.md +69 -2
- package/pipeline/commands/multi-agent/sync/SKILL.md +128 -5
- package/pipeline/commands/multi-agent/testflight-validation/SKILL.md +219 -0
- package/pipeline/commands/multi-agent/update/SKILL.md +7 -4
- package/pipeline/multi-agent-refs/_input-parser.md +1 -1
- package/pipeline/multi-agent-refs/cross-cli-contract.md +51 -17
- package/pipeline/multi-agent-refs/features/model-fallback.md +29 -0
- package/pipeline/multi-agent-refs/features/review-multi-repo.md +1 -1
- package/pipeline/multi-agent-refs/phases/log-format.md +1 -1
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +14 -2
- package/pipeline/multi-agent-refs/phases/phase-4-review.md +36 -5
- package/pipeline/multi-agent-refs/progress-contract.md +1 -1
- package/pipeline/multi-agent-refs/tracker-contract.md +17 -1
- package/pipeline/schemas/prefs.schema.json +296 -62
- package/pipeline/schemas/reviewer-output.schema.json +1 -1
- package/pipeline/schemas/triage-output.schema.json +1 -1
- package/pipeline/scripts/cost-table.json +15 -1
- package/pipeline/scripts/smoke-cross-cli-behavior.sh +25 -10
- package/pipeline/scripts/uninstall.mjs +105 -9
- package/pipeline/scripts/update-check.sh +2 -1
- package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +48 -1
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +87 -6
- package/pipeline/skills/shared/core/multi-agent-testflight-validation/SKILL.md +120 -0
package/install/index.mjs
CHANGED
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
* multi-agent-pipeline installer entry point.
|
|
3
3
|
*
|
|
4
4
|
* Parses flags, dispatches to the per-target installers (Claude Code / Copilot
|
|
5
|
-
* CLI), prints a summary, and fires opt-in telemetry. The pipeline
|
|
6
|
-
* on these
|
|
5
|
+
* CLI / Codex CLI), prints a summary, and fires opt-in telemetry. The pipeline
|
|
6
|
+
* runs natively on these three CLIs only.
|
|
7
7
|
*
|
|
8
8
|
* @module install/index
|
|
9
9
|
*/
|
|
@@ -16,6 +16,7 @@ import { join } from "path";
|
|
|
16
16
|
|
|
17
17
|
import { installClaude } from "./claude.mjs";
|
|
18
18
|
import { installCopilot } from "./copilot.mjs";
|
|
19
|
+
import { installCodex } from "./codex.mjs";
|
|
19
20
|
import { setDryRun, writeFile } from "./_common.mjs";
|
|
20
21
|
import { sendInstallTelemetry } from "./_telemetry.mjs";
|
|
21
22
|
|
|
@@ -25,7 +26,7 @@ const PIPELINE_ROOT = dirname(__dirname);
|
|
|
25
26
|
const PIPELINE_SRC = `${PIPELINE_ROOT}/pipeline`;
|
|
26
27
|
|
|
27
28
|
/** Tool flags surfaced in CLI help. */
|
|
28
|
-
const TOOL_FLAGS = ["--claude", "--copilot"];
|
|
29
|
+
const TOOL_FLAGS = ["--claude", "--copilot", "--codex"];
|
|
29
30
|
|
|
30
31
|
/**
|
|
31
32
|
* @param {string[]} argv
|
|
@@ -69,6 +70,8 @@ export async function runInstall(argv) {
|
|
|
69
70
|
|
|
70
71
|
const forCopilot = flags.includes("--copilot") || flags.includes("--all");
|
|
71
72
|
|
|
73
|
+
const forCodex = flags.includes("--codex") || flags.includes("--all");
|
|
74
|
+
|
|
72
75
|
const useSymlinks = flags.includes("--link");
|
|
73
76
|
const indexOnly = flags.includes("--index-only");
|
|
74
77
|
const platformFlag = parsePlatformFlag(flags);
|
|
@@ -89,10 +92,11 @@ export async function runInstall(argv) {
|
|
|
89
92
|
|
|
90
93
|
if (forClaude) installClaude(installerCtx);
|
|
91
94
|
if (forCopilot) installCopilot(installerCtx);
|
|
95
|
+
if (forCodex) installCodex(installerCtx);
|
|
92
96
|
|
|
93
97
|
// Version marker: update-check.sh falls back to this for npx-only installs
|
|
94
98
|
// that have no pipeline repo clone to read a version from.
|
|
95
|
-
writeVersionMarkers({ home: HOME, forClaude, forCopilot });
|
|
99
|
+
writeVersionMarkers({ home: HOME, forClaude, forCopilot, forCodex });
|
|
96
100
|
|
|
97
101
|
if (dryRun) {
|
|
98
102
|
console.log("");
|
|
@@ -101,7 +105,7 @@ export async function runInstall(argv) {
|
|
|
101
105
|
return;
|
|
102
106
|
}
|
|
103
107
|
|
|
104
|
-
printSummary({ forClaude, forCopilot });
|
|
108
|
+
printSummary({ forClaude, forCopilot, forCodex });
|
|
105
109
|
|
|
106
110
|
// Fire-and-forget; install is already done.
|
|
107
111
|
sendInstallTelemetry({ packageRoot: PIPELINE_ROOT, flags });
|
|
@@ -111,9 +115,9 @@ export async function runInstall(argv) {
|
|
|
111
115
|
* Stamp the installed package version into each target so the advisory
|
|
112
116
|
* update check works without a repo clone. Failure is non-fatal.
|
|
113
117
|
*
|
|
114
|
-
* @param {{home: string, forClaude: boolean, forCopilot: boolean}} opts
|
|
118
|
+
* @param {{home: string, forClaude: boolean, forCopilot: boolean, forCodex?: boolean}} opts
|
|
115
119
|
*/
|
|
116
|
-
export function writeVersionMarkers({ home, forClaude, forCopilot }) {
|
|
120
|
+
export function writeVersionMarkers({ home, forClaude, forCopilot, forCodex }) {
|
|
117
121
|
let version;
|
|
118
122
|
try {
|
|
119
123
|
version = JSON.parse(readFileSync(join(PIPELINE_ROOT, "package.json"), "utf-8")).version;
|
|
@@ -124,6 +128,7 @@ export function writeVersionMarkers({ home, forClaude, forCopilot }) {
|
|
|
124
128
|
const targets = [];
|
|
125
129
|
if (forClaude) targets.push(join(home, ".claude", ".pipeline-version"));
|
|
126
130
|
if (forCopilot) targets.push(join(home, ".copilot", ".pipeline-version"));
|
|
131
|
+
if (forCodex) targets.push(join(home, ".codex", ".pipeline-version"));
|
|
127
132
|
for (const target of targets) {
|
|
128
133
|
try {
|
|
129
134
|
writeFile(target, `${version}\n`);
|
|
@@ -150,7 +155,7 @@ function parsePlatformFlag(flags) {
|
|
|
150
155
|
}
|
|
151
156
|
|
|
152
157
|
function printSummary(opts) {
|
|
153
|
-
const { forClaude, forCopilot } = opts;
|
|
158
|
+
const { forClaude, forCopilot, forCodex } = opts;
|
|
154
159
|
|
|
155
160
|
console.log(" Installation complete!");
|
|
156
161
|
console.log("");
|
|
@@ -168,9 +173,20 @@ function printSummary(opts) {
|
|
|
168
173
|
console.log(" Just describe your task - Copilot will follow the pipeline");
|
|
169
174
|
console.log("");
|
|
170
175
|
}
|
|
176
|
+
if (forCodex) {
|
|
177
|
+
console.log(" Codex CLI:");
|
|
178
|
+
console.log(' /multi-agent "MOBILE-123" Start a task (prompt form)');
|
|
179
|
+
console.log(' $multi-agent "MOBILE-123" Start a task (skill form)');
|
|
180
|
+
console.log(" Sub-commands are refs, not skills - the orchestrator loads them on demand");
|
|
181
|
+
console.log(" Run `codex login` first if you have not already");
|
|
182
|
+
console.log("");
|
|
183
|
+
}
|
|
171
184
|
|
|
172
185
|
console.log(" For UI testing, also install the mobile MCP server:");
|
|
173
|
-
|
|
186
|
+
if (forCodex) {
|
|
187
|
+
console.log(" Codex: registered automatically (codex mcp add dev-toolkit)");
|
|
188
|
+
}
|
|
189
|
+
console.log(" Otherwise add to your MCP config: npx @mmerterden/dev-toolkit-mcp");
|
|
174
190
|
console.log("");
|
|
175
191
|
console.log(" Uninstall everything: npx @mmerterden/multi-agent-pipeline uninstall");
|
|
176
192
|
console.log(" (Personal access tokens in keychain are preserved.)");
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# Multi-Agent Development Pipeline
|
|
2
|
+
|
|
3
|
+
This block is managed by `multi-agent-pipeline`. Edit anything outside it freely;
|
|
4
|
+
the installer replaces only the span up to the end marker.
|
|
5
|
+
|
|
6
|
+
The pipeline is an 8-phase development workflow (analysis, planning, TDD dev,
|
|
7
|
+
parallel review + triage, test, commit, report). It is invoked as `/multi-agent`
|
|
8
|
+
or `$multi-agent`, and the orchestrator spec lives at
|
|
9
|
+
`$HOME/.codex/skills/multi-agent/SKILL.md`.
|
|
10
|
+
|
|
11
|
+
## Host adaptation (Codex CLI)
|
|
12
|
+
|
|
13
|
+
| Concern | On Codex |
|
|
14
|
+
| ---------------- | ---------------------------------------------------------------------------------------------------------------------------- |
|
|
15
|
+
| Orchestrator | one skill, `multi-agent` - it reads per-command specs on demand from `$HOME/.codex/multi-agent-refs/commands/<cmd>/SKILL.md` |
|
|
16
|
+
| Phase specs | `$HOME/.codex/multi-agent-refs/phases/phase-N-*.md`, read only when that phase runs |
|
|
17
|
+
| Personas | `$HOME/.codex/agents/*.toml` (generated from the pipeline's persona set) |
|
|
18
|
+
| Progress display | the native `update_plan` tool, one plan step per phase, plus `phase-tracker.sh` state on disk |
|
|
19
|
+
| Shared state | task logs and preferences stay under `$HOME/.claude/` - that path is the cross-host state root, not a Claude Code dependency |
|
|
20
|
+
|
|
21
|
+
## Sub-agent delegation is authorized for this pipeline
|
|
22
|
+
|
|
23
|
+
Codex does not spawn sub-agents unless instructions explicitly ask for it. The
|
|
24
|
+
pipeline's Phase 1 (parallel codebase scan) and Phase 4 (parallel review) depend
|
|
25
|
+
on real fan-out, so **delegation is authorized whenever a `/multi-agent` or
|
|
26
|
+
`$multi-agent` run is active**, under these rules:
|
|
27
|
+
|
|
28
|
+
1. **Always pass `fork_turns: "none"`** (or a positive integer) on a
|
|
29
|
+
`spawn_agent` that sets `model` or `reasoning_effort`. A full-history fork
|
|
30
|
+
inherits the parent model and **silently discards the override**, which would
|
|
31
|
+
collapse the whole review panel onto one model with no error.
|
|
32
|
+
2. **At most 3 concurrent children.** There are 4 concurrency slots including the
|
|
33
|
+
orchestrator, so a 3-reviewer panel is the ceiling; a 4th child queues rather
|
|
34
|
+
than running in parallel.
|
|
35
|
+
3. Outside an active pipeline run, the default applies: do not delegate unless
|
|
36
|
+
asked.
|
|
37
|
+
|
|
38
|
+
## Skills-block budget
|
|
39
|
+
|
|
40
|
+
Codex assembles every discovered skill's name + description into one prompt block
|
|
41
|
+
and **silently drops entries when that block overflows**. The pipeline therefore
|
|
42
|
+
contributes exactly **one** skill (`multi-agent`) and keeps its 42 sub-command
|
|
43
|
+
specs as reference files that cost nothing until read. Do not convert those specs
|
|
44
|
+
into peer skills: doing so evicts other skills, including ones from installed
|
|
45
|
+
plugins, with no error surfaced.
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@mmerterden/multi-agent-pipeline",
|
|
3
|
-
"version": "
|
|
4
|
-
"description": "8-phase AI development pipeline with full orchestration on Claude Code and
|
|
3
|
+
"version": "13.0.0",
|
|
4
|
+
"description": "8-phase AI development pipeline with full orchestration on Claude Code, Copilot CLI and Codex CLI. Analysis, planning, TDD, CLI-aware parallel review with consensus surfacing + Fable triage, default-FAIL evidence gates, secret + intent guards, per-phase cost ledger, persistent learnings memory, wiki generation, commit automation. Token-preserving uninstall.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "index.js",
|
|
7
7
|
"exports": {
|
|
@@ -42,7 +42,9 @@
|
|
|
42
42
|
"frontend",
|
|
43
43
|
"jira",
|
|
44
44
|
"github-issues",
|
|
45
|
-
"automation"
|
|
45
|
+
"automation",
|
|
46
|
+
"codex",
|
|
47
|
+
"codex-cli"
|
|
46
48
|
],
|
|
47
49
|
"author": "Mert Erden",
|
|
48
50
|
"license": "MIT",
|
|
@@ -19,6 +19,7 @@
|
|
|
19
19
|
4. Review -> deterministic gates + parallel review + Fable triage
|
|
20
20
|
- Claude Code: Opus + Sonnet (2 paralel)
|
|
21
21
|
- Copilot CLI: GPT-5.4 + Opus + Sonnet (3 paralel)
|
|
22
|
+
- Codex CLI: gpt-5.6 (xhigh) + gpt-5.4 + gpt-5.6 (medium) (3 paralel)
|
|
22
23
|
|
|
23
24
|
### Strict Rules
|
|
24
25
|
|
|
@@ -48,7 +48,7 @@ Classification schema lives in `$HOME/.claude/multi-agent-refs/_input-parser.md`
|
|
|
48
48
|
| 7 | `issue` | full picker | account → repo (multi) → issue → maturity → dev-context |
|
|
49
49
|
| 8 | Free-text | `freetext` | account → repo (single) → dev-context (maturity skip) |
|
|
50
50
|
|
|
51
|
-
**Rule**: Whatever the type, **account is always asked** (autopilot picks a default). After issue fetch, **maturity check is mandatory** - blockers halt the pipeline.
|
|
51
|
+
**Rule**: Whatever the type, **account is always asked** (autopilot picks a default). After issue fetch, **maturity check is mandatory** - blockers halt the pipeline. Picker `label` + `header` render in English (`promptLanguage` is locked to `en`); the `question` and each option's `description` render in `outputLanguage`, per the canonical matrix in `multi-agent-refs/rules.md`.
|
|
52
52
|
|
|
53
53
|
Lib scripts (`~/.claude/lib/`):
|
|
54
54
|
- `account-resolver.sh` - keychain account inventory
|
|
@@ -57,19 +57,44 @@ The agent CANNOT make these Phase 0 decisions automatically; it suggests, then w
|
|
|
57
57
|
- Show the chosen provider in the picker label (e.g. "Bitbucket account?" not "GitHub account?")
|
|
58
58
|
3. Project root (state inheritance is FORBIDDEN - picker runs every run)
|
|
59
59
|
4. dev-context picker (extra repos / submodules - read `.gitmodules` and suggest)
|
|
60
|
-
5. **
|
|
61
|
-
-
|
|
62
|
-
|
|
60
|
+
5. **Remote reachability gate (run before branch picker)**: Test remote with
|
|
61
|
+
`git ls-remote --heads origin <baseBranch>` (5s timeout), capturing **stderr**.
|
|
62
|
+
|
|
63
|
+
**Classify the failure before naming a cause.** A non-zero exit is not evidence
|
|
64
|
+
of a network problem, and telling the user to enable a VPN for an auth failure
|
|
65
|
+
sends them to fix something that was never broken. Match stderr, in this order:
|
|
66
|
+
|
|
67
|
+
| stderr contains | Cause | What to offer |
|
|
68
|
+
|---|---|---|
|
|
69
|
+
| `could not read Password`, `Authentication failed`, `terminal prompts disabled`, `Invalid username or password`, `403` | **credential**, not network | the remote URL usually embeds a username with no credential-helper entry. Offer: store the PAT in the credential helper (`git credential approve`, or re-run `/multi-agent:setup`), or switch the remote to SSH. A VPN cannot fix this. |
|
|
70
|
+
| `Could not resolve host`, `Operation timed out`, `Connection refused`, `Network is unreachable`, or the 5s timeout fired with no output | **network** | the VPN/DNS prompt below |
|
|
71
|
+
| `Repository not found`, `does not appear to be a git repository`, `404` | **wrong remote** | show `git remote -v` and ask which remote is correct |
|
|
72
|
+
| anything else | **unknown** | print the stderr line verbatim and ask; never assert a cause you did not observe |
|
|
73
|
+
|
|
74
|
+
Outcomes:
|
|
75
|
+
- **Reachable** → proceed to base branch picker, fetch latest, create worktree
|
|
76
|
+
from `origin/<baseBranch>`
|
|
77
|
+
- **Credential / wrong remote** → STOP with the matching remedy above. Do **not**
|
|
78
|
+
offer "continue from the local ref": the base being stale is unrelated to the
|
|
79
|
+
actual failure, so accepting staleness here trades correctness for nothing.
|
|
80
|
+
- **Network** → STOP and ask:
|
|
63
81
|
```
|
|
64
|
-
[N/Total]
|
|
65
|
-
|
|
82
|
+
[N/Total] Remote gate
|
|
83
|
+
Observed: <the stderr line, verbatim>
|
|
84
|
+
Classified: <host> unreachable (network).
|
|
66
85
|
Options:
|
|
67
86
|
1. Enable VPN and retry (recommended - branch will be fresh)
|
|
68
87
|
2. Continue from local origin/<baseBranch> ref (may be stale)
|
|
69
88
|
3. Cancel
|
|
70
89
|
Confirm? [1/2/3]
|
|
71
90
|
```
|
|
72
|
-
- User picks `2` → log warning + record `"baseRefFreshness": "stale"` in
|
|
91
|
+
- User picks `2` → log warning + record `"baseRefFreshness": "stale"` in
|
|
92
|
+
`agent-state.json`, proceed from local ref. Phase 6 push needs network anyway,
|
|
93
|
+
so re-prompt there if still unreachable.
|
|
94
|
+
|
|
95
|
+
Always show what was **observed** next to what was **classified**. The previous
|
|
96
|
+
wording asserted `Detected: <host> unreachable (VPN/DNS)` for every failure mode,
|
|
97
|
+
including a plain missing-credential error that returns in under a second.
|
|
73
98
|
6. Base branch picker (show recents, take an explicit pick)
|
|
74
99
|
7. Branch name (suggest, allow edit)
|
|
75
100
|
8. Maturity flags acknowledge (read from the issue body's Progress table)
|
|
@@ -56,7 +56,7 @@ Phases 1-3 (Analysis / Planning / Dev) are skipped by design - `finish` treats
|
|
|
56
56
|
- interactive: present them and ask (`AskUserQuestion`) whether to fix now (loop back through a minimal Phase-3-style TDD fix) or proceed;
|
|
57
57
|
- `autopilot` (or `prefs.global.finish.autoFix == true`): auto-fix accepted blocking/important findings, then re-review the fix, before advancing.
|
|
58
58
|
- **Phase 5 Build+Test** - the **automated success gate** (this is what "build+test success" means here; the interactive device user-test is `/multi-agent:manual-test`). Stack-aware: build via `figma-config.build` (iOS scheme / Android gradle / detected backend/frontend build) and run the existing test suite if present (`swift test` / `xcodebuild test` / `./gradlew test` / `pytest` / `npm test` / `vitest`). Require success to advance; on failure, surface logs and (interactive) stop or (autopilot) attempt a bounded fix loop. **If the repo has no tests, report "no tests present" - never fabricate test results.**
|
|
59
|
-
- **Phase 6 Commit/PR** - per `$HOME/.claude/multi-agent-refs/phases/phase-6-commit.md`: stage + commit any remaining local changes with a conventional message (`{type}(scope): desc [{JIRA_KEY}-{id}]`), push, and open a PR **only if one does not already exist** for the branch. PR body per `$HOME/.claude/
|
|
59
|
+
- **Phase 6 Commit/PR** - per `$HOME/.claude/multi-agent-refs/phases/phase-6-commit.md`: stage + commit any remaining local changes with a conventional message (`{type}(scope): desc [{JIRA_KEY}-{id}]`), push, and open a PR **only if one does not already exist** for the branch. PR body per `$HOME/.claude/rules/pipeline-output-formatting.md` and `$HOME/.claude/rules/git-conventions.md` - `Ref: #N` / `Related: #N`, never `Closes/Fixes/Resolves`; NO AI/bot attribution anywhere.
|
|
60
60
|
- **Phase 7 Report** - per `$HOME/.claude/multi-agent-refs/phases/phase-7-report.md` + `channels.md`: produce the **technical analysis** and **test scenarios**, then post to the configured channels. Default content for `finish`: a Jira **comment** carrying the technical analysis + the test scenarios (and, when the PR was opened, the PR description). Every body runs through the humanizer; bot/tool/AI signatures are FORBIDDEN in comments.
|
|
61
61
|
|
|
62
62
|
## Modes
|
|
@@ -15,7 +15,7 @@ Two-axis language preference for the pipeline:
|
|
|
15
15
|
| `promptLanguage` | Interactive prompts during a pipeline run (account picker, project picker, dev-context picker, base-branch picker, branch-name picker, maturity ack, channels picker, Phase 5 test prompt, Phase 6 local-checkout prompt) | `prefs.global.promptLanguage` | **Fixed to `en`** - never toggled by this skill |
|
|
16
16
|
| `outputLanguage` | Assistant's explanations, status updates, error messages, and pipeline-generated reports rendered to the user (NOT external payloads) | `prefs.global.outputLanguage` | Toggled by this skill |
|
|
17
17
|
|
|
18
|
-
**Why promptLanguage is fixed:**
|
|
18
|
+
**Why promptLanguage is fixed:** it governs the picker's structural chrome only - `AskUserQuestion` `label` (button text) and `header` (chip), host error UI, and internal contract identifiers. Those stay English so tooling reads the same across CLIs. Everything a user actually reads follows `outputLanguage`, including the picker `question` and each option's `description`, per the canonical per-field matrix in `multi-agent-refs/rules.md`. A picker whose question is English on a Turkish run is a bug, not the contract.
|
|
19
19
|
|
|
20
20
|
**Always English regardless of either field**: commit messages, PR titles/bodies, Jira comments, wiki pages, reviewer/triage system prompts, agent-log.md payloads, confirmation / error UI exposed by the CLI host.
|
|
21
21
|
|
|
@@ -109,5 +109,5 @@ Render in the **new** outputLanguage:
|
|
|
109
109
|
|
|
110
110
|
- `setup.md` Step 0 - first-run language picker (asks only outputLanguage; promptLanguage seeded as `en`)
|
|
111
111
|
- `prefs.schema.json` - `global.promptLanguage` (fixed `"en"`) and `global.outputLanguage` definitions
|
|
112
|
-
- `phase-5-test.md`, `phase-6-commit.md` -
|
|
112
|
+
- `phase-5-test.md`, `phase-6-commit.md` - interactive prompts follow the per-field matrix: `question` + `description` in `outputLanguage`, `label` + `header` English
|
|
113
113
|
- `help.md` - reads outputLanguage to render its own help text
|
|
@@ -24,7 +24,7 @@ Ask BEFORE anything else, so every subsequent setup prompt and the rest of the p
|
|
|
24
24
|
|
|
25
25
|
| Axis | What it controls | Configurable? |
|
|
26
26
|
|---|---|---|
|
|
27
|
-
| `promptLanguage` |
|
|
27
|
+
| `promptLanguage` | The picker's structural UI chrome only: `AskUserQuestion` `label` (button text) and `header` (chip), plus host error UI and internal contract identifiers | **No - fixed to `en`.** Button and chip text stays English so tooling reads the same across CLIs. |
|
|
28
28
|
| `outputLanguage` | The assistant's non-interactive explanations, status updates, error messages, and pipeline-generated reports rendered to the user | Yes - set here and changeable later via `/multi-agent:language <en\|tr>` |
|
|
29
29
|
|
|
30
30
|
`promptLanguage` is seeded as `"en"` and never offered to the user. **External payloads always stay English** (commits, PR titles/bodies, Jira comments, wiki content, reviewer/triage prompts, agent-log.md).
|
|
@@ -51,7 +51,16 @@ Select (1-2):
|
|
|
51
51
|
- `tr`: `outputLanguage=<Y>. promptLanguage "en" sabit. Dış çıktılar (PR açıklamaları, Jira yorumları, review istemleri) her durumda İngilizce kalır.`
|
|
52
52
|
- **Unknown input** → re-ask once, then abort setup with a hint pointing to `/multi-agent:language`.
|
|
53
53
|
|
|
54
|
-
From this point forward, every prompt in Steps 1-6 below
|
|
54
|
+
From this point forward, every prompt in Steps 1-6 below follows the canonical
|
|
55
|
+
per-field matrix in `$HOME/.claude/multi-agent-refs/rules.md` ("Language Application"):
|
|
56
|
+
the `question` text and each option's `description` render in `outputLanguage`, while
|
|
57
|
+
`label` and `header` stay English. The wizard's status text and post-setup summary
|
|
58
|
+
render in `outputLanguage`. External payloads remain English.
|
|
59
|
+
|
|
60
|
+
> This paragraph used to say every prompt renders in English "per the fixed
|
|
61
|
+
> `promptLanguage`", which contradicted the canonical matrix and produced
|
|
62
|
+
> half-English pickers on Turkish runs: `promptLanguage` governs only the button and
|
|
63
|
+
> chip chrome, never the question a user reads.
|
|
55
64
|
|
|
56
65
|
### Step 0.5 - Credential Backend Check
|
|
57
66
|
|
|
@@ -102,6 +111,12 @@ These are the RECOMMENDED key names. When creating NEW keys, use these. But exis
|
|
|
102
111
|
| `graylog` | `${USER}_Graylog_Access_Token` | `graylog` |
|
|
103
112
|
| `firebase` | `${USER}_Firebase_Access_Json` | `firebase` (any variant: `sa`, `service`, `account`, `access`, `json`) |
|
|
104
113
|
| `jenkins` | `${USER}_Jenkins_Access_Token` | `jenkins` |
|
|
114
|
+
| `appstore_connect_key_id` | `${USER}_AppStoreConnect_Key_Id` | (`appstore` or `asc` or `app_store`) + (`key` or `keyid`) |
|
|
115
|
+
| `appstore_connect_issuer_id` | `${USER}_AppStoreConnect_Issuer_Id` | (`appstore` or `asc` or `app_store`) + `issuer` |
|
|
116
|
+
| `appstore_connect_apple_id` | `${USER}_AppStoreConnect_Apple_Id` | (`appstore` or `asc`) + (`apple` or `account` or `user`) |
|
|
117
|
+
| `appstore_connect_password_item` | `${USER}_AppStoreConnect_Password_Item` | (`appstore` or `asc` or `altool`) + (`password` or `app_specific`) |
|
|
118
|
+
|
|
119
|
+
> The four App Store Connect entries are **iOS-only and optional**: skip them all and the pipeline still works, it just reports Gate 2 of `/multi-agent:testflight-validation` as `SKIPPED` (never as a pass). They mirror the Figma 3-tier shape - Tier 1 = API key (`appstore_connect_key_id` + `appstore_connect_issuer_id`), Tier 2 = Apple ID + app-specific password (`appstore_connect_apple_id` + `appstore_connect_password_item`), Tier 3 = nothing configured. **Offer Tier 2 first when the user says they cannot create an API key**: creating one needs an Admin or App Manager role in App Store Connect, while an app-specific password is generated by the account holder at `appleid.apple.com` with no team permission at all. Two of these hold identifiers rather than secrets (key id, issuer id) and one holds a keychain ITEM NAME, not a password - they still go through the mapping layer so every credential is read the same way. Onboarding mechanics in Step 3b.
|
|
105
120
|
|
|
106
121
|
**1c. Resolution logic (per service):**
|
|
107
122
|
|
|
@@ -352,6 +367,58 @@ This builds `platformIdentityRouting` incrementally - no separate Step 7 neede
|
|
|
352
367
|
|
|
353
368
|
---
|
|
354
369
|
|
|
370
|
+
### Step 3b - App Store Connect onboarding (iOS only, optional)
|
|
371
|
+
|
|
372
|
+
Runs inside Step 3 alongside the other missing credentials, not as a late add-on:
|
|
373
|
+
a user who already has an App Store Connect credential in their keychain gets it
|
|
374
|
+
mapped by Step 1 discovery like any other token, and only the genuinely missing
|
|
375
|
+
pieces reach this flow.
|
|
376
|
+
|
|
377
|
+
Three of the four entries do not go through the normal Token Save Flow, because
|
|
378
|
+
what they hold is not a pasteable secret:
|
|
379
|
+
|
|
380
|
+
| Entry | Holds | Flow |
|
|
381
|
+
|---|---|---|
|
|
382
|
+
| `appstore_connect_key_id` | an identifier | plain value, not a secret; still mapped so it is read through the mapping layer |
|
|
383
|
+
| `appstore_connect_issuer_id` | an identifier | same |
|
|
384
|
+
| `appstore_connect_apple_id` | an email address | same |
|
|
385
|
+
| `appstore_connect_password_item` | a keychain ITEM NAME | the password lives in Apple's own keychain item, referenced as `-p @keychain:<item>` and never read by the pipeline |
|
|
386
|
+
|
|
387
|
+
Ask which tier to configure (picker): **API key** / **Apple ID + app-specific
|
|
388
|
+
password** / **Skip**. Lead with the second when the user says they cannot create
|
|
389
|
+
an API key.
|
|
390
|
+
|
|
391
|
+
**API key.** The private key is a FILE and is never copied into the credential
|
|
392
|
+
store. It must sit in a directory `altool` already searches:
|
|
393
|
+
|
|
394
|
+
```bash
|
|
395
|
+
ls ~/.appstoreconnect/private_keys/AuthKey_*.p8 2>/dev/null \
|
|
396
|
+
|| echo "MISSING: put AuthKey_<keyId>.p8 in ~/.appstoreconnect/private_keys/"
|
|
397
|
+
```
|
|
398
|
+
|
|
399
|
+
**Apple ID + app-specific password.** Use Apple's own keychain helper. The secret
|
|
400
|
+
never enters chat and never becomes a shell argument, per the Token Save Flow rule:
|
|
401
|
+
|
|
402
|
+
```bash
|
|
403
|
+
# the user exports AC_PASSWORD_ONCE in their own shell, for this one command
|
|
404
|
+
xcrun altool --store-password-in-keychain-item "<item-name>" \
|
|
405
|
+
-u "<apple-id>" -p @env:AC_PASSWORD_ONCE
|
|
406
|
+
```
|
|
407
|
+
|
|
408
|
+
Then map only `<item-name>` as `appstore_connect_password_item`.
|
|
409
|
+
|
|
410
|
+
**Multi-provider accounts.** A corporate Apple ID often belongs to several
|
|
411
|
+
providers, and `altool` fails opaquely without one. Resolve it once with
|
|
412
|
+
`ios_testflight_validate({list_providers: true, <credentials just configured>})`
|
|
413
|
+
and store the answer under
|
|
414
|
+
`prefs.projects[<key>].appStoreConnect.providerPublicId` - per-project, since a
|
|
415
|
+
user can ship for more than one team.
|
|
416
|
+
|
|
417
|
+
**Verify + expiry.** Re-run the `list_providers` probe and report the resolved
|
|
418
|
+
tier. A credential that resolves but is rejected (401/403) follows the
|
|
419
|
+
Expired-token decision in `refs/keychain.md` Rule 1 - Regenerate / Use a
|
|
420
|
+
different token / Skip and continue - never a silent drop.
|
|
421
|
+
|
|
355
422
|
### Step 3.5 - Host Prompt (embedded in Token Save Flow)
|
|
356
423
|
|
|
357
424
|
**Not a standalone step** - runs inline at the end of the Token Save Flow whenever the saved token belongs to a **hosted service** (Jira, Confluence, Bitbucket, Fortify, Graylog) AND the host is not yet in `prefs.global.hosts`. Firebase tokens skip this step - Crashlytics is always on Google's fixed domains and `project_id` is embedded in the SA JSON.
|
|
@@ -27,6 +27,7 @@ When invoked, it synchronizes all targets in order. It detects what changed, upd
|
|
|
27
27
|
|---|-------|-----|-----|
|
|
28
28
|
| 1 | Claude Code (source of truth) | `~/.claude/commands/multi-agent.md` + `~/.claude/commands/multi-agent/` + `~/.claude/agents/` + `~/.claude/scripts/` + `~/.claude/lib/` | source |
|
|
29
29
|
| 2 | Copilot CLI | `~/.copilot/copilot-instructions.md` + `~/.copilot/skills/` | <- from Claude |
|
|
30
|
+
| 2b | Codex CLI | `~/.codex/AGENTS.md` + `~/.codex/skills/multi-agent/` + `~/.codex/multi-agent-refs/` + `~/.codex/agents/*.toml` | <- from Claude (path-rewritten) |
|
|
30
31
|
| 3 | multi-agent-pipeline repo | `~/multi-agent-pipeline/pipeline/` | <- from Claude (genericized) |
|
|
31
32
|
| 4 | Website | `{owner}/{website-host}` | <- version + features |
|
|
32
33
|
| 5 | dev-toolkit MCP server | resolved from `prefs.global.devToolkit` or the `mcpServers` registration | own repo: gate, commit, publish |
|
|
@@ -58,7 +59,8 @@ Run every step automatically:
|
|
|
58
59
|
```
|
|
59
60
|
Step 1: PLATFORM Detect macOS / Linux / Windows (Git Bash / WSL); export PLATFORM env
|
|
60
61
|
Step 1.5: DETECT Compare timestamps, find stale targets
|
|
61
|
-
Step 2: COPILOT Claude Code -> Copilot CLI (instructions +
|
|
62
|
+
Step 2: COPILOT Claude Code -> Copilot CLI (instructions + 43 sub-command skills)
|
|
63
|
+
Step 2b: CODEX Claude Code -> Codex CLI (1 router skill + 43 specs as refs + 8 agent TOML)
|
|
62
64
|
Step 3: REPO Claude Code -> pipeline repo (genericized, personal data scrub, bash -n on all sh)
|
|
63
65
|
Step 3c: PLUGINS pipeline shared/external -> multi-agent-plugins marketplace (rebuild knowledge/,
|
|
64
66
|
bump changed plugins' patch version, commit + push the plugins repo)
|
|
@@ -169,6 +171,50 @@ If nothing is stale → report "All targets up to date" and stop.
|
|
|
169
171
|
|
|
170
172
|
---
|
|
171
173
|
|
|
174
|
+
## Codex Sync (Step 2b)
|
|
175
|
+
|
|
176
|
+
Unlike the Copilot step, this one does **not** hand-copy files. The Codex tree is a
|
|
177
|
+
*transform* of the Claude tree, not a mirror of it, and the transform is real work:
|
|
178
|
+
|
|
179
|
+
- the 43 sub-command specs become reference files, because Codex silently truncates
|
|
180
|
+
its skills block (see `cross-cli-contract.md` 2.6 for the measurement)
|
|
181
|
+
- every `$HOME/.claude/...` reference to a CLI-owned tree is retargeted, with
|
|
182
|
+
`agents/<persona>.md` becoming `.toml` and the dispatcher becoming the router skill
|
|
183
|
+
- the 8 personas are regenerated as TOML with a model + reasoning-effort mapping
|
|
184
|
+
- shared state (`logs/`, prefs, `knowledge/`) is deliberately NOT retargeted
|
|
185
|
+
|
|
186
|
+
That logic lives in `install/codex.mjs` and is gate-locked by
|
|
187
|
+
`smoke-codex-install.sh`. Re-describing it here in prose would give the pipeline two
|
|
188
|
+
definitions of the same transform, and the prose copy would be the one that rots. So
|
|
189
|
+
the step runs the installer's Codex target:
|
|
190
|
+
|
|
191
|
+
```bash
|
|
192
|
+
cd "$HOME/multi-agent-pipeline"
|
|
193
|
+
node install.js --codex
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
**Verify** (the installer is quiet about correctness, only about counts):
|
|
197
|
+
|
|
198
|
+
```bash
|
|
199
|
+
# exactly one pipeline skill: more than one means the skills block will truncate
|
|
200
|
+
ls -1 "$HOME/.codex/skills" | grep -c '^multi-agent$' # want 1
|
|
201
|
+
# every sub-command reachable as a ref
|
|
202
|
+
find "$HOME/.codex/multi-agent-refs/commands" -name SKILL.md | wc -l # want the command count
|
|
203
|
+
# no reference left pointing at a tree a Codex-only install does not have
|
|
204
|
+
grep -rhoE '(\$HOME|~)/\.claude/(agents|scripts|lib|schemas|commands|multi-agent-refs|rules)' \
|
|
205
|
+
"$HOME/.codex/skills" "$HOME/.codex/multi-agent-refs" | sort -u # want empty
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
**MCP**: the installer runs `codex mcp add dev-toolkit` itself. It is idempotent, and
|
|
209
|
+
it is skipped with a warning when `codex` is not on `PATH` - never hand-edit
|
|
210
|
+
`~/.codex/config.toml`, which Codex owns (marketplace and plugin state live there
|
|
211
|
+
too).
|
|
212
|
+
|
|
213
|
+
**Order matters**: run this AFTER Step 3 (REPO), because the installer reads the repo
|
|
214
|
+
tree. Running it before means Codex gets the previous revision.
|
|
215
|
+
|
|
216
|
+
---
|
|
217
|
+
|
|
172
218
|
## Stack-Plugin Sync (Step 3c)
|
|
173
219
|
|
|
174
220
|
Stack skills are distributed as versioned plugins in the `mmerterden/multi-agent-plugins` marketplace. The pipeline's `pipeline/skills/shared/external/` is the **single authoring source**; the marketplace is a derived, versioned publish artifact. This step rebuilds it and publishes only when something changed.
|
|
@@ -283,7 +329,79 @@ git tag "v$NEW" && git push origin main --tags
|
|
|
283
329
|
|
|
284
330
|
Publish to the registry declared in `publishConfig` through a throwaway userconfig. Never edit `~/.npmrc`, and never a bare `npm publish` - it lands on whichever registry the ambient config happens to name.
|
|
285
331
|
|
|
286
|
-
**The token depends on the registry
|
|
332
|
+
**The token depends on the registry AND on the package scope.** The registry picks
|
|
333
|
+
the credential *type*; the scope picks the *account*. Resolving on host alone
|
|
334
|
+
misroutes on any machine with more than one GitHub identity, which is the common
|
|
335
|
+
case for anyone with a work and a personal account.
|
|
336
|
+
|
|
337
|
+
| Registry | Credential type | Account chosen by |
|
|
338
|
+
|---|---|---|
|
|
339
|
+
| `npm.pkg.github.com` | GitHub PAT with `write:packages` (Classic; fine-grained PATs are not fully supported for Packages) | the package scope, e.g. `@{owner}` → the `{owner}` account's PAT |
|
|
340
|
+
| `registry.npmjs.org` | npm token (logical key `npm`) | the npm account that owns the scope |
|
|
341
|
+
|
|
342
|
+
Resolve the scope first, then the key:
|
|
343
|
+
|
|
344
|
+
```bash
|
|
345
|
+
SCOPE=$(node -p "(require('./package.json').name.match(/^@([^/]+)/)||[])[1] || ''")
|
|
346
|
+
# Prefer a scope-specific mapping; fall back to the generic key only when the
|
|
347
|
+
# machine has exactly one GitHub identity.
|
|
348
|
+
KEY=$(node -e '
|
|
349
|
+
const fs=require("fs"),os=require("os"),p=os.homedir()+"/.claude/multi-agent-preferences.json";
|
|
350
|
+
const km=(JSON.parse(fs.readFileSync(p,"utf8")).global||{}).keychainMapping||{};
|
|
351
|
+
const scope=process.argv[1];
|
|
352
|
+
process.stdout.write(km[`github_${scope}`] ? `github_${scope}` : "github");
|
|
353
|
+
' "$SCOPE")
|
|
354
|
+
```
|
|
355
|
+
|
|
356
|
+
**Pre-flight the scope, do not learn it from a 403.** GitHub tells you a token's
|
|
357
|
+
scopes on any authenticated request, so check before uploading rather than after:
|
|
358
|
+
|
|
359
|
+
```bash
|
|
360
|
+
scope_ok() { # $1 = token; prints the login, non-zero when write:packages is absent
|
|
361
|
+
local hdrs; hdrs=$(curl -sI -H "Authorization: token $1" https://api.github.com/user)
|
|
362
|
+
printf '%s' "$hdrs" | grep -i '^x-oauth-scopes:' | grep -q 'write:packages'
|
|
363
|
+
}
|
|
364
|
+
```
|
|
365
|
+
|
|
366
|
+
**Candidate order for a `npm.pkg.github.com` publish.** The scope matters more than
|
|
367
|
+
where the token is stored, and the two are not correlated: on a machine with a work
|
|
368
|
+
and a personal identity, the hand-made PAT in the credential store may be the wrong
|
|
369
|
+
account or the right account without `write:packages`, while the `gh` CLI's own
|
|
370
|
+
OAuth token for that account often has it.
|
|
371
|
+
|
|
372
|
+
1. `github_<scope>` from `keychainMapping` (a PAT deliberately onboarded for this scope)
|
|
373
|
+
2. `gh auth token -u <scope>` - gh's stored OAuth token for that account
|
|
374
|
+
3. the generic `github` key - only when the machine has one GitHub identity
|
|
375
|
+
|
|
376
|
+
Take the first candidate that passes `scope_ok` AND whose `login` matches the
|
|
377
|
+
package scope. If none qualifies, stop before `npm publish` and report which
|
|
378
|
+
candidates were tried, what login each resolved to, and which scope was missing.
|
|
379
|
+
That report is the actionable output; a 403 body is not.
|
|
380
|
+
|
|
381
|
+
Measured on this machine, which is why the order is what it is:
|
|
382
|
+
|
|
383
|
+
| Candidate | login | scopes |
|
|
384
|
+
|---|---|---|
|
|
385
|
+
| `keychainMapping.github` | corporate EMU account | cannot publish to a personal scope under any grant |
|
|
386
|
+
| `mmerterden_Github_Auth_Token` | personal | `admin:public_key, gist, read:org, repo` - no `write:packages` |
|
|
387
|
+
| `gh auth token -u mmerterden` | personal | `gist, read:org, repo, workflow, write:packages` ✓ |
|
|
388
|
+
|
|
389
|
+
**Two 403s mean two different things, and neither says "wrong token" plainly:**
|
|
390
|
+
|
|
391
|
+
- `Unauthorized: As an Enterprise Managed User, you cannot access this content` -
|
|
392
|
+
the resolved token belongs to a corporate EMU account, which cannot publish to a
|
|
393
|
+
personal scope at all. The mapping points at the wrong identity. Map the personal
|
|
394
|
+
account's PAT under `github_<scope>` and re-run; do not "fix" this by granting
|
|
395
|
+
the EMU token more scopes, because no scope makes an EMU account able to write
|
|
396
|
+
to a personal namespace.
|
|
397
|
+
- `The token provided does not match expected scopes` - right account, missing
|
|
398
|
+
permission. The PAT needs **Classic** with `write:packages` (plus `repo` for a
|
|
399
|
+
private package). Regenerate it at
|
|
400
|
+
`https://github.com/settings/tokens/new?scopes=write:packages,read:packages,repo`
|
|
401
|
+
and re-onboard via `/multi-agent:setup`.
|
|
402
|
+
|
|
403
|
+
Report which of the two it was. "Permission denied" alone sends the user to
|
|
404
|
+
regenerate a token that was never the problem.
|
|
287
405
|
|
|
288
406
|
```bash
|
|
289
407
|
NPMRC=$(mktemp); trap 'rm -f "$NPMRC"' EXIT
|
|
@@ -350,6 +468,7 @@ When invoked with the `release` argument:
|
|
|
350
468
|
7. DEV-TOOLKIT Ship the companion MCP server if it moved (Step 3d gates, then publish)
|
|
351
469
|
8. WEBSITE Version + features -> {website-host}
|
|
352
470
|
9. COPILOT Copilot CLI instructions + skills sync
|
|
471
|
+
9b. CODEX Codex CLI router skill + refs + agent TOML (node install.js --codex)
|
|
353
472
|
10. Report Summary: version, touched repos, deploy status
|
|
354
473
|
```
|
|
355
474
|
|
|
@@ -357,20 +476,24 @@ When invoked with the `release` argument:
|
|
|
357
476
|
|
|
358
477
|
## Sub-Command Sync (Claude Code <-> Copilot CLI Skills)
|
|
359
478
|
|
|
360
|
-
|
|
479
|
+
> Codex takes the Step 2b path instead; see that section.
|
|
480
|
+
|
|
481
|
+
This runs on the Claude <-> Copilot axis. Codex is NOT synced here: it receives the
|
|
482
|
+
same 43 specs as reference files rather than as peer skills, via Step 2b - see
|
|
483
|
+
`cross-cli-contract.md` 2.6 for why the parity axis differs per host.
|
|
361
484
|
|
|
362
485
|
| Claude Code | Copilot CLI |
|
|
363
486
|
|-------------|-------------|
|
|
364
487
|
| `~/.claude/commands/multi-agent/{cmd}/SKILL.md` | `~/.copilot/skills/multi-agent-{cmd}/SKILL.md` |
|
|
365
488
|
|
|
366
|
-
**
|
|
489
|
+
**43 commands are synced** (canonical inventory - must match `cross-cli-contract.md` section 1; drift = contract violation):
|
|
367
490
|
|
|
368
491
|
```
|
|
369
492
|
analysis, analysis-resolve, autopilot, build-optimize, channels, create-jira, design-check, dev,
|
|
370
493
|
dev-autopilot, dev-local, dev-local-autopilot, diff-explain, finish, forget, garbage-collect,
|
|
371
494
|
help, issue, jira, kill, language, local,
|
|
372
495
|
local-autopilot, log, manual-test, prune-logs, purge, refactor, resume, review, review-issue, review-jira,
|
|
373
|
-
routines, save, scan, search, setup, stack, status, sync, test, uninstall, update
|
|
496
|
+
routines, save, scan, search, setup, stack, status, sync, test, testflight-validation, uninstall, update
|
|
374
497
|
```
|
|
375
498
|
|
|
376
499
|
**NOT synced**: `$HOME/.claude/multi-agent-refs/*` - lazy-load references, Claude Code specific
|