@mmerterden/multi-agent-pipeline 12.11.0 → 13.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/CHANGELOG.md +120 -0
  2. package/README.md +24 -7
  3. package/index.js +5 -2
  4. package/install/_codex-agents.mjs +211 -0
  5. package/install/_codex-instructions.mjs +33 -0
  6. package/install/_managed-block.mjs +99 -0
  7. package/install/codex.mjs +478 -0
  8. package/install/copilot.mjs +34 -80
  9. package/install/index.mjs +25 -9
  10. package/install/templates/codex-instructions.md +45 -0
  11. package/package.json +5 -3
  12. package/pipeline/claude-md-template.md +1 -0
  13. package/pipeline/commands/multi-agent/SKILL.md +1 -1
  14. package/pipeline/commands/multi-agent/dev/SKILL.md +31 -6
  15. package/pipeline/commands/multi-agent/finish/SKILL.md +1 -1
  16. package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
  17. package/pipeline/commands/multi-agent/setup/SKILL.md +69 -2
  18. package/pipeline/commands/multi-agent/sync/SKILL.md +128 -5
  19. package/pipeline/commands/multi-agent/testflight-validation/SKILL.md +219 -0
  20. package/pipeline/commands/multi-agent/update/SKILL.md +7 -4
  21. package/pipeline/multi-agent-refs/_input-parser.md +1 -1
  22. package/pipeline/multi-agent-refs/cross-cli-contract.md +51 -17
  23. package/pipeline/multi-agent-refs/features/model-fallback.md +29 -0
  24. package/pipeline/multi-agent-refs/features/review-multi-repo.md +1 -1
  25. package/pipeline/multi-agent-refs/phases/log-format.md +1 -1
  26. package/pipeline/multi-agent-refs/phases/phase-0-init.md +14 -2
  27. package/pipeline/multi-agent-refs/phases/phase-4-review.md +36 -5
  28. package/pipeline/multi-agent-refs/progress-contract.md +1 -1
  29. package/pipeline/multi-agent-refs/tracker-contract.md +17 -1
  30. package/pipeline/schemas/prefs.schema.json +296 -62
  31. package/pipeline/schemas/reviewer-output.schema.json +1 -1
  32. package/pipeline/schemas/triage-output.schema.json +1 -1
  33. package/pipeline/scripts/cost-table.json +15 -1
  34. package/pipeline/scripts/smoke-cross-cli-behavior.sh +25 -10
  35. package/pipeline/scripts/uninstall.mjs +105 -9
  36. package/pipeline/scripts/update-check.sh +2 -1
  37. package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
  38. package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +48 -1
  39. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +87 -6
  40. package/pipeline/skills/shared/core/multi-agent-testflight-validation/SKILL.md +120 -0
package/install/index.mjs CHANGED
@@ -2,8 +2,8 @@
2
2
  * multi-agent-pipeline installer entry point.
3
3
  *
4
4
  * Parses flags, dispatches to the per-target installers (Claude Code / Copilot
5
- * CLI), prints a summary, and fires opt-in telemetry. The pipeline runs natively
6
- * on these two CLIs only.
5
+ * CLI / Codex CLI), prints a summary, and fires opt-in telemetry. The pipeline
6
+ * runs natively on these three CLIs only.
7
7
  *
8
8
  * @module install/index
9
9
  */
@@ -16,6 +16,7 @@ import { join } from "path";
16
16
 
17
17
  import { installClaude } from "./claude.mjs";
18
18
  import { installCopilot } from "./copilot.mjs";
19
+ import { installCodex } from "./codex.mjs";
19
20
  import { setDryRun, writeFile } from "./_common.mjs";
20
21
  import { sendInstallTelemetry } from "./_telemetry.mjs";
21
22
 
@@ -25,7 +26,7 @@ const PIPELINE_ROOT = dirname(__dirname);
25
26
  const PIPELINE_SRC = `${PIPELINE_ROOT}/pipeline`;
26
27
 
27
28
  /** Tool flags surfaced in CLI help. */
28
- const TOOL_FLAGS = ["--claude", "--copilot"];
29
+ const TOOL_FLAGS = ["--claude", "--copilot", "--codex"];
29
30
 
30
31
  /**
31
32
  * @param {string[]} argv
@@ -69,6 +70,8 @@ export async function runInstall(argv) {
69
70
 
70
71
  const forCopilot = flags.includes("--copilot") || flags.includes("--all");
71
72
 
73
+ const forCodex = flags.includes("--codex") || flags.includes("--all");
74
+
72
75
  const useSymlinks = flags.includes("--link");
73
76
  const indexOnly = flags.includes("--index-only");
74
77
  const platformFlag = parsePlatformFlag(flags);
@@ -89,10 +92,11 @@ export async function runInstall(argv) {
89
92
 
90
93
  if (forClaude) installClaude(installerCtx);
91
94
  if (forCopilot) installCopilot(installerCtx);
95
+ if (forCodex) installCodex(installerCtx);
92
96
 
93
97
  // Version marker: update-check.sh falls back to this for npx-only installs
94
98
  // that have no pipeline repo clone to read a version from.
95
- writeVersionMarkers({ home: HOME, forClaude, forCopilot });
99
+ writeVersionMarkers({ home: HOME, forClaude, forCopilot, forCodex });
96
100
 
97
101
  if (dryRun) {
98
102
  console.log("");
@@ -101,7 +105,7 @@ export async function runInstall(argv) {
101
105
  return;
102
106
  }
103
107
 
104
- printSummary({ forClaude, forCopilot });
108
+ printSummary({ forClaude, forCopilot, forCodex });
105
109
 
106
110
  // Fire-and-forget; install is already done.
107
111
  sendInstallTelemetry({ packageRoot: PIPELINE_ROOT, flags });
@@ -111,9 +115,9 @@ export async function runInstall(argv) {
111
115
  * Stamp the installed package version into each target so the advisory
112
116
  * update check works without a repo clone. Failure is non-fatal.
113
117
  *
114
- * @param {{home: string, forClaude: boolean, forCopilot: boolean}} opts
118
+ * @param {{home: string, forClaude: boolean, forCopilot: boolean, forCodex?: boolean}} opts
115
119
  */
116
- export function writeVersionMarkers({ home, forClaude, forCopilot }) {
120
+ export function writeVersionMarkers({ home, forClaude, forCopilot, forCodex }) {
117
121
  let version;
118
122
  try {
119
123
  version = JSON.parse(readFileSync(join(PIPELINE_ROOT, "package.json"), "utf-8")).version;
@@ -124,6 +128,7 @@ export function writeVersionMarkers({ home, forClaude, forCopilot }) {
124
128
  const targets = [];
125
129
  if (forClaude) targets.push(join(home, ".claude", ".pipeline-version"));
126
130
  if (forCopilot) targets.push(join(home, ".copilot", ".pipeline-version"));
131
+ if (forCodex) targets.push(join(home, ".codex", ".pipeline-version"));
127
132
  for (const target of targets) {
128
133
  try {
129
134
  writeFile(target, `${version}\n`);
@@ -150,7 +155,7 @@ function parsePlatformFlag(flags) {
150
155
  }
151
156
 
152
157
  function printSummary(opts) {
153
- const { forClaude, forCopilot } = opts;
158
+ const { forClaude, forCopilot, forCodex } = opts;
154
159
 
155
160
  console.log(" Installation complete!");
156
161
  console.log("");
@@ -168,9 +173,20 @@ function printSummary(opts) {
168
173
  console.log(" Just describe your task - Copilot will follow the pipeline");
169
174
  console.log("");
170
175
  }
176
+ if (forCodex) {
177
+ console.log(" Codex CLI:");
178
+ console.log(' /multi-agent "MOBILE-123" Start a task (prompt form)');
179
+ console.log(' $multi-agent "MOBILE-123" Start a task (skill form)');
180
+ console.log(" Sub-commands are refs, not skills - the orchestrator loads them on demand");
181
+ console.log(" Run `codex login` first if you have not already");
182
+ console.log("");
183
+ }
171
184
 
172
185
  console.log(" For UI testing, also install the mobile MCP server:");
173
- console.log(" Add to your MCP config: npx @mmerterden/dev-toolkit-mcp");
186
+ if (forCodex) {
187
+ console.log(" Codex: registered automatically (codex mcp add dev-toolkit)");
188
+ }
189
+ console.log(" Otherwise add to your MCP config: npx @mmerterden/dev-toolkit-mcp");
174
190
  console.log("");
175
191
  console.log(" Uninstall everything: npx @mmerterden/multi-agent-pipeline uninstall");
176
192
  console.log(" (Personal access tokens in keychain are preserved.)");
@@ -0,0 +1,45 @@
1
+ # Multi-Agent Development Pipeline
2
+
3
+ This block is managed by `multi-agent-pipeline`. Edit anything outside it freely;
4
+ the installer replaces only the span up to the end marker.
5
+
6
+ The pipeline is an 8-phase development workflow (analysis, planning, TDD dev,
7
+ parallel review + triage, test, commit, report). It is invoked as `/multi-agent`
8
+ or `$multi-agent`, and the orchestrator spec lives at
9
+ `$HOME/.codex/skills/multi-agent/SKILL.md`.
10
+
11
+ ## Host adaptation (Codex CLI)
12
+
13
+ | Concern | On Codex |
14
+ | ---------------- | ---------------------------------------------------------------------------------------------------------------------------- |
15
+ | Orchestrator | one skill, `multi-agent` - it reads per-command specs on demand from `$HOME/.codex/multi-agent-refs/commands/<cmd>/SKILL.md` |
16
+ | Phase specs | `$HOME/.codex/multi-agent-refs/phases/phase-N-*.md`, read only when that phase runs |
17
+ | Personas | `$HOME/.codex/agents/*.toml` (generated from the pipeline's persona set) |
18
+ | Progress display | the native `update_plan` tool, one plan step per phase, plus `phase-tracker.sh` state on disk |
19
+ | Shared state | task logs and preferences stay under `$HOME/.claude/` - that path is the cross-host state root, not a Claude Code dependency |
20
+
21
+ ## Sub-agent delegation is authorized for this pipeline
22
+
23
+ Codex does not spawn sub-agents unless instructions explicitly ask for it. The
24
+ pipeline's Phase 1 (parallel codebase scan) and Phase 4 (parallel review) depend
25
+ on real fan-out, so **delegation is authorized whenever a `/multi-agent` or
26
+ `$multi-agent` run is active**, under these rules:
27
+
28
+ 1. **Always pass `fork_turns: "none"`** (or a positive integer) on a
29
+ `spawn_agent` that sets `model` or `reasoning_effort`. A full-history fork
30
+ inherits the parent model and **silently discards the override**, which would
31
+ collapse the whole review panel onto one model with no error.
32
+ 2. **At most 3 concurrent children.** There are 4 concurrency slots including the
33
+ orchestrator, so a 3-reviewer panel is the ceiling; a 4th child queues rather
34
+ than running in parallel.
35
+ 3. Outside an active pipeline run, the default applies: do not delegate unless
36
+ asked.
37
+
38
+ ## Skills-block budget
39
+
40
+ Codex assembles every discovered skill's name + description into one prompt block
41
+ and **silently drops entries when that block overflows**. The pipeline therefore
42
+ contributes exactly **one** skill (`multi-agent`) and keeps its 42 sub-command
43
+ specs as reference files that cost nothing until read. Do not convert those specs
44
+ into peer skills: doing so evicts other skills, including ones from installed
45
+ plugins, with no error surfaced.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@mmerterden/multi-agent-pipeline",
3
- "version": "12.11.0",
4
- "description": "8-phase AI development pipeline with full orchestration on Claude Code and Copilot CLI. Analysis, planning, TDD, CLI-aware parallel review with consensus surfacing + Fable triage, default-FAIL evidence gates, secret + intent guards, per-phase cost ledger, persistent learnings memory, wiki generation, commit automation. Token-preserving uninstall.",
3
+ "version": "13.0.0",
4
+ "description": "8-phase AI development pipeline with full orchestration on Claude Code, Copilot CLI and Codex CLI. Analysis, planning, TDD, CLI-aware parallel review with consensus surfacing + Fable triage, default-FAIL evidence gates, secret + intent guards, per-phase cost ledger, persistent learnings memory, wiki generation, commit automation. Token-preserving uninstall.",
5
5
  "type": "module",
6
6
  "main": "index.js",
7
7
  "exports": {
@@ -42,7 +42,9 @@
42
42
  "frontend",
43
43
  "jira",
44
44
  "github-issues",
45
- "automation"
45
+ "automation",
46
+ "codex",
47
+ "codex-cli"
46
48
  ],
47
49
  "author": "Mert Erden",
48
50
  "license": "MIT",
@@ -19,6 +19,7 @@
19
19
  4. Review -> deterministic gates + parallel review + Fable triage
20
20
  - Claude Code: Opus + Sonnet (2 paralel)
21
21
  - Copilot CLI: GPT-5.4 + Opus + Sonnet (3 paralel)
22
+ - Codex CLI: gpt-5.6 (xhigh) + gpt-5.4 + gpt-5.6 (medium) (3 paralel)
22
23
 
23
24
  ### Strict Rules
24
25
 
@@ -48,7 +48,7 @@ Classification schema lives in `$HOME/.claude/multi-agent-refs/_input-parser.md`
48
48
  | 7 | `issue` | full picker | account → repo (multi) → issue → maturity → dev-context |
49
49
  | 8 | Free-text | `freetext` | account → repo (single) → dev-context (maturity skip) |
50
50
 
51
- **Rule**: Whatever the type, **account is always asked** (autopilot picks a default). After issue fetch, **maturity check is mandatory** - blockers halt the pipeline. All picker UI strings render in English (`promptLanguage` is locked to `en`).
51
+ **Rule**: Whatever the type, **account is always asked** (autopilot picks a default). After issue fetch, **maturity check is mandatory** - blockers halt the pipeline. Picker `label` + `header` render in English (`promptLanguage` is locked to `en`); the `question` and each option's `description` render in `outputLanguage`, per the canonical matrix in `multi-agent-refs/rules.md`.
52
52
 
53
53
  Lib scripts (`~/.claude/lib/`):
54
54
  - `account-resolver.sh` - keychain account inventory
@@ -57,19 +57,44 @@ The agent CANNOT make these Phase 0 decisions automatically; it suggests, then w
57
57
  - Show the chosen provider in the picker label (e.g. "Bitbucket account?" not "GitHub account?")
58
58
  3. Project root (state inheritance is FORBIDDEN - picker runs every run)
59
59
  4. dev-context picker (extra repos / submodules - read `.gitmodules` and suggest)
60
- 5. **Network reachability gate (run before branch picker)**: Test remote with `git ls-remote --heads origin <baseBranch>` (5s timeout). One of three outcomes:
61
- - **Reachable** → proceed to base branch picker, fetch latest, create worktree from `origin/<baseBranch>`
62
- - **Unreachable** (DNS resolve fail / timeout) → STOP and ask the user:
60
+ 5. **Remote reachability gate (run before branch picker)**: Test remote with
61
+ `git ls-remote --heads origin <baseBranch>` (5s timeout), capturing **stderr**.
62
+
63
+ **Classify the failure before naming a cause.** A non-zero exit is not evidence
64
+ of a network problem, and telling the user to enable a VPN for an auth failure
65
+ sends them to fix something that was never broken. Match stderr, in this order:
66
+
67
+ | stderr contains | Cause | What to offer |
68
+ |---|---|---|
69
+ | `could not read Password`, `Authentication failed`, `terminal prompts disabled`, `Invalid username or password`, `403` | **credential**, not network | the remote URL usually embeds a username with no credential-helper entry. Offer: store the PAT in the credential helper (`git credential approve`, or re-run `/multi-agent:setup`), or switch the remote to SSH. A VPN cannot fix this. |
70
+ | `Could not resolve host`, `Operation timed out`, `Connection refused`, `Network is unreachable`, or the 5s timeout fired with no output | **network** | the VPN/DNS prompt below |
71
+ | `Repository not found`, `does not appear to be a git repository`, `404` | **wrong remote** | show `git remote -v` and ask which remote is correct |
72
+ | anything else | **unknown** | print the stderr line verbatim and ask; never assert a cause you did not observe |
73
+
74
+ Outcomes:
75
+ - **Reachable** → proceed to base branch picker, fetch latest, create worktree
76
+ from `origin/<baseBranch>`
77
+ - **Credential / wrong remote** → STOP with the matching remedy above. Do **not**
78
+ offer "continue from the local ref": the base being stale is unrelated to the
79
+ actual failure, so accepting staleness here trades correctness for nothing.
80
+ - **Network** → STOP and ask:
63
81
  ```
64
- [N/Total] Network gate
65
- Detected: <host> unreachable (VPN/DNS).
82
+ [N/Total] Remote gate
83
+ Observed: <the stderr line, verbatim>
84
+ Classified: <host> unreachable (network).
66
85
  Options:
67
86
  1. Enable VPN and retry (recommended - branch will be fresh)
68
87
  2. Continue from local origin/<baseBranch> ref (may be stale)
69
88
  3. Cancel
70
89
  Confirm? [1/2/3]
71
90
  ```
72
- - User picks `2` → log warning + record `"baseRefFreshness": "stale"` in `agent-state.json`, proceed from local ref. Phase 6 push needs network anyway, so re-prompt VPN there if still unreachable.
91
+ - User picks `2` → log warning + record `"baseRefFreshness": "stale"` in
92
+ `agent-state.json`, proceed from local ref. Phase 6 push needs network anyway,
93
+ so re-prompt there if still unreachable.
94
+
95
+ Always show what was **observed** next to what was **classified**. The previous
96
+ wording asserted `Detected: <host> unreachable (VPN/DNS)` for every failure mode,
97
+ including a plain missing-credential error that returns in under a second.
73
98
  6. Base branch picker (show recents, take an explicit pick)
74
99
  7. Branch name (suggest, allow edit)
75
100
  8. Maturity flags acknowledge (read from the issue body's Progress table)
@@ -56,7 +56,7 @@ Phases 1-3 (Analysis / Planning / Dev) are skipped by design - `finish` treats
56
56
  - interactive: present them and ask (`AskUserQuestion`) whether to fix now (loop back through a minimal Phase-3-style TDD fix) or proceed;
57
57
  - `autopilot` (or `prefs.global.finish.autoFix == true`): auto-fix accepted blocking/important findings, then re-review the fix, before advancing.
58
58
  - **Phase 5 Build+Test** - the **automated success gate** (this is what "build+test success" means here; the interactive device user-test is `/multi-agent:manual-test`). Stack-aware: build via `figma-config.build` (iOS scheme / Android gradle / detected backend/frontend build) and run the existing test suite if present (`swift test` / `xcodebuild test` / `./gradlew test` / `pytest` / `npm test` / `vitest`). Require success to advance; on failure, surface logs and (interactive) stop or (autopilot) attempt a bounded fix loop. **If the repo has no tests, report "no tests present" - never fabricate test results.**
59
- - **Phase 6 Commit/PR** - per `$HOME/.claude/multi-agent-refs/phases/phase-6-commit.md`: stage + commit any remaining local changes with a conventional message (`{type}(scope): desc [{JIRA_KEY}-{id}]`), push, and open a PR **only if one does not already exist** for the branch. PR body per `$HOME/.claude/multi-agent-refs/pipeline-output-formatting` and `rules/git-conventions` - `Ref: #N` / `Related: #N`, never `Closes/Fixes/Resolves`; NO AI/bot attribution anywhere.
59
+ - **Phase 6 Commit/PR** - per `$HOME/.claude/multi-agent-refs/phases/phase-6-commit.md`: stage + commit any remaining local changes with a conventional message (`{type}(scope): desc [{JIRA_KEY}-{id}]`), push, and open a PR **only if one does not already exist** for the branch. PR body per `$HOME/.claude/rules/pipeline-output-formatting.md` and `$HOME/.claude/rules/git-conventions.md` - `Ref: #N` / `Related: #N`, never `Closes/Fixes/Resolves`; NO AI/bot attribution anywhere.
60
60
  - **Phase 7 Report** - per `$HOME/.claude/multi-agent-refs/phases/phase-7-report.md` + `channels.md`: produce the **technical analysis** and **test scenarios**, then post to the configured channels. Default content for `finish`: a Jira **comment** carrying the technical analysis + the test scenarios (and, when the PR was opened, the PR description). Every body runs through the humanizer; bot/tool/AI signatures are FORBIDDEN in comments.
61
61
 
62
62
  ## Modes
@@ -15,7 +15,7 @@ Two-axis language preference for the pipeline:
15
15
  | `promptLanguage` | Interactive prompts during a pipeline run (account picker, project picker, dev-context picker, base-branch picker, branch-name picker, maturity ack, channels picker, Phase 5 test prompt, Phase 6 local-checkout prompt) | `prefs.global.promptLanguage` | **Fixed to `en`** - never toggled by this skill |
16
16
  | `outputLanguage` | Assistant's explanations, status updates, error messages, and pipeline-generated reports rendered to the user (NOT external payloads) | `prefs.global.outputLanguage` | Toggled by this skill |
17
17
 
18
- **Why promptLanguage is fixed:** Pipeline picker UI, confirmation prompts, and error messages are authored in English to keep tooling consistent across CLIs. Mixing Turkish prompts into otherwise-English skill output looks inconsistent. Only the assistant's free-form replies follow `outputLanguage`.
18
+ **Why promptLanguage is fixed:** it governs the picker's structural chrome only - `AskUserQuestion` `label` (button text) and `header` (chip), host error UI, and internal contract identifiers. Those stay English so tooling reads the same across CLIs. Everything a user actually reads follows `outputLanguage`, including the picker `question` and each option's `description`, per the canonical per-field matrix in `multi-agent-refs/rules.md`. A picker whose question is English on a Turkish run is a bug, not the contract.
19
19
 
20
20
  **Always English regardless of either field**: commit messages, PR titles/bodies, Jira comments, wiki pages, reviewer/triage system prompts, agent-log.md payloads, confirmation / error UI exposed by the CLI host.
21
21
 
@@ -109,5 +109,5 @@ Render in the **new** outputLanguage:
109
109
 
110
110
  - `setup.md` Step 0 - first-run language picker (asks only outputLanguage; promptLanguage seeded as `en`)
111
111
  - `prefs.schema.json` - `global.promptLanguage` (fixed `"en"`) and `global.outputLanguage` definitions
112
- - `phase-5-test.md`, `phase-6-commit.md` - read promptLanguage for their interactive prompts (always English)
112
+ - `phase-5-test.md`, `phase-6-commit.md` - interactive prompts follow the per-field matrix: `question` + `description` in `outputLanguage`, `label` + `header` English
113
113
  - `help.md` - reads outputLanguage to render its own help text
@@ -24,7 +24,7 @@ Ask BEFORE anything else, so every subsequent setup prompt and the rest of the p
24
24
 
25
25
  | Axis | What it controls | Configurable? |
26
26
  |---|---|---|
27
- | `promptLanguage` | Interactive pickers and prompts during a pipeline run (account, project, dev-context, branch, channels, Phase 5 test, Phase 6 confirmations) | **No - fixed to `en`.** Picker UI is always English to keep tooling consistent across CLIs. |
27
+ | `promptLanguage` | The picker's structural UI chrome only: `AskUserQuestion` `label` (button text) and `header` (chip), plus host error UI and internal contract identifiers | **No - fixed to `en`.** Button and chip text stays English so tooling reads the same across CLIs. |
28
28
  | `outputLanguage` | The assistant's non-interactive explanations, status updates, error messages, and pipeline-generated reports rendered to the user | Yes - set here and changeable later via `/multi-agent:language <en\|tr>` |
29
29
 
30
30
  `promptLanguage` is seeded as `"en"` and never offered to the user. **External payloads always stay English** (commits, PR titles/bodies, Jira comments, wiki content, reviewer/triage prompts, agent-log.md).
@@ -51,7 +51,16 @@ Select (1-2):
51
51
  - `tr`: `outputLanguage=<Y>. promptLanguage "en" sabit. Dış çıktılar (PR açıklamaları, Jira yorumları, review istemleri) her durumda İngilizce kalır.`
52
52
  - **Unknown input** → re-ask once, then abort setup with a hint pointing to `/multi-agent:language`.
53
53
 
54
- From this point forward, every prompt in Steps 1-6 below (discovery summaries, Token Save Flow Steps A-D, identity binding, repo picker) renders in English (per the fixed `promptLanguage`). The setup wizard's own status text and the post-setup summary render in `outputLanguage`. External payloads remain English.
54
+ From this point forward, every prompt in Steps 1-6 below follows the canonical
55
+ per-field matrix in `$HOME/.claude/multi-agent-refs/rules.md` ("Language Application"):
56
+ the `question` text and each option's `description` render in `outputLanguage`, while
57
+ `label` and `header` stay English. The wizard's status text and post-setup summary
58
+ render in `outputLanguage`. External payloads remain English.
59
+
60
+ > This paragraph used to say every prompt renders in English "per the fixed
61
+ > `promptLanguage`", which contradicted the canonical matrix and produced
62
+ > half-English pickers on Turkish runs: `promptLanguage` governs only the button and
63
+ > chip chrome, never the question a user reads.
55
64
 
56
65
  ### Step 0.5 - Credential Backend Check
57
66
 
@@ -102,6 +111,12 @@ These are the RECOMMENDED key names. When creating NEW keys, use these. But exis
102
111
  | `graylog` | `${USER}_Graylog_Access_Token` | `graylog` |
103
112
  | `firebase` | `${USER}_Firebase_Access_Json` | `firebase` (any variant: `sa`, `service`, `account`, `access`, `json`) |
104
113
  | `jenkins` | `${USER}_Jenkins_Access_Token` | `jenkins` |
114
+ | `appstore_connect_key_id` | `${USER}_AppStoreConnect_Key_Id` | (`appstore` or `asc` or `app_store`) + (`key` or `keyid`) |
115
+ | `appstore_connect_issuer_id` | `${USER}_AppStoreConnect_Issuer_Id` | (`appstore` or `asc` or `app_store`) + `issuer` |
116
+ | `appstore_connect_apple_id` | `${USER}_AppStoreConnect_Apple_Id` | (`appstore` or `asc`) + (`apple` or `account` or `user`) |
117
+ | `appstore_connect_password_item` | `${USER}_AppStoreConnect_Password_Item` | (`appstore` or `asc` or `altool`) + (`password` or `app_specific`) |
118
+
119
+ > The four App Store Connect entries are **iOS-only and optional**: skip them all and the pipeline still works, it just reports Gate 2 of `/multi-agent:testflight-validation` as `SKIPPED` (never as a pass). They mirror the Figma 3-tier shape - Tier 1 = API key (`appstore_connect_key_id` + `appstore_connect_issuer_id`), Tier 2 = Apple ID + app-specific password (`appstore_connect_apple_id` + `appstore_connect_password_item`), Tier 3 = nothing configured. **Offer Tier 2 first when the user says they cannot create an API key**: creating one needs an Admin or App Manager role in App Store Connect, while an app-specific password is generated by the account holder at `appleid.apple.com` with no team permission at all. Two of these hold identifiers rather than secrets (key id, issuer id) and one holds a keychain ITEM NAME, not a password - they still go through the mapping layer so every credential is read the same way. Onboarding mechanics in Step 3b.
105
120
 
106
121
  **1c. Resolution logic (per service):**
107
122
 
@@ -352,6 +367,58 @@ This builds `platformIdentityRouting` incrementally - no separate Step 7 neede
352
367
 
353
368
  ---
354
369
 
370
+ ### Step 3b - App Store Connect onboarding (iOS only, optional)
371
+
372
+ Runs inside Step 3 alongside the other missing credentials, not as a late add-on:
373
+ a user who already has an App Store Connect credential in their keychain gets it
374
+ mapped by Step 1 discovery like any other token, and only the genuinely missing
375
+ pieces reach this flow.
376
+
377
+ Three of the four entries do not go through the normal Token Save Flow, because
378
+ what they hold is not a pasteable secret:
379
+
380
+ | Entry | Holds | Flow |
381
+ |---|---|---|
382
+ | `appstore_connect_key_id` | an identifier | plain value, not a secret; still mapped so it is read through the mapping layer |
383
+ | `appstore_connect_issuer_id` | an identifier | same |
384
+ | `appstore_connect_apple_id` | an email address | same |
385
+ | `appstore_connect_password_item` | a keychain ITEM NAME | the password lives in Apple's own keychain item, referenced as `-p @keychain:<item>` and never read by the pipeline |
386
+
387
+ Ask which tier to configure (picker): **API key** / **Apple ID + app-specific
388
+ password** / **Skip**. Lead with the second when the user says they cannot create
389
+ an API key.
390
+
391
+ **API key.** The private key is a FILE and is never copied into the credential
392
+ store. It must sit in a directory `altool` already searches:
393
+
394
+ ```bash
395
+ ls ~/.appstoreconnect/private_keys/AuthKey_*.p8 2>/dev/null \
396
+ || echo "MISSING: put AuthKey_<keyId>.p8 in ~/.appstoreconnect/private_keys/"
397
+ ```
398
+
399
+ **Apple ID + app-specific password.** Use Apple's own keychain helper. The secret
400
+ never enters chat and never becomes a shell argument, per the Token Save Flow rule:
401
+
402
+ ```bash
403
+ # the user exports AC_PASSWORD_ONCE in their own shell, for this one command
404
+ xcrun altool --store-password-in-keychain-item "<item-name>" \
405
+ -u "<apple-id>" -p @env:AC_PASSWORD_ONCE
406
+ ```
407
+
408
+ Then map only `<item-name>` as `appstore_connect_password_item`.
409
+
410
+ **Multi-provider accounts.** A corporate Apple ID often belongs to several
411
+ providers, and `altool` fails opaquely without one. Resolve it once with
412
+ `ios_testflight_validate({list_providers: true, <credentials just configured>})`
413
+ and store the answer under
414
+ `prefs.projects[<key>].appStoreConnect.providerPublicId` - per-project, since a
415
+ user can ship for more than one team.
416
+
417
+ **Verify + expiry.** Re-run the `list_providers` probe and report the resolved
418
+ tier. A credential that resolves but is rejected (401/403) follows the
419
+ Expired-token decision in `refs/keychain.md` Rule 1 - Regenerate / Use a
420
+ different token / Skip and continue - never a silent drop.
421
+
355
422
  ### Step 3.5 - Host Prompt (embedded in Token Save Flow)
356
423
 
357
424
  **Not a standalone step** - runs inline at the end of the Token Save Flow whenever the saved token belongs to a **hosted service** (Jira, Confluence, Bitbucket, Fortify, Graylog) AND the host is not yet in `prefs.global.hosts`. Firebase tokens skip this step - Crashlytics is always on Google's fixed domains and `project_id` is embedded in the SA JSON.
@@ -27,6 +27,7 @@ When invoked, it synchronizes all targets in order. It detects what changed, upd
27
27
  |---|-------|-----|-----|
28
28
  | 1 | Claude Code (source of truth) | `~/.claude/commands/multi-agent.md` + `~/.claude/commands/multi-agent/` + `~/.claude/agents/` + `~/.claude/scripts/` + `~/.claude/lib/` | source |
29
29
  | 2 | Copilot CLI | `~/.copilot/copilot-instructions.md` + `~/.copilot/skills/` | <- from Claude |
30
+ | 2b | Codex CLI | `~/.codex/AGENTS.md` + `~/.codex/skills/multi-agent/` + `~/.codex/multi-agent-refs/` + `~/.codex/agents/*.toml` | <- from Claude (path-rewritten) |
30
31
  | 3 | multi-agent-pipeline repo | `~/multi-agent-pipeline/pipeline/` | <- from Claude (genericized) |
31
32
  | 4 | Website | `{owner}/{website-host}` | <- version + features |
32
33
  | 5 | dev-toolkit MCP server | resolved from `prefs.global.devToolkit` or the `mcpServers` registration | own repo: gate, commit, publish |
@@ -58,7 +59,8 @@ Run every step automatically:
58
59
  ```
59
60
  Step 1: PLATFORM Detect macOS / Linux / Windows (Git Bash / WSL); export PLATFORM env
60
61
  Step 1.5: DETECT Compare timestamps, find stale targets
61
- Step 2: COPILOT Claude Code -> Copilot CLI (instructions + 42 sub-command skills)
62
+ Step 2: COPILOT Claude Code -> Copilot CLI (instructions + 43 sub-command skills)
63
+ Step 2b: CODEX Claude Code -> Codex CLI (1 router skill + 43 specs as refs + 8 agent TOML)
62
64
  Step 3: REPO Claude Code -> pipeline repo (genericized, personal data scrub, bash -n on all sh)
63
65
  Step 3c: PLUGINS pipeline shared/external -> multi-agent-plugins marketplace (rebuild knowledge/,
64
66
  bump changed plugins' patch version, commit + push the plugins repo)
@@ -169,6 +171,50 @@ If nothing is stale → report "All targets up to date" and stop.
169
171
 
170
172
  ---
171
173
 
174
+ ## Codex Sync (Step 2b)
175
+
176
+ Unlike the Copilot step, this one does **not** hand-copy files. The Codex tree is a
177
+ *transform* of the Claude tree, not a mirror of it, and the transform is real work:
178
+
179
+ - the 43 sub-command specs become reference files, because Codex silently truncates
180
+ its skills block (see `cross-cli-contract.md` 2.6 for the measurement)
181
+ - every `$HOME/.claude/...` reference to a CLI-owned tree is retargeted, with
182
+ `agents/<persona>.md` becoming `.toml` and the dispatcher becoming the router skill
183
+ - the 8 personas are regenerated as TOML with a model + reasoning-effort mapping
184
+ - shared state (`logs/`, prefs, `knowledge/`) is deliberately NOT retargeted
185
+
186
+ That logic lives in `install/codex.mjs` and is gate-locked by
187
+ `smoke-codex-install.sh`. Re-describing it here in prose would give the pipeline two
188
+ definitions of the same transform, and the prose copy would be the one that rots. So
189
+ the step runs the installer's Codex target:
190
+
191
+ ```bash
192
+ cd "$HOME/multi-agent-pipeline"
193
+ node install.js --codex
194
+ ```
195
+
196
+ **Verify** (the installer is quiet about correctness, only about counts):
197
+
198
+ ```bash
199
+ # exactly one pipeline skill: more than one means the skills block will truncate
200
+ ls -1 "$HOME/.codex/skills" | grep -c '^multi-agent$' # want 1
201
+ # every sub-command reachable as a ref
202
+ find "$HOME/.codex/multi-agent-refs/commands" -name SKILL.md | wc -l # want the command count
203
+ # no reference left pointing at a tree a Codex-only install does not have
204
+ grep -rhoE '(\$HOME|~)/\.claude/(agents|scripts|lib|schemas|commands|multi-agent-refs|rules)' \
205
+ "$HOME/.codex/skills" "$HOME/.codex/multi-agent-refs" | sort -u # want empty
206
+ ```
207
+
208
+ **MCP**: the installer runs `codex mcp add dev-toolkit` itself. It is idempotent, and
209
+ it is skipped with a warning when `codex` is not on `PATH` - never hand-edit
210
+ `~/.codex/config.toml`, which Codex owns (marketplace and plugin state live there
211
+ too).
212
+
213
+ **Order matters**: run this AFTER Step 3 (REPO), because the installer reads the repo
214
+ tree. Running it before means Codex gets the previous revision.
215
+
216
+ ---
217
+
172
218
  ## Stack-Plugin Sync (Step 3c)
173
219
 
174
220
  Stack skills are distributed as versioned plugins in the `mmerterden/multi-agent-plugins` marketplace. The pipeline's `pipeline/skills/shared/external/` is the **single authoring source**; the marketplace is a derived, versioned publish artifact. This step rebuilds it and publishes only when something changed.
@@ -283,7 +329,79 @@ git tag "v$NEW" && git push origin main --tags
283
329
 
284
330
  Publish to the registry declared in `publishConfig` through a throwaway userconfig. Never edit `~/.npmrc`, and never a bare `npm publish` - it lands on whichever registry the ambient config happens to name.
285
331
 
286
- **The token depends on the registry**: `npm.pkg.github.com` authenticates with a GitHub PAT carrying `write:packages` (logical key `github`), `registry.npmjs.org` with an npm token (logical key `npm`). Picking the wrong one fails with a 401 that reads like a missing package.
332
+ **The token depends on the registry AND on the package scope.** The registry picks
333
+ the credential *type*; the scope picks the *account*. Resolving on host alone
334
+ misroutes on any machine with more than one GitHub identity, which is the common
335
+ case for anyone with a work and a personal account.
336
+
337
+ | Registry | Credential type | Account chosen by |
338
+ |---|---|---|
339
+ | `npm.pkg.github.com` | GitHub PAT with `write:packages` (Classic; fine-grained PATs are not fully supported for Packages) | the package scope, e.g. `@{owner}` → the `{owner}` account's PAT |
340
+ | `registry.npmjs.org` | npm token (logical key `npm`) | the npm account that owns the scope |
341
+
342
+ Resolve the scope first, then the key:
343
+
344
+ ```bash
345
+ SCOPE=$(node -p "(require('./package.json').name.match(/^@([^/]+)/)||[])[1] || ''")
346
+ # Prefer a scope-specific mapping; fall back to the generic key only when the
347
+ # machine has exactly one GitHub identity.
348
+ KEY=$(node -e '
349
+ const fs=require("fs"),os=require("os"),p=os.homedir()+"/.claude/multi-agent-preferences.json";
350
+ const km=(JSON.parse(fs.readFileSync(p,"utf8")).global||{}).keychainMapping||{};
351
+ const scope=process.argv[1];
352
+ process.stdout.write(km[`github_${scope}`] ? `github_${scope}` : "github");
353
+ ' "$SCOPE")
354
+ ```
355
+
356
+ **Pre-flight the scope, do not learn it from a 403.** GitHub tells you a token's
357
+ scopes on any authenticated request, so check before uploading rather than after:
358
+
359
+ ```bash
360
+ scope_ok() { # $1 = token; prints the login, non-zero when write:packages is absent
361
+ local hdrs; hdrs=$(curl -sI -H "Authorization: token $1" https://api.github.com/user)
362
+ printf '%s' "$hdrs" | grep -i '^x-oauth-scopes:' | grep -q 'write:packages'
363
+ }
364
+ ```
365
+
366
+ **Candidate order for a `npm.pkg.github.com` publish.** The scope matters more than
367
+ where the token is stored, and the two are not correlated: on a machine with a work
368
+ and a personal identity, the hand-made PAT in the credential store may be the wrong
369
+ account or the right account without `write:packages`, while the `gh` CLI's own
370
+ OAuth token for that account often has it.
371
+
372
+ 1. `github_<scope>` from `keychainMapping` (a PAT deliberately onboarded for this scope)
373
+ 2. `gh auth token -u <scope>` - gh's stored OAuth token for that account
374
+ 3. the generic `github` key - only when the machine has one GitHub identity
375
+
376
+ Take the first candidate that passes `scope_ok` AND whose `login` matches the
377
+ package scope. If none qualifies, stop before `npm publish` and report which
378
+ candidates were tried, what login each resolved to, and which scope was missing.
379
+ That report is the actionable output; a 403 body is not.
380
+
381
+ Measured on this machine, which is why the order is what it is:
382
+
383
+ | Candidate | login | scopes |
384
+ |---|---|---|
385
+ | `keychainMapping.github` | corporate EMU account | cannot publish to a personal scope under any grant |
386
+ | `mmerterden_Github_Auth_Token` | personal | `admin:public_key, gist, read:org, repo` - no `write:packages` |
387
+ | `gh auth token -u mmerterden` | personal | `gist, read:org, repo, workflow, write:packages` ✓ |
388
+
389
+ **Two 403s mean two different things, and neither says "wrong token" plainly:**
390
+
391
+ - `Unauthorized: As an Enterprise Managed User, you cannot access this content` -
392
+ the resolved token belongs to a corporate EMU account, which cannot publish to a
393
+ personal scope at all. The mapping points at the wrong identity. Map the personal
394
+ account's PAT under `github_<scope>` and re-run; do not "fix" this by granting
395
+ the EMU token more scopes, because no scope makes an EMU account able to write
396
+ to a personal namespace.
397
+ - `The token provided does not match expected scopes` - right account, missing
398
+ permission. The PAT needs **Classic** with `write:packages` (plus `repo` for a
399
+ private package). Regenerate it at
400
+ `https://github.com/settings/tokens/new?scopes=write:packages,read:packages,repo`
401
+ and re-onboard via `/multi-agent:setup`.
402
+
403
+ Report which of the two it was. "Permission denied" alone sends the user to
404
+ regenerate a token that was never the problem.
287
405
 
288
406
  ```bash
289
407
  NPMRC=$(mktemp); trap 'rm -f "$NPMRC"' EXIT
@@ -350,6 +468,7 @@ When invoked with the `release` argument:
350
468
  7. DEV-TOOLKIT Ship the companion MCP server if it moved (Step 3d gates, then publish)
351
469
  8. WEBSITE Version + features -> {website-host}
352
470
  9. COPILOT Copilot CLI instructions + skills sync
471
+ 9b. CODEX Codex CLI router skill + refs + agent TOML (node install.js --codex)
353
472
  10. Report Summary: version, touched repos, deploy status
354
473
  ```
355
474
 
@@ -357,20 +476,24 @@ When invoked with the `release` argument:
357
476
 
358
477
  ## Sub-Command Sync (Claude Code <-> Copilot CLI Skills)
359
478
 
360
- This runs on the Claude <-> Copilot axis - the two CLIs the pipeline supports natively.
479
+ > Codex takes the Step 2b path instead; see that section.
480
+
481
+ This runs on the Claude <-> Copilot axis. Codex is NOT synced here: it receives the
482
+ same 43 specs as reference files rather than as peer skills, via Step 2b - see
483
+ `cross-cli-contract.md` 2.6 for why the parity axis differs per host.
361
484
 
362
485
  | Claude Code | Copilot CLI |
363
486
  |-------------|-------------|
364
487
  | `~/.claude/commands/multi-agent/{cmd}/SKILL.md` | `~/.copilot/skills/multi-agent-{cmd}/SKILL.md` |
365
488
 
366
- **42 commands are synced** (canonical inventory - must match `cross-cli-contract.md` section 1; drift = contract violation):
489
+ **43 commands are synced** (canonical inventory - must match `cross-cli-contract.md` section 1; drift = contract violation):
367
490
 
368
491
  ```
369
492
  analysis, analysis-resolve, autopilot, build-optimize, channels, create-jira, design-check, dev,
370
493
  dev-autopilot, dev-local, dev-local-autopilot, diff-explain, finish, forget, garbage-collect,
371
494
  help, issue, jira, kill, language, local,
372
495
  local-autopilot, log, manual-test, prune-logs, purge, refactor, resume, review, review-issue, review-jira,
373
- routines, save, scan, search, setup, stack, status, sync, test, uninstall, update
496
+ routines, save, scan, search, setup, stack, status, sync, test, testflight-validation, uninstall, update
374
497
  ```
375
498
 
376
499
  **NOT synced**: `$HOME/.claude/multi-agent-refs/*` - lazy-load references, Claude Code specific