@mmerterden/multi-agent-pipeline 12.11.0 → 13.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/CHANGELOG.md +120 -0
  2. package/README.md +24 -7
  3. package/index.js +5 -2
  4. package/install/_codex-agents.mjs +211 -0
  5. package/install/_codex-instructions.mjs +33 -0
  6. package/install/_managed-block.mjs +99 -0
  7. package/install/codex.mjs +478 -0
  8. package/install/copilot.mjs +34 -80
  9. package/install/index.mjs +25 -9
  10. package/install/templates/codex-instructions.md +45 -0
  11. package/package.json +5 -3
  12. package/pipeline/claude-md-template.md +1 -0
  13. package/pipeline/commands/multi-agent/SKILL.md +1 -1
  14. package/pipeline/commands/multi-agent/dev/SKILL.md +31 -6
  15. package/pipeline/commands/multi-agent/finish/SKILL.md +1 -1
  16. package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
  17. package/pipeline/commands/multi-agent/setup/SKILL.md +69 -2
  18. package/pipeline/commands/multi-agent/sync/SKILL.md +128 -5
  19. package/pipeline/commands/multi-agent/testflight-validation/SKILL.md +219 -0
  20. package/pipeline/commands/multi-agent/update/SKILL.md +7 -4
  21. package/pipeline/multi-agent-refs/_input-parser.md +1 -1
  22. package/pipeline/multi-agent-refs/cross-cli-contract.md +51 -17
  23. package/pipeline/multi-agent-refs/features/model-fallback.md +29 -0
  24. package/pipeline/multi-agent-refs/features/review-multi-repo.md +1 -1
  25. package/pipeline/multi-agent-refs/phases/log-format.md +1 -1
  26. package/pipeline/multi-agent-refs/phases/phase-0-init.md +14 -2
  27. package/pipeline/multi-agent-refs/phases/phase-4-review.md +36 -5
  28. package/pipeline/multi-agent-refs/progress-contract.md +1 -1
  29. package/pipeline/multi-agent-refs/tracker-contract.md +17 -1
  30. package/pipeline/schemas/prefs.schema.json +296 -62
  31. package/pipeline/schemas/reviewer-output.schema.json +1 -1
  32. package/pipeline/schemas/triage-output.schema.json +1 -1
  33. package/pipeline/scripts/cost-table.json +15 -1
  34. package/pipeline/scripts/smoke-cross-cli-behavior.sh +25 -10
  35. package/pipeline/scripts/uninstall.mjs +105 -9
  36. package/pipeline/scripts/update-check.sh +2 -1
  37. package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
  38. package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +48 -1
  39. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +87 -6
  40. package/pipeline/skills/shared/core/multi-agent-testflight-validation/SKILL.md +120 -0
@@ -3,7 +3,7 @@
3
3
  "$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/reviewer-output.schema.json",
4
4
  "version": "1.0.0",
5
5
  "title": "Multi-Agent Pipeline - Phase 4 reviewer output",
6
- "description": "Contract for a single code-reviewer subagent's JSON output in Phase 4 Step 2. Reviewer set is CLI-aware: Claude Code dispatches 2 parallel reviewers (Opus, Sonnet); Copilot CLI dispatches 3 (GPT-5.4, Opus, Sonnet). Every reviewer must return an object matching this shape before Opus triage merges them.",
6
+ "description": "Contract for a single code-reviewer subagent's JSON output in Phase 4 Step 2. Reviewer set is CLI-aware: Claude Code dispatches 2 parallel reviewers (Fable, Sonnet); Copilot CLI dispatches 3 (Opus, GPT-5.4, Sonnet); Codex CLI dispatches 3 (gpt-5.6 at xhigh, gpt-5.4, gpt-5.6 at medium). Every reviewer must return an object matching this shape before Opus triage merges them.",
7
7
  "type": "object",
8
8
  "additionalProperties": false,
9
9
  "required": ["findings", "approved"],
@@ -86,7 +86,7 @@
86
86
  "reviewerCount": {
87
87
  "type": "integer",
88
88
  "minimum": 1,
89
- "description": "Reviewers dispatched this iteration (Claude Code = 2, Copilot CLI = 3)."
89
+ "description": "Reviewers dispatched this iteration (Claude Code = 2, Copilot CLI = 3, Codex CLI = 3)."
90
90
  },
91
91
  "verdict": {
92
92
  "type": "string",
@@ -32,7 +32,21 @@
32
32
  "outPerMtok": 30.0,
33
33
  "cacheReadPerMtok": 1.0,
34
34
  "modelId": "gpt-5.4",
35
- "note": "Copilot CLI Reviewer 2 - approximate; verify against OpenAI pricing page before relying on totals."
35
+ "note": "Copilot CLI Reviewer 2 and Codex CLI Reviewer 2 - approximate; verify against OpenAI pricing page before relying on totals."
36
+ },
37
+ "gpt-5.6": {
38
+ "inPerMtok": 10.0,
39
+ "outPerMtok": 30.0,
40
+ "cacheReadPerMtok": 1.0,
41
+ "modelId": "gpt-5.6",
42
+ "note": "Codex CLI top tier - Reviewer 1 at xhigh effort, Reviewer 3 at medium, triage at max, and the fable/opus rungs of the Codex persona tier map. Reasoning effort changes output volume, not the per-token rate, so one entry covers every effort level. Approximate; verify against OpenAI pricing before relying on totals."
43
+ },
44
+ "gpt-5.6-terra": {
45
+ "inPerMtok": 1.0,
46
+ "outPerMtok": 5.0,
47
+ "cacheReadPerMtok": 0.1,
48
+ "modelId": "gpt-5.6-terra",
49
+ "note": "Codex CLI floor tier - speed-optimised, maps the haiku rung of the Codex persona tier map (task-clarifier). Approximate; verify against OpenAI pricing before relying on totals."
36
50
  }
37
51
  }
38
52
  }
@@ -185,11 +185,12 @@ else
185
185
  fi
186
186
 
187
187
  # ──────────────────────────────────────────────────────────────────────────
188
- # CLI-aware reviewer-count contract: Claude Code = 2 reviewers (Opus + Sonnet),
189
- # Copilot CLI = 3 (GPT-5.4 + Opus + Sonnet). Nothing in code enforces the count
190
- # (the orchestrator dispatches per the doc), so lock the CONTRACT here against
191
- # drift across the phase doc, the schema, and the consensus block.
192
- echo "→ reviewer-count contract (Claude=2, Copilot=3)"
188
+ # CLI-aware reviewer-count contract: Claude Code = 2 (Fable + Sonnet), Copilot CLI
189
+ # = 3 (Opus + GPT-5.4 + Sonnet), Codex CLI = 3 (gpt-5.6 xhigh + gpt-5.4 + gpt-5.6
190
+ # medium). Nothing in code enforces the count (the orchestrator dispatches per the
191
+ # doc), so lock the CONTRACT here against drift across the phase doc, the schema,
192
+ # and the consensus block.
193
+ echo "→ reviewer-count contract (Claude=2, Copilot=3, Codex=3)"
193
194
  # `/multi-agent:update` runs this smoke on the user's machine, where the tree is
194
195
  # ~/.claude/{schemas,multi-agent-refs} and NOT <root>/pipeline/*. Resolving the
195
196
  # root by hand here reported three phantom failures in an install, which reads as
@@ -202,16 +203,30 @@ TRSCHEMA="${MA_SCHEMAS:+$MA_SCHEMAS/triage-output.schema.json}"
202
203
 
203
204
  if [ -z "$P4" ] || [ ! -f "$P4" ]; then
204
205
  echo " ↷ SKIP: phase-4-review.md not present in this $MA_LAYOUT layout"
205
- elif grep -qiE "Claude Code (dispatches|=|:) ?2|2-model" "$P4" && grep -qiE "Copilot CLI (dispatches|=|:) ?3|3-model" "$P4"; then
206
- pass "phase-4-review declares Claude=2 / Copilot=3 reviewers"
206
+ elif grep -qiE "Claude Code (dispatches|=|:) ?2|2-model" "$P4" \
207
+ && grep -qiE "Copilot CLI (dispatches|=|:) ?3|3-model" "$P4" \
208
+ && grep -qiE "Codex CLI (dispatches|=|:) ?3|Claude Code 2, Copilot CLI 3, Codex CLI 3" "$P4"; then
209
+ pass "phase-4-review declares Claude=2 / Copilot=3 / Codex=3 reviewers"
207
210
  else
208
- fail "phase-4-review does not declare the CLI-aware reviewer count"
211
+ fail "phase-4-review does not declare the CLI-aware reviewer count for all three hosts"
212
+ fi
213
+
214
+ # The two Codex constraints are silent-failure shaped, so the contract has to name
215
+ # them: a spawn without fork_turns collapses the panel onto one model, and the
216
+ # 4-slot ceiling (orchestrator included) is why the count is 3 and not more.
217
+ if [ -n "$P4" ] && [ -f "$P4" ]; then
218
+ if grep -q 'fork_turns' "$P4" && grep -qiE "concurrency|slots" "$P4"; then
219
+ pass "phase-4-review documents the fork_turns override rule + the concurrency ceiling"
220
+ else
221
+ fail "phase-4-review must document fork_turns and the Codex concurrency ceiling"
222
+ fi
209
223
  fi
210
224
 
211
225
  if [ -z "$REVSCHEMA" ] || [ ! -f "$REVSCHEMA" ]; then
212
226
  echo " ↷ SKIP: reviewer-output.schema.json not present in this $MA_LAYOUT layout"
213
- elif grep -qi "Copilot CLI" "$REVSCHEMA" && grep -qi "Claude" "$REVSCHEMA"; then
214
- pass "reviewer-output schema notes the CLI-aware reviewer set"
227
+ elif grep -qi "Copilot CLI" "$REVSCHEMA" && grep -qi "Claude" "$REVSCHEMA" \
228
+ && grep -qi "Codex" "$REVSCHEMA"; then
229
+ pass "reviewer-output schema notes the CLI-aware reviewer set (all three hosts)"
215
230
  else
216
231
  fail "reviewer-output schema missing the CLI-aware note"
217
232
  fi
@@ -36,6 +36,7 @@
36
36
  */
37
37
 
38
38
  import { existsSync, readdirSync, readFileSync, rmSync, writeFileSync } from "fs";
39
+ import { execFileSync } from "child_process";
39
40
  import { join } from "path";
40
41
  import { pathToFileURL } from "url";
41
42
  import { createInterface } from "readline";
@@ -123,6 +124,54 @@ function rmIfExists(path) {
123
124
  return true;
124
125
  }
125
126
 
127
+ /**
128
+ * Remove generated Codex agent TOML files, identified by the header the
129
+ * installer stamps on them. A user-authored `.toml` sharing a persona name is
130
+ * left alone - `~/.codex/agents/` is a co-owned directory.
131
+ *
132
+ * @param {string} agentsDir
133
+ * @returns {number} files removed
134
+ */
135
+ function removeGeneratedCodexAgents(agentsDir) {
136
+ if (!existsSync(agentsDir)) return 0;
137
+ const MARKER = "# Generated by multi-agent-pipeline install --codex.";
138
+ let removed = 0;
139
+ for (const name of readdirSync(agentsDir)) {
140
+ if (!name.endsWith(".toml")) continue;
141
+ const path = join(agentsDir, name);
142
+ try {
143
+ if (!readFileSync(path, "utf-8").startsWith(MARKER)) continue;
144
+ } catch {
145
+ continue;
146
+ }
147
+ if (rmIfExists(path)) removed++;
148
+ }
149
+ if (removed > 0) console.log(` removed ${removed} generated agent file(s) under ${agentsDir}`);
150
+ return removed;
151
+ }
152
+
153
+ /**
154
+ * Hand the dev-toolkit MCP registration back to Codex.
155
+ *
156
+ * Best-effort: a missing `codex` binary or an already-absent entry is not an
157
+ * uninstall failure. Never edits `config.toml` directly - Codex keeps
158
+ * marketplace and plugin state in the same file.
159
+ */
160
+ function deregisterCodexMcpServer() {
161
+ if (dryRun) {
162
+ report("would run", "codex mcp remove dev-toolkit");
163
+ return;
164
+ }
165
+ try {
166
+ execFileSync("codex", ["mcp", "remove", "dev-toolkit"], { stdio: "pipe", timeout: 20_000 });
167
+ console.log(" removed: dev-toolkit MCP registration");
168
+ } catch {
169
+ console.log(
170
+ " note: could not run `codex mcp remove dev-toolkit` (codex not on PATH, or not registered)",
171
+ );
172
+ }
173
+ }
174
+
126
175
  /**
127
176
  * Remove every directory under `parent` whose name matches a predicate.
128
177
  * @param {string} parent
@@ -160,6 +209,32 @@ function rmMatchingFiles(parent, predicate) {
160
209
  // (v11.4.1+). Must stay in sync with INSTRUCTIONS_END_MARKER there.
161
210
  const COPILOT_END_MARKER = "<!-- multi-agent-pipeline:copilot-instructions:end -->";
162
211
 
212
+ /**
213
+ * Every end marker the installers write. Both host instruction files share the
214
+ * same START marker but terminate with their own comment, so a single-marker
215
+ * lookup silently falls through to the legacy heading-bounded path - and since
216
+ * the managed body has no top-level heading after the first, that path treats
217
+ * ALL trailing user content as part of the block and deletes it.
218
+ */
219
+ const MANAGED_END_MARKERS = [
220
+ COPILOT_END_MARKER,
221
+ "<!-- multi-agent-pipeline:codex-instructions:end -->",
222
+ ];
223
+
224
+ /**
225
+ * Locate whichever managed end marker terminates this block.
226
+ *
227
+ * @param {string} fromStart - content from the start marker onward
228
+ * @returns {{index: number, length: number}} index -1 when none is present
229
+ */
230
+ function findManagedEndMarker(fromStart) {
231
+ for (const marker of MANAGED_END_MARKERS) {
232
+ const index = fromStart.indexOf(marker);
233
+ if (index >= 0) return { index, length: marker.length };
234
+ }
235
+ return { index: -1, length: 0 };
236
+ }
237
+
163
238
  /**
164
239
  * Legacy copilot-instructions files (written before the end marker existed)
165
240
  * have no explicit terminator. Bound the pipeline span at the next top-level
@@ -205,10 +280,10 @@ function stripManagedBlock(filePath) {
205
280
  if (idx >= 0) {
206
281
  const before = content.slice(0, idx).trimEnd();
207
282
  const fromStart = content.slice(idx);
208
- const endIdx = fromStart.indexOf(COPILOT_END_MARKER);
283
+ const end = findManagedEndMarker(fromStart);
209
284
  const rawTrailing =
210
- endIdx >= 0
211
- ? fromStart.slice(endIdx + COPILOT_END_MARKER.length)
285
+ end.index >= 0
286
+ ? fromStart.slice(end.index + end.length)
212
287
  : legacyTrailingContent(fromStart);
213
288
  const trailing = rawTrailing.replace(/^[\r\n]+/, "").trimEnd();
214
289
  let remaining = before;
@@ -332,7 +407,7 @@ export async function main() {
332
407
  );
333
408
  if (forCodex)
334
409
  console.log(
335
- " - OpenAI Codex CLI (~/.codex/prompts/multi-agent.md + AGENTS.md block + config.toml mcp block)",
410
+ " - Codex CLI (~/.codex: skills/multi-agent, multi-agent-refs, agents/*.toml, prompts/multi-agent.md, scripts, lib, schemas, rules, AGENTS.md block, dev-toolkit MCP entry)",
336
411
  );
337
412
  console.log("");
338
413
  if (allData) {
@@ -478,14 +553,35 @@ export async function main() {
478
553
 
479
554
  if (forCodex && HOME) {
480
555
  console.log("");
481
- console.log(" [OpenAI Codex CLI - legacy cleanup] Removing from ~/.codex...");
556
+ console.log(" [Codex CLI] Removing from ~/.codex...");
482
557
  const CODEX = join(HOME, ".codex");
558
+
559
+ // Wholly pipeline-owned trees.
560
+ rmIfExists(join(CODEX, "multi-agent-refs"));
561
+ rmIfExists(join(CODEX, "scripts"));
562
+ rmIfExists(join(CODEX, "lib"));
563
+ rmIfExists(join(CODEX, "schemas"));
564
+ rmIfExists(join(CODEX, "rules"));
565
+ rmIfExists(join(CODEX, ".pipeline-version"));
566
+
567
+ // Co-owned dirs: remove only what the pipeline wrote. `skills/` holds
568
+ // Codex's own `.system` set plus user skills; `prompts/` and `agents/` hold
569
+ // user-authored files.
570
+ const skills = join(CODEX, "skills");
571
+ const n = rmMatchingDirs(
572
+ skills,
573
+ (name) => name === "multi-agent" || name.startsWith("multi-agent-"),
574
+ );
575
+ if (n > 0) console.log(` removed ${n} skill dir(s) under ${skills}`);
483
576
  rmIfExists(join(CODEX, "prompts", "multi-agent.md"));
577
+ removeGeneratedCodexAgents(join(CODEX, "agents"));
578
+
484
579
  stripManagedBlock(join(CODEX, "AGENTS.md"));
485
- if (existsSync(join(CODEX, "config.toml")))
486
- console.log(
487
- " note: ~/.codex/config.toml left untouched (may hold user MCP servers) - remove pipeline entries manually if present",
488
- );
580
+
581
+ // The MCP entry was written by `codex mcp add`, so hand it back the same
582
+ // way rather than editing config.toml - Codex also stores marketplace and
583
+ // plugin state in that file.
584
+ deregisterCodexMcpServer();
489
585
  }
490
586
 
491
587
  console.log("");
@@ -48,7 +48,8 @@ if [ -z "$LOCAL_VERSION" ]; then
48
48
  fi
49
49
  # npx-only installs have no repo clone; the installer stamps the version here.
50
50
  if [ -z "$LOCAL_VERSION" ]; then
51
- for marker in "$HOME/.claude/.pipeline-version" "$HOME/.copilot/.pipeline-version"; do
51
+ for marker in "$HOME/.claude/.pipeline-version" "$HOME/.copilot/.pipeline-version" \
52
+ "$HOME/.codex/.pipeline-version"; do
52
53
  if [ -f "$marker" ]; then
53
54
  LOCAL_VERSION=$(head -1 "$marker" 2>/dev/null | tr -d '[:space:]')
54
55
  [ -n "$LOCAL_VERSION" ] && break
@@ -17,7 +17,7 @@ Two-axis language preference for the pipeline:
17
17
  | `promptLanguage` | Interactive prompts during a pipeline run (account picker, project picker, dev-context picker, base-branch picker, branch-name picker, maturity ack, channels picker, Phase 5 test prompt, Phase 6 local-checkout prompt) | `prefs.global.promptLanguage` | **Fixed to `en`** - never toggled by this skill |
18
18
  | `outputLanguage` | Assistant's explanations, status updates, error messages, and pipeline-generated reports rendered to the user (NOT external payloads) | `prefs.global.outputLanguage` | Toggled by this skill |
19
19
 
20
- **Why promptLanguage is fixed:** Pipeline picker UI, confirmation prompts, and error messages are authored in English to keep tooling consistent across CLIs. Mixing Turkish prompts into otherwise-English skill output looks inconsistent. Only the assistant's free-form replies follow `outputLanguage`.
20
+ **Why promptLanguage is fixed:** it governs the picker's structural chrome only - `AskUserQuestion` `label` (button text) and `header` (chip), host error UI, and internal contract identifiers. Those stay English so tooling reads the same across CLIs. Everything a user actually reads follows `outputLanguage`, including the picker `question` and each option's `description`, per the canonical per-field matrix in `multi-agent-refs/rules.md`. A picker whose question is English on a Turkish run is a bug, not the contract.
21
21
 
22
22
  **Always English regardless of either field**: commit messages, PR titles/bodies, Jira comments, wiki pages, reviewer/triage system prompts, agent-log.md payloads, skill-picker / confirmation / error UI exposed by the CLI host.
23
23
 
@@ -83,5 +83,5 @@ Render in the new outputLanguage:
83
83
 
84
84
  - `multi-agent-setup` - first-run language picker (asks only outputLanguage; promptLanguage seeded as `en`)
85
85
  - `prefs.schema.json` - `global.promptLanguage` (fixed `"en"`) and `global.outputLanguage`
86
- - Phase 5 / Phase 6 - read promptLanguage for interactive prompts (always English)
86
+ - Phase 5 / Phase 6 - interactive prompts follow the per-field matrix: `question` + `description` in `outputLanguage`, `label` + `header` English
87
87
  - `multi-agent-help` - reads outputLanguage for its own copy
@@ -15,7 +15,7 @@ Ask BEFORE anything else. The pipeline has two language fields:
15
15
 
16
16
  | Field | Controls | Configurable? |
17
17
  |---|---|---|
18
- | `promptLanguage` | Interactive pickers and prompts during pipeline runs (account, project, branch, channels, Phase 5/6 confirmations) | **No - fixed to `en`.** Picker UI is always English. |
18
+ | `promptLanguage` | The picker's structural UI chrome only: `AskUserQuestion` `label` + `header`, host error UI, internal contract identifiers | **No - fixed to `en`.** Button and chip text stays English. |
19
19
  | `outputLanguage` | The assistant's non-interactive explanations, status updates, error messages, and pipeline-generated reports | Yes - set here, change later via `/multi-agent-language <en\|tr>` |
20
20
 
21
21
  `promptLanguage` is seeded as `"en"` and never offered to the user. External payloads (commits, PR/Jira/wiki, reviewer/triage prompts, agent-log.md) always stay English regardless.
@@ -278,6 +278,53 @@ pbcopy < /dev/null
278
278
  # ~/.claude/settings.json -> mcpServers -> figma
279
279
  ```
280
280
 
281
+ ### App Store Connect Credentials (optional, iOS only)
282
+
283
+ Unlocks Gate 2 of `multi-agent-testflight-validation` - Apple's own
284
+ `altool --validate-app`, the only check that sees an unregistered bundle ID, a
285
+ profile that does not match the App Store Connect app record, or a version+build
286
+ pair already used. Skipping is a first-class answer: the validation command still
287
+ runs its static audit and guideline review and reports Gate 2 as `SKIPPED` with
288
+ the reason, never as a pass.
289
+
290
+ Ask (picker): **API key** / **Apple ID + app-specific password** / **Skip**.
291
+
292
+ **API key** - needs an Admin or App Manager role in App Store Connect. Map the key
293
+ id and issuer id as `appstore_connect_key_id` / `appstore_connect_issuer_id`. The
294
+ private key is a FILE and never enters the credential store; it goes to one of the
295
+ directories altool searches:
296
+
297
+ ```bash
298
+ ls ~/.appstoreconnect/private_keys/AuthKey_*.p8 2>/dev/null \
299
+ || echo "MISSING: put AuthKey_<keyId>.p8 in ~/.appstoreconnect/private_keys/"
300
+ ```
301
+
302
+ **Apple ID + app-specific password** - usable by **any Apple ID holder, no elevated
303
+ role**, which is the realistic path when API-key creation is not permitted on the
304
+ account. Lead with this option when the user says they cannot create an API key.
305
+ The password goes into Apple's own keychain helper, which is what
306
+ `altool -p @keychain:<item>` reads - never into chat, never into an argument:
307
+
308
+ ```bash
309
+ # export AC_PASSWORD_ONCE in your own shell first, for this one command
310
+ xcrun altool --store-password-in-keychain-item "<item-name>" \
311
+ -u "<apple-id>" -p @env:AC_PASSWORD_ONCE
312
+ ```
313
+
314
+ Map only the ITEM NAME as `appstore_connect_password_item`, plus the Apple ID as
315
+ `appstore_connect_apple_id`. The password stays in the keychain and is referenced,
316
+ never read by the pipeline.
317
+
318
+ **Multi-provider accounts.** A corporate Apple ID often belongs to several
319
+ providers and altool fails opaquely without one. Resolve it once with
320
+ `ios_testflight_validate({list_providers: true, ...})` and store the id under
321
+ `prefs.projects[<key>].appStoreConnect.providerPublicId` - per-project, since a
322
+ user can ship for more than one team.
323
+
324
+ A credential that resolves but is rejected (401/403) follows the Expired-token
325
+ decision (Regenerate / Use a different token / Skip and continue) rather than
326
+ being silently dropped.
327
+
281
328
  ### Security Rules
282
329
 
283
330
  - Never paste tokens into terminal or chat - use `pbpaste` for input
@@ -20,6 +20,7 @@ When invoked, it synchronizes all targets in order. It detects what changed, upd
20
20
  |---|-------|-----|-----|
21
21
  | 1 | Claude Code (source of truth) | `~/.claude/commands/multi-agent.md` + `~/.claude/commands/multi-agent/` + `~/.claude/agents/` + `~/.claude/scripts/` | source |
22
22
  | 2 | Copilot CLI | `~/.copilot/copilot-instructions.md` + `~/.copilot/skills/` | <- from Claude |
23
+ | 2b | Codex CLI | `~/.codex/AGENTS.md` + `~/.codex/skills/multi-agent/` + `~/.codex/multi-agent-refs/` + `~/.codex/agents/*.toml` | <- from Claude (path-rewritten) |
23
24
  | 3 | multi-agent-pipeline repo | `~/multi-agent-pipeline/pipeline/` | <- from Claude (genericized) |
24
25
  | 4 | Website | `{owner}/{website-host}` | <- version + features |
25
26
  | 5 | dev-toolkit MCP server | resolved from `prefs.global.devToolkit` or the `mcpServers` registration | own repo: gate, commit, publish |
@@ -30,7 +31,8 @@ Run all steps automatically:
30
31
 
31
32
  ```
32
33
  Step 1: DETECT Compare timestamps, find stale targets
33
- Step 2: COPILOT Claude Code -> Copilot CLI (instructions + 42 sub-command skills)
34
+ Step 2: COPILOT Claude Code -> Copilot CLI (instructions + 43 sub-command skills)
35
+ Step 2b: CODEX Claude Code -> Codex CLI (1 router skill + 43 specs as refs + 8 agent TOML)
34
36
  Step 3: REPO Claude Code -> pipeline repo (genericized, personal data scrub)
35
37
  Step 3d: DEV-TOOLKIT Companion MCP server -> detect movement, ship gates, commit + publish
36
38
  Step 4: WEBSITE Version + phase/model counts -> {website-host} (i18n + projects.ts)
@@ -93,6 +95,41 @@ If nothing is stale -> report "All targets up to date" and stop.
93
95
  - `~/.claude/settings.json`
94
96
 
95
97
 
98
+ ## Codex Sync (Step 2b)
99
+
100
+ This step does **not** hand-copy files. The Codex tree is a *transform* of the Claude
101
+ tree, not a mirror: the 43 sub-command specs become reference files (Codex silently
102
+ truncates its skills block - see `cross-cli-contract.md` 2.6), every reference to a
103
+ CLI-owned tree is retargeted (`agents/<persona>.md` becomes `.toml`, the dispatcher
104
+ becomes the router skill), the 8 personas are regenerated as TOML with a model +
105
+ reasoning-effort map, and shared state (`logs/`, prefs, `knowledge/`) is deliberately
106
+ left under `~/.claude`.
107
+
108
+ That transform lives in `install/codex.mjs` and is gate-locked by
109
+ `smoke-codex-install.sh`. Describing it again in prose would give the pipeline two
110
+ definitions of the same thing, and the prose one would rot. So the step runs the
111
+ installer:
112
+
113
+ ```bash
114
+ cd "$HOME/multi-agent-pipeline" && node install.js --codex
115
+ ```
116
+
117
+ Verify:
118
+
119
+ ```bash
120
+ ls -1 "$HOME/.codex/skills" | grep -c '^multi-agent$' # want 1
121
+ find "$HOME/.codex/multi-agent-refs/commands" -name SKILL.md | wc -l # want the command count
122
+ grep -rhoE '(\$HOME|~)/\.claude/(agents|scripts|lib|schemas|commands|multi-agent-refs|rules)' \
123
+ "$HOME/.codex/skills" "$HOME/.codex/multi-agent-refs" | sort -u # want empty
124
+ ```
125
+
126
+ MCP is registered by the installer via `codex mcp add dev-toolkit` (idempotent,
127
+ skipped with a warning when `codex` is absent). Never hand-edit
128
+ `~/.codex/config.toml` - Codex owns it. Run this AFTER the REPO step, since the
129
+ installer reads the repo tree.
130
+
131
+ ---
132
+
96
133
  ## Dev-Toolkit Sync (Step 3d)
97
134
 
98
135
  The companion MCP server (`dev-toolkit-mcp`) is its own repo with its own registry. Pipeline skills call its tools and several declare a minimum version, so when it moves it has to ship.
@@ -114,13 +151,19 @@ When the toolkit ships its own gate script (`npm run gates` / `scripts/gates.sh`
114
151
 
115
152
  **Version bump**: patch for fixes and docs, minor for a new tool, major for a removed or renamed tool.
116
153
 
117
- **Ship**: commit with that repo's own convention, tag `v<version>`, push with `--tags`, then publish to the registry from `publishConfig` using a throwaway userconfig - never edit `~/.npmrc`, never a bare `npm publish`. The token depends on the registry: `npm.pkg.github.com` needs a GitHub PAT with `write:packages` (logical key `github`), `registry.npmjs.org` needs the `npm` key.
154
+ **Ship**: commit with that repo's own convention, tag `v<version>`, push with `--tags`, then publish to the registry from `publishConfig` using a throwaway userconfig - never edit `~/.npmrc`, never a bare `npm publish`. The token depends on the registry AND on the package scope: the registry picks the credential type, the scope picks the account. `npm.pkg.github.com` needs a **Classic** GitHub PAT with `write:packages`; `registry.npmjs.org` needs the `npm` key. Resolving on host alone misroutes on any machine with a work and a personal GitHub identity - prefer a scope-specific mapping (`github_<scope>`) and fall back to the generic `github` key only when there is one identity.
118
155
 
119
156
  ```bash
120
157
  NPMRC=$(mktemp); trap 'rm -f "$NPMRC"' EXIT
121
158
  REG=$(node -p "require('./package.json').publishConfig?.registry || 'https://registry.npmjs.org'")
122
159
  HOST=${REG#https://}; HOST=${HOST%/}
123
- case "$HOST" in npm.pkg.github.com*) KEY=github ;; *) KEY=npm ;; esac
160
+ SCOPE=$(node -p "(require('./package.json').name.match(/^@([^/]+)/)||[])[1] || ''")
161
+ case "$HOST" in
162
+ npm.pkg.github.com*) KEY=github; [ -n "$SCOPE" ] && KEY="github_$SCOPE" ;;
163
+ *) KEY=npm ;;
164
+ esac
165
+ # Fall back to the generic key when no scope-specific mapping exists.
166
+ bash "$HOME/.copilot/lib/credential-store.sh" get "$KEY" >/dev/null 2>&1 || KEY=github
124
167
  TOKEN=$(bash "$HOME/.claude/lib/credential-store.sh" get "$KEY")
125
168
  [ -n "$TOKEN" ] || echo "ABORT: no '$KEY' token - onboard it via /multi-agent:setup before publishing"
126
169
  printf '%s\n' "registry=$REG" "//$HOST/:_authToken=$TOKEN" > "$NPMRC"
@@ -167,24 +210,27 @@ When invoked with the `release` argument:
167
210
  7. DEV-TOOLKIT Ship the companion MCP server if it moved (Step 3d gates, then publish)
168
211
  8. WEBSITE Version + features -> {website-host}
169
212
  9. COPILOT Copilot CLI instructions + skills sync
213
+ 9b. CODEX Codex CLI router skill + refs + agent TOML (node install.js --codex)
170
214
  10. Report Summary: version, touched repos, deploy status
171
215
  ```
172
216
 
173
217
 
174
218
  ## Sub-Command Sync (Claude Code <-> Copilot CLI Skills)
175
219
 
220
+ > Codex takes the Step 2b path instead; see that section.
221
+
176
222
  | Claude Code | Copilot CLI |
177
223
  |-------------|-------------|
178
224
  | `~/.claude/commands/multi-agent/{cmd}.md` | `~/.copilot/skills/multi-agent-{cmd}/SKILL.md` |
179
225
 
180
- **42 commands are synced** (canonical inventory - must match `cross-cli-contract.md` section 1; drift = contract violation):
226
+ **43 commands are synced** (canonical inventory - must match `cross-cli-contract.md` section 1; drift = contract violation):
181
227
 
182
228
  ```
183
229
  analysis, analysis-resolve, autopilot, build-optimize, channels, create-jira, design-check, dev,
184
230
  dev-autopilot, dev-local, dev-local-autopilot, diff-explain, finish, forget, garbage-collect,
185
231
  help, issue, jira, kill, language, local,
186
232
  local-autopilot, log, manual-test, prune-logs, purge, refactor, resume, review, review-issue, review-jira,
187
- routines, save, scan, search, setup, stack, status, sync, test, uninstall, update
233
+ routines, save, scan, search, setup, stack, status, sync, test, testflight-validation, uninstall, update
188
234
  ```
189
235
 
190
236
  **NOT synced**: `refs/*` - Lazy-load references, Claude Code specific
@@ -200,8 +246,43 @@ routines, save, scan, search, setup, stack, status, sync, test, uninstall, updat
200
246
  | `gh` CLI (personal) | `{owner}` gh auth (Keychain) |
201
247
  | `gh` CLI (work) | `${USER}_{work-gh-alias}` gh auth (Keychain) |
202
248
  | npm publish | `NODE_AUTH_TOKEN` - `GITHUB_TOKEN` in CI, Keychain PAT locally |
203
- | dev-toolkit publish (Step 3d) | `npm` logical Keychain key -> throwaway `--userconfig`, registry from that repo's `publishConfig` |
249
+ | dev-toolkit publish (Step 3d) | scope-resolved key (`github_<scope>`, else `github`, else `npm`) -> throwaway `--userconfig`, registry from that repo's `publishConfig` |
204
250
 
205
251
  ```bash
206
252
  gh auth switch --user {owner} # for personal repos
207
253
  ```
254
+
255
+
256
+ ## Pre-flight the token scope
257
+
258
+ GitHub reports a token's scopes on any authenticated request, so check before
259
+ uploading instead of reading it out of a 403:
260
+
261
+ ```bash
262
+ curl -sI -H "Authorization: token $TOKEN" https://api.github.com/user \
263
+ | grep -i '^x-oauth-scopes:' | grep -q 'write:packages'
264
+ ```
265
+
266
+ Candidate order for `npm.pkg.github.com`, because storage location and scope are
267
+ not correlated: `github_<scope>` from the mapping, then `gh auth token -u <scope>`
268
+ (gh's own OAuth token often carries `write:packages` when a hand-made PAT does
269
+ not), then the generic `github` key only on a single-identity machine. Take the
270
+ first candidate whose scopes include `write:packages` AND whose login matches the
271
+ package scope. If none qualifies, stop before publishing and report each candidate
272
+ with the login it resolved to and the scope it was missing - that is the
273
+ actionable output, not a 403 body.
274
+
275
+ ## Publish 403s: two causes, two fixes
276
+
277
+ Neither message says "wrong token" plainly, and they need opposite responses:
278
+
279
+ - `Unauthorized: As an Enterprise Managed User, you cannot access this content` -
280
+ the resolved token belongs to a corporate EMU account, which cannot publish to a
281
+ personal scope at all. No scope grant fixes this; map the personal account's PAT
282
+ under `github_<scope>`.
283
+ - `The token provided does not match expected scopes` - right account, missing
284
+ permission. Regenerate as a **Classic** PAT with `write:packages`
285
+ (plus `repo` for a private package) and re-onboard through setup.
286
+
287
+ Always report which of the two occurred. "Permission denied" alone sends the user
288
+ to regenerate a token that was never the problem.
@@ -0,0 +1,120 @@
1
+ ---
2
+ name: multi-agent-testflight-validation
3
+ language: en
4
+ description: "Pre-submission validation for a TestFlight / App Store build (iOS, local-only). Three gates: static archive audit, Apple's own `altool --validate-app`, and a Review-Guidelines check. ITMS codes are mapped to the rule each implies. Validates only, never uploads. Use when a build is about to go to TestFlight, or a submission was rejected and you need why."
5
+ user-invocable: true
6
+ argument-hint: "[repo] - empty = pick from prefs; repo name or path; --ipa=<path>; --archive=<path>; --resume"
7
+ ---
8
+
9
+ # multi-agent-testflight-validation - pre-submission validation
10
+
11
+ Catch, before you upload, what App Store Connect would send back after you do.
12
+
13
+ **Local-only**: no commits, no push, no PR, no channels. **It never uploads** - only
14
+ `--validate-app` is ever invoked, never `--upload-app`.
15
+
16
+ ## Why three gates
17
+
18
+ Each sees something the others structurally cannot, so reporting one as "the check"
19
+ is how a build passes locally and gets rejected anyway.
20
+
21
+ | Gate | Runs | Needs | Sees | Blind to |
22
+ |---|---|---|---|---|
23
+ | **1. Static** | `ios_app_store_audit` (18 rules) | `.xcarchive` | privacy manifest, required-reason API, Info.plist, signing, entitlements, embedded SDK, IPv6, debug leak | anything account-dependent |
24
+ | **2. Authoritative** | `ios_testflight_validate` → `altool --validate-app` | `.ipa` + credentials | unregistered bundle ID, profile/app-record mismatch, **version+build already used**, entitlement not provisioned | the Review Guidelines |
25
+ | **3. Guideline** | `app-store-review` skill + repo evidence | repo checkout | ATT, privacy policy, account deletion, IAP, purpose-string wording | anything not in source |
26
+
27
+ Most "we passed validation and still got rejected" cases are Gate 3 findings:
28
+ Apple's validator does not read the Review Guidelines.
29
+
30
+ ## Flow
31
+
32
+ **0. Input.** `(empty)` → ask · `my-ios-app`/path → that repo · `--ipa=<path>` →
33
+ Mode B without Gate 1 · `--archive=<path>` → Mode B with all gates · `--resume`.
34
+ State at `~/.claude/logs/multi-agent/<task_id>/agent-state.json`,
35
+ `taskId = TFV-<repo>-<yyyymmddHHMM>`. Register phases with
36
+ `bash ~/.copilot/scripts/phase-tracker.sh` and render with
37
+ `phase-tracker.sh render` at every boundary.
38
+
39
+ **1. Pickers.** Repo (iOS entries in `prefs.projects`; a single match auto-resolves,
40
+ and the breadcrumb says so) → branch → mode. Print
41
+ `Step <i>/<n>: <what this decides>` for each. On `git fetch` failure do **not**
42
+ silently use a cached ref, and **classify the failure before naming a cause**:
43
+ `could not read Password` / `Authentication failed` / `403` is a credential
44
+ problem, not a network one, and a VPN cannot fix it; `Could not resolve host` /
45
+ `Operation timed out` is the network. Show the stderr line verbatim next to your
46
+ classification, then offer the remedy that matches it. Only the network case gets
47
+ the continue-on-a-stale-ref option: for a credential failure, staleness is
48
+ unrelated to what broke.
49
+
50
+ **2. Pre-flight.** `xcrun --find altool` and `xcodebuild -version`; a missing
51
+ prerequisite halts. Resolve credentials and name the active tier in the report:
52
+ tier 1 ASC API key (key id + issuer id from the keychain via
53
+ `prefs.global.keychainMapping`, `.p8` at
54
+ `~/.appstoreconnect/private_keys/AuthKey_<keyId>.p8`), tier 2 Apple ID +
55
+ app-specific password by keychain reference, tier 3 none → **Gate 2 is `SKIPPED`
56
+ and the verdict line says so**. Credentials come from `/multi-agent:setup`; never
57
+ prompt for a secret value in chat. Tier 2 needs no elevated App Store Connect
58
+ role, which matters when API-key creation is not permitted on the account.
59
+
60
+ **3. Obtain the build.**
61
+ - Mode B `.xcarchive` → Gate 1 runs; export with `ios_export_ipa` to reach Gate 2.
62
+ - Mode B `.ipa` only → **Gate 1 `SKIPPED (needs .xcarchive)`**. The static audit
63
+ reads archive structure an `.ipa` does not carry. Do not present a two-gate run
64
+ as a full pass.
65
+ - Mode A → worktree at `{projectRoot}/{worktreeBasePath}/{taskId}` (**never under
66
+ `$HOME`**) → `ios_xcodebuild({action: "archive", configuration: "Release",
67
+ destination: "generic/platform=iOS"})` - the simulator default would produce a
68
+ non-distributable archive - then `ios_export_ipa({method:
69
+ "app-store-connect", team_id, ...})`. Leave `allow_provisioning_updates` off
70
+ unless asked: it lets xcodebuild create or modify profiles in the developer
71
+ account, which a validation run has no business doing.
72
+
73
+ **4. Gate 1.** `ios_app_store_audit({archive_path, rules: "all"})`. `error` blocks,
74
+ `warning` advises. Keep each ITMS code - Gate 2 may return the same one, and that
75
+ agreement tells the user it is real.
76
+
77
+ **5. Gate 2.** `ios_testflight_validate({ipa_path, platform: "ios", <creds>})`.
78
+ Render the verdict as returned: `PASS` / `FAIL` (each issue with ITMS code, mapped
79
+ guideline, hint) / `SKIPPED` (with the reason, **never as a pass**). This is the
80
+ only gate that catches an already-used build number - the most common wasted
81
+ upload - so when it fails on that, name the next free build number.
82
+
83
+ **6. Gate 3.** Load `app-store-review` and check, with an evidence path each:
84
+ purpose strings (present, specific, matching actual use) · `PrivacyInfo.xcprivacy`
85
+ (exists, declares required-reason APIs, matches linked SDKs) · ATT before any
86
+ tracking · in-app account deletion when accounts are created · privacy policy
87
+ reachable · IAP through StoreKit with no external purchase path · Sign in with
88
+ Apple alongside third-party social login. Mark `pass` / `fail` /
89
+ `not-applicable`, and `not-applicable` needs a reason - an unexamined area is not
90
+ a pass.
91
+
92
+ **7. Report.** `~/TestFlightChecks/<repo>-<branch>-<timestamp>/report.md` plus a
93
+ printed summary:
94
+
95
+ ```
96
+ Verdict: <N> of 3 gates cleared[, <M> skipped]
97
+ Build: <path> · <bundle id> <version> (<build>)
98
+ Auth: tier <1|2|none> · <method>
99
+
100
+ Gate 1 static audit PASS | FAIL (<n> blocking, <n> advisory) | SKIPPED (<reason>)
101
+ Gate 2 Apple validation PASS | FAIL (<n> issues) | SKIPPED (<reason>)
102
+ Gate 3 guideline review PASS | FAIL (<n> findings) | <n> not-applicable
103
+
104
+ Blocking - fix before uploading
105
+ [ITMS-90683] Info.plist: NSCameraUsageDescription missing
106
+ guideline 5.1.1 Data Collection and Storage
107
+ <hint> · <file:line>
108
+
109
+ Not run
110
+ Gate 1: needs an .xcarchive; only an .ipa was supplied
111
+ ```
112
+
113
+ A skipped gate is never folded into the pass count. Every blocking finding carries
114
+ a file path or an ITMS code. No AI or assistant attribution anywhere; real
115
+ newlines, no HTML entities.
116
+
117
+ **8. Offer, do not act.** Print the exact `xcrun altool --upload-app` command for
118
+ when the gates are clear (uploading stays an explicit human act), plus
119
+ `--resume` after fixes and `/multi-agent:fix-bug` for code-level Gate 3 findings.
120
+ Never upload, never bump the build number, never commit.