@mmerterden/multi-agent-pipeline 12.11.0 → 13.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/CHANGELOG.md +179 -0
  2. package/README.md +24 -7
  3. package/index.js +5 -2
  4. package/install/_codex-agents.mjs +211 -0
  5. package/install/_codex-instructions.mjs +33 -0
  6. package/install/_managed-block.mjs +99 -0
  7. package/install/codex.mjs +478 -0
  8. package/install/copilot.mjs +34 -80
  9. package/install/index.mjs +25 -9
  10. package/install/templates/codex-instructions.md +45 -0
  11. package/package.json +5 -3
  12. package/pipeline/claude-md-template.md +1 -0
  13. package/pipeline/commands/multi-agent/SKILL.md +1 -1
  14. package/pipeline/commands/multi-agent/dev/SKILL.md +52 -6
  15. package/pipeline/commands/multi-agent/dev-local/SKILL.md +21 -0
  16. package/pipeline/commands/multi-agent/finish/SKILL.md +1 -1
  17. package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
  18. package/pipeline/commands/multi-agent/setup/SKILL.md +69 -2
  19. package/pipeline/commands/multi-agent/sync/SKILL.md +128 -5
  20. package/pipeline/commands/multi-agent/testflight-validation/SKILL.md +219 -0
  21. package/pipeline/commands/multi-agent/update/SKILL.md +7 -4
  22. package/pipeline/multi-agent-refs/_input-parser.md +1 -1
  23. package/pipeline/multi-agent-refs/component-dispatch.md +40 -7
  24. package/pipeline/multi-agent-refs/cross-cli-contract.md +51 -17
  25. package/pipeline/multi-agent-refs/features/model-fallback.md +29 -0
  26. package/pipeline/multi-agent-refs/features/review-multi-repo.md +1 -1
  27. package/pipeline/multi-agent-refs/phases/log-format.md +1 -1
  28. package/pipeline/multi-agent-refs/phases/phase-0-init.md +43 -2
  29. package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +24 -1
  30. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +32 -0
  31. package/pipeline/multi-agent-refs/phases/phase-4-review.md +62 -5
  32. package/pipeline/multi-agent-refs/progress-contract.md +1 -1
  33. package/pipeline/multi-agent-refs/tracker-contract.md +17 -1
  34. package/pipeline/schemas/prefs.schema.json +296 -62
  35. package/pipeline/schemas/reviewer-output.schema.json +1 -1
  36. package/pipeline/schemas/triage-output.schema.json +1 -1
  37. package/pipeline/scripts/cost-table.json +15 -1
  38. package/pipeline/scripts/phase0-exit-gate.mjs +185 -0
  39. package/pipeline/scripts/smoke-cross-cli-behavior.sh +25 -10
  40. package/pipeline/scripts/uninstall.mjs +105 -9
  41. package/pipeline/scripts/update-check.sh +2 -1
  42. package/pipeline/skills/shared/core/multi-agent-dev/SKILL.md +21 -0
  43. package/pipeline/skills/shared/core/multi-agent-dev-local/SKILL.md +21 -0
  44. package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
  45. package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +48 -1
  46. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +87 -6
  47. package/pipeline/skills/shared/core/multi-agent-testflight-validation/SKILL.md +120 -0
@@ -3,7 +3,7 @@
3
3
  "$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/reviewer-output.schema.json",
4
4
  "version": "1.0.0",
5
5
  "title": "Multi-Agent Pipeline - Phase 4 reviewer output",
6
- "description": "Contract for a single code-reviewer subagent's JSON output in Phase 4 Step 2. Reviewer set is CLI-aware: Claude Code dispatches 2 parallel reviewers (Opus, Sonnet); Copilot CLI dispatches 3 (GPT-5.4, Opus, Sonnet). Every reviewer must return an object matching this shape before Opus triage merges them.",
6
+ "description": "Contract for a single code-reviewer subagent's JSON output in Phase 4 Step 2. Reviewer set is CLI-aware: Claude Code dispatches 2 parallel reviewers (Fable, Sonnet); Copilot CLI dispatches 3 (Opus, GPT-5.4, Sonnet); Codex CLI dispatches 3 (gpt-5.6 at xhigh, gpt-5.4, gpt-5.6 at medium). Every reviewer must return an object matching this shape before Opus triage merges them.",
7
7
  "type": "object",
8
8
  "additionalProperties": false,
9
9
  "required": ["findings", "approved"],
@@ -86,7 +86,7 @@
86
86
  "reviewerCount": {
87
87
  "type": "integer",
88
88
  "minimum": 1,
89
- "description": "Reviewers dispatched this iteration (Claude Code = 2, Copilot CLI = 3)."
89
+ "description": "Reviewers dispatched this iteration (Claude Code = 2, Copilot CLI = 3, Codex CLI = 3)."
90
90
  },
91
91
  "verdict": {
92
92
  "type": "string",
@@ -32,7 +32,21 @@
32
32
  "outPerMtok": 30.0,
33
33
  "cacheReadPerMtok": 1.0,
34
34
  "modelId": "gpt-5.4",
35
- "note": "Copilot CLI Reviewer 2 - approximate; verify against OpenAI pricing page before relying on totals."
35
+ "note": "Copilot CLI Reviewer 2 and Codex CLI Reviewer 2 - approximate; verify against OpenAI pricing page before relying on totals."
36
+ },
37
+ "gpt-5.6": {
38
+ "inPerMtok": 10.0,
39
+ "outPerMtok": 30.0,
40
+ "cacheReadPerMtok": 1.0,
41
+ "modelId": "gpt-5.6",
42
+ "note": "Codex CLI top tier - Reviewer 1 at xhigh effort, Reviewer 3 at medium, triage at max, and the fable/opus rungs of the Codex persona tier map. Reasoning effort changes output volume, not the per-token rate, so one entry covers every effort level. Approximate; verify against OpenAI pricing before relying on totals."
43
+ },
44
+ "gpt-5.6-terra": {
45
+ "inPerMtok": 1.0,
46
+ "outPerMtok": 5.0,
47
+ "cacheReadPerMtok": 0.1,
48
+ "modelId": "gpt-5.6-terra",
49
+ "note": "Codex CLI floor tier - speed-optimised, maps the haiku rung of the Codex persona tier map (task-clarifier). Approximate; verify against OpenAI pricing before relying on totals."
36
50
  }
37
51
  }
38
52
  }
@@ -0,0 +1,185 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * phase0-exit-gate.mjs - Phase 0 may not be marked completed until it has
4
+ * actually produced its own output.
5
+ *
6
+ * Why this exists. Phase 0 Step 7 says to classify the task and "persist to
7
+ * `agent-state.json.taskType`", and Phase 3 branches on that field to route a
8
+ * Figma-driven task to the component skills (`create-screen` / `create-component`
9
+ * in the stack plugin) instead of generic development. On a real run the tracker
10
+ * showed Phase 0 `completed` while the task directory held only
11
+ * `tracker-state.json` - no `agent-state.json` at all. With no `taskType`, the
12
+ * dispatch could not fire, so a Figma-driven screen was built as generic work: no
13
+ * token-compliance check, no Code Connect publish, no 14-item component review,
14
+ * and spacing guessed at 16 where the frame said `Spacing/12`. Half the commits on
15
+ * that branch were rework.
16
+ *
17
+ * A phase that reports success without its output is worse than one that fails:
18
+ * every later phase then reasons from a field that is not there. So this is a
19
+ * gate, not a lint - the spec already said what to write, and prose alone did
20
+ * not make it happen.
21
+ *
22
+ * Usage:
23
+ * node phase0-exit-gate.mjs <task_id> [--input "<original user input>"] [--json]
24
+ *
25
+ * Exit codes: 0 = pass, 1 = gate failure (Phase 0 must not be closed), 2 = usage
26
+ *
27
+ * @module pipeline/scripts/phase0-exit-gate
28
+ */
29
+
30
+ import { existsSync, readFileSync } from "fs";
31
+ import { join } from "path";
32
+ import { homedir } from "os";
33
+
34
+ const FIGMA_URL = /figma\.com\/(design|make|file)\//i;
35
+
36
+ /** Fields that can carry the user's original request text. */
37
+ const INPUT_TEXT_FIELDS = [
38
+ ["input", "summary"],
39
+ ["input", "raw"],
40
+ ["issue", "body"],
41
+ ["issue", "title"],
42
+ ["jira", "summary"],
43
+ ["jira", "description"],
44
+ ["task", "description"],
45
+ ];
46
+
47
+ /**
48
+ * Pull a nested value without throwing on a missing branch.
49
+ * @param {object} obj
50
+ * @param {string[]} path
51
+ * @returns {unknown}
52
+ */
53
+ function at(obj, path) {
54
+ return path.reduce((acc, k) => (acc && typeof acc === "object" ? acc[k] : undefined), obj);
55
+ }
56
+
57
+ /**
58
+ * Collect every string in the state that could hold the original request, so a
59
+ * Figma reference is found wherever the input parser happened to store it.
60
+ *
61
+ * @param {object} state
62
+ * @returns {string}
63
+ */
64
+ export function inputTextOf(state) {
65
+ const parts = [];
66
+ for (const path of INPUT_TEXT_FIELDS) {
67
+ const v = at(state, path);
68
+ if (typeof v === "string") parts.push(v);
69
+ }
70
+ // evidence[].url / figmaFrames[] are where the analysis phase parks references.
71
+ for (const key of ["figmaFrames", "designRefs"]) {
72
+ const v = state[key];
73
+ if (Array.isArray(v)) parts.push(v.map((x) => (typeof x === "string" ? x : JSON.stringify(x))).join(" "));
74
+ }
75
+ return parts.join("\n");
76
+ }
77
+
78
+ /**
79
+ * Evaluate the gate against a parsed state object.
80
+ *
81
+ * Kept pure and exported so the smoke can drive it without laying down a task
82
+ * directory: a gate that can only be tested by reproducing a full run does not
83
+ * get tested.
84
+ *
85
+ * @param {object|null} state - parsed agent-state.json, or null when absent
86
+ * @param {string} extraInput - input text supplied on the command line
87
+ * @returns {{ok: boolean, failures: string[], taskType: string|undefined, figmaSeen: boolean}}
88
+ */
89
+ export function evaluate(state, extraInput = "") {
90
+ const failures = [];
91
+
92
+ if (state === null) {
93
+ return {
94
+ ok: false,
95
+ failures: [
96
+ "agent-state.json is missing. Phase 0 owns this file; without it taskType, " +
97
+ "maturity, account and repo resolution are all unreadable by later phases.",
98
+ ],
99
+ taskType: undefined,
100
+ figmaSeen: FIGMA_URL.test(extraInput),
101
+ };
102
+ }
103
+
104
+ const taskType = typeof state.taskType === "string" ? state.taskType.trim() : "";
105
+ if (!taskType) {
106
+ failures.push(
107
+ "agent-state.json has no taskType. Phase 3 branches on it: without the field " +
108
+ "a component task is dispatched as generic development (Step 7 of phase-0-init).",
109
+ );
110
+ }
111
+
112
+ const haystack = `${inputTextOf(state)}\n${extraInput}`;
113
+ const figmaSeen = FIGMA_URL.test(haystack);
114
+
115
+ if (figmaSeen && taskType && taskType !== "component") {
116
+ failures.push(
117
+ `the input carries a Figma URL but taskType is "${taskType}". Step 7 rule 1 makes ` +
118
+ `"component" mandatory here - otherwise the run skips the stack plugin's ` +
119
+ `token-compliance check, Code Connect publish and component review.`,
120
+ );
121
+ }
122
+
123
+ // The Figma access chain records which tier answered. A component task with no
124
+ // recorded tier means nothing verified that the design was actually reachable,
125
+ // which is how a run ends up guessing spacing.
126
+ if (figmaSeen) {
127
+ const tier = at(state, ["figmaAccess", "tier"]);
128
+ if (tier === undefined || tier === null || tier === "") {
129
+ failures.push(
130
+ "the input carries a Figma URL but state.figmaAccess.tier is unset. The 3-tier " +
131
+ "access chain must record which tier answered, so a later phase can tell " +
132
+ "'design confirmed' from 'design never fetched'.",
133
+ );
134
+ }
135
+ }
136
+
137
+ return { ok: failures.length === 0, failures, taskType: taskType || undefined, figmaSeen };
138
+ }
139
+
140
+ /** @param {string} taskId */
141
+ function statePathFor(taskId) {
142
+ return join(homedir(), ".claude", "logs", "multi-agent", taskId, "agent-state.json");
143
+ }
144
+
145
+ function main(argv) {
146
+ const args = argv.slice(2);
147
+ const taskId = args.find((a) => !a.startsWith("--"));
148
+ if (!taskId) {
149
+ console.error("usage: phase0-exit-gate.mjs <task_id> [--input \"<text>\"] [--json]");
150
+ return 2;
151
+ }
152
+ const inputIdx = args.indexOf("--input");
153
+ const extraInput = inputIdx >= 0 ? (args[inputIdx + 1] ?? "") : "";
154
+ const asJson = args.includes("--json");
155
+
156
+ const path = statePathFor(taskId);
157
+ let state = null;
158
+ if (existsSync(path)) {
159
+ try {
160
+ state = JSON.parse(readFileSync(path, "utf-8"));
161
+ } catch (e) {
162
+ console.error(`phase0-exit-gate: ${path} is not valid JSON: ${e.message}`);
163
+ return 1;
164
+ }
165
+ }
166
+
167
+ const result = evaluate(state, extraInput);
168
+
169
+ if (asJson) {
170
+ console.log(JSON.stringify({ taskId, statePath: path, ...result }, null, 2));
171
+ } else if (result.ok) {
172
+ console.log(
173
+ `phase0-exit-gate: PASS (taskType=${result.taskType}${result.figmaSeen ? ", figma reference present" : ""})`,
174
+ );
175
+ } else {
176
+ console.error("phase0-exit-gate: FAIL - Phase 0 must not be marked completed");
177
+ for (const f of result.failures) console.error(` - ${f}`);
178
+ console.error(` state: ${path}`);
179
+ }
180
+ return result.ok ? 0 : 1;
181
+ }
182
+
183
+ if (import.meta.url === `file://${process.argv[1]}`) {
184
+ process.exit(main(process.argv));
185
+ }
@@ -185,11 +185,12 @@ else
185
185
  fi
186
186
 
187
187
  # ──────────────────────────────────────────────────────────────────────────
188
- # CLI-aware reviewer-count contract: Claude Code = 2 reviewers (Opus + Sonnet),
189
- # Copilot CLI = 3 (GPT-5.4 + Opus + Sonnet). Nothing in code enforces the count
190
- # (the orchestrator dispatches per the doc), so lock the CONTRACT here against
191
- # drift across the phase doc, the schema, and the consensus block.
192
- echo "→ reviewer-count contract (Claude=2, Copilot=3)"
188
+ # CLI-aware reviewer-count contract: Claude Code = 2 (Fable + Sonnet), Copilot CLI
189
+ # = 3 (Opus + GPT-5.4 + Sonnet), Codex CLI = 3 (gpt-5.6 xhigh + gpt-5.4 + gpt-5.6
190
+ # medium). Nothing in code enforces the count (the orchestrator dispatches per the
191
+ # doc), so lock the CONTRACT here against drift across the phase doc, the schema,
192
+ # and the consensus block.
193
+ echo "→ reviewer-count contract (Claude=2, Copilot=3, Codex=3)"
193
194
  # `/multi-agent:update` runs this smoke on the user's machine, where the tree is
194
195
  # ~/.claude/{schemas,multi-agent-refs} and NOT <root>/pipeline/*. Resolving the
195
196
  # root by hand here reported three phantom failures in an install, which reads as
@@ -202,16 +203,30 @@ TRSCHEMA="${MA_SCHEMAS:+$MA_SCHEMAS/triage-output.schema.json}"
202
203
 
203
204
  if [ -z "$P4" ] || [ ! -f "$P4" ]; then
204
205
  echo " ↷ SKIP: phase-4-review.md not present in this $MA_LAYOUT layout"
205
- elif grep -qiE "Claude Code (dispatches|=|:) ?2|2-model" "$P4" && grep -qiE "Copilot CLI (dispatches|=|:) ?3|3-model" "$P4"; then
206
- pass "phase-4-review declares Claude=2 / Copilot=3 reviewers"
206
+ elif grep -qiE "Claude Code (dispatches|=|:) ?2|2-model" "$P4" \
207
+ && grep -qiE "Copilot CLI (dispatches|=|:) ?3|3-model" "$P4" \
208
+ && grep -qiE "Codex CLI (dispatches|=|:) ?3|Claude Code 2, Copilot CLI 3, Codex CLI 3" "$P4"; then
209
+ pass "phase-4-review declares Claude=2 / Copilot=3 / Codex=3 reviewers"
207
210
  else
208
- fail "phase-4-review does not declare the CLI-aware reviewer count"
211
+ fail "phase-4-review does not declare the CLI-aware reviewer count for all three hosts"
212
+ fi
213
+
214
+ # The two Codex constraints are silent-failure shaped, so the contract has to name
215
+ # them: a spawn without fork_turns collapses the panel onto one model, and the
216
+ # 4-slot ceiling (orchestrator included) is why the count is 3 and not more.
217
+ if [ -n "$P4" ] && [ -f "$P4" ]; then
218
+ if grep -q 'fork_turns' "$P4" && grep -qiE "concurrency|slots" "$P4"; then
219
+ pass "phase-4-review documents the fork_turns override rule + the concurrency ceiling"
220
+ else
221
+ fail "phase-4-review must document fork_turns and the Codex concurrency ceiling"
222
+ fi
209
223
  fi
210
224
 
211
225
  if [ -z "$REVSCHEMA" ] || [ ! -f "$REVSCHEMA" ]; then
212
226
  echo " ↷ SKIP: reviewer-output.schema.json not present in this $MA_LAYOUT layout"
213
- elif grep -qi "Copilot CLI" "$REVSCHEMA" && grep -qi "Claude" "$REVSCHEMA"; then
214
- pass "reviewer-output schema notes the CLI-aware reviewer set"
227
+ elif grep -qi "Copilot CLI" "$REVSCHEMA" && grep -qi "Claude" "$REVSCHEMA" \
228
+ && grep -qi "Codex" "$REVSCHEMA"; then
229
+ pass "reviewer-output schema notes the CLI-aware reviewer set (all three hosts)"
215
230
  else
216
231
  fail "reviewer-output schema missing the CLI-aware note"
217
232
  fi
@@ -36,6 +36,7 @@
36
36
  */
37
37
 
38
38
  import { existsSync, readdirSync, readFileSync, rmSync, writeFileSync } from "fs";
39
+ import { execFileSync } from "child_process";
39
40
  import { join } from "path";
40
41
  import { pathToFileURL } from "url";
41
42
  import { createInterface } from "readline";
@@ -123,6 +124,54 @@ function rmIfExists(path) {
123
124
  return true;
124
125
  }
125
126
 
127
+ /**
128
+ * Remove generated Codex agent TOML files, identified by the header the
129
+ * installer stamps on them. A user-authored `.toml` sharing a persona name is
130
+ * left alone - `~/.codex/agents/` is a co-owned directory.
131
+ *
132
+ * @param {string} agentsDir
133
+ * @returns {number} files removed
134
+ */
135
+ function removeGeneratedCodexAgents(agentsDir) {
136
+ if (!existsSync(agentsDir)) return 0;
137
+ const MARKER = "# Generated by multi-agent-pipeline install --codex.";
138
+ let removed = 0;
139
+ for (const name of readdirSync(agentsDir)) {
140
+ if (!name.endsWith(".toml")) continue;
141
+ const path = join(agentsDir, name);
142
+ try {
143
+ if (!readFileSync(path, "utf-8").startsWith(MARKER)) continue;
144
+ } catch {
145
+ continue;
146
+ }
147
+ if (rmIfExists(path)) removed++;
148
+ }
149
+ if (removed > 0) console.log(` removed ${removed} generated agent file(s) under ${agentsDir}`);
150
+ return removed;
151
+ }
152
+
153
+ /**
154
+ * Hand the dev-toolkit MCP registration back to Codex.
155
+ *
156
+ * Best-effort: a missing `codex` binary or an already-absent entry is not an
157
+ * uninstall failure. Never edits `config.toml` directly - Codex keeps
158
+ * marketplace and plugin state in the same file.
159
+ */
160
+ function deregisterCodexMcpServer() {
161
+ if (dryRun) {
162
+ report("would run", "codex mcp remove dev-toolkit");
163
+ return;
164
+ }
165
+ try {
166
+ execFileSync("codex", ["mcp", "remove", "dev-toolkit"], { stdio: "pipe", timeout: 20_000 });
167
+ console.log(" removed: dev-toolkit MCP registration");
168
+ } catch {
169
+ console.log(
170
+ " note: could not run `codex mcp remove dev-toolkit` (codex not on PATH, or not registered)",
171
+ );
172
+ }
173
+ }
174
+
126
175
  /**
127
176
  * Remove every directory under `parent` whose name matches a predicate.
128
177
  * @param {string} parent
@@ -160,6 +209,32 @@ function rmMatchingFiles(parent, predicate) {
160
209
  // (v11.4.1+). Must stay in sync with INSTRUCTIONS_END_MARKER there.
161
210
  const COPILOT_END_MARKER = "<!-- multi-agent-pipeline:copilot-instructions:end -->";
162
211
 
212
+ /**
213
+ * Every end marker the installers write. Both host instruction files share the
214
+ * same START marker but terminate with their own comment, so a single-marker
215
+ * lookup silently falls through to the legacy heading-bounded path - and since
216
+ * the managed body has no top-level heading after the first, that path treats
217
+ * ALL trailing user content as part of the block and deletes it.
218
+ */
219
+ const MANAGED_END_MARKERS = [
220
+ COPILOT_END_MARKER,
221
+ "<!-- multi-agent-pipeline:codex-instructions:end -->",
222
+ ];
223
+
224
+ /**
225
+ * Locate whichever managed end marker terminates this block.
226
+ *
227
+ * @param {string} fromStart - content from the start marker onward
228
+ * @returns {{index: number, length: number}} index -1 when none is present
229
+ */
230
+ function findManagedEndMarker(fromStart) {
231
+ for (const marker of MANAGED_END_MARKERS) {
232
+ const index = fromStart.indexOf(marker);
233
+ if (index >= 0) return { index, length: marker.length };
234
+ }
235
+ return { index: -1, length: 0 };
236
+ }
237
+
163
238
  /**
164
239
  * Legacy copilot-instructions files (written before the end marker existed)
165
240
  * have no explicit terminator. Bound the pipeline span at the next top-level
@@ -205,10 +280,10 @@ function stripManagedBlock(filePath) {
205
280
  if (idx >= 0) {
206
281
  const before = content.slice(0, idx).trimEnd();
207
282
  const fromStart = content.slice(idx);
208
- const endIdx = fromStart.indexOf(COPILOT_END_MARKER);
283
+ const end = findManagedEndMarker(fromStart);
209
284
  const rawTrailing =
210
- endIdx >= 0
211
- ? fromStart.slice(endIdx + COPILOT_END_MARKER.length)
285
+ end.index >= 0
286
+ ? fromStart.slice(end.index + end.length)
212
287
  : legacyTrailingContent(fromStart);
213
288
  const trailing = rawTrailing.replace(/^[\r\n]+/, "").trimEnd();
214
289
  let remaining = before;
@@ -332,7 +407,7 @@ export async function main() {
332
407
  );
333
408
  if (forCodex)
334
409
  console.log(
335
- " - OpenAI Codex CLI (~/.codex/prompts/multi-agent.md + AGENTS.md block + config.toml mcp block)",
410
+ " - Codex CLI (~/.codex: skills/multi-agent, multi-agent-refs, agents/*.toml, prompts/multi-agent.md, scripts, lib, schemas, rules, AGENTS.md block, dev-toolkit MCP entry)",
336
411
  );
337
412
  console.log("");
338
413
  if (allData) {
@@ -478,14 +553,35 @@ export async function main() {
478
553
 
479
554
  if (forCodex && HOME) {
480
555
  console.log("");
481
- console.log(" [OpenAI Codex CLI - legacy cleanup] Removing from ~/.codex...");
556
+ console.log(" [Codex CLI] Removing from ~/.codex...");
482
557
  const CODEX = join(HOME, ".codex");
558
+
559
+ // Wholly pipeline-owned trees.
560
+ rmIfExists(join(CODEX, "multi-agent-refs"));
561
+ rmIfExists(join(CODEX, "scripts"));
562
+ rmIfExists(join(CODEX, "lib"));
563
+ rmIfExists(join(CODEX, "schemas"));
564
+ rmIfExists(join(CODEX, "rules"));
565
+ rmIfExists(join(CODEX, ".pipeline-version"));
566
+
567
+ // Co-owned dirs: remove only what the pipeline wrote. `skills/` holds
568
+ // Codex's own `.system` set plus user skills; `prompts/` and `agents/` hold
569
+ // user-authored files.
570
+ const skills = join(CODEX, "skills");
571
+ const n = rmMatchingDirs(
572
+ skills,
573
+ (name) => name === "multi-agent" || name.startsWith("multi-agent-"),
574
+ );
575
+ if (n > 0) console.log(` removed ${n} skill dir(s) under ${skills}`);
483
576
  rmIfExists(join(CODEX, "prompts", "multi-agent.md"));
577
+ removeGeneratedCodexAgents(join(CODEX, "agents"));
578
+
484
579
  stripManagedBlock(join(CODEX, "AGENTS.md"));
485
- if (existsSync(join(CODEX, "config.toml")))
486
- console.log(
487
- " note: ~/.codex/config.toml left untouched (may hold user MCP servers) - remove pipeline entries manually if present",
488
- );
580
+
581
+ // The MCP entry was written by `codex mcp add`, so hand it back the same
582
+ // way rather than editing config.toml - Codex also stores marketplace and
583
+ // plugin state in that file.
584
+ deregisterCodexMcpServer();
489
585
  }
490
586
 
491
587
  console.log("");
@@ -48,7 +48,8 @@ if [ -z "$LOCAL_VERSION" ]; then
48
48
  fi
49
49
  # npx-only installs have no repo clone; the installer stamps the version here.
50
50
  if [ -z "$LOCAL_VERSION" ]; then
51
- for marker in "$HOME/.claude/.pipeline-version" "$HOME/.copilot/.pipeline-version"; do
51
+ for marker in "$HOME/.claude/.pipeline-version" "$HOME/.copilot/.pipeline-version" \
52
+ "$HOME/.codex/.pipeline-version"; do
52
53
  if [ -f "$marker" ]; then
53
54
  LOCAL_VERSION=$(head -1 "$marker" 2>/dev/null | tr -d '[:space:]')
54
55
  [ -n "$LOCAL_VERSION" ] && break
@@ -62,3 +62,24 @@ Phase 7: Report → Channels (Jira / Confluence / PR / Wiki)
62
62
  | Phase 5 User Test | ✅ | ✅ (same) |
63
63
  | Phase 7 channels (Jira / Confluence / PR / Wiki) | ✅ | ✅ (same) |
64
64
  | Duration | ~10-15 min | ~5-7 min |
65
+
66
+ ## Analysis doc supplied to a fast mode (warn before starting)
67
+
68
+ The `--dev` family skips Phase 1 (Analysis) and Phase 2 (Planning) by design. So when
69
+ the input references an analysis document - a Confluence URL, a local analysis file,
70
+ or the user says "I ran analysis for this" - there is **no phase that turns it into a
71
+ plan**. The doc becomes raw context for one Dev pass, and work comes out ordered by
72
+ whatever the model read first: the bottom of the dependency chain lands, the screen
73
+ wiring does not.
74
+
75
+ Say so before starting, once, and offer the choice:
76
+
77
+ ```
78
+ This mode skips Analysis and Planning, so the analysis document will not be turned
79
+ into a task breakdown. For analysis-driven screen work, /multi-agent or
80
+ /multi-agent:local run both phases.
81
+ 1. Continue with --dev (doc as context only)
82
+ 2. Switch to the full pipeline
83
+ ```
84
+
85
+ Autopilot picks 1 and logs the warning rather than asking.
@@ -34,3 +34,24 @@ Routes to the orchestrator with the `--dev --local` flags. The pipeline contract
34
34
  multi-agent-dev-local "PROJ-12345" # Jira
35
35
  multi-agent-dev-local "Bug: LoginView dark mode" # Free-text
36
36
  ```
37
+
38
+ ## Analysis doc supplied to a fast mode (warn before starting)
39
+
40
+ The `--dev` family skips Phase 1 (Analysis) and Phase 2 (Planning) by design. So when
41
+ the input references an analysis document - a Confluence URL, a local analysis file,
42
+ or the user says "I ran analysis for this" - there is **no phase that turns it into a
43
+ plan**. The doc becomes raw context for one Dev pass, and work comes out ordered by
44
+ whatever the model read first: the bottom of the dependency chain lands, the screen
45
+ wiring does not.
46
+
47
+ Say so before starting, once, and offer the choice:
48
+
49
+ ```
50
+ This mode skips Analysis and Planning, so the analysis document will not be turned
51
+ into a task breakdown. For analysis-driven screen work, /multi-agent or
52
+ /multi-agent:local run both phases.
53
+ 1. Continue with --dev (doc as context only)
54
+ 2. Switch to the full pipeline
55
+ ```
56
+
57
+ Autopilot picks 1 and logs the warning rather than asking.
@@ -17,7 +17,7 @@ Two-axis language preference for the pipeline:
17
17
  | `promptLanguage` | Interactive prompts during a pipeline run (account picker, project picker, dev-context picker, base-branch picker, branch-name picker, maturity ack, channels picker, Phase 5 test prompt, Phase 6 local-checkout prompt) | `prefs.global.promptLanguage` | **Fixed to `en`** - never toggled by this skill |
18
18
  | `outputLanguage` | Assistant's explanations, status updates, error messages, and pipeline-generated reports rendered to the user (NOT external payloads) | `prefs.global.outputLanguage` | Toggled by this skill |
19
19
 
20
- **Why promptLanguage is fixed:** Pipeline picker UI, confirmation prompts, and error messages are authored in English to keep tooling consistent across CLIs. Mixing Turkish prompts into otherwise-English skill output looks inconsistent. Only the assistant's free-form replies follow `outputLanguage`.
20
+ **Why promptLanguage is fixed:** it governs the picker's structural chrome only - `AskUserQuestion` `label` (button text) and `header` (chip), host error UI, and internal contract identifiers. Those stay English so tooling reads the same across CLIs. Everything a user actually reads follows `outputLanguage`, including the picker `question` and each option's `description`, per the canonical per-field matrix in `multi-agent-refs/rules.md`. A picker whose question is English on a Turkish run is a bug, not the contract.
21
21
 
22
22
  **Always English regardless of either field**: commit messages, PR titles/bodies, Jira comments, wiki pages, reviewer/triage system prompts, agent-log.md payloads, skill-picker / confirmation / error UI exposed by the CLI host.
23
23
 
@@ -83,5 +83,5 @@ Render in the new outputLanguage:
83
83
 
84
84
  - `multi-agent-setup` - first-run language picker (asks only outputLanguage; promptLanguage seeded as `en`)
85
85
  - `prefs.schema.json` - `global.promptLanguage` (fixed `"en"`) and `global.outputLanguage`
86
- - Phase 5 / Phase 6 - read promptLanguage for interactive prompts (always English)
86
+ - Phase 5 / Phase 6 - interactive prompts follow the per-field matrix: `question` + `description` in `outputLanguage`, `label` + `header` English
87
87
  - `multi-agent-help` - reads outputLanguage for its own copy
@@ -15,7 +15,7 @@ Ask BEFORE anything else. The pipeline has two language fields:
15
15
 
16
16
  | Field | Controls | Configurable? |
17
17
  |---|---|---|
18
- | `promptLanguage` | Interactive pickers and prompts during pipeline runs (account, project, branch, channels, Phase 5/6 confirmations) | **No - fixed to `en`.** Picker UI is always English. |
18
+ | `promptLanguage` | The picker's structural UI chrome only: `AskUserQuestion` `label` + `header`, host error UI, internal contract identifiers | **No - fixed to `en`.** Button and chip text stays English. |
19
19
  | `outputLanguage` | The assistant's non-interactive explanations, status updates, error messages, and pipeline-generated reports | Yes - set here, change later via `/multi-agent-language <en\|tr>` |
20
20
 
21
21
  `promptLanguage` is seeded as `"en"` and never offered to the user. External payloads (commits, PR/Jira/wiki, reviewer/triage prompts, agent-log.md) always stay English regardless.
@@ -278,6 +278,53 @@ pbcopy < /dev/null
278
278
  # ~/.claude/settings.json -> mcpServers -> figma
279
279
  ```
280
280
 
281
+ ### App Store Connect Credentials (optional, iOS only)
282
+
283
+ Unlocks Gate 2 of `multi-agent-testflight-validation` - Apple's own
284
+ `altool --validate-app`, the only check that sees an unregistered bundle ID, a
285
+ profile that does not match the App Store Connect app record, or a version+build
286
+ pair already used. Skipping is a first-class answer: the validation command still
287
+ runs its static audit and guideline review and reports Gate 2 as `SKIPPED` with
288
+ the reason, never as a pass.
289
+
290
+ Ask (picker): **API key** / **Apple ID + app-specific password** / **Skip**.
291
+
292
+ **API key** - needs an Admin or App Manager role in App Store Connect. Map the key
293
+ id and issuer id as `appstore_connect_key_id` / `appstore_connect_issuer_id`. The
294
+ private key is a FILE and never enters the credential store; it goes to one of the
295
+ directories altool searches:
296
+
297
+ ```bash
298
+ ls ~/.appstoreconnect/private_keys/AuthKey_*.p8 2>/dev/null \
299
+ || echo "MISSING: put AuthKey_<keyId>.p8 in ~/.appstoreconnect/private_keys/"
300
+ ```
301
+
302
+ **Apple ID + app-specific password** - usable by **any Apple ID holder, no elevated
303
+ role**, which is the realistic path when API-key creation is not permitted on the
304
+ account. Lead with this option when the user says they cannot create an API key.
305
+ The password goes into Apple's own keychain helper, which is what
306
+ `altool -p @keychain:<item>` reads - never into chat, never into an argument:
307
+
308
+ ```bash
309
+ # export AC_PASSWORD_ONCE in your own shell first, for this one command
310
+ xcrun altool --store-password-in-keychain-item "<item-name>" \
311
+ -u "<apple-id>" -p @env:AC_PASSWORD_ONCE
312
+ ```
313
+
314
+ Map only the ITEM NAME as `appstore_connect_password_item`, plus the Apple ID as
315
+ `appstore_connect_apple_id`. The password stays in the keychain and is referenced,
316
+ never read by the pipeline.
317
+
318
+ **Multi-provider accounts.** A corporate Apple ID often belongs to several
319
+ providers and altool fails opaquely without one. Resolve it once with
320
+ `ios_testflight_validate({list_providers: true, ...})` and store the id under
321
+ `prefs.projects[<key>].appStoreConnect.providerPublicId` - per-project, since a
322
+ user can ship for more than one team.
323
+
324
+ A credential that resolves but is rejected (401/403) follows the Expired-token
325
+ decision (Regenerate / Use a different token / Skip and continue) rather than
326
+ being silently dropped.
327
+
281
328
  ### Security Rules
282
329
 
283
330
  - Never paste tokens into terminal or chat - use `pbpaste` for input