@mmerterden/multi-agent-pipeline 12.11.0 → 13.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +179 -0
- package/README.md +24 -7
- package/index.js +5 -2
- package/install/_codex-agents.mjs +211 -0
- package/install/_codex-instructions.mjs +33 -0
- package/install/_managed-block.mjs +99 -0
- package/install/codex.mjs +478 -0
- package/install/copilot.mjs +34 -80
- package/install/index.mjs +25 -9
- package/install/templates/codex-instructions.md +45 -0
- package/package.json +5 -3
- package/pipeline/claude-md-template.md +1 -0
- package/pipeline/commands/multi-agent/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/dev/SKILL.md +52 -6
- package/pipeline/commands/multi-agent/dev-local/SKILL.md +21 -0
- package/pipeline/commands/multi-agent/finish/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/setup/SKILL.md +69 -2
- package/pipeline/commands/multi-agent/sync/SKILL.md +128 -5
- package/pipeline/commands/multi-agent/testflight-validation/SKILL.md +219 -0
- package/pipeline/commands/multi-agent/update/SKILL.md +7 -4
- package/pipeline/multi-agent-refs/_input-parser.md +1 -1
- package/pipeline/multi-agent-refs/component-dispatch.md +40 -7
- package/pipeline/multi-agent-refs/cross-cli-contract.md +51 -17
- package/pipeline/multi-agent-refs/features/model-fallback.md +29 -0
- package/pipeline/multi-agent-refs/features/review-multi-repo.md +1 -1
- package/pipeline/multi-agent-refs/phases/log-format.md +1 -1
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +43 -2
- package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +24 -1
- package/pipeline/multi-agent-refs/phases/phase-3-dev.md +32 -0
- package/pipeline/multi-agent-refs/phases/phase-4-review.md +62 -5
- package/pipeline/multi-agent-refs/progress-contract.md +1 -1
- package/pipeline/multi-agent-refs/tracker-contract.md +17 -1
- package/pipeline/schemas/prefs.schema.json +296 -62
- package/pipeline/schemas/reviewer-output.schema.json +1 -1
- package/pipeline/schemas/triage-output.schema.json +1 -1
- package/pipeline/scripts/cost-table.json +15 -1
- package/pipeline/scripts/phase0-exit-gate.mjs +185 -0
- package/pipeline/scripts/smoke-cross-cli-behavior.sh +25 -10
- package/pipeline/scripts/uninstall.mjs +105 -9
- package/pipeline/scripts/update-check.sh +2 -1
- package/pipeline/skills/shared/core/multi-agent-dev/SKILL.md +21 -0
- package/pipeline/skills/shared/core/multi-agent-dev-local/SKILL.md +21 -0
- package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +48 -1
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +87 -6
- package/pipeline/skills/shared/core/multi-agent-testflight-validation/SKILL.md +120 -0
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
"$id": "https://github.com/mmerterden/multi-agent-pipeline/pipeline/schemas/reviewer-output.schema.json",
|
|
4
4
|
"version": "1.0.0",
|
|
5
5
|
"title": "Multi-Agent Pipeline - Phase 4 reviewer output",
|
|
6
|
-
"description": "Contract for a single code-reviewer subagent's JSON output in Phase 4 Step 2. Reviewer set is CLI-aware: Claude Code dispatches 2 parallel reviewers (
|
|
6
|
+
"description": "Contract for a single code-reviewer subagent's JSON output in Phase 4 Step 2. Reviewer set is CLI-aware: Claude Code dispatches 2 parallel reviewers (Fable, Sonnet); Copilot CLI dispatches 3 (Opus, GPT-5.4, Sonnet); Codex CLI dispatches 3 (gpt-5.6 at xhigh, gpt-5.4, gpt-5.6 at medium). Every reviewer must return an object matching this shape before Opus triage merges them.",
|
|
7
7
|
"type": "object",
|
|
8
8
|
"additionalProperties": false,
|
|
9
9
|
"required": ["findings", "approved"],
|
|
@@ -86,7 +86,7 @@
|
|
|
86
86
|
"reviewerCount": {
|
|
87
87
|
"type": "integer",
|
|
88
88
|
"minimum": 1,
|
|
89
|
-
"description": "Reviewers dispatched this iteration (Claude Code = 2, Copilot CLI = 3)."
|
|
89
|
+
"description": "Reviewers dispatched this iteration (Claude Code = 2, Copilot CLI = 3, Codex CLI = 3)."
|
|
90
90
|
},
|
|
91
91
|
"verdict": {
|
|
92
92
|
"type": "string",
|
|
@@ -32,7 +32,21 @@
|
|
|
32
32
|
"outPerMtok": 30.0,
|
|
33
33
|
"cacheReadPerMtok": 1.0,
|
|
34
34
|
"modelId": "gpt-5.4",
|
|
35
|
-
"note": "Copilot CLI Reviewer 2 - approximate; verify against OpenAI pricing page before relying on totals."
|
|
35
|
+
"note": "Copilot CLI Reviewer 2 and Codex CLI Reviewer 2 - approximate; verify against OpenAI pricing page before relying on totals."
|
|
36
|
+
},
|
|
37
|
+
"gpt-5.6": {
|
|
38
|
+
"inPerMtok": 10.0,
|
|
39
|
+
"outPerMtok": 30.0,
|
|
40
|
+
"cacheReadPerMtok": 1.0,
|
|
41
|
+
"modelId": "gpt-5.6",
|
|
42
|
+
"note": "Codex CLI top tier - Reviewer 1 at xhigh effort, Reviewer 3 at medium, triage at max, and the fable/opus rungs of the Codex persona tier map. Reasoning effort changes output volume, not the per-token rate, so one entry covers every effort level. Approximate; verify against OpenAI pricing before relying on totals."
|
|
43
|
+
},
|
|
44
|
+
"gpt-5.6-terra": {
|
|
45
|
+
"inPerMtok": 1.0,
|
|
46
|
+
"outPerMtok": 5.0,
|
|
47
|
+
"cacheReadPerMtok": 0.1,
|
|
48
|
+
"modelId": "gpt-5.6-terra",
|
|
49
|
+
"note": "Codex CLI floor tier - speed-optimised, maps the haiku rung of the Codex persona tier map (task-clarifier). Approximate; verify against OpenAI pricing before relying on totals."
|
|
36
50
|
}
|
|
37
51
|
}
|
|
38
52
|
}
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* phase0-exit-gate.mjs - Phase 0 may not be marked completed until it has
|
|
4
|
+
* actually produced its own output.
|
|
5
|
+
*
|
|
6
|
+
* Why this exists. Phase 0 Step 7 says to classify the task and "persist to
|
|
7
|
+
* `agent-state.json.taskType`", and Phase 3 branches on that field to route a
|
|
8
|
+
* Figma-driven task to the component skills (`create-screen` / `create-component`
|
|
9
|
+
* in the stack plugin) instead of generic development. On a real run the tracker
|
|
10
|
+
* showed Phase 0 `completed` while the task directory held only
|
|
11
|
+
* `tracker-state.json` - no `agent-state.json` at all. With no `taskType`, the
|
|
12
|
+
* dispatch could not fire, so a Figma-driven screen was built as generic work: no
|
|
13
|
+
* token-compliance check, no Code Connect publish, no 14-item component review,
|
|
14
|
+
* and spacing guessed at 16 where the frame said `Spacing/12`. Half the commits on
|
|
15
|
+
* that branch were rework.
|
|
16
|
+
*
|
|
17
|
+
* A phase that reports success without its output is worse than one that fails:
|
|
18
|
+
* every later phase then reasons from a field that is not there. So this is a
|
|
19
|
+
* gate, not a lint - the spec already said what to write, and prose alone did
|
|
20
|
+
* not make it happen.
|
|
21
|
+
*
|
|
22
|
+
* Usage:
|
|
23
|
+
* node phase0-exit-gate.mjs <task_id> [--input "<original user input>"] [--json]
|
|
24
|
+
*
|
|
25
|
+
* Exit codes: 0 = pass, 1 = gate failure (Phase 0 must not be closed), 2 = usage
|
|
26
|
+
*
|
|
27
|
+
* @module pipeline/scripts/phase0-exit-gate
|
|
28
|
+
*/
|
|
29
|
+
|
|
30
|
+
import { existsSync, readFileSync } from "fs";
|
|
31
|
+
import { join } from "path";
|
|
32
|
+
import { homedir } from "os";
|
|
33
|
+
|
|
34
|
+
const FIGMA_URL = /figma\.com\/(design|make|file)\//i;
|
|
35
|
+
|
|
36
|
+
/** Fields that can carry the user's original request text. */
|
|
37
|
+
const INPUT_TEXT_FIELDS = [
|
|
38
|
+
["input", "summary"],
|
|
39
|
+
["input", "raw"],
|
|
40
|
+
["issue", "body"],
|
|
41
|
+
["issue", "title"],
|
|
42
|
+
["jira", "summary"],
|
|
43
|
+
["jira", "description"],
|
|
44
|
+
["task", "description"],
|
|
45
|
+
];
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Pull a nested value without throwing on a missing branch.
|
|
49
|
+
* @param {object} obj
|
|
50
|
+
* @param {string[]} path
|
|
51
|
+
* @returns {unknown}
|
|
52
|
+
*/
|
|
53
|
+
function at(obj, path) {
|
|
54
|
+
return path.reduce((acc, k) => (acc && typeof acc === "object" ? acc[k] : undefined), obj);
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Collect every string in the state that could hold the original request, so a
|
|
59
|
+
* Figma reference is found wherever the input parser happened to store it.
|
|
60
|
+
*
|
|
61
|
+
* @param {object} state
|
|
62
|
+
* @returns {string}
|
|
63
|
+
*/
|
|
64
|
+
export function inputTextOf(state) {
|
|
65
|
+
const parts = [];
|
|
66
|
+
for (const path of INPUT_TEXT_FIELDS) {
|
|
67
|
+
const v = at(state, path);
|
|
68
|
+
if (typeof v === "string") parts.push(v);
|
|
69
|
+
}
|
|
70
|
+
// evidence[].url / figmaFrames[] are where the analysis phase parks references.
|
|
71
|
+
for (const key of ["figmaFrames", "designRefs"]) {
|
|
72
|
+
const v = state[key];
|
|
73
|
+
if (Array.isArray(v)) parts.push(v.map((x) => (typeof x === "string" ? x : JSON.stringify(x))).join(" "));
|
|
74
|
+
}
|
|
75
|
+
return parts.join("\n");
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Evaluate the gate against a parsed state object.
|
|
80
|
+
*
|
|
81
|
+
* Kept pure and exported so the smoke can drive it without laying down a task
|
|
82
|
+
* directory: a gate that can only be tested by reproducing a full run does not
|
|
83
|
+
* get tested.
|
|
84
|
+
*
|
|
85
|
+
* @param {object|null} state - parsed agent-state.json, or null when absent
|
|
86
|
+
* @param {string} extraInput - input text supplied on the command line
|
|
87
|
+
* @returns {{ok: boolean, failures: string[], taskType: string|undefined, figmaSeen: boolean}}
|
|
88
|
+
*/
|
|
89
|
+
export function evaluate(state, extraInput = "") {
|
|
90
|
+
const failures = [];
|
|
91
|
+
|
|
92
|
+
if (state === null) {
|
|
93
|
+
return {
|
|
94
|
+
ok: false,
|
|
95
|
+
failures: [
|
|
96
|
+
"agent-state.json is missing. Phase 0 owns this file; without it taskType, " +
|
|
97
|
+
"maturity, account and repo resolution are all unreadable by later phases.",
|
|
98
|
+
],
|
|
99
|
+
taskType: undefined,
|
|
100
|
+
figmaSeen: FIGMA_URL.test(extraInput),
|
|
101
|
+
};
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
const taskType = typeof state.taskType === "string" ? state.taskType.trim() : "";
|
|
105
|
+
if (!taskType) {
|
|
106
|
+
failures.push(
|
|
107
|
+
"agent-state.json has no taskType. Phase 3 branches on it: without the field " +
|
|
108
|
+
"a component task is dispatched as generic development (Step 7 of phase-0-init).",
|
|
109
|
+
);
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
const haystack = `${inputTextOf(state)}\n${extraInput}`;
|
|
113
|
+
const figmaSeen = FIGMA_URL.test(haystack);
|
|
114
|
+
|
|
115
|
+
if (figmaSeen && taskType && taskType !== "component") {
|
|
116
|
+
failures.push(
|
|
117
|
+
`the input carries a Figma URL but taskType is "${taskType}". Step 7 rule 1 makes ` +
|
|
118
|
+
`"component" mandatory here - otherwise the run skips the stack plugin's ` +
|
|
119
|
+
`token-compliance check, Code Connect publish and component review.`,
|
|
120
|
+
);
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
// The Figma access chain records which tier answered. A component task with no
|
|
124
|
+
// recorded tier means nothing verified that the design was actually reachable,
|
|
125
|
+
// which is how a run ends up guessing spacing.
|
|
126
|
+
if (figmaSeen) {
|
|
127
|
+
const tier = at(state, ["figmaAccess", "tier"]);
|
|
128
|
+
if (tier === undefined || tier === null || tier === "") {
|
|
129
|
+
failures.push(
|
|
130
|
+
"the input carries a Figma URL but state.figmaAccess.tier is unset. The 3-tier " +
|
|
131
|
+
"access chain must record which tier answered, so a later phase can tell " +
|
|
132
|
+
"'design confirmed' from 'design never fetched'.",
|
|
133
|
+
);
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
return { ok: failures.length === 0, failures, taskType: taskType || undefined, figmaSeen };
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/** @param {string} taskId */
|
|
141
|
+
function statePathFor(taskId) {
|
|
142
|
+
return join(homedir(), ".claude", "logs", "multi-agent", taskId, "agent-state.json");
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
function main(argv) {
|
|
146
|
+
const args = argv.slice(2);
|
|
147
|
+
const taskId = args.find((a) => !a.startsWith("--"));
|
|
148
|
+
if (!taskId) {
|
|
149
|
+
console.error("usage: phase0-exit-gate.mjs <task_id> [--input \"<text>\"] [--json]");
|
|
150
|
+
return 2;
|
|
151
|
+
}
|
|
152
|
+
const inputIdx = args.indexOf("--input");
|
|
153
|
+
const extraInput = inputIdx >= 0 ? (args[inputIdx + 1] ?? "") : "";
|
|
154
|
+
const asJson = args.includes("--json");
|
|
155
|
+
|
|
156
|
+
const path = statePathFor(taskId);
|
|
157
|
+
let state = null;
|
|
158
|
+
if (existsSync(path)) {
|
|
159
|
+
try {
|
|
160
|
+
state = JSON.parse(readFileSync(path, "utf-8"));
|
|
161
|
+
} catch (e) {
|
|
162
|
+
console.error(`phase0-exit-gate: ${path} is not valid JSON: ${e.message}`);
|
|
163
|
+
return 1;
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
const result = evaluate(state, extraInput);
|
|
168
|
+
|
|
169
|
+
if (asJson) {
|
|
170
|
+
console.log(JSON.stringify({ taskId, statePath: path, ...result }, null, 2));
|
|
171
|
+
} else if (result.ok) {
|
|
172
|
+
console.log(
|
|
173
|
+
`phase0-exit-gate: PASS (taskType=${result.taskType}${result.figmaSeen ? ", figma reference present" : ""})`,
|
|
174
|
+
);
|
|
175
|
+
} else {
|
|
176
|
+
console.error("phase0-exit-gate: FAIL - Phase 0 must not be marked completed");
|
|
177
|
+
for (const f of result.failures) console.error(` - ${f}`);
|
|
178
|
+
console.error(` state: ${path}`);
|
|
179
|
+
}
|
|
180
|
+
return result.ok ? 0 : 1;
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
if (import.meta.url === `file://${process.argv[1]}`) {
|
|
184
|
+
process.exit(main(process.argv));
|
|
185
|
+
}
|
|
@@ -185,11 +185,12 @@ else
|
|
|
185
185
|
fi
|
|
186
186
|
|
|
187
187
|
# ──────────────────────────────────────────────────────────────────────────
|
|
188
|
-
# CLI-aware reviewer-count contract: Claude Code = 2
|
|
189
|
-
#
|
|
190
|
-
#
|
|
191
|
-
# drift across the phase doc, the schema,
|
|
192
|
-
|
|
188
|
+
# CLI-aware reviewer-count contract: Claude Code = 2 (Fable + Sonnet), Copilot CLI
|
|
189
|
+
# = 3 (Opus + GPT-5.4 + Sonnet), Codex CLI = 3 (gpt-5.6 xhigh + gpt-5.4 + gpt-5.6
|
|
190
|
+
# medium). Nothing in code enforces the count (the orchestrator dispatches per the
|
|
191
|
+
# doc), so lock the CONTRACT here against drift across the phase doc, the schema,
|
|
192
|
+
# and the consensus block.
|
|
193
|
+
echo "→ reviewer-count contract (Claude=2, Copilot=3, Codex=3)"
|
|
193
194
|
# `/multi-agent:update` runs this smoke on the user's machine, where the tree is
|
|
194
195
|
# ~/.claude/{schemas,multi-agent-refs} and NOT <root>/pipeline/*. Resolving the
|
|
195
196
|
# root by hand here reported three phantom failures in an install, which reads as
|
|
@@ -202,16 +203,30 @@ TRSCHEMA="${MA_SCHEMAS:+$MA_SCHEMAS/triage-output.schema.json}"
|
|
|
202
203
|
|
|
203
204
|
if [ -z "$P4" ] || [ ! -f "$P4" ]; then
|
|
204
205
|
echo " ↷ SKIP: phase-4-review.md not present in this $MA_LAYOUT layout"
|
|
205
|
-
elif grep -qiE "Claude Code (dispatches|=|:) ?2|2-model" "$P4"
|
|
206
|
-
|
|
206
|
+
elif grep -qiE "Claude Code (dispatches|=|:) ?2|2-model" "$P4" \
|
|
207
|
+
&& grep -qiE "Copilot CLI (dispatches|=|:) ?3|3-model" "$P4" \
|
|
208
|
+
&& grep -qiE "Codex CLI (dispatches|=|:) ?3|Claude Code 2, Copilot CLI 3, Codex CLI 3" "$P4"; then
|
|
209
|
+
pass "phase-4-review declares Claude=2 / Copilot=3 / Codex=3 reviewers"
|
|
207
210
|
else
|
|
208
|
-
fail "phase-4-review does not declare the CLI-aware reviewer count"
|
|
211
|
+
fail "phase-4-review does not declare the CLI-aware reviewer count for all three hosts"
|
|
212
|
+
fi
|
|
213
|
+
|
|
214
|
+
# The two Codex constraints are silent-failure shaped, so the contract has to name
|
|
215
|
+
# them: a spawn without fork_turns collapses the panel onto one model, and the
|
|
216
|
+
# 4-slot ceiling (orchestrator included) is why the count is 3 and not more.
|
|
217
|
+
if [ -n "$P4" ] && [ -f "$P4" ]; then
|
|
218
|
+
if grep -q 'fork_turns' "$P4" && grep -qiE "concurrency|slots" "$P4"; then
|
|
219
|
+
pass "phase-4-review documents the fork_turns override rule + the concurrency ceiling"
|
|
220
|
+
else
|
|
221
|
+
fail "phase-4-review must document fork_turns and the Codex concurrency ceiling"
|
|
222
|
+
fi
|
|
209
223
|
fi
|
|
210
224
|
|
|
211
225
|
if [ -z "$REVSCHEMA" ] || [ ! -f "$REVSCHEMA" ]; then
|
|
212
226
|
echo " ↷ SKIP: reviewer-output.schema.json not present in this $MA_LAYOUT layout"
|
|
213
|
-
elif grep -qi "Copilot CLI" "$REVSCHEMA" && grep -qi "Claude" "$REVSCHEMA"
|
|
214
|
-
|
|
227
|
+
elif grep -qi "Copilot CLI" "$REVSCHEMA" && grep -qi "Claude" "$REVSCHEMA" \
|
|
228
|
+
&& grep -qi "Codex" "$REVSCHEMA"; then
|
|
229
|
+
pass "reviewer-output schema notes the CLI-aware reviewer set (all three hosts)"
|
|
215
230
|
else
|
|
216
231
|
fail "reviewer-output schema missing the CLI-aware note"
|
|
217
232
|
fi
|
|
@@ -36,6 +36,7 @@
|
|
|
36
36
|
*/
|
|
37
37
|
|
|
38
38
|
import { existsSync, readdirSync, readFileSync, rmSync, writeFileSync } from "fs";
|
|
39
|
+
import { execFileSync } from "child_process";
|
|
39
40
|
import { join } from "path";
|
|
40
41
|
import { pathToFileURL } from "url";
|
|
41
42
|
import { createInterface } from "readline";
|
|
@@ -123,6 +124,54 @@ function rmIfExists(path) {
|
|
|
123
124
|
return true;
|
|
124
125
|
}
|
|
125
126
|
|
|
127
|
+
/**
|
|
128
|
+
* Remove generated Codex agent TOML files, identified by the header the
|
|
129
|
+
* installer stamps on them. A user-authored `.toml` sharing a persona name is
|
|
130
|
+
* left alone - `~/.codex/agents/` is a co-owned directory.
|
|
131
|
+
*
|
|
132
|
+
* @param {string} agentsDir
|
|
133
|
+
* @returns {number} files removed
|
|
134
|
+
*/
|
|
135
|
+
function removeGeneratedCodexAgents(agentsDir) {
|
|
136
|
+
if (!existsSync(agentsDir)) return 0;
|
|
137
|
+
const MARKER = "# Generated by multi-agent-pipeline install --codex.";
|
|
138
|
+
let removed = 0;
|
|
139
|
+
for (const name of readdirSync(agentsDir)) {
|
|
140
|
+
if (!name.endsWith(".toml")) continue;
|
|
141
|
+
const path = join(agentsDir, name);
|
|
142
|
+
try {
|
|
143
|
+
if (!readFileSync(path, "utf-8").startsWith(MARKER)) continue;
|
|
144
|
+
} catch {
|
|
145
|
+
continue;
|
|
146
|
+
}
|
|
147
|
+
if (rmIfExists(path)) removed++;
|
|
148
|
+
}
|
|
149
|
+
if (removed > 0) console.log(` removed ${removed} generated agent file(s) under ${agentsDir}`);
|
|
150
|
+
return removed;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/**
|
|
154
|
+
* Hand the dev-toolkit MCP registration back to Codex.
|
|
155
|
+
*
|
|
156
|
+
* Best-effort: a missing `codex` binary or an already-absent entry is not an
|
|
157
|
+
* uninstall failure. Never edits `config.toml` directly - Codex keeps
|
|
158
|
+
* marketplace and plugin state in the same file.
|
|
159
|
+
*/
|
|
160
|
+
function deregisterCodexMcpServer() {
|
|
161
|
+
if (dryRun) {
|
|
162
|
+
report("would run", "codex mcp remove dev-toolkit");
|
|
163
|
+
return;
|
|
164
|
+
}
|
|
165
|
+
try {
|
|
166
|
+
execFileSync("codex", ["mcp", "remove", "dev-toolkit"], { stdio: "pipe", timeout: 20_000 });
|
|
167
|
+
console.log(" removed: dev-toolkit MCP registration");
|
|
168
|
+
} catch {
|
|
169
|
+
console.log(
|
|
170
|
+
" note: could not run `codex mcp remove dev-toolkit` (codex not on PATH, or not registered)",
|
|
171
|
+
);
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
|
|
126
175
|
/**
|
|
127
176
|
* Remove every directory under `parent` whose name matches a predicate.
|
|
128
177
|
* @param {string} parent
|
|
@@ -160,6 +209,32 @@ function rmMatchingFiles(parent, predicate) {
|
|
|
160
209
|
// (v11.4.1+). Must stay in sync with INSTRUCTIONS_END_MARKER there.
|
|
161
210
|
const COPILOT_END_MARKER = "<!-- multi-agent-pipeline:copilot-instructions:end -->";
|
|
162
211
|
|
|
212
|
+
/**
|
|
213
|
+
* Every end marker the installers write. Both host instruction files share the
|
|
214
|
+
* same START marker but terminate with their own comment, so a single-marker
|
|
215
|
+
* lookup silently falls through to the legacy heading-bounded path - and since
|
|
216
|
+
* the managed body has no top-level heading after the first, that path treats
|
|
217
|
+
* ALL trailing user content as part of the block and deletes it.
|
|
218
|
+
*/
|
|
219
|
+
const MANAGED_END_MARKERS = [
|
|
220
|
+
COPILOT_END_MARKER,
|
|
221
|
+
"<!-- multi-agent-pipeline:codex-instructions:end -->",
|
|
222
|
+
];
|
|
223
|
+
|
|
224
|
+
/**
|
|
225
|
+
* Locate whichever managed end marker terminates this block.
|
|
226
|
+
*
|
|
227
|
+
* @param {string} fromStart - content from the start marker onward
|
|
228
|
+
* @returns {{index: number, length: number}} index -1 when none is present
|
|
229
|
+
*/
|
|
230
|
+
function findManagedEndMarker(fromStart) {
|
|
231
|
+
for (const marker of MANAGED_END_MARKERS) {
|
|
232
|
+
const index = fromStart.indexOf(marker);
|
|
233
|
+
if (index >= 0) return { index, length: marker.length };
|
|
234
|
+
}
|
|
235
|
+
return { index: -1, length: 0 };
|
|
236
|
+
}
|
|
237
|
+
|
|
163
238
|
/**
|
|
164
239
|
* Legacy copilot-instructions files (written before the end marker existed)
|
|
165
240
|
* have no explicit terminator. Bound the pipeline span at the next top-level
|
|
@@ -205,10 +280,10 @@ function stripManagedBlock(filePath) {
|
|
|
205
280
|
if (idx >= 0) {
|
|
206
281
|
const before = content.slice(0, idx).trimEnd();
|
|
207
282
|
const fromStart = content.slice(idx);
|
|
208
|
-
const
|
|
283
|
+
const end = findManagedEndMarker(fromStart);
|
|
209
284
|
const rawTrailing =
|
|
210
|
-
|
|
211
|
-
? fromStart.slice(
|
|
285
|
+
end.index >= 0
|
|
286
|
+
? fromStart.slice(end.index + end.length)
|
|
212
287
|
: legacyTrailingContent(fromStart);
|
|
213
288
|
const trailing = rawTrailing.replace(/^[\r\n]+/, "").trimEnd();
|
|
214
289
|
let remaining = before;
|
|
@@ -332,7 +407,7 @@ export async function main() {
|
|
|
332
407
|
);
|
|
333
408
|
if (forCodex)
|
|
334
409
|
console.log(
|
|
335
|
-
" -
|
|
410
|
+
" - Codex CLI (~/.codex: skills/multi-agent, multi-agent-refs, agents/*.toml, prompts/multi-agent.md, scripts, lib, schemas, rules, AGENTS.md block, dev-toolkit MCP entry)",
|
|
336
411
|
);
|
|
337
412
|
console.log("");
|
|
338
413
|
if (allData) {
|
|
@@ -478,14 +553,35 @@ export async function main() {
|
|
|
478
553
|
|
|
479
554
|
if (forCodex && HOME) {
|
|
480
555
|
console.log("");
|
|
481
|
-
console.log(" [
|
|
556
|
+
console.log(" [Codex CLI] Removing from ~/.codex...");
|
|
482
557
|
const CODEX = join(HOME, ".codex");
|
|
558
|
+
|
|
559
|
+
// Wholly pipeline-owned trees.
|
|
560
|
+
rmIfExists(join(CODEX, "multi-agent-refs"));
|
|
561
|
+
rmIfExists(join(CODEX, "scripts"));
|
|
562
|
+
rmIfExists(join(CODEX, "lib"));
|
|
563
|
+
rmIfExists(join(CODEX, "schemas"));
|
|
564
|
+
rmIfExists(join(CODEX, "rules"));
|
|
565
|
+
rmIfExists(join(CODEX, ".pipeline-version"));
|
|
566
|
+
|
|
567
|
+
// Co-owned dirs: remove only what the pipeline wrote. `skills/` holds
|
|
568
|
+
// Codex's own `.system` set plus user skills; `prompts/` and `agents/` hold
|
|
569
|
+
// user-authored files.
|
|
570
|
+
const skills = join(CODEX, "skills");
|
|
571
|
+
const n = rmMatchingDirs(
|
|
572
|
+
skills,
|
|
573
|
+
(name) => name === "multi-agent" || name.startsWith("multi-agent-"),
|
|
574
|
+
);
|
|
575
|
+
if (n > 0) console.log(` removed ${n} skill dir(s) under ${skills}`);
|
|
483
576
|
rmIfExists(join(CODEX, "prompts", "multi-agent.md"));
|
|
577
|
+
removeGeneratedCodexAgents(join(CODEX, "agents"));
|
|
578
|
+
|
|
484
579
|
stripManagedBlock(join(CODEX, "AGENTS.md"));
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
580
|
+
|
|
581
|
+
// The MCP entry was written by `codex mcp add`, so hand it back the same
|
|
582
|
+
// way rather than editing config.toml - Codex also stores marketplace and
|
|
583
|
+
// plugin state in that file.
|
|
584
|
+
deregisterCodexMcpServer();
|
|
489
585
|
}
|
|
490
586
|
|
|
491
587
|
console.log("");
|
|
@@ -48,7 +48,8 @@ if [ -z "$LOCAL_VERSION" ]; then
|
|
|
48
48
|
fi
|
|
49
49
|
# npx-only installs have no repo clone; the installer stamps the version here.
|
|
50
50
|
if [ -z "$LOCAL_VERSION" ]; then
|
|
51
|
-
for marker in "$HOME/.claude/.pipeline-version" "$HOME/.copilot/.pipeline-version"
|
|
51
|
+
for marker in "$HOME/.claude/.pipeline-version" "$HOME/.copilot/.pipeline-version" \
|
|
52
|
+
"$HOME/.codex/.pipeline-version"; do
|
|
52
53
|
if [ -f "$marker" ]; then
|
|
53
54
|
LOCAL_VERSION=$(head -1 "$marker" 2>/dev/null | tr -d '[:space:]')
|
|
54
55
|
[ -n "$LOCAL_VERSION" ] && break
|
|
@@ -62,3 +62,24 @@ Phase 7: Report → Channels (Jira / Confluence / PR / Wiki)
|
|
|
62
62
|
| Phase 5 User Test | ✅ | ✅ (same) |
|
|
63
63
|
| Phase 7 channels (Jira / Confluence / PR / Wiki) | ✅ | ✅ (same) |
|
|
64
64
|
| Duration | ~10-15 min | ~5-7 min |
|
|
65
|
+
|
|
66
|
+
## Analysis doc supplied to a fast mode (warn before starting)
|
|
67
|
+
|
|
68
|
+
The `--dev` family skips Phase 1 (Analysis) and Phase 2 (Planning) by design. So when
|
|
69
|
+
the input references an analysis document - a Confluence URL, a local analysis file,
|
|
70
|
+
or the user says "I ran analysis for this" - there is **no phase that turns it into a
|
|
71
|
+
plan**. The doc becomes raw context for one Dev pass, and work comes out ordered by
|
|
72
|
+
whatever the model read first: the bottom of the dependency chain lands, the screen
|
|
73
|
+
wiring does not.
|
|
74
|
+
|
|
75
|
+
Say so before starting, once, and offer the choice:
|
|
76
|
+
|
|
77
|
+
```
|
|
78
|
+
This mode skips Analysis and Planning, so the analysis document will not be turned
|
|
79
|
+
into a task breakdown. For analysis-driven screen work, /multi-agent or
|
|
80
|
+
/multi-agent:local run both phases.
|
|
81
|
+
1. Continue with --dev (doc as context only)
|
|
82
|
+
2. Switch to the full pipeline
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
Autopilot picks 1 and logs the warning rather than asking.
|
|
@@ -34,3 +34,24 @@ Routes to the orchestrator with the `--dev --local` flags. The pipeline contract
|
|
|
34
34
|
multi-agent-dev-local "PROJ-12345" # Jira
|
|
35
35
|
multi-agent-dev-local "Bug: LoginView dark mode" # Free-text
|
|
36
36
|
```
|
|
37
|
+
|
|
38
|
+
## Analysis doc supplied to a fast mode (warn before starting)
|
|
39
|
+
|
|
40
|
+
The `--dev` family skips Phase 1 (Analysis) and Phase 2 (Planning) by design. So when
|
|
41
|
+
the input references an analysis document - a Confluence URL, a local analysis file,
|
|
42
|
+
or the user says "I ran analysis for this" - there is **no phase that turns it into a
|
|
43
|
+
plan**. The doc becomes raw context for one Dev pass, and work comes out ordered by
|
|
44
|
+
whatever the model read first: the bottom of the dependency chain lands, the screen
|
|
45
|
+
wiring does not.
|
|
46
|
+
|
|
47
|
+
Say so before starting, once, and offer the choice:
|
|
48
|
+
|
|
49
|
+
```
|
|
50
|
+
This mode skips Analysis and Planning, so the analysis document will not be turned
|
|
51
|
+
into a task breakdown. For analysis-driven screen work, /multi-agent or
|
|
52
|
+
/multi-agent:local run both phases.
|
|
53
|
+
1. Continue with --dev (doc as context only)
|
|
54
|
+
2. Switch to the full pipeline
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Autopilot picks 1 and logs the warning rather than asking.
|
|
@@ -17,7 +17,7 @@ Two-axis language preference for the pipeline:
|
|
|
17
17
|
| `promptLanguage` | Interactive prompts during a pipeline run (account picker, project picker, dev-context picker, base-branch picker, branch-name picker, maturity ack, channels picker, Phase 5 test prompt, Phase 6 local-checkout prompt) | `prefs.global.promptLanguage` | **Fixed to `en`** - never toggled by this skill |
|
|
18
18
|
| `outputLanguage` | Assistant's explanations, status updates, error messages, and pipeline-generated reports rendered to the user (NOT external payloads) | `prefs.global.outputLanguage` | Toggled by this skill |
|
|
19
19
|
|
|
20
|
-
**Why promptLanguage is fixed:**
|
|
20
|
+
**Why promptLanguage is fixed:** it governs the picker's structural chrome only - `AskUserQuestion` `label` (button text) and `header` (chip), host error UI, and internal contract identifiers. Those stay English so tooling reads the same across CLIs. Everything a user actually reads follows `outputLanguage`, including the picker `question` and each option's `description`, per the canonical per-field matrix in `multi-agent-refs/rules.md`. A picker whose question is English on a Turkish run is a bug, not the contract.
|
|
21
21
|
|
|
22
22
|
**Always English regardless of either field**: commit messages, PR titles/bodies, Jira comments, wiki pages, reviewer/triage system prompts, agent-log.md payloads, skill-picker / confirmation / error UI exposed by the CLI host.
|
|
23
23
|
|
|
@@ -83,5 +83,5 @@ Render in the new outputLanguage:
|
|
|
83
83
|
|
|
84
84
|
- `multi-agent-setup` - first-run language picker (asks only outputLanguage; promptLanguage seeded as `en`)
|
|
85
85
|
- `prefs.schema.json` - `global.promptLanguage` (fixed `"en"`) and `global.outputLanguage`
|
|
86
|
-
- Phase 5 / Phase 6 -
|
|
86
|
+
- Phase 5 / Phase 6 - interactive prompts follow the per-field matrix: `question` + `description` in `outputLanguage`, `label` + `header` English
|
|
87
87
|
- `multi-agent-help` - reads outputLanguage for its own copy
|
|
@@ -15,7 +15,7 @@ Ask BEFORE anything else. The pipeline has two language fields:
|
|
|
15
15
|
|
|
16
16
|
| Field | Controls | Configurable? |
|
|
17
17
|
|---|---|---|
|
|
18
|
-
| `promptLanguage` |
|
|
18
|
+
| `promptLanguage` | The picker's structural UI chrome only: `AskUserQuestion` `label` + `header`, host error UI, internal contract identifiers | **No - fixed to `en`.** Button and chip text stays English. |
|
|
19
19
|
| `outputLanguage` | The assistant's non-interactive explanations, status updates, error messages, and pipeline-generated reports | Yes - set here, change later via `/multi-agent-language <en\|tr>` |
|
|
20
20
|
|
|
21
21
|
`promptLanguage` is seeded as `"en"` and never offered to the user. External payloads (commits, PR/Jira/wiki, reviewer/triage prompts, agent-log.md) always stay English regardless.
|
|
@@ -278,6 +278,53 @@ pbcopy < /dev/null
|
|
|
278
278
|
# ~/.claude/settings.json -> mcpServers -> figma
|
|
279
279
|
```
|
|
280
280
|
|
|
281
|
+
### App Store Connect Credentials (optional, iOS only)
|
|
282
|
+
|
|
283
|
+
Unlocks Gate 2 of `multi-agent-testflight-validation` - Apple's own
|
|
284
|
+
`altool --validate-app`, the only check that sees an unregistered bundle ID, a
|
|
285
|
+
profile that does not match the App Store Connect app record, or a version+build
|
|
286
|
+
pair already used. Skipping is a first-class answer: the validation command still
|
|
287
|
+
runs its static audit and guideline review and reports Gate 2 as `SKIPPED` with
|
|
288
|
+
the reason, never as a pass.
|
|
289
|
+
|
|
290
|
+
Ask (picker): **API key** / **Apple ID + app-specific password** / **Skip**.
|
|
291
|
+
|
|
292
|
+
**API key** - needs an Admin or App Manager role in App Store Connect. Map the key
|
|
293
|
+
id and issuer id as `appstore_connect_key_id` / `appstore_connect_issuer_id`. The
|
|
294
|
+
private key is a FILE and never enters the credential store; it goes to one of the
|
|
295
|
+
directories altool searches:
|
|
296
|
+
|
|
297
|
+
```bash
|
|
298
|
+
ls ~/.appstoreconnect/private_keys/AuthKey_*.p8 2>/dev/null \
|
|
299
|
+
|| echo "MISSING: put AuthKey_<keyId>.p8 in ~/.appstoreconnect/private_keys/"
|
|
300
|
+
```
|
|
301
|
+
|
|
302
|
+
**Apple ID + app-specific password** - usable by **any Apple ID holder, no elevated
|
|
303
|
+
role**, which is the realistic path when API-key creation is not permitted on the
|
|
304
|
+
account. Lead with this option when the user says they cannot create an API key.
|
|
305
|
+
The password goes into Apple's own keychain helper, which is what
|
|
306
|
+
`altool -p @keychain:<item>` reads - never into chat, never into an argument:
|
|
307
|
+
|
|
308
|
+
```bash
|
|
309
|
+
# export AC_PASSWORD_ONCE in your own shell first, for this one command
|
|
310
|
+
xcrun altool --store-password-in-keychain-item "<item-name>" \
|
|
311
|
+
-u "<apple-id>" -p @env:AC_PASSWORD_ONCE
|
|
312
|
+
```
|
|
313
|
+
|
|
314
|
+
Map only the ITEM NAME as `appstore_connect_password_item`, plus the Apple ID as
|
|
315
|
+
`appstore_connect_apple_id`. The password stays in the keychain and is referenced,
|
|
316
|
+
never read by the pipeline.
|
|
317
|
+
|
|
318
|
+
**Multi-provider accounts.** A corporate Apple ID often belongs to several
|
|
319
|
+
providers and altool fails opaquely without one. Resolve it once with
|
|
320
|
+
`ios_testflight_validate({list_providers: true, ...})` and store the id under
|
|
321
|
+
`prefs.projects[<key>].appStoreConnect.providerPublicId` - per-project, since a
|
|
322
|
+
user can ship for more than one team.
|
|
323
|
+
|
|
324
|
+
A credential that resolves but is rejected (401/403) follows the Expired-token
|
|
325
|
+
decision (Regenerate / Use a different token / Skip and continue) rather than
|
|
326
|
+
being silently dropped.
|
|
327
|
+
|
|
281
328
|
### Security Rules
|
|
282
329
|
|
|
283
330
|
- Never paste tokens into terminal or chat - use `pbpaste` for input
|