@mmerterden/multi-agent-pipeline 17.6.0 → 19.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (272) hide show
  1. package/CHANGELOG.md +310 -0
  2. package/README.md +76 -18
  3. package/README.tr.md +55 -16
  4. package/docs/adr/0002-instruction-driven-flag.md +1 -0
  5. package/docs/adr/0005-lazy-phase-docs.md +11 -1
  6. package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
  7. package/docs/adr/0010-own-code-graph.md +1 -0
  8. package/docs/adr/0011-dormant-ci.md +25 -1
  9. package/docs/adr/0014-six-phase-consolidation.md +134 -0
  10. package/docs/adr/README.md +2 -1
  11. package/docs/architecture.md +37 -38
  12. package/docs/best-practices.md +1 -1
  13. package/docs/ecosystem.md +37 -26
  14. package/docs/engineering.md +1 -1
  15. package/docs/facts.json +45 -0
  16. package/docs/features.md +54 -53
  17. package/docs/performance.md +5 -5
  18. package/docs/recovery-guide.md +9 -9
  19. package/docs/server-readiness.md +188 -0
  20. package/docs/token-budget-history.md +3 -1
  21. package/index.js +18 -3
  22. package/install/_codex-agents.mjs +1 -1
  23. package/install/_common.mjs +42 -17
  24. package/install/_dev-only-files.mjs +8 -0
  25. package/install/_unattended-profile.mjs +113 -0
  26. package/install/index.mjs +48 -0
  27. package/install/templates/claude-hooks.json +1 -1
  28. package/install/templates/codex-instructions.md +1 -1
  29. package/install/templates/copilot-instructions.md +28 -28
  30. package/manifest.json +1065 -0
  31. package/package.json +6 -3
  32. package/pipeline/agents/dev-critic.md +3 -3
  33. package/pipeline/commands/figma-to-swiftui.md +1 -1
  34. package/pipeline/commands/multi-agent/SKILL.md +8 -8
  35. package/pipeline/commands/multi-agent/analysis/SKILL.md +9 -9
  36. package/pipeline/commands/multi-agent/autopilot/SKILL.md +7 -7
  37. package/pipeline/commands/multi-agent/channels/SKILL.md +15 -15
  38. package/pipeline/commands/multi-agent/diff-explain/SKILL.md +6 -6
  39. package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
  40. package/pipeline/commands/multi-agent/graph/SKILL.md +1 -1
  41. package/pipeline/commands/multi-agent/help/SKILL.md +62 -62
  42. package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
  43. package/pipeline/commands/multi-agent/local/SKILL.md +11 -11
  44. package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +13 -13
  45. package/pipeline/commands/multi-agent/log/SKILL.md +2 -2
  46. package/pipeline/commands/multi-agent/manual-test/SKILL.md +9 -9
  47. package/pipeline/commands/multi-agent/model/SKILL.md +69 -0
  48. package/pipeline/commands/multi-agent/refactor/SKILL.md +3 -3
  49. package/pipeline/commands/multi-agent/resume/SKILL.md +4 -4
  50. package/pipeline/commands/multi-agent/resume-local/SKILL.md +19 -17
  51. package/pipeline/commands/multi-agent/review/SKILL.md +1 -1
  52. package/pipeline/commands/multi-agent/route-off/SKILL.md +36 -0
  53. package/pipeline/commands/multi-agent/route-on/SKILL.md +74 -0
  54. package/pipeline/commands/multi-agent/route-status/SKILL.md +56 -0
  55. package/pipeline/commands/multi-agent/setup/SKILL.md +2 -2
  56. package/pipeline/commands/multi-agent/status/SKILL.md +54 -23
  57. package/pipeline/commands/multi-agent/steer/SKILL.md +2 -2
  58. package/pipeline/commands/multi-agent/sync/SKILL.md +12 -13
  59. package/pipeline/commands/multi-agent/test/SKILL.md +1 -1
  60. package/pipeline/lib/_jira-auth.sh +8 -0
  61. package/pipeline/lib/analysis-jira-write.sh +32 -0
  62. package/pipeline/lib/ask-choice.sh +13 -2
  63. package/pipeline/lib/autopilot-state.sh +8 -0
  64. package/pipeline/lib/credential-inventory.sh +1 -1
  65. package/pipeline/lib/fatal.mjs +129 -0
  66. package/pipeline/lib/fetch-fortify.sh +1 -1
  67. package/pipeline/lib/figma-mcp-refresh.sh +18 -0
  68. package/pipeline/lib/figma-screenshot.sh +18 -0
  69. package/pipeline/lib/invoked-directly.mjs +43 -0
  70. package/pipeline/lib/jira-publish.sh +42 -0
  71. package/pipeline/lib/md2confluence-v3.py +47 -0
  72. package/pipeline/lib/model-rung.sh +142 -0
  73. package/pipeline/lib/outbound-gate.mjs +175 -0
  74. package/pipeline/lib/phase-schema.mjs +88 -0
  75. package/pipeline/lib/plan-todos.sh +32 -11
  76. package/pipeline/lib/post-pr-review.sh +77 -8
  77. package/pipeline/lib/repo-hygiene.sh +8 -3
  78. package/pipeline/lib/require-jq.sh +40 -0
  79. package/pipeline/lib/route-state.sh +161 -0
  80. package/pipeline/lib/run-paths.sh +335 -0
  81. package/pipeline/multi-agent-refs/_account-picker.md +1 -1
  82. package/pipeline/multi-agent-refs/_dev-context.md +1 -1
  83. package/pipeline/multi-agent-refs/_input-parser.md +1 -1
  84. package/pipeline/multi-agent-refs/analysis/evidence.md +0 -9
  85. package/pipeline/multi-agent-refs/analysis/intake.md +1 -1
  86. package/pipeline/multi-agent-refs/analysis/locked.md +21 -22
  87. package/pipeline/multi-agent-refs/analysis/render.md +1 -1
  88. package/pipeline/multi-agent-refs/analysis/synthesis.md +12 -6
  89. package/pipeline/multi-agent-refs/android-guide.md +1 -1
  90. package/pipeline/multi-agent-refs/audit-guide.md +13 -13
  91. package/pipeline/multi-agent-refs/channels/issue-comment.md +2 -2
  92. package/pipeline/multi-agent-refs/channels/jira.md +3 -3
  93. package/pipeline/multi-agent-refs/channels/pr.md +4 -4
  94. package/pipeline/multi-agent-refs/channels/wiki.md +1 -1
  95. package/pipeline/multi-agent-refs/component-dispatch.md +3 -3
  96. package/pipeline/multi-agent-refs/cross-cli-contract.md +31 -6
  97. package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +74 -4
  98. package/pipeline/multi-agent-refs/features/code-graph.md +5 -5
  99. package/pipeline/multi-agent-refs/features/cost-analysis.md +93 -0
  100. package/pipeline/multi-agent-refs/features/design-conformance.md +1 -1
  101. package/pipeline/multi-agent-refs/features/dev-critic.md +3 -3
  102. package/pipeline/multi-agent-refs/features/doctor.md +47 -2
  103. package/pipeline/multi-agent-refs/features/external-context-injection.md +3 -3
  104. package/pipeline/multi-agent-refs/features/maturity-followup.md +3 -3
  105. package/pipeline/multi-agent-refs/features/model-fallback.md +5 -5
  106. package/pipeline/multi-agent-refs/features/plan-todos.md +1 -1
  107. package/pipeline/multi-agent-refs/features/repo-map.md +1 -1
  108. package/pipeline/multi-agent-refs/features/review-delta.md +3 -3
  109. package/pipeline/multi-agent-refs/features/review-multi-repo.md +1 -1
  110. package/pipeline/multi-agent-refs/features/scope-check.md +4 -4
  111. package/pipeline/multi-agent-refs/features/skill-conformance.md +2 -2
  112. package/pipeline/multi-agent-refs/features/stack-skill-routing.md +1 -1
  113. package/pipeline/multi-agent-refs/features/verify-by-test.md +4 -4
  114. package/pipeline/multi-agent-refs/features/verify.md +83 -0
  115. package/pipeline/multi-agent-refs/features/visual-evidence.md +19 -19
  116. package/pipeline/multi-agent-refs/features/worktree-finalize.md +6 -6
  117. package/pipeline/multi-agent-refs/issue-jira-triad.md +10 -10
  118. package/pipeline/multi-agent-refs/knowledge.md +11 -11
  119. package/pipeline/multi-agent-refs/multi-repo-integration-build.md +13 -13
  120. package/pipeline/multi-agent-refs/payload-contracts.md +8 -8
  121. package/pipeline/multi-agent-refs/phases/log-format.md +10 -10
  122. package/pipeline/multi-agent-refs/phases/modes.md +30 -30
  123. package/pipeline/multi-agent-refs/phases/operations.md +21 -10
  124. package/pipeline/multi-agent-refs/phases/phase-0-init.md +25 -25
  125. package/pipeline/multi-agent-refs/phases/phase-1-plan.md +599 -0
  126. package/pipeline/multi-agent-refs/phases/{phase-3-dev.md → phase-2-dev.md} +129 -49
  127. package/pipeline/multi-agent-refs/phases/{phase-4-review.md → phase-3-review.md} +225 -107
  128. package/pipeline/multi-agent-refs/phases/{phase-6-commit.md → phase-4-commit.md} +23 -23
  129. package/pipeline/multi-agent-refs/phases/{phase-7-report.md → phase-5-report.md} +29 -29
  130. package/pipeline/multi-agent-refs/phases.md +44 -48
  131. package/pipeline/multi-agent-refs/picker-contract.md +1 -1
  132. package/pipeline/multi-agent-refs/progress-contract.md +6 -6
  133. package/pipeline/multi-agent-refs/readiness-review.md +1 -1
  134. package/pipeline/multi-agent-refs/rules.md +7 -7
  135. package/pipeline/multi-agent-refs/swiftui-guide.md +2 -2
  136. package/pipeline/multi-agent-refs/tracker-contract.md +31 -32
  137. package/pipeline/multi-agent-refs/unattended-contract.md +129 -0
  138. package/pipeline/multi-agent-refs/wiki-capture.md +14 -14
  139. package/pipeline/preferences-template.json +9 -1
  140. package/pipeline/rules/outside-the-pipeline.md +1 -1
  141. package/pipeline/schemas/agent-state.schema.json +50 -50
  142. package/pipeline/schemas/analysis-output.schema.json +2 -2
  143. package/pipeline/schemas/autopilot-config.schema.json +1 -1
  144. package/pipeline/schemas/code-graph.schema.json +1 -1
  145. package/pipeline/schemas/criteria-manifest.schema.json +1 -1
  146. package/pipeline/schemas/dev-critic-output.schema.json +1 -1
  147. package/pipeline/schemas/diff-risk.schema.json +1 -1
  148. package/pipeline/schemas/migrations/prefs-2.4.0-to-2.5.0.mjs +2 -2
  149. package/pipeline/schemas/migrations/prefs-2.6.0-to-2.7.0.mjs +31 -0
  150. package/pipeline/schemas/migrations/state-2.1.0-to-2.2.0.mjs +129 -0
  151. package/pipeline/schemas/phases.json +105 -0
  152. package/pipeline/schemas/plan-todos.schema.json +5 -5
  153. package/pipeline/schemas/planning-output.schema.json +1 -1
  154. package/pipeline/schemas/prefs.schema.json +100 -56
  155. package/pipeline/schemas/reviewer-output.schema.json +3 -3
  156. package/pipeline/schemas/route-config.schema.json +74 -0
  157. package/pipeline/schemas/scope-check.schema.json +1 -1
  158. package/pipeline/schemas/test-gap.schema.json +1 -1
  159. package/pipeline/schemas/token-budget.json +12 -18
  160. package/pipeline/schemas/triage-output.schema.json +6 -6
  161. package/pipeline/scripts/README.md +3 -3
  162. package/pipeline/scripts/_code-graph.mjs +2 -2
  163. package/pipeline/scripts/_run-paths.mjs +372 -0
  164. package/pipeline/scripts/_smoke-root.sh +1 -1
  165. package/pipeline/scripts/aggregate-metrics.mjs +65 -65
  166. package/pipeline/scripts/autopilot-arming.mjs +2 -1
  167. package/pipeline/scripts/autopilot-intake.mjs +2 -1
  168. package/pipeline/scripts/autopilot-runner.mjs +206 -2
  169. package/pipeline/scripts/build-references.mjs +2 -1
  170. package/pipeline/scripts/build-stack-plugins.mjs +10 -2
  171. package/pipeline/scripts/capture-evidence.sh +7 -2
  172. package/pipeline/scripts/capture-flush.sh +8 -8
  173. package/pipeline/scripts/capture-resume.sh +3 -3
  174. package/pipeline/scripts/classify-plan-safety.mjs +3 -2
  175. package/pipeline/scripts/cost-analyze.mjs +600 -0
  176. package/pipeline/scripts/cost-budget-check.mjs +4 -12
  177. package/pipeline/scripts/council-view.mjs +2 -1
  178. package/pipeline/scripts/crush-json.mjs +2 -1
  179. package/pipeline/scripts/diff-explain.mjs +7 -10
  180. package/pipeline/scripts/diff-risk-score.mjs +2 -1
  181. package/pipeline/scripts/doctor.mjs +140 -6
  182. package/pipeline/scripts/evidence-gate.mjs +9 -3
  183. package/pipeline/scripts/feedback-send.mjs +12 -2
  184. package/pipeline/scripts/gc-abandoned.sh +32 -16
  185. package/pipeline/scripts/gc-tmp.sh +1 -1
  186. package/pipeline/scripts/gc-worktrees.sh +12 -5
  187. package/pipeline/scripts/gen-facts.mjs +175 -0
  188. package/pipeline/scripts/gen-mode-dispatch.mjs +32 -37
  189. package/pipeline/scripts/gen-ref-toc.mjs +1 -1
  190. package/pipeline/scripts/github-ssh-setup.sh +64 -7
  191. package/pipeline/scripts/graph-mermaid.mjs +4 -2
  192. package/pipeline/scripts/graph-report.mjs +1 -1
  193. package/pipeline/scripts/jira-attach.sh +1 -1
  194. package/pipeline/scripts/keychain-save.sh +101 -30
  195. package/pipeline/scripts/learn-from-transcripts.mjs +3 -2
  196. package/pipeline/scripts/learning-curve.mjs +36 -31
  197. package/pipeline/scripts/log-metric.sh +17 -4
  198. package/pipeline/scripts/make-manifest.mjs +199 -0
  199. package/pipeline/scripts/memory-save.sh +1 -1
  200. package/pipeline/scripts/migrate-prefs.mjs +24 -6
  201. package/pipeline/scripts/migrate-state.mjs +94 -4
  202. package/pipeline/scripts/phase-banner.sh +26 -22
  203. package/pipeline/scripts/phase-tracker.sh +48 -10
  204. package/pipeline/scripts/plan-coverage-gate.mjs +8 -4
  205. package/pipeline/scripts/pre-commit-check.sh +7 -0
  206. package/pipeline/scripts/pre-push-check.sh +7 -0
  207. package/pipeline/scripts/purge.sh +23 -6
  208. package/pipeline/scripts/render-agent-log-cost.sh +10 -3
  209. package/pipeline/scripts/render-cost-summary.sh +9 -2
  210. package/pipeline/scripts/render-work-summary.sh +14 -7
  211. package/pipeline/scripts/review-file-filter.mjs +5 -3
  212. package/pipeline/scripts/review-scope.mjs +2 -1
  213. package/pipeline/scripts/routine-registry.mjs +2 -1
  214. package/pipeline/scripts/run-aggregator.mjs +26 -20
  215. package/pipeline/scripts/run-metrics.mjs +4 -2
  216. package/pipeline/scripts/runs-index.mjs +353 -0
  217. package/pipeline/scripts/scorecard-snapshot.mjs +178 -0
  218. package/pipeline/scripts/search-logs.sh +18 -0
  219. package/pipeline/scripts/smoke-cross-cli-behavior.sh +6 -6
  220. package/pipeline/scripts/smoke-schema-validation.sh +26 -7
  221. package/pipeline/scripts/test-gap-scan.mjs +2 -1
  222. package/pipeline/scripts/test-integrity-gate.mjs +2 -1
  223. package/pipeline/scripts/token-budget-report.mjs +13 -2
  224. package/pipeline/scripts/triage-memory.mjs +2 -2
  225. package/pipeline/scripts/update-issue-progress.sh +56 -7
  226. package/pipeline/scripts/usage-report.mjs +12 -1
  227. package/pipeline/scripts/validate-analysis-doc.mjs +75 -18
  228. package/pipeline/scripts/validate-code-graph.mjs +6 -3
  229. package/pipeline/scripts/validate-complaint-doc.mjs +2 -1
  230. package/pipeline/scripts/validate-diff-risk.mjs +6 -3
  231. package/pipeline/scripts/validate-planning.mjs +1 -1
  232. package/pipeline/scripts/validate-reviewer.mjs +1 -1
  233. package/pipeline/scripts/validate-state.mjs +45 -5
  234. package/pipeline/scripts/validate-test-gap.mjs +6 -3
  235. package/pipeline/scripts/validate-triage.mjs +6 -4
  236. package/pipeline/scripts/verify-citations.mjs +4 -2
  237. package/pipeline/scripts/verify.mjs +327 -0
  238. package/pipeline/scripts/worktree-finalize.sh +18 -9
  239. package/pipeline/scripts/write-state.mjs +154 -15
  240. package/pipeline/skills/.skill-manifest.json +37 -21
  241. package/pipeline/skills/.skills-index.json +104 -5
  242. package/pipeline/skills/shared/README.md +15 -6
  243. package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +2 -2
  244. package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +2 -2
  245. package/pipeline/skills/shared/core/multi-agent/SKILL.md +69 -71
  246. package/pipeline/skills/shared/core/multi-agent-autopilot/SKILL.md +3 -3
  247. package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +14 -14
  248. package/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md +5 -5
  249. package/pipeline/skills/shared/core/multi-agent-graph/SKILL.md +1 -1
  250. package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +25 -23
  251. package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
  252. package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +2 -2
  253. package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +8 -8
  254. package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +6 -6
  255. package/pipeline/skills/shared/core/multi-agent-model/SKILL.md +71 -0
  256. package/pipeline/skills/shared/core/multi-agent-refactor/SKILL.md +3 -3
  257. package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +1 -1
  258. package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +7 -7
  259. package/pipeline/skills/shared/core/multi-agent-route-off/SKILL.md +39 -0
  260. package/pipeline/skills/shared/core/multi-agent-route-on/SKILL.md +76 -0
  261. package/pipeline/skills/shared/core/multi-agent-route-status/SKILL.md +59 -0
  262. package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +1 -1
  263. package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +35 -11
  264. package/pipeline/skills/shared/core/multi-agent-steer/SKILL.md +2 -2
  265. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +6 -5
  266. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/package_app.sh +4 -1
  267. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/setup_dev_signing.sh +4 -1
  268. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/sign-and-notarize.sh +2 -1
  269. package/pipeline/skills/skills-index.md +13 -4
  270. package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +0 -263
  271. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +0 -344
  272. package/pipeline/multi-agent-refs/phases/phase-5-test.md +0 -182
@@ -0,0 +1,353 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * @file runs-index.mjs - one deterministic answer to "what runs exist and
4
+ * where is each one".
5
+ *
6
+ * `/multi-agent:status` used to answer this by telling the model to go and find
7
+ * the files itself: scan three hard-coded `.worktrees/` paths, then `find` the
8
+ * log tree at one depth, then merge. Three problems with that. It cannot be
9
+ * called by anything that is not a model, the depth was wrong for half the
10
+ * layouts, and two invocations could disagree because nothing pinned the
11
+ * traversal. A UI, a gate, or a second phase reading the same question got a
12
+ * different answer than the terminal did.
13
+ *
14
+ * This is the producer. `--json` and the human table are rendered from the SAME
15
+ * in-memory records, so a dashboard and a terminal cannot disagree - the rule
16
+ * autopilot-status.sh already follows for the autopilot half of the picture.
17
+ *
18
+ * Grouping follows the contract in commands/multi-agent/status/SKILL.md 3b,
19
+ * including its final clause: a run with no `status` is placed in no group at
20
+ * all. Unknown is not a finding, and calling it dead is the same false claim in
21
+ * the other direction.
22
+ *
23
+ * Read-only. Never writes, never migrates, never deletes.
24
+ *
25
+ * Usage:
26
+ * node runs-index.mjs # human table, grouped
27
+ * node runs-index.mjs --json # the same records as JSON
28
+ * node runs-index.mjs --group waiting # one group only
29
+ * node runs-index.mjs --task-id <id> # one run
30
+ *
31
+ * Exit codes:
32
+ * 0 - answered (including "no runs")
33
+ * 2 - usage error
34
+ *
35
+ * @module pipeline/scripts/runs-index
36
+ */
37
+
38
+ import { existsSync, readFileSync } from "node:fs";
39
+ import { join } from "node:path";
40
+ import { costUsd } from "./_cost.mjs";
41
+ import { listRuns, logsRoot, resolveRunDir, taskIdVariants } from "./_run-paths.mjs";
42
+ import { runMain } from "../lib/fatal.mjs";
43
+ import { invokedDirectly } from "../lib/invoked-directly.mjs";
44
+
45
+ /**
46
+ * The phase from which a run counts as "waiting on you", read from the phase
47
+ * contract rather than written here. This threshold moved once already (it was
48
+ * 6 under the eight-phase contract, it is 4 under six) and nothing connected it
49
+ * to the renumbering, so it would have silently regrouped every run.
50
+ */
51
+ const PHASE_WAITING_FROM = JSON.parse(
52
+ readFileSync(new URL("../schemas/phases.json", import.meta.url), "utf8"),
53
+ ).thresholds.waitingFromPhase;
54
+
55
+ const GROUPS = {
56
+ waiting: "Waiting on you",
57
+ stopped: "Stopped mid-development",
58
+ question: "Left at a question",
59
+ unknown: "Status not recorded",
60
+ };
61
+
62
+ function parseArgs(argv) {
63
+ const flags = { json: false, group: null, taskId: null };
64
+ for (let i = 0; i < argv.length; i++) {
65
+ const a = argv[i];
66
+ if (a === "--json") flags.json = true;
67
+ else if (a === "--group") flags.group = argv[++i];
68
+ else if (a.startsWith("--group=")) flags.group = a.slice(8);
69
+ else if (a === "--task-id") flags.taskId = argv[++i];
70
+ else if (a.startsWith("--task-id=")) flags.taskId = a.slice(10);
71
+ else if (a === "-h" || a === "--help") flags.help = true;
72
+ else {
73
+ process.stderr.write(`runs-index: unknown argument ${a}\n`);
74
+ process.exit(2);
75
+ }
76
+ }
77
+ if (flags.group && !Object.hasOwn(GROUPS, flags.group)) {
78
+ process.stderr.write(
79
+ `runs-index: unknown group ${flags.group} (want ${Object.keys(GROUPS).join(", ")})\n`,
80
+ );
81
+ process.exit(2);
82
+ }
83
+ return flags;
84
+ }
85
+
86
+ function readJson(path) {
87
+ if (!path || !existsSync(path)) return null;
88
+ try {
89
+ return JSON.parse(readFileSync(path, "utf-8"));
90
+ } catch {
91
+ // A truncated state file is a fact about the run, not a reason to refuse
92
+ // the whole index. It surfaces as `stateReadable: false`.
93
+ return null;
94
+ }
95
+ }
96
+
97
+ function firstFile(dir, name) {
98
+ for (const base of [dir, join(dir, "artifacts")]) {
99
+ const p = join(base, name);
100
+ if (existsSync(p)) return p;
101
+ }
102
+ return null;
103
+ }
104
+
105
+ let COST_TABLE = null;
106
+ function rateFor(model) {
107
+ if (!COST_TABLE) {
108
+ COST_TABLE = readJson(new URL("./cost-table.json", import.meta.url).pathname) ?? { prices: {} };
109
+ }
110
+ if (!model) return null;
111
+ const prices = COST_TABLE.prices ?? {};
112
+ // phase-tracker.sh stores the short tier name ("opus", "fable"), which is the
113
+ // table's own key - the same lookup run-aggregator.mjs does. A caller that
114
+ // stored the full model id instead is matched on the table's `modelId`
115
+ // rather than guessed at by prefix: "gpt-5.6" is a prefix of "gpt-5.6-terra"
116
+ // and those are two different prices.
117
+ if (prices[model]) return prices[model];
118
+ for (const rate of Object.values(prices)) {
119
+ if (rate?.modelId === model) return rate;
120
+ }
121
+ return null;
122
+ }
123
+
124
+ /**
125
+ * Phase rows plus token/cost totals, from the tracker document.
126
+ *
127
+ * @param {object|null} tracker
128
+ */
129
+ function summarisePhases(tracker) {
130
+ const phases = Array.isArray(tracker?.phases) ? tracker.phases : [];
131
+ let tokensIn = 0;
132
+ let tokensOut = 0;
133
+ let tokensCached = 0;
134
+ let usd = 0;
135
+ const rows = phases.map((p) => {
136
+ // phase-tracker.sh writes flat `tokens_in` / `tokens_out` / `tokens_cached`
137
+ // on each phase, not a nested `tokens` object.
138
+ const tin = Number(p.tokens_in ?? 0) || 0;
139
+ const tout = Number(p.tokens_out ?? 0) || 0;
140
+ const tcached = Number(p.tokens_cached ?? 0) || 0;
141
+ tokensIn += tin;
142
+ tokensOut += tout;
143
+ tokensCached += tcached;
144
+ const c = costUsd(rateFor(p.model), tin, tout, tcached);
145
+ if (typeof c === "number") usd += c;
146
+ return {
147
+ id: String(p.id ?? ""),
148
+ name: p.name ?? "",
149
+ status: p.status ?? "pending",
150
+ model: p.model ?? null,
151
+ startedAt: p.started_at ?? null,
152
+ completedAt: p.completed_at ?? null,
153
+ now: p.now ?? null,
154
+ subs: Array.isArray(p.subs) ? p.subs.length : 0,
155
+ };
156
+ });
157
+ return {
158
+ phases: rows,
159
+ startedAt: tracker?.started_at ?? null,
160
+ tokens: { in: tokensIn, out: tokensOut, cached: tokensCached },
161
+ estUsd: Number(usd.toFixed(4)),
162
+ };
163
+ }
164
+
165
+ /**
166
+ * The group a run belongs to, per status/SKILL.md 3b.
167
+ *
168
+ * @param {object|null} state
169
+ * @returns {"waiting"|"stopped"|"question"|"unknown"}
170
+ */
171
+ function groupOf(state) {
172
+ const status = state?.status;
173
+ if (!status) return "unknown";
174
+ const phase = Number(state?.currentPhase);
175
+ const prUrl = state?.pr?.url ?? state?.prUrl ?? null;
176
+ if (status === "awaiting_input" || status === "awaiting-user-test-main-checkout")
177
+ return "waiting";
178
+ if (prUrl) return "waiting";
179
+ if (Number.isFinite(phase) && phase >= PHASE_WAITING_FROM) return "waiting";
180
+ if (Number.isFinite(phase) && phase === 0) return "question";
181
+ return "stopped";
182
+ }
183
+
184
+ /**
185
+ * Every run, enriched, in one stable order.
186
+ *
187
+ * @returns {object[]}
188
+ */
189
+ export function buildIndex() {
190
+ return listRuns().map((run) => {
191
+ const statePath = firstFile(run.dir, "agent-state.json");
192
+ const trackerPath = firstFile(run.dir, "tracker-state.json");
193
+ const state = readJson(statePath);
194
+ const tracker = readJson(trackerPath);
195
+ const phases = summarisePhases(tracker);
196
+ const prUrl = state?.pr?.url ?? state?.prUrl ?? null;
197
+ return {
198
+ taskId: run.taskId,
199
+ project: run.project ?? run.projectHint ?? null,
200
+ dir: run.dir,
201
+ layout: run.layout,
202
+ duplicateOf: run.duplicateOf,
203
+ salvaged: Boolean(statePath && statePath.includes(`${run.dir}/artifacts/`)),
204
+ // What KIND of record this is, before asking whether it is healthy.
205
+ //
206
+ // 74 of the 103 runs on this machine have a tracker file and no agent
207
+ // state, and the first version of this index called every one of them
208
+ // `stateReadable: false` - so a panel built on it announced "74 runs
209
+ // could not be read" about records that are not damaged and never had
210
+ // agent state to begin with. `/multi-agent:analysis` says so in its own
211
+ // description ("no worktree, no commits, no dev chain"), and design-check
212
+ // is the same shape.
213
+ //
214
+ // Derived from what is on disk, not from the task id: `ANALYSIS-` and
215
+ // `DC-` are naming conventions, and a convention is not a contract.
216
+ kind: statePath ? "pipeline" : trackerPath ? "tracker-only" : "empty",
217
+ // Now the health question, and it only applies where state was expected.
218
+ // A tracker-only record is `true` because there is nothing it failed to
219
+ // read; "could not read" and "there is none" are different answers and
220
+ // only the first one asks anyone to do something.
221
+ stateReadable: statePath ? Boolean(state) : true,
222
+ status: state?.status ?? null,
223
+ currentPhase: Number.isFinite(Number(state?.currentPhase))
224
+ ? Number(state.currentPhase)
225
+ : null,
226
+ branch: state?.branch ?? state?.branchName ?? null,
227
+ baseBranch: state?.baseBranch ?? null,
228
+ startedAt: state?.startedAt ?? phases.startedAt,
229
+ worktreePath: state?.worktreePath ?? null,
230
+ prUrl,
231
+ autopilot: Boolean(state?.autopilot),
232
+ schemaVersion: state?.schemaVersion ?? null,
233
+ rev: Number.isInteger(state?.rev) ? state.rev : null,
234
+ group: groupOf(state),
235
+ phases: phases.phases,
236
+ tokens: phases.tokens,
237
+ estUsd: phases.estUsd,
238
+ // How many tracker writes went through without the lock. Non-zero means
239
+ // two writers were in the critical section and one of their token
240
+ // deltas was dropped, so the numbers on this run are LOW. Reporting the
241
+ // count rather than the loss, because the size of what was dropped is
242
+ // exactly what nobody measured.
243
+ unlockedWrites: Number.isInteger(tracker?.unlockedWrites) ? tracker.unlockedWrites : 0,
244
+ };
245
+ });
246
+ }
247
+
248
+ function fmtDuration(startedAt) {
249
+ if (!startedAt) return "-";
250
+ const t = Date.parse(startedAt);
251
+ if (!Number.isFinite(t)) return "-";
252
+ const mins = Math.max(0, Math.round((Date.now() - t) / 60000));
253
+ if (mins < 60) return `${mins}m`;
254
+ const h = Math.floor(mins / 60);
255
+ if (h < 48) return `${h}h`;
256
+ return `${Math.floor(h / 24)}d`;
257
+ }
258
+
259
+ function renderHuman(records) {
260
+ const out = [];
261
+ const total = records.length;
262
+ out.push(`multi-agent runs - ${total} under ${logsRoot()}`);
263
+
264
+ const dupes = records.filter((r) => r.duplicateOf);
265
+ if (dupes.length) {
266
+ out.push(
267
+ ` ${dupes.length} run(s) also have a copy in the other directory layout; ` +
268
+ `the newer one is shown. migrate-state.mjs --all reports them.`,
269
+ );
270
+ }
271
+
272
+ for (const key of Object.keys(GROUPS)) {
273
+ const rows = records.filter((r) => r.group === key);
274
+ if (!rows.length) continue;
275
+ out.push("");
276
+ out.push(`${GROUPS[key]} (${rows.length})`);
277
+ out.push(
278
+ " ID Phase Status Branch Age ~USD",
279
+ );
280
+ for (const r of rows) {
281
+ const phase = r.currentPhase === null ? " -" : `${r.currentPhase}/7`;
282
+ out.push(
283
+ " " +
284
+ [
285
+ r.taskId.padEnd(30).slice(0, 30),
286
+ String(phase).padStart(5),
287
+ (r.status ?? "-").padEnd(17).slice(0, 17),
288
+ (r.branch ?? "-").padEnd(24).slice(0, 24),
289
+ fmtDuration(r.startedAt).padStart(5),
290
+ (r.estUsd ? r.estUsd.toFixed(2) : "-").padStart(6),
291
+ ].join(" "),
292
+ );
293
+ }
294
+ }
295
+
296
+ const hints = {
297
+ waiting: "resume #N - the work landed, it needs your answer",
298
+ stopped: "resume #N or kill #N",
299
+ question: "garbage-collect --abandoned - nothing was built",
300
+ };
301
+ const present = Object.keys(hints).filter((k) => records.some((r) => r.group === k));
302
+ if (present.length) {
303
+ out.push("");
304
+ for (const k of present) out.push(` ${GROUPS[k]}: ${hints[k]}`);
305
+ }
306
+ return out.join("\n");
307
+ }
308
+
309
+ function main() {
310
+ const flags = parseArgs(process.argv.slice(2));
311
+ if (flags.help) {
312
+ process.stdout.write(
313
+ "Usage: runs-index.mjs [--json] [--group waiting|stopped|question|unknown] [--task-id <id>]\n",
314
+ );
315
+ return 0;
316
+ }
317
+
318
+ let records = buildIndex();
319
+
320
+ if (flags.taskId) {
321
+ const wanted = new Set(taskIdVariants(flags.taskId));
322
+ records = records.filter((r) => wanted.has(r.taskId));
323
+ if (!records.length) {
324
+ const dir = resolveRunDir(flags.taskId);
325
+ process.stderr.write(
326
+ `runs-index: no run for ${flags.taskId}${dir ? ` (directory ${dir} has no markers)` : ""}\n`,
327
+ );
328
+ return 0;
329
+ }
330
+ }
331
+ if (flags.group) records = records.filter((r) => r.group === flags.group);
332
+
333
+ if (flags.json) {
334
+ process.stdout.write(
335
+ JSON.stringify({ logsRoot: logsRoot(), count: records.length, runs: records }, null, 2) +
336
+ "\n",
337
+ );
338
+ } else {
339
+ process.stdout.write(renderHuman(records) + "\n");
340
+ }
341
+ return 0;
342
+ }
343
+
344
+ if (invokedDirectly(import.meta.url)) {
345
+ // process.exitCode, not process.exit(): stdout to a PIPE is asynchronous and
346
+ // process.exit() throws away whatever has not drained. 103 runs render to
347
+ // ~200KB of JSON, so a caller doing `runs-index.mjs --json | jq` received the
348
+ // first 64KB and a parse error. A terminal hid it, because stdout to a TTY is
349
+ // synchronous - and a terminal is where this was tested.
350
+ runMain("runs-index", () => {
351
+ process.exitCode = main();
352
+ });
353
+ }
@@ -0,0 +1,178 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * scorecard-snapshot.mjs - keep what the scorecard said, and say what moved.
4
+ *
5
+ * The scorecard answers "does every mechanical claim hold right now". It
6
+ * cannot answer "is this better or worse than last week", and that second
7
+ * question is the one that catches slow rot: a metric that has been failing
8
+ * for a month reads identically to one that broke an hour ago.
9
+ *
10
+ * WHAT THIS DELIBERATELY DOES NOT DO: collapse the result into a 0-100 score.
11
+ * ruflo's scorecard does, and the discipline worth taking from it is
12
+ * "measurable and comparable over time", not the number. The scorecard already
13
+ * reports twelve measured metrics AND four it refuses to measure - a single
14
+ * figure would hide both halves, and an unmeasured category would silently
15
+ * count as zero or as full marks depending on an arithmetic choice nobody
16
+ * would ever read. A diff keeps every metric answering for itself.
17
+ *
18
+ * Snapshots live in ~/.claude/state/scorecard/<iso>.json and are never pruned
19
+ * by this script: they are small, and a history that deletes itself cannot
20
+ * answer the question it was kept for.
21
+ *
22
+ * Usage:
23
+ * scorecard-snapshot.mjs --save run the scorecard, store a snapshot
24
+ * scorecard-snapshot.mjs --diff latest vs the one before it
25
+ * scorecard-snapshot.mjs --diff a.json b.json two snapshots by path
26
+ * scorecard-snapshot.mjs --list what is stored
27
+ *
28
+ * Exit codes:
29
+ * 0 - nothing regressed (or a save succeeded)
30
+ * 1 - at least one metric that used to pass now fails
31
+ * 2 - usage error, or not enough snapshots to compare
32
+ */
33
+
34
+ import { execFileSync } from "node:child_process";
35
+ import { existsSync, mkdirSync, readdirSync, readFileSync, writeFileSync } from "node:fs";
36
+ import { homedir } from "node:os";
37
+ import { dirname, join } from "node:path";
38
+ import { fileURLToPath } from "node:url";
39
+ import { runMain } from "../lib/fatal.mjs";
40
+
41
+ const ROOT = join(dirname(fileURLToPath(import.meta.url)), "..", "..");
42
+ const STORE = process.env.SCORECARD_STORE || join(homedir(), ".claude", "state", "scorecard");
43
+
44
+ /** Metric identity. Category plus metric name, because neither is unique alone. */
45
+ const keyOf = (r) => `${r.category} :: ${r.metric}`;
46
+
47
+ function runScorecard() {
48
+ // `--json` prints the report on stdout; a failing metric is a non-zero exit
49
+ // and is exactly the case worth snapshotting, so the status is captured
50
+ // rather than thrown.
51
+ try {
52
+ return JSON.parse(
53
+ execFileSync("node", ["pipeline/scripts/scorecard.mjs", "--json"], {
54
+ cwd: ROOT,
55
+ encoding: "utf-8",
56
+ maxBuffer: 32 * 1024 * 1024,
57
+ }),
58
+ );
59
+ } catch (err) {
60
+ const out = err?.stdout;
61
+ if (typeof out === "string" && out.trim().startsWith("{")) return JSON.parse(out);
62
+ throw new Error(`scorecard did not produce JSON: ${err?.message ?? err}`, { cause: err });
63
+ }
64
+ }
65
+
66
+ function snapshots() {
67
+ if (!existsSync(STORE)) return [];
68
+ return (
69
+ readdirSync(STORE)
70
+ .filter((f) => f.endsWith(".json"))
71
+ // ISO-8601 sorts lexically in time order, which is the whole reason the
72
+ // file is named after the timestamp rather than a counter.
73
+ .sort()
74
+ .map((f) => join(STORE, f))
75
+ );
76
+ }
77
+
78
+ function save() {
79
+ const report = runScorecard();
80
+ mkdirSync(STORE, { recursive: true });
81
+ const at = new Date().toISOString().replace(/[:.]/g, "-");
82
+ const path = join(STORE, `${at}.json`);
83
+ writeFileSync(path, JSON.stringify({ at: new Date().toISOString(), ...report }, null, 2) + "\n");
84
+ process.stdout.write(`scorecard snapshot: ${path}\n`);
85
+ process.stdout.write(` ${report.passed} passed, ${report.failed} failed\n`);
86
+ return 0;
87
+ }
88
+
89
+ function diff(aPath, bPath) {
90
+ const a = JSON.parse(readFileSync(aPath, "utf-8"));
91
+ const b = JSON.parse(readFileSync(bPath, "utf-8"));
92
+ // UNMEASURED rows carry no `ok` at all - they are the categories the
93
+ // scorecard refuses to score. Comparing verdicts across them would invent
94
+ // one, which is the thing that section exists to avoid.
95
+ const measured = (rs) => (rs || []).filter((r) => r.kind !== "UNMEASURED");
96
+ const ma = new Map(measured(a.results).map((r) => [keyOf(r), r]));
97
+ const mb = new Map(measured(b.results).map((r) => [keyOf(r), r]));
98
+
99
+ const regressed = [];
100
+ const fixed = [];
101
+ const added = [];
102
+ const removed = [];
103
+ const changed = [];
104
+
105
+ for (const [k, rb] of mb) {
106
+ const ra = ma.get(k);
107
+ if (!ra) {
108
+ added.push(rb);
109
+ continue;
110
+ }
111
+ if (ra.ok && !rb.ok) regressed.push({ k, ra, rb });
112
+ else if (!ra.ok && rb.ok) fixed.push({ k, ra, rb });
113
+ // A metric whose verdict held but whose DETAIL moved is the early warning:
114
+ // coverage sliding from 72 to 69 passes the floor and is still the thing
115
+ // that will fail next month.
116
+ else if (ra.detail !== rb.detail) changed.push({ k, ra, rb });
117
+ }
118
+ for (const [k, ra] of ma) if (!mb.has(k)) removed.push({ k, ra });
119
+
120
+ const out = [];
121
+ out.push(`scorecard diff - ${a.at ?? aPath} → ${b.at ?? bPath}`);
122
+ out.push(` ${a.passed}/${a.passed + a.failed} → ${b.passed}/${b.passed + b.failed} passing`);
123
+ const section = (title, rows, fmt) => {
124
+ if (!rows.length) return;
125
+ out.push("");
126
+ out.push(`${title} (${rows.length})`);
127
+ for (const r of rows) out.push(` ${fmt(r)}`);
128
+ };
129
+ section("REGRESSED", regressed, ({ k, rb }) => `${k}\n now: ${rb.detail ?? "(no detail)"}`);
130
+ section("FIXED", fixed, ({ k, rb }) => `${k}\n now: ${rb.detail ?? "(no detail)"}`);
131
+ section("NEW METRIC", added, (r) => `${keyOf(r)} ${r.ok ? "passing" : "FAILING"}`);
132
+ section("METRIC GONE", removed, ({ k }) => `${k} - no longer measured`);
133
+ section(
134
+ "SAME VERDICT, DIFFERENT NUMBERS",
135
+ changed,
136
+ ({ k, ra, rb }) =>
137
+ `${k}\n was: ${ra.detail ?? "(none)"}\n now: ${rb.detail ?? "(none)"}`,
138
+ );
139
+
140
+ if (!regressed.length && !fixed.length && !added.length && !removed.length && !changed.length) {
141
+ out.push("");
142
+ out.push(" nothing moved");
143
+ }
144
+ process.stdout.write(out.join("\n") + "\n");
145
+ return regressed.length ? 1 : 0;
146
+ }
147
+
148
+ function main() {
149
+ const args = process.argv.slice(2);
150
+ if (args.includes("--save")) return save();
151
+ if (args.includes("--list")) {
152
+ const s = snapshots();
153
+ process.stdout.write(s.length ? s.join("\n") + "\n" : `no snapshots under ${STORE}\n`);
154
+ return 0;
155
+ }
156
+ if (args.includes("--diff")) {
157
+ const paths = args.filter((a) => !a.startsWith("--"));
158
+ if (paths.length === 2) return diff(paths[0], paths[1]);
159
+ const s = snapshots();
160
+ if (s.length < 2) {
161
+ process.stderr.write(
162
+ `need two snapshots to compare, found ${s.length} under ${STORE}\n` +
163
+ ` take one with: node pipeline/scripts/scorecard-snapshot.mjs --save\n`,
164
+ );
165
+ return 2;
166
+ }
167
+ return diff(s[s.length - 2], s[s.length - 1]);
168
+ }
169
+ process.stderr.write("usage: scorecard-snapshot.mjs --save | --diff [a.json b.json] | --list\n");
170
+ return 2;
171
+ }
172
+
173
+ // `process.exitCode`, never `process.exit()`: stdout to a pipe is asynchronous
174
+ // and exit() discards whatever has not drained, which truncated two scripts in
175
+ // this repo at a buffer boundary and only ever when piped.
176
+ runMain("scorecard-snapshot", () => {
177
+ process.exitCode = main();
178
+ });
@@ -30,6 +30,24 @@
30
30
 
31
31
  set -uo pipefail
32
32
 
33
+ # jq is not optional on this path. Without the guard below a missing binary
34
+ # renders as EMPTY DATA and the work continues on it; see lib/require-jq.sh.
35
+ for _rq in "$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)/require-jq.sh" \
36
+ "$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")/../lib" 2>/dev/null && pwd)/require-jq.sh" \
37
+ "$HOME/.claude/lib/require-jq.sh" \
38
+ "$HOME/.copilot/lib/require-jq.sh" \
39
+ "$HOME/.codex/lib/require-jq.sh"; do
40
+ [ -f "$_rq" ] || continue
41
+ # shellcheck source=/dev/null
42
+ . "$_rq" && break
43
+ done
44
+ unset _rq
45
+ if ! command -v ma_require_jq >/dev/null 2>&1; then
46
+ # The helper itself is missing, which is an install problem, not a jq one.
47
+ ma_require_jq() { command -v jq >/dev/null 2>&1 || { echo "jq not found - cannot ${1:-continue}." >&2; return 1; }; }
48
+ fi
49
+ ma_require_jq "search the logs" || exit 3
50
+
33
51
  SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
34
52
 
35
53
  ROOT="$HOME/.claude/logs/multi-agent"
@@ -204,21 +204,21 @@ echo "→ reviewer-count contract (Claude=3, Copilot=3, Codex=3)"
204
204
  # a broken installation. _smoke-root.sh handles both layouts.
205
205
  # shellcheck source=pipeline/scripts/_smoke-root.sh
206
206
  . "$(dirname "${BASH_SOURCE[0]}")/_smoke-root.sh"
207
- P4="${MA_REFS:+$MA_REFS/phases/phase-4-review.md}"
207
+ P4="${MA_REFS:+$MA_REFS/phases/phase-3-review.md}"
208
208
  REVSCHEMA="${MA_SCHEMAS:+$MA_SCHEMAS/reviewer-output.schema.json}"
209
209
  TRSCHEMA="${MA_SCHEMAS:+$MA_SCHEMAS/triage-output.schema.json}"
210
210
 
211
211
  if [ -z "$P4" ] || [ ! -f "$P4" ]; then
212
- echo " ↷ SKIP: phase-4-review.md not present in this $MA_LAYOUT layout"
212
+ echo " ↷ SKIP: phase-3-review.md not present in this $MA_LAYOUT layout"
213
213
  # Two independent statements, because they can drift apart: the count sentence is
214
214
  # the contract, and the matrix is what a reader dispatches from. Three regexes over
215
215
  # overlapping prose used to stand in for this and let the count sentence 300 lines
216
216
  # further down go stale for a whole release without failing.
217
217
  elif grep -qF "Claude Code 3, Copilot CLI 3, Codex CLI 3" "$P4" \
218
218
  && grep -qE '^\| Reviewer 3 .*\|.*\|.*\|.*\|' "$P4"; then
219
- pass "phase-4-review declares Claude=3 / Copilot=3 / Codex=3 reviewers, and the matrix has all three host columns"
219
+ pass "phase-3-review declares Claude=3 / Copilot=3 / Codex=3 reviewers, and the matrix has all three host columns"
220
220
  else
221
- fail "phase-4-review does not declare the CLI-aware reviewer count for all three hosts"
221
+ fail "phase-3-review does not declare the CLI-aware reviewer count for all three hosts"
222
222
  fi
223
223
 
224
224
  # The two Codex constraints are silent-failure shaped, so the contract has to name
@@ -226,9 +226,9 @@ fi
226
226
  # 4-slot ceiling (orchestrator included) is why the count is 3 and not more.
227
227
  if [ -n "$P4" ] && [ -f "$P4" ]; then
228
228
  if grep -q 'fork_turns' "$P4" && grep -qiE "concurrency|slots" "$P4"; then
229
- pass "phase-4-review documents the fork_turns override rule + the concurrency ceiling"
229
+ pass "phase-3-review documents the fork_turns override rule + the concurrency ceiling"
230
230
  else
231
- fail "phase-4-review must document fork_turns and the Codex concurrency ceiling"
231
+ fail "phase-3-review must document fork_turns and the Codex concurrency ceiling"
232
232
  fi
233
233
  fi
234
234
 
@@ -24,6 +24,26 @@ TEMPLATE="$SMOKE_DIR/../preferences-template.json"
24
24
  [ -f "$TEMPLATE" ] || TEMPLATE="$HOME/multi-agent-pipeline/pipeline/preferences-template.json"
25
25
  LIVE_PREFS="$HOME/.claude/multi-agent-preferences.json"
26
26
 
27
+ # The migration target, derived the way migrate-prefs.mjs derives it: the last
28
+ # entry of the schema's schemaVersion enum, read from the schema that sits
29
+ # beside the migrator being checked. It used to be grepped as a
30
+ # `TARGET_VERSION = "x.y.z"` literal out of the migrator, and when that literal
31
+ # was replaced by the schema read - precisely because a literal had drifted a
32
+ # minor behind - the grep started matching nothing and both checks that depend
33
+ # on it reported "no reference point". A gate that reads the source of truth
34
+ # cannot go stale against it; a gate that reads a transcription of it can.
35
+ migration_target() {
36
+ local schema="$1/../schemas/prefs.schema.json"
37
+ [ -f "$schema" ] || schema="$PREFS_SCHEMA"
38
+ [ -f "$schema" ] || return 1
39
+ node -e "
40
+ const sv = JSON.parse(require('fs').readFileSync('$schema','utf8'))?.properties?.schemaVersion;
41
+ const v = sv?.const ?? sv?.enum?.at(-1);
42
+ if (!v) process.exit(1);
43
+ process.stdout.write(v);
44
+ " 2>/dev/null
45
+ }
46
+
27
47
  # ──────────────────────────────────────────────────────────────────────────
28
48
  echo "→ 1. Schema files parse as JSON"
29
49
  for f in "$PREFS_SCHEMA" "$STATE_SCHEMA"; do
@@ -145,10 +165,9 @@ if [ -f "$TEMPLATE" ]; then
145
165
  # every old entry in the migrator's accepted set became load-bearing purely to
146
166
  # rescue the template this gate was holding back. A template behind the target
147
167
  # is a defect, not the expected shape.
148
- TEMPLATE_TARGET=$(grep -oE 'TARGET_VERSION = "[0-9.]+"' "$SMOKE_DIR/migrate-prefs.mjs" 2>/dev/null \
149
- | grep -oE '[0-9]+\.[0-9]+\.[0-9]+')
168
+ TEMPLATE_TARGET=$(migration_target "$SMOKE_DIR")
150
169
  if [ -z "$TEMPLATE_TARGET" ]; then
151
- fail "cannot read TARGET_VERSION from migrate-prefs.mjs - the check has no reference point"
170
+ fail "cannot derive the migration target from prefs.schema.json - the check has no reference point"
152
171
  elif [ "$TVER" = "$TEMPLATE_TARGET" ]; then
153
172
  pass "template schemaVersion: $TVER (at the migration target)"
154
173
  else
@@ -201,17 +220,17 @@ if [ -f "$LIVE_PREFS" ]; then
201
220
  # correctly migrated to 2.4.0 hit the else arm and FAILED as "unknown". The
202
221
  # gate was rejecting the only fully-migrated state it exists to encourage.
203
222
  #
204
- # Reading the target from the migrator, and the accepted set from the schema
205
- # enum, means this can never disagree with them again.
223
+ # Reading the target and the accepted set from the same schema enum the
224
+ # migrator reads means this can never disagree with it again.
206
225
  MIGRATOR="$SMOKE_DIR/migrate-prefs.mjs"
207
- TARGET=$(grep -oE 'TARGET_VERSION = "[0-9.]+"' "$MIGRATOR" 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+')
226
+ TARGET=$(migration_target "$(dirname "$MIGRATOR")")
208
227
  KNOWN=$(node -e "
209
228
  const s = require('$PREFS_SCHEMA');
210
229
  process.stdout.write((s.properties.schemaVersion.enum || []).join(' '));
211
230
  " 2>/dev/null)
212
231
 
213
232
  if [ -z "$TARGET" ]; then
214
- fail "cannot read TARGET_VERSION from migrate-prefs.mjs - the check has no reference point"
233
+ fail "cannot derive the migration target from prefs.schema.json - the check has no reference point"
215
234
  elif [ "$LVER" = "$TARGET" ]; then
216
235
  pass "live prefs at the migration target (v$TARGET)"
217
236
  elif [ "$LVER" = "none" ]; then
@@ -392,8 +392,9 @@ function main() {
392
392
  gaps,
393
393
  };
394
394
 
395
+ // Returning, not exiting: `gaps` grows with the repo and process.exit() would
396
+ // cut the payload at the pipe buffer.
395
397
  process.stdout.write(PRETTY ? JSON.stringify(out, null, 2) + "\n" : JSON.stringify(out) + "\n");
396
- process.exit(0);
397
398
  }
398
399
 
399
400
  try {
@@ -29,6 +29,7 @@
29
29
  * only on a setup error (unreadable/invalid input is treated as "no findings").
30
30
  */
31
31
  import { readFileSync } from "node:fs";
32
+ import { runMain } from "../lib/fatal.mjs";
32
33
 
33
34
  const args = process.argv.slice(2);
34
35
  const fileFlag = args.indexOf("--file");
@@ -92,4 +93,4 @@ function main() {
92
93
  );
93
94
  }
94
95
 
95
- main();
96
+ runMain("test-integrity-gate", main);