@mmerterden/multi-agent-pipeline 17.6.0 → 18.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/CHANGELOG.md +127 -0
  2. package/README.md +43 -1
  3. package/README.tr.md +41 -0
  4. package/docs/adr/0011-dormant-ci.md +25 -1
  5. package/docs/server-readiness.md +188 -0
  6. package/index.js +16 -1
  7. package/install/_common.mjs +42 -17
  8. package/install/_dev-only-files.mjs +8 -0
  9. package/install/_unattended-profile.mjs +113 -0
  10. package/install/index.mjs +48 -0
  11. package/manifest.json +1049 -0
  12. package/package.json +5 -2
  13. package/pipeline/commands/multi-agent/status/SKILL.md +52 -21
  14. package/pipeline/lib/_jira-auth.sh +8 -0
  15. package/pipeline/lib/analysis-jira-write.sh +32 -0
  16. package/pipeline/lib/ask-choice.sh +13 -2
  17. package/pipeline/lib/autopilot-state.sh +8 -0
  18. package/pipeline/lib/fatal.mjs +129 -0
  19. package/pipeline/lib/figma-mcp-refresh.sh +18 -0
  20. package/pipeline/lib/figma-screenshot.sh +18 -0
  21. package/pipeline/lib/invoked-directly.mjs +43 -0
  22. package/pipeline/lib/jira-publish.sh +42 -0
  23. package/pipeline/lib/md2confluence-v3.py +47 -0
  24. package/pipeline/lib/outbound-gate.mjs +175 -0
  25. package/pipeline/lib/plan-todos.sh +27 -6
  26. package/pipeline/lib/post-pr-review.sh +77 -8
  27. package/pipeline/lib/repo-hygiene.sh +8 -3
  28. package/pipeline/lib/require-jq.sh +40 -0
  29. package/pipeline/lib/run-paths.sh +335 -0
  30. package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +70 -0
  31. package/pipeline/multi-agent-refs/features/cost-analysis.md +93 -0
  32. package/pipeline/multi-agent-refs/features/doctor.md +45 -0
  33. package/pipeline/multi-agent-refs/features/verify.md +83 -0
  34. package/pipeline/multi-agent-refs/phases/operations.md +13 -2
  35. package/pipeline/multi-agent-refs/phases/phase-0-init.md +1 -1
  36. package/pipeline/multi-agent-refs/unattended-contract.md +129 -0
  37. package/pipeline/scripts/_run-paths.mjs +372 -0
  38. package/pipeline/scripts/aggregate-metrics.mjs +64 -64
  39. package/pipeline/scripts/autopilot-arming.mjs +2 -1
  40. package/pipeline/scripts/autopilot-intake.mjs +2 -1
  41. package/pipeline/scripts/autopilot-runner.mjs +206 -2
  42. package/pipeline/scripts/build-references.mjs +2 -1
  43. package/pipeline/scripts/build-stack-plugins.mjs +10 -2
  44. package/pipeline/scripts/capture-evidence.sh +7 -2
  45. package/pipeline/scripts/classify-plan-safety.mjs +2 -1
  46. package/pipeline/scripts/cost-analyze.mjs +600 -0
  47. package/pipeline/scripts/cost-budget-check.mjs +4 -12
  48. package/pipeline/scripts/council-view.mjs +2 -1
  49. package/pipeline/scripts/crush-json.mjs +2 -1
  50. package/pipeline/scripts/diff-explain.mjs +6 -9
  51. package/pipeline/scripts/diff-risk-score.mjs +2 -1
  52. package/pipeline/scripts/doctor.mjs +138 -4
  53. package/pipeline/scripts/evidence-gate.mjs +9 -3
  54. package/pipeline/scripts/feedback-send.mjs +12 -2
  55. package/pipeline/scripts/gc-abandoned.sh +29 -13
  56. package/pipeline/scripts/gc-worktrees.sh +11 -4
  57. package/pipeline/scripts/github-ssh-setup.sh +64 -7
  58. package/pipeline/scripts/graph-mermaid.mjs +4 -2
  59. package/pipeline/scripts/keychain-save.sh +101 -30
  60. package/pipeline/scripts/learn-from-transcripts.mjs +2 -1
  61. package/pipeline/scripts/learning-curve.mjs +34 -29
  62. package/pipeline/scripts/make-manifest.mjs +199 -0
  63. package/pipeline/scripts/migrate-prefs.mjs +2 -1
  64. package/pipeline/scripts/migrate-state.mjs +94 -4
  65. package/pipeline/scripts/phase-banner.sh +6 -2
  66. package/pipeline/scripts/phase-tracker.sh +41 -3
  67. package/pipeline/scripts/plan-coverage-gate.mjs +6 -2
  68. package/pipeline/scripts/pre-commit-check.sh +7 -0
  69. package/pipeline/scripts/pre-push-check.sh +7 -0
  70. package/pipeline/scripts/purge.sh +23 -6
  71. package/pipeline/scripts/render-agent-log-cost.sh +9 -2
  72. package/pipeline/scripts/render-cost-summary.sh +9 -2
  73. package/pipeline/scripts/render-work-summary.sh +11 -4
  74. package/pipeline/scripts/review-file-filter.mjs +4 -2
  75. package/pipeline/scripts/review-scope.mjs +2 -1
  76. package/pipeline/scripts/routine-registry.mjs +2 -1
  77. package/pipeline/scripts/run-aggregator.mjs +13 -14
  78. package/pipeline/scripts/run-metrics.mjs +3 -1
  79. package/pipeline/scripts/runs-index.mjs +343 -0
  80. package/pipeline/scripts/scorecard-snapshot.mjs +178 -0
  81. package/pipeline/scripts/search-logs.sh +18 -0
  82. package/pipeline/scripts/test-gap-scan.mjs +2 -1
  83. package/pipeline/scripts/test-integrity-gate.mjs +2 -1
  84. package/pipeline/scripts/update-issue-progress.sh +56 -7
  85. package/pipeline/scripts/usage-report.mjs +12 -1
  86. package/pipeline/scripts/validate-analysis-doc.mjs +2 -1
  87. package/pipeline/scripts/validate-code-graph.mjs +6 -3
  88. package/pipeline/scripts/validate-complaint-doc.mjs +2 -1
  89. package/pipeline/scripts/validate-diff-risk.mjs +6 -3
  90. package/pipeline/scripts/validate-test-gap.mjs +6 -3
  91. package/pipeline/scripts/validate-triage.mjs +3 -1
  92. package/pipeline/scripts/verify-citations.mjs +4 -2
  93. package/pipeline/scripts/verify.mjs +327 -0
  94. package/pipeline/scripts/worktree-finalize.sh +13 -4
  95. package/pipeline/scripts/write-state.mjs +154 -15
  96. package/pipeline/skills/.skill-manifest.json +2 -2
  97. package/pipeline/skills/.skills-index.json +56 -1
  98. package/pipeline/skills/shared/README.md +8 -3
  99. package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +33 -9
  100. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/package_app.sh +4 -1
  101. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/setup_dev_signing.sh +4 -1
  102. package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/sign-and-notarize.sh +2 -1
  103. package/pipeline/skills/skills-index.md +6 -1
@@ -0,0 +1,343 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * @file runs-index.mjs - one deterministic answer to "what runs exist and
4
+ * where is each one".
5
+ *
6
+ * `/multi-agent:status` used to answer this by telling the model to go and find
7
+ * the files itself: scan three hard-coded `.worktrees/` paths, then `find` the
8
+ * log tree at one depth, then merge. Three problems with that. It cannot be
9
+ * called by anything that is not a model, the depth was wrong for half the
10
+ * layouts, and two invocations could disagree because nothing pinned the
11
+ * traversal. A UI, a gate, or a second phase reading the same question got a
12
+ * different answer than the terminal did.
13
+ *
14
+ * This is the producer. `--json` and the human table are rendered from the SAME
15
+ * in-memory records, so a dashboard and a terminal cannot disagree - the rule
16
+ * autopilot-status.sh already follows for the autopilot half of the picture.
17
+ *
18
+ * Grouping follows the contract in commands/multi-agent/status/SKILL.md 3b,
19
+ * including its final clause: a run with no `status` is placed in no group at
20
+ * all. Unknown is not a finding, and calling it dead is the same false claim in
21
+ * the other direction.
22
+ *
23
+ * Read-only. Never writes, never migrates, never deletes.
24
+ *
25
+ * Usage:
26
+ * node runs-index.mjs # human table, grouped
27
+ * node runs-index.mjs --json # the same records as JSON
28
+ * node runs-index.mjs --group waiting # one group only
29
+ * node runs-index.mjs --task-id <id> # one run
30
+ *
31
+ * Exit codes:
32
+ * 0 - answered (including "no runs")
33
+ * 2 - usage error
34
+ *
35
+ * @module pipeline/scripts/runs-index
36
+ */
37
+
38
+ import { existsSync, readFileSync } from "node:fs";
39
+ import { join } from "node:path";
40
+ import { costUsd } from "./_cost.mjs";
41
+ import { listRuns, logsRoot, resolveRunDir, taskIdVariants } from "./_run-paths.mjs";
42
+ import { runMain } from "../lib/fatal.mjs";
43
+ import { invokedDirectly } from "../lib/invoked-directly.mjs";
44
+
45
+ const GROUPS = {
46
+ waiting: "Waiting on you",
47
+ stopped: "Stopped mid-development",
48
+ question: "Left at a question",
49
+ unknown: "Status not recorded",
50
+ };
51
+
52
+ function parseArgs(argv) {
53
+ const flags = { json: false, group: null, taskId: null };
54
+ for (let i = 0; i < argv.length; i++) {
55
+ const a = argv[i];
56
+ if (a === "--json") flags.json = true;
57
+ else if (a === "--group") flags.group = argv[++i];
58
+ else if (a.startsWith("--group=")) flags.group = a.slice(8);
59
+ else if (a === "--task-id") flags.taskId = argv[++i];
60
+ else if (a.startsWith("--task-id=")) flags.taskId = a.slice(10);
61
+ else if (a === "-h" || a === "--help") flags.help = true;
62
+ else {
63
+ process.stderr.write(`runs-index: unknown argument ${a}\n`);
64
+ process.exit(2);
65
+ }
66
+ }
67
+ if (flags.group && !Object.hasOwn(GROUPS, flags.group)) {
68
+ process.stderr.write(
69
+ `runs-index: unknown group ${flags.group} (want ${Object.keys(GROUPS).join(", ")})\n`,
70
+ );
71
+ process.exit(2);
72
+ }
73
+ return flags;
74
+ }
75
+
76
+ function readJson(path) {
77
+ if (!path || !existsSync(path)) return null;
78
+ try {
79
+ return JSON.parse(readFileSync(path, "utf-8"));
80
+ } catch {
81
+ // A truncated state file is a fact about the run, not a reason to refuse
82
+ // the whole index. It surfaces as `stateReadable: false`.
83
+ return null;
84
+ }
85
+ }
86
+
87
+ function firstFile(dir, name) {
88
+ for (const base of [dir, join(dir, "artifacts")]) {
89
+ const p = join(base, name);
90
+ if (existsSync(p)) return p;
91
+ }
92
+ return null;
93
+ }
94
+
95
+ let COST_TABLE = null;
96
+ function rateFor(model) {
97
+ if (!COST_TABLE) {
98
+ COST_TABLE = readJson(new URL("./cost-table.json", import.meta.url).pathname) ?? { prices: {} };
99
+ }
100
+ if (!model) return null;
101
+ const prices = COST_TABLE.prices ?? {};
102
+ // phase-tracker.sh stores the short tier name ("opus", "fable"), which is the
103
+ // table's own key - the same lookup run-aggregator.mjs does. A caller that
104
+ // stored the full model id instead is matched on the table's `modelId`
105
+ // rather than guessed at by prefix: "gpt-5.6" is a prefix of "gpt-5.6-terra"
106
+ // and those are two different prices.
107
+ if (prices[model]) return prices[model];
108
+ for (const rate of Object.values(prices)) {
109
+ if (rate?.modelId === model) return rate;
110
+ }
111
+ return null;
112
+ }
113
+
114
+ /**
115
+ * Phase rows plus token/cost totals, from the tracker document.
116
+ *
117
+ * @param {object|null} tracker
118
+ */
119
+ function summarisePhases(tracker) {
120
+ const phases = Array.isArray(tracker?.phases) ? tracker.phases : [];
121
+ let tokensIn = 0;
122
+ let tokensOut = 0;
123
+ let tokensCached = 0;
124
+ let usd = 0;
125
+ const rows = phases.map((p) => {
126
+ // phase-tracker.sh writes flat `tokens_in` / `tokens_out` / `tokens_cached`
127
+ // on each phase, not a nested `tokens` object.
128
+ const tin = Number(p.tokens_in ?? 0) || 0;
129
+ const tout = Number(p.tokens_out ?? 0) || 0;
130
+ const tcached = Number(p.tokens_cached ?? 0) || 0;
131
+ tokensIn += tin;
132
+ tokensOut += tout;
133
+ tokensCached += tcached;
134
+ const c = costUsd(rateFor(p.model), tin, tout, tcached);
135
+ if (typeof c === "number") usd += c;
136
+ return {
137
+ id: String(p.id ?? ""),
138
+ name: p.name ?? "",
139
+ status: p.status ?? "pending",
140
+ model: p.model ?? null,
141
+ startedAt: p.started_at ?? null,
142
+ completedAt: p.completed_at ?? null,
143
+ now: p.now ?? null,
144
+ subs: Array.isArray(p.subs) ? p.subs.length : 0,
145
+ };
146
+ });
147
+ return {
148
+ phases: rows,
149
+ startedAt: tracker?.started_at ?? null,
150
+ tokens: { in: tokensIn, out: tokensOut, cached: tokensCached },
151
+ estUsd: Number(usd.toFixed(4)),
152
+ };
153
+ }
154
+
155
+ /**
156
+ * The group a run belongs to, per status/SKILL.md 3b.
157
+ *
158
+ * @param {object|null} state
159
+ * @returns {"waiting"|"stopped"|"question"|"unknown"}
160
+ */
161
+ function groupOf(state) {
162
+ const status = state?.status;
163
+ if (!status) return "unknown";
164
+ const phase = Number(state?.currentPhase);
165
+ const prUrl = state?.pr?.url ?? state?.prUrl ?? null;
166
+ if (status === "awaiting_input" || status === "awaiting-user-test-main-checkout")
167
+ return "waiting";
168
+ if (prUrl) return "waiting";
169
+ if (Number.isFinite(phase) && phase >= 6) return "waiting";
170
+ if (Number.isFinite(phase) && phase === 0) return "question";
171
+ return "stopped";
172
+ }
173
+
174
+ /**
175
+ * Every run, enriched, in one stable order.
176
+ *
177
+ * @returns {object[]}
178
+ */
179
+ export function buildIndex() {
180
+ return listRuns().map((run) => {
181
+ const statePath = firstFile(run.dir, "agent-state.json");
182
+ const trackerPath = firstFile(run.dir, "tracker-state.json");
183
+ const state = readJson(statePath);
184
+ const tracker = readJson(trackerPath);
185
+ const phases = summarisePhases(tracker);
186
+ const prUrl = state?.pr?.url ?? state?.prUrl ?? null;
187
+ return {
188
+ taskId: run.taskId,
189
+ project: run.project ?? run.projectHint ?? null,
190
+ dir: run.dir,
191
+ layout: run.layout,
192
+ duplicateOf: run.duplicateOf,
193
+ salvaged: Boolean(statePath && statePath.includes(`${run.dir}/artifacts/`)),
194
+ // What KIND of record this is, before asking whether it is healthy.
195
+ //
196
+ // 74 of the 103 runs on this machine have a tracker file and no agent
197
+ // state, and the first version of this index called every one of them
198
+ // `stateReadable: false` - so a panel built on it announced "74 runs
199
+ // could not be read" about records that are not damaged and never had
200
+ // agent state to begin with. `/multi-agent:analysis` says so in its own
201
+ // description ("no worktree, no commits, no dev chain"), and design-check
202
+ // is the same shape.
203
+ //
204
+ // Derived from what is on disk, not from the task id: `ANALYSIS-` and
205
+ // `DC-` are naming conventions, and a convention is not a contract.
206
+ kind: statePath ? "pipeline" : trackerPath ? "tracker-only" : "empty",
207
+ // Now the health question, and it only applies where state was expected.
208
+ // A tracker-only record is `true` because there is nothing it failed to
209
+ // read; "could not read" and "there is none" are different answers and
210
+ // only the first one asks anyone to do something.
211
+ stateReadable: statePath ? Boolean(state) : true,
212
+ status: state?.status ?? null,
213
+ currentPhase: Number.isFinite(Number(state?.currentPhase))
214
+ ? Number(state.currentPhase)
215
+ : null,
216
+ branch: state?.branch ?? state?.branchName ?? null,
217
+ baseBranch: state?.baseBranch ?? null,
218
+ startedAt: state?.startedAt ?? phases.startedAt,
219
+ worktreePath: state?.worktreePath ?? null,
220
+ prUrl,
221
+ autopilot: Boolean(state?.autopilot),
222
+ schemaVersion: state?.schemaVersion ?? null,
223
+ rev: Number.isInteger(state?.rev) ? state.rev : null,
224
+ group: groupOf(state),
225
+ phases: phases.phases,
226
+ tokens: phases.tokens,
227
+ estUsd: phases.estUsd,
228
+ // How many tracker writes went through without the lock. Non-zero means
229
+ // two writers were in the critical section and one of their token
230
+ // deltas was dropped, so the numbers on this run are LOW. Reporting the
231
+ // count rather than the loss, because the size of what was dropped is
232
+ // exactly what nobody measured.
233
+ unlockedWrites: Number.isInteger(tracker?.unlockedWrites) ? tracker.unlockedWrites : 0,
234
+ };
235
+ });
236
+ }
237
+
238
+ function fmtDuration(startedAt) {
239
+ if (!startedAt) return "-";
240
+ const t = Date.parse(startedAt);
241
+ if (!Number.isFinite(t)) return "-";
242
+ const mins = Math.max(0, Math.round((Date.now() - t) / 60000));
243
+ if (mins < 60) return `${mins}m`;
244
+ const h = Math.floor(mins / 60);
245
+ if (h < 48) return `${h}h`;
246
+ return `${Math.floor(h / 24)}d`;
247
+ }
248
+
249
+ function renderHuman(records) {
250
+ const out = [];
251
+ const total = records.length;
252
+ out.push(`multi-agent runs - ${total} under ${logsRoot()}`);
253
+
254
+ const dupes = records.filter((r) => r.duplicateOf);
255
+ if (dupes.length) {
256
+ out.push(
257
+ ` ${dupes.length} run(s) also have a copy in the other directory layout; ` +
258
+ `the newer one is shown. migrate-state.mjs --all reports them.`,
259
+ );
260
+ }
261
+
262
+ for (const key of Object.keys(GROUPS)) {
263
+ const rows = records.filter((r) => r.group === key);
264
+ if (!rows.length) continue;
265
+ out.push("");
266
+ out.push(`${GROUPS[key]} (${rows.length})`);
267
+ out.push(
268
+ " ID Phase Status Branch Age ~USD",
269
+ );
270
+ for (const r of rows) {
271
+ const phase = r.currentPhase === null ? " -" : `${r.currentPhase}/7`;
272
+ out.push(
273
+ " " +
274
+ [
275
+ r.taskId.padEnd(30).slice(0, 30),
276
+ String(phase).padStart(5),
277
+ (r.status ?? "-").padEnd(17).slice(0, 17),
278
+ (r.branch ?? "-").padEnd(24).slice(0, 24),
279
+ fmtDuration(r.startedAt).padStart(5),
280
+ (r.estUsd ? r.estUsd.toFixed(2) : "-").padStart(6),
281
+ ].join(" "),
282
+ );
283
+ }
284
+ }
285
+
286
+ const hints = {
287
+ waiting: "resume #N - the work landed, it needs your answer",
288
+ stopped: "resume #N or kill #N",
289
+ question: "garbage-collect --abandoned - nothing was built",
290
+ };
291
+ const present = Object.keys(hints).filter((k) => records.some((r) => r.group === k));
292
+ if (present.length) {
293
+ out.push("");
294
+ for (const k of present) out.push(` ${GROUPS[k]}: ${hints[k]}`);
295
+ }
296
+ return out.join("\n");
297
+ }
298
+
299
+ function main() {
300
+ const flags = parseArgs(process.argv.slice(2));
301
+ if (flags.help) {
302
+ process.stdout.write(
303
+ "Usage: runs-index.mjs [--json] [--group waiting|stopped|question|unknown] [--task-id <id>]\n",
304
+ );
305
+ return 0;
306
+ }
307
+
308
+ let records = buildIndex();
309
+
310
+ if (flags.taskId) {
311
+ const wanted = new Set(taskIdVariants(flags.taskId));
312
+ records = records.filter((r) => wanted.has(r.taskId));
313
+ if (!records.length) {
314
+ const dir = resolveRunDir(flags.taskId);
315
+ process.stderr.write(
316
+ `runs-index: no run for ${flags.taskId}${dir ? ` (directory ${dir} has no markers)` : ""}\n`,
317
+ );
318
+ return 0;
319
+ }
320
+ }
321
+ if (flags.group) records = records.filter((r) => r.group === flags.group);
322
+
323
+ if (flags.json) {
324
+ process.stdout.write(
325
+ JSON.stringify({ logsRoot: logsRoot(), count: records.length, runs: records }, null, 2) +
326
+ "\n",
327
+ );
328
+ } else {
329
+ process.stdout.write(renderHuman(records) + "\n");
330
+ }
331
+ return 0;
332
+ }
333
+
334
+ if (invokedDirectly(import.meta.url)) {
335
+ // process.exitCode, not process.exit(): stdout to a PIPE is asynchronous and
336
+ // process.exit() throws away whatever has not drained. 103 runs render to
337
+ // ~200KB of JSON, so a caller doing `runs-index.mjs --json | jq` received the
338
+ // first 64KB and a parse error. A terminal hid it, because stdout to a TTY is
339
+ // synchronous - and a terminal is where this was tested.
340
+ runMain("runs-index", () => {
341
+ process.exitCode = main();
342
+ });
343
+ }
@@ -0,0 +1,178 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * scorecard-snapshot.mjs - keep what the scorecard said, and say what moved.
4
+ *
5
+ * The scorecard answers "does every mechanical claim hold right now". It
6
+ * cannot answer "is this better or worse than last week", and that second
7
+ * question is the one that catches slow rot: a metric that has been failing
8
+ * for a month reads identically to one that broke an hour ago.
9
+ *
10
+ * WHAT THIS DELIBERATELY DOES NOT DO: collapse the result into a 0-100 score.
11
+ * ruflo's scorecard does, and the discipline worth taking from it is
12
+ * "measurable and comparable over time", not the number. The scorecard already
13
+ * reports twelve measured metrics AND four it refuses to measure - a single
14
+ * figure would hide both halves, and an unmeasured category would silently
15
+ * count as zero or as full marks depending on an arithmetic choice nobody
16
+ * would ever read. A diff keeps every metric answering for itself.
17
+ *
18
+ * Snapshots live in ~/.claude/state/scorecard/<iso>.json and are never pruned
19
+ * by this script: they are small, and a history that deletes itself cannot
20
+ * answer the question it was kept for.
21
+ *
22
+ * Usage:
23
+ * scorecard-snapshot.mjs --save run the scorecard, store a snapshot
24
+ * scorecard-snapshot.mjs --diff latest vs the one before it
25
+ * scorecard-snapshot.mjs --diff a.json b.json two snapshots by path
26
+ * scorecard-snapshot.mjs --list what is stored
27
+ *
28
+ * Exit codes:
29
+ * 0 - nothing regressed (or a save succeeded)
30
+ * 1 - at least one metric that used to pass now fails
31
+ * 2 - usage error, or not enough snapshots to compare
32
+ */
33
+
34
+ import { execFileSync } from "node:child_process";
35
+ import { existsSync, mkdirSync, readdirSync, readFileSync, writeFileSync } from "node:fs";
36
+ import { homedir } from "node:os";
37
+ import { dirname, join } from "node:path";
38
+ import { fileURLToPath } from "node:url";
39
+ import { runMain } from "../lib/fatal.mjs";
40
+
41
+ const ROOT = join(dirname(fileURLToPath(import.meta.url)), "..", "..");
42
+ const STORE = process.env.SCORECARD_STORE || join(homedir(), ".claude", "state", "scorecard");
43
+
44
+ /** Metric identity. Category plus metric name, because neither is unique alone. */
45
+ const keyOf = (r) => `${r.category} :: ${r.metric}`;
46
+
47
+ function runScorecard() {
48
+ // `--json` prints the report on stdout; a failing metric is a non-zero exit
49
+ // and is exactly the case worth snapshotting, so the status is captured
50
+ // rather than thrown.
51
+ try {
52
+ return JSON.parse(
53
+ execFileSync("node", ["pipeline/scripts/scorecard.mjs", "--json"], {
54
+ cwd: ROOT,
55
+ encoding: "utf-8",
56
+ maxBuffer: 32 * 1024 * 1024,
57
+ }),
58
+ );
59
+ } catch (err) {
60
+ const out = err?.stdout;
61
+ if (typeof out === "string" && out.trim().startsWith("{")) return JSON.parse(out);
62
+ throw new Error(`scorecard did not produce JSON: ${err?.message ?? err}`, { cause: err });
63
+ }
64
+ }
65
+
66
+ function snapshots() {
67
+ if (!existsSync(STORE)) return [];
68
+ return (
69
+ readdirSync(STORE)
70
+ .filter((f) => f.endsWith(".json"))
71
+ // ISO-8601 sorts lexically in time order, which is the whole reason the
72
+ // file is named after the timestamp rather than a counter.
73
+ .sort()
74
+ .map((f) => join(STORE, f))
75
+ );
76
+ }
77
+
78
+ function save() {
79
+ const report = runScorecard();
80
+ mkdirSync(STORE, { recursive: true });
81
+ const at = new Date().toISOString().replace(/[:.]/g, "-");
82
+ const path = join(STORE, `${at}.json`);
83
+ writeFileSync(path, JSON.stringify({ at: new Date().toISOString(), ...report }, null, 2) + "\n");
84
+ process.stdout.write(`scorecard snapshot: ${path}\n`);
85
+ process.stdout.write(` ${report.passed} passed, ${report.failed} failed\n`);
86
+ return 0;
87
+ }
88
+
89
+ function diff(aPath, bPath) {
90
+ const a = JSON.parse(readFileSync(aPath, "utf-8"));
91
+ const b = JSON.parse(readFileSync(bPath, "utf-8"));
92
+ // UNMEASURED rows carry no `ok` at all - they are the categories the
93
+ // scorecard refuses to score. Comparing verdicts across them would invent
94
+ // one, which is the thing that section exists to avoid.
95
+ const measured = (rs) => (rs || []).filter((r) => r.kind !== "UNMEASURED");
96
+ const ma = new Map(measured(a.results).map((r) => [keyOf(r), r]));
97
+ const mb = new Map(measured(b.results).map((r) => [keyOf(r), r]));
98
+
99
+ const regressed = [];
100
+ const fixed = [];
101
+ const added = [];
102
+ const removed = [];
103
+ const changed = [];
104
+
105
+ for (const [k, rb] of mb) {
106
+ const ra = ma.get(k);
107
+ if (!ra) {
108
+ added.push(rb);
109
+ continue;
110
+ }
111
+ if (ra.ok && !rb.ok) regressed.push({ k, ra, rb });
112
+ else if (!ra.ok && rb.ok) fixed.push({ k, ra, rb });
113
+ // A metric whose verdict held but whose DETAIL moved is the early warning:
114
+ // coverage sliding from 72 to 69 passes the floor and is still the thing
115
+ // that will fail next month.
116
+ else if (ra.detail !== rb.detail) changed.push({ k, ra, rb });
117
+ }
118
+ for (const [k, ra] of ma) if (!mb.has(k)) removed.push({ k, ra });
119
+
120
+ const out = [];
121
+ out.push(`scorecard diff - ${a.at ?? aPath} → ${b.at ?? bPath}`);
122
+ out.push(` ${a.passed}/${a.passed + a.failed} → ${b.passed}/${b.passed + b.failed} passing`);
123
+ const section = (title, rows, fmt) => {
124
+ if (!rows.length) return;
125
+ out.push("");
126
+ out.push(`${title} (${rows.length})`);
127
+ for (const r of rows) out.push(` ${fmt(r)}`);
128
+ };
129
+ section("REGRESSED", regressed, ({ k, rb }) => `${k}\n now: ${rb.detail ?? "(no detail)"}`);
130
+ section("FIXED", fixed, ({ k, rb }) => `${k}\n now: ${rb.detail ?? "(no detail)"}`);
131
+ section("NEW METRIC", added, (r) => `${keyOf(r)} ${r.ok ? "passing" : "FAILING"}`);
132
+ section("METRIC GONE", removed, ({ k }) => `${k} - no longer measured`);
133
+ section(
134
+ "SAME VERDICT, DIFFERENT NUMBERS",
135
+ changed,
136
+ ({ k, ra, rb }) =>
137
+ `${k}\n was: ${ra.detail ?? "(none)"}\n now: ${rb.detail ?? "(none)"}`,
138
+ );
139
+
140
+ if (!regressed.length && !fixed.length && !added.length && !removed.length && !changed.length) {
141
+ out.push("");
142
+ out.push(" nothing moved");
143
+ }
144
+ process.stdout.write(out.join("\n") + "\n");
145
+ return regressed.length ? 1 : 0;
146
+ }
147
+
148
+ function main() {
149
+ const args = process.argv.slice(2);
150
+ if (args.includes("--save")) return save();
151
+ if (args.includes("--list")) {
152
+ const s = snapshots();
153
+ process.stdout.write(s.length ? s.join("\n") + "\n" : `no snapshots under ${STORE}\n`);
154
+ return 0;
155
+ }
156
+ if (args.includes("--diff")) {
157
+ const paths = args.filter((a) => !a.startsWith("--"));
158
+ if (paths.length === 2) return diff(paths[0], paths[1]);
159
+ const s = snapshots();
160
+ if (s.length < 2) {
161
+ process.stderr.write(
162
+ `need two snapshots to compare, found ${s.length} under ${STORE}\n` +
163
+ ` take one with: node pipeline/scripts/scorecard-snapshot.mjs --save\n`,
164
+ );
165
+ return 2;
166
+ }
167
+ return diff(s[s.length - 2], s[s.length - 1]);
168
+ }
169
+ process.stderr.write("usage: scorecard-snapshot.mjs --save | --diff [a.json b.json] | --list\n");
170
+ return 2;
171
+ }
172
+
173
+ // `process.exitCode`, never `process.exit()`: stdout to a pipe is asynchronous
174
+ // and exit() discards whatever has not drained, which truncated two scripts in
175
+ // this repo at a buffer boundary and only ever when piped.
176
+ runMain("scorecard-snapshot", () => {
177
+ process.exitCode = main();
178
+ });
@@ -30,6 +30,24 @@
30
30
 
31
31
  set -uo pipefail
32
32
 
33
+ # jq is not optional on this path. Without the guard below a missing binary
34
+ # renders as EMPTY DATA and the work continues on it; see lib/require-jq.sh.
35
+ for _rq in "$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)/require-jq.sh" \
36
+ "$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")/../lib" 2>/dev/null && pwd)/require-jq.sh" \
37
+ "$HOME/.claude/lib/require-jq.sh" \
38
+ "$HOME/.copilot/lib/require-jq.sh" \
39
+ "$HOME/.codex/lib/require-jq.sh"; do
40
+ [ -f "$_rq" ] || continue
41
+ # shellcheck source=/dev/null
42
+ . "$_rq" && break
43
+ done
44
+ unset _rq
45
+ if ! command -v ma_require_jq >/dev/null 2>&1; then
46
+ # The helper itself is missing, which is an install problem, not a jq one.
47
+ ma_require_jq() { command -v jq >/dev/null 2>&1 || { echo "jq not found - cannot ${1:-continue}." >&2; return 1; }; }
48
+ fi
49
+ ma_require_jq "search the logs" || exit 3
50
+
33
51
  SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
34
52
 
35
53
  ROOT="$HOME/.claude/logs/multi-agent"
@@ -392,8 +392,9 @@ function main() {
392
392
  gaps,
393
393
  };
394
394
 
395
+ // Returning, not exiting: `gaps` grows with the repo and process.exit() would
396
+ // cut the payload at the pipe buffer.
395
397
  process.stdout.write(PRETTY ? JSON.stringify(out, null, 2) + "\n" : JSON.stringify(out) + "\n");
396
- process.exit(0);
397
398
  }
398
399
 
399
400
  try {
@@ -29,6 +29,7 @@
29
29
  * only on a setup error (unreadable/invalid input is treated as "no findings").
30
30
  */
31
31
  import { readFileSync } from "node:fs";
32
+ import { runMain } from "../lib/fatal.mjs";
32
33
 
33
34
  const args = process.argv.slice(2);
34
35
  const fileFlag = args.indexOf("--file");
@@ -92,4 +93,4 @@ function main() {
92
93
  );
93
94
  }
94
95
 
95
- main();
96
+ runMain("test-integrity-gate", main);
@@ -29,6 +29,54 @@
29
29
 
30
30
  set -euo pipefail
31
31
 
32
+ # jq is not optional on this path. Without the guard below a missing binary
33
+ # renders as EMPTY DATA and the work continues on it; see lib/require-jq.sh.
34
+ for _rq in "$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)/require-jq.sh" \
35
+ "$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")/../lib" 2>/dev/null && pwd)/require-jq.sh" \
36
+ "$HOME/.claude/lib/require-jq.sh" \
37
+ "$HOME/.copilot/lib/require-jq.sh" \
38
+ "$HOME/.codex/lib/require-jq.sh"; do
39
+ [ -f "$_rq" ] || continue
40
+ # shellcheck source=/dev/null
41
+ . "$_rq" && break
42
+ done
43
+ unset _rq
44
+ if ! command -v ma_require_jq >/dev/null 2>&1; then
45
+ # The helper itself is missing, which is an install problem, not a jq one.
46
+ ma_require_jq() { command -v jq >/dev/null 2>&1 || { echo "jq not found - cannot ${1:-continue}." >&2; return 1; }; }
47
+ fi
48
+ ma_require_jq "update the issue" || exit 3
49
+
50
+ # Outbound leak gate. Every byte below is composed at runtime out of command
51
+ # output, error text and file excerpts, any of which can carry a token that was
52
+ # on this machine a second earlier - and once it is in a comment it is in
53
+ # someone else's database. The repo's own leak scanner looks at FILES IN THE
54
+ # REPO and never sees this text. See lib/outbound-gate.mjs.
55
+ ma_outbound_gate() {
56
+ local body_file="$1" og=""
57
+ for c in "$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)/outbound-gate.mjs" \
58
+ "$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")/../lib" 2>/dev/null && pwd)/outbound-gate.mjs" \
59
+ "$HOME/.claude/lib/outbound-gate.mjs" \
60
+ "$HOME/.copilot/lib/outbound-gate.mjs" \
61
+ "$HOME/.codex/lib/outbound-gate.mjs"; do
62
+ [ -f "$c" ] && { og="$c"; break; }
63
+ done
64
+ # Missing gate is NOT an open door: refusing to publish beats publishing
65
+ # unchecked, and the only way this file is absent is a broken install.
66
+ if [ -z "$og" ]; then
67
+ echo "outbound-gate.mjs not found - refusing to publish unchecked text." >&2
68
+ return 7
69
+ fi
70
+ node "$og" --file "$body_file"
71
+ }
72
+
73
+ # Run-state path resolution: pipeline/lib/run-paths.sh owns the two layouts
74
+ # (nested <root>/<project>/<id>/ and flat <root>/<id>/) and every id spelling.
75
+ _MA_RP_HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
76
+ # shellcheck source=/dev/null
77
+ . "$_MA_RP_HERE/../lib/run-paths.sh" 2>/dev/null || . "$HOME/.claude/lib/run-paths.sh"
78
+
79
+
32
80
  # Rewrite the `### Progress` block of the body file ($2) with $1 and print the
33
81
  # result to stdout. The replaced region is bounded at the FIRST of:
34
82
  # - a line starting with `<!--` (the legend comment, printed and kept), or
@@ -90,13 +138,10 @@ fi
90
138
  # per-project directory ($HOME/.claude/logs/multi-agent/{project}/{taskId}/)
91
139
  # - not the ~/.claude/projects/*/state/ path this used to glob, which no
92
140
  # writer in the pipeline ever populates.
93
- AGENT_STATE=""
94
- for candidate in "$HOME"/.claude/logs/multi-agent/*/"$TASK_ID"/agent-state.json; do
95
- if [ -f "$candidate" ]; then
96
- AGENT_STATE="$candidate"
97
- break
98
- fi
99
- done
141
+ # This globbed ONLY the nested layout. phase-tracker.sh and most Phase 0 paths
142
+ # write flat, so on a real machine the majority of runs resolved to nothing and
143
+ # this script reported "no agent-state.json" for a run whose state existed.
144
+ AGENT_STATE="$(ma_resolve_run_file "$TASK_ID" agent-state.json 2>/dev/null || true)"
100
145
 
101
146
  if [ -z "$AGENT_STATE" ]; then
102
147
  echo "update-issue-progress: no agent-state.json found for task=$TASK_ID" >&2
@@ -154,6 +199,10 @@ if cmp -s "$TMP_BODY" "$TMP_NEW"; then
154
199
  fi
155
200
 
156
201
  # Apply.
202
+ if ! ma_outbound_gate "$TMP_NEW"; then
203
+ echo "update-issue-progress: outbound gate refused the new body; #$ISSUE_NUM left unchanged" >&2
204
+ exit 7
205
+ fi
157
206
  if ! gh issue edit "$ISSUE_NUM" --repo "$ORG_REPO" --body-file "$TMP_NEW" >/dev/null; then
158
207
  echo "update-issue-progress: gh issue edit failed for #$ISSUE_NUM" >&2
159
208
  exit 4
@@ -609,7 +609,18 @@ async function main() {
609
609
  // Windows-safe entry-point check: compare file URLs, never string-split a path.
610
610
  const invokedDirectly = process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href;
611
611
  if (invokedDirectly) {
612
- main().catch(() => process.exit(0));
612
+ // Exit 0 stays: usage telemetry must never be the reason a run fails, and a
613
+ // non-zero exit here would propagate into whatever called it.
614
+ //
615
+ // But the silence goes. `catch(() => process.exit(0))` reported success for
616
+ // every failure, so a telemetry path that had been broken for weeks looked
617
+ // exactly like one that worked - there was no surface on which anyone could
618
+ // notice. One line on stderr costs nothing and is the difference between a
619
+ // degraded feature and an invisible one.
620
+ main().catch((err) => {
621
+ process.stderr.write(`usage-report: not sent - ${err?.message ?? err}\n`);
622
+ process.exit(0);
623
+ });
613
624
  }
614
625
 
615
626
  export {