@feigi/fleet-ctl 3.19.8 → 3.21.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,521 @@
1
+ // pr-cost.mjs: booking, the per-cell figures, and the guard's exit contract,
2
+ // over synthetic member-outcomes / tier-outcomes / ticket-features TSVs and a
3
+ // stub `gh` that answers the one `gh pr list` from a fixture file.
4
+ import { test } from "node:test";
5
+ import assert from "node:assert/strict";
6
+ import { spawnSync } from "node:child_process";
7
+ import { existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs";
8
+ import { join } from "node:path";
9
+ import { fileURLToPath } from "node:url";
10
+ import { tempDir } from "./temp-dir.mjs";
11
+ import { writeExecStub } from "./exec-stub.mjs";
12
+ import { gitEnv } from "./git-env.mjs";
13
+ import { COLUMNS as MEMBER_COLUMNS, formatTsv, parseTsv as parseMemberTsv } from "./member-outcomes.mjs";
14
+ import { COLUMNS as TIER_COLUMNS, formatRow, parseTierOutcomes } from "./tier-outcomes.mjs";
15
+ import {
16
+ computeReport, parseFeatures, trips, guardFile, formatReport, crossCheck,
17
+ FEATURE_COLUMNS, WINDOW_START, MIN_N,
18
+ } from "./pr-cost.mjs";
19
+ import { readCostGuard, routerRows } from "./fleet-tick.mjs";
20
+
21
+ const SCRIPT = fileURLToPath(new URL("./pr-cost.mjs", import.meta.url));
22
+ const DAY = WINDOW_START;
23
+ const SESSION = "2026-10-03T10-00-00-000Z_01a00000-0000-7000-8000-000000000000";
24
+
25
+ // One world: member rows, tier rows, Pulls and PR states, built up by helpers
26
+ // and rendered into the three files the CLI reads.
27
+ function world() {
28
+ return { members: [], tiers: [], features: [], prs: [] };
29
+ }
30
+ const member = (o) => ({
31
+ session: SESSION, run_date: DAY, role: "implementer", member: o.agent, model: "claude-opus-5",
32
+ effort: "high", ticket: "", pr: "", tokensCacheCreate: 0, tokensOut: 0, wallS: 0, turns: 1,
33
+ harness: "omp", subagentType: "", tokensIn: 0, tokensCacheRead: 0, tokensCacheWrite1h: 0, cost: 0, ...o,
34
+ });
35
+ const pull = (o) => ({
36
+ run_date: DAY, session: SESSION, policy_cell: "slow-high", exploration_draw: "", sizing_src: "rule",
37
+ sizing_pre: "", router_usd: "", brief_chars: "100", criteria: "1", comments: "0", age_days: "0",
38
+ paths: "0", test_paths: "0", xrefs: "0", kind: "enhancement", ...o,
39
+ });
40
+ const tier = (o) => ({
41
+ run_date: DAY, class: "", tier: "", closed_own_ticket: "yes", minted_false_claim: "no", note: "n",
42
+ sizing: "light", profile: "small", loc: "10", files: "1", ...o,
43
+ });
44
+
45
+ // A whole ruled PR: one Pull at `cell` costing `usd`, its verdict, its state.
46
+ function addPr(w, { ticket, pr, cell, usd, fail = false, state = "MERGED", agent = `impl-${ticket}` }) {
47
+ const level = cell.split("-")[1];
48
+ w.features.push(pull({ ticket: String(ticket), agent, chosen_cell: cell }));
49
+ w.members.push(member({ agent, cost: usd, effort: level, subagentType: `fleet-implementer-${cell}`, pr: String(pr) }));
50
+ w.tiers.push(tier({ pr: String(pr), ticket: String(ticket), minted_false_claim: fail ? "yes" : "no" }));
51
+ w.prs.push({ number: pr, state, mergedAt: state === "MERGED" ? `${DAY}T12:00:00Z` : null });
52
+ }
53
+ // `n` merged PRs at `cell`, `failures` of them failing the floor, each costing `usd`.
54
+ let nextTicket = 1000;
55
+ function addCell(w, cell, n, failures, usd) {
56
+ for (let i = 0; i < n; i++) {
57
+ const t = nextTicket++;
58
+ addPr(w, { ticket: t, pr: t + 5000, cell, usd, fail: i < failures });
59
+ }
60
+ }
61
+
62
+ const parsed = (w) => ({
63
+ members: parseMemberTsv(formatTsv(w.members)),
64
+ tiers: parseTierOutcomes(w.tiers.map(formatRow).join("\n")),
65
+ features: parseFeatures(featuresText(w)),
66
+ prs: w.prs,
67
+ });
68
+ const featuresText = (w) => [FEATURE_COLUMNS.join("\t"), ...w.features.map((f) => FEATURE_COLUMNS.map((c) => f[c] ?? "").join("\t"))].join("\n") + "\n";
69
+
70
+ // `at(dir)` may replace the cwd and argv, and add env, for a run that needs the scratch dir's own paths.
71
+ function runCli(w, args = ["--guard"], { features, at } = {}) {
72
+ const dir = tempDir("pr-cost-");
73
+ mkdirSync(join(dir, "docs", "metrics"), { recursive: true });
74
+ mkdirSync(join(dir, "bin"));
75
+ writeFileSync(join(dir, "docs", "metrics", "member-outcomes.tsv"), `# ${MEMBER_COLUMNS.join("\t")}\n${formatTsv(w.members)}`);
76
+ writeFileSync(join(dir, "docs", "metrics", "tier-outcomes.tsv"), `# ${TIER_COLUMNS.join("\t")}\n${w.tiers.map(formatRow).join("\n")}\n`);
77
+ writeFileSync(join(dir, "docs", "metrics", "ticket-features.tsv"), features ?? featuresText(w));
78
+ writeFileSync(join(dir, "prs.json"), JSON.stringify(w.prs));
79
+ writeExecStub(join(dir, "bin", "gh"), `#!/bin/sh\n[ "$1 $2" = "pr list" ] || exit 9\nprintf '%s\\n' "$@" > "$FIXTURE_PRS.args"\nprintf '%s' "\${GIT_DIR-unset}" > "$FIXTURE_PRS.gitdir"\ncat "$FIXTURE_PRS"\n`);
80
+ const run = at?.(dir) ?? { cwd: dir, args };
81
+ const r = spawnSync(process.execPath, [SCRIPT, ...run.args], {
82
+ cwd: run.cwd, encoding: "utf8",
83
+ env: { ...process.env, PATH: `${join(dir, "bin")}:${process.env.PATH}`, FIXTURE_PRS: join(dir, "prs.json"), ...run.env },
84
+ });
85
+ const guardPath = join(dir, ".fleet", "cost-guard.json");
86
+ const argsPath = join(dir, "prs.json.args");
87
+ return {
88
+ ...r,
89
+ guard: existsSync(guardPath) ? JSON.parse(readFileSync(guardPath, "utf8")) : null,
90
+ ghArgs: existsSync(argsPath) ? readFileSync(argsPath, "utf8").trim().split("\n") : null,
91
+ ghGitDir: existsSync(join(dir, "prs.json.gitdir")) ? readFileSync(join(dir, "prs.json.gitdir"), "utf8") : null,
92
+ guardPath,
93
+ dir,
94
+ };
95
+ }
96
+
97
+ // ---------------------------------------------------------------------------
98
+ // The guard's exit contract.
99
+
100
+ test("--guard: a cell exactly 15 points worse than the baseline trips, exits 3 and is named in tripped[]", () => {
101
+ const w = world();
102
+ addCell(w, "slow-high", MIN_N, 6, 10); // 30% fail, $10
103
+ addCell(w, "task-high", MIN_N, 9, 5); // 45% fail: +15 points, exactly the margin
104
+ addCell(w, "smol-high", MIN_N, 8, 3); // 40% fail: +10 points, under it
105
+ const r = runCli(w);
106
+ assert.equal(r.status, 3, r.stderr);
107
+ assert.deepEqual(r.guard.tripped, ["task-high"]);
108
+ assert.equal(r.guard.verdict, "tripped");
109
+ assert.deepEqual(r.guard.baseline, { cell: "slow-high", n: 20, mean_usd: 10, fail_rate: 0.3 });
110
+ assert.equal(r.guard.min_n, MIN_N);
111
+ // The file is the router row's input: fleet-tick reads it as a verdict.
112
+ const read = readCostGuard(r.guardPath);
113
+ assert.equal(read.status, "ok");
114
+ assert.deepEqual(routerRows({ router: read }).map((x) => [x.action, x.detail]),
115
+ [["DEFAULT-ONLY", `cost guard: task-high $5.00 vs $10.00, fail 45% vs 30%, n=20/20; guard computed ${r.guard.computed_at}`]]);
116
+ assert.deepEqual(r.guard.cells.find((c) => c.cell === "task-high"), { cell: "task-high", n: 20, mean_usd: 5, fail_rate: 0.45 });
117
+ assert.equal(r.guard.window_start, WINDOW_START);
118
+ assert.ok(!Number.isNaN(Date.parse(r.guard.computed_at)));
119
+ assert.match(r.stdout, /^task-high\t20\t20\t11\t0\.45\t5\t5\t0\t[^\t]*\ttripped$/m);
120
+ assert.match(r.stdout, /^smol-high\t.*\tok$/m);
121
+ });
122
+
123
+ test("--guard: a cell at least as dear as the baseline trips on $ alone, at an equal fail rate", () => {
124
+ const w = world();
125
+ addCell(w, "slow-high", MIN_N, 4, 10);
126
+ addCell(w, "task-high", MIN_N, 4, 10);
127
+ const r = runCli(w);
128
+ assert.equal(r.status, 3, r.stderr);
129
+ assert.deepEqual(r.guard.tripped, ["task-high"]);
130
+ });
131
+
132
+ test("--guard: the $ leg compares unrounded means, so a cell cheaper by under half a cent does not trip on rounding", () => {
133
+ const w = world();
134
+ addCell(w, "slow-high", MIN_N, 4, 10.004);
135
+ addCell(w, "task-high", MIN_N, 4, 10.001);
136
+ const r = runCli(w);
137
+ assert.equal(r.status, 0, r.stderr);
138
+ assert.deepEqual(r.guard.cells.map((c) => c.mean_usd), [10, 10], "both print as $10.00");
139
+ assert.deepEqual(r.guard.tripped, []);
140
+ });
141
+
142
+ test("--guard: every cell within the margin and cheaper exits 0, verdict ok", () => {
143
+ const w = world();
144
+ addCell(w, "slow-high", MIN_N, 6, 10);
145
+ addCell(w, "task-high", MIN_N, 8, 5);
146
+ const r = runCli(w);
147
+ assert.equal(r.status, 0, r.stderr);
148
+ assert.deepEqual(r.guard.tripped, []);
149
+ assert.equal(r.guard.verdict, "ok");
150
+ });
151
+
152
+ test("--guard: a baseline under n=20 is no verdict, exit 4, and judges no cell however bad", () => {
153
+ const w = world();
154
+ addCell(w, "slow-high", MIN_N - 1, 0, 10);
155
+ addCell(w, "task-high", MIN_N, MIN_N, 50);
156
+ const r = runCli(w);
157
+ assert.equal(r.status, 4, r.stderr);
158
+ assert.equal(r.guard.verdict, "none");
159
+ assert.deepEqual(r.guard.tripped, []);
160
+ assert.equal(r.guard.baseline.n, 19);
161
+ assert.match(r.stdout, /^# verdict: none \(baseline slow-high n=19\/20\)/m);
162
+ });
163
+
164
+ test("--guard: a cell under n=20 is never tripped, however bad, while the baseline has a verdict", () => {
165
+ const w = world();
166
+ addCell(w, "slow-high", MIN_N, 0, 10);
167
+ addCell(w, "smol-high", MIN_N - 1, MIN_N - 1, 50);
168
+ const r = runCli(w);
169
+ assert.equal(r.status, 0, r.stderr);
170
+ assert.deepEqual(r.guard.tripped, []);
171
+ });
172
+
173
+ test("--guard: an unreadable input is exit 2 and writes no guard file", () => {
174
+ const w = world();
175
+ addCell(w, "slow-high", MIN_N, 0, 10);
176
+ const bad = runCli(w, ["--guard"], { features: `${FEATURE_COLUMNS.join("\t")}\n${DAY}\tshort-row\n` });
177
+ assert.equal(bad.status, 2);
178
+ assert.match(bad.stderr, /ticket-features\.tsv row 1: 2 fields, expected 18/);
179
+ assert.equal(bad.guard, null);
180
+ const noCell = runCli(w, ["--guard"], { features: featuresText(w).replace("\tslow-high\t\t", "\tfast-high\t\t") });
181
+ assert.equal(noCell.status, 2);
182
+ assert.match(noCell.stderr, /chosen_cell "fast-high" is not a cell/);
183
+ });
184
+
185
+ test("--guard: a missing ticket-features.tsv is exit 2 naming its producer, and writes no guard file", () => {
186
+ const w = world();
187
+ addCell(w, "slow-high", MIN_N, 0, 10);
188
+ const r = runCli(w, null, {
189
+ at: (dir) => {
190
+ rmSync(join(dir, "docs", "metrics", "ticket-features.tsv"));
191
+ return { cwd: dir, args: ["--guard"] };
192
+ },
193
+ });
194
+ assert.equal(r.status, 2);
195
+ assert.match(r.stderr, /cannot read docs\/metrics\/ticket-features\.tsv: ENOENT — the router script writes it at dispatch/);
196
+ assert.equal(r.guard, null);
197
+ });
198
+
199
+ test("--guard: a gh pr list page at its cap is refused rather than read as complete", () => {
200
+ const w = world();
201
+ addCell(w, "slow-high", 1, 0, 10);
202
+ for (let i = w.prs.length; i < 1000; i++) w.prs.push({ number: 90000 + i, state: "CLOSED", mergedAt: null });
203
+ const r = runCli(w);
204
+ assert.equal(r.status, 2);
205
+ assert.match(r.stderr, /returned 1000 PRs, its cap/);
206
+ });
207
+
208
+ test("--guard: an explicit --pricing that does not exist is exit 2, while an absent default is simply no cross-check", () => {
209
+ const w = world();
210
+ addCell(w, "slow-high", 1, 0, 10);
211
+ const named = runCli(w, null, { at: (dir) => ({ cwd: dir, args: ["--guard", "--pricing", join(dir, "nope.json")] }) });
212
+ assert.equal(named.status, 2);
213
+ assert.match(named.stderr, /cannot read .*nope\.json: ENOENT/);
214
+ assert.equal(named.guard, null);
215
+ const absent = runCli(w, ["--json"]);
216
+ assert.equal(absent.status, 0, absent.stderr);
217
+ assert.equal(JSON.parse(absent.stdout).cross_check, null);
218
+ const present = runCli(w, null, {
219
+ at: (dir) => {
220
+ writeFileSync(join(dir, "p.json"), JSON.stringify({ models: {} }));
221
+ return { cwd: dir, args: ["--json", "--pricing", join(dir, "p.json")] };
222
+ },
223
+ });
224
+ assert.deepEqual(JSON.parse(present.stdout).cross_check, { rows: 0, skipped: 1, ratio: null });
225
+ });
226
+
227
+ test("--guard: with no --out the file lands in the main workspace, where fleet-tick reads it, even from a linked worktree", () => {
228
+ const w = world();
229
+ addCell(w, "slow-high", MIN_N - 1, 0, 10);
230
+ const r = runCli(w, null, {
231
+ at: (dir) => {
232
+ const git = (cwd, ...a) => {
233
+ const g = spawnSync("git", ["-c", "user.email=t@t", "-c", "user.name=t", "-c", "commit.gpgsign=false", ...a], { cwd, encoding: "utf8", env: gitEnv() });
234
+ assert.equal(g.status, 0, g.stderr);
235
+ };
236
+ const main = join(dir, "main");
237
+ mkdirSync(main);
238
+ git(main, "init", "-q");
239
+ git(main, "commit", "-q", "--allow-empty", "-m", "x");
240
+ git(main, "worktree", "add", "-q", join(dir, "wt"), "-b", "side");
241
+ const m = join(dir, "docs", "metrics");
242
+ return {
243
+ cwd: join(dir, "wt"),
244
+ args: ["--guard", "--member-outcomes", join(m, "member-outcomes.tsv"), "--tier-outcomes", join(m, "tier-outcomes.tsv"),
245
+ "--ticket-features", join(m, "ticket-features.tsv")],
246
+ };
247
+ },
248
+ });
249
+ assert.equal(r.status, 4, r.stderr);
250
+ const file = join(r.dir, "main", ".fleet", "cost-guard.json");
251
+ assert.equal(readCostGuard(file).status, "ok", "the tick's reader accepts the file where it looks for it");
252
+ assert.equal(existsSync(join(r.dir, "wt", ".fleet")), false, "nothing is written under the worktree's own cwd");
253
+ });
254
+
255
+ test("an ambient GIT_DIR naming another repository cannot move the guard file out of the repository pr-cost runs in", () => {
256
+ const w = world();
257
+ addCell(w, "slow-high", MIN_N - 1, 0, 10);
258
+ let other;
259
+ const git = (cwd, env, ...a) => spawnSync("git", ["-c", "user.email=t@t", "-c", "user.name=t", "-c", "commit.gpgsign=false", ...a], { cwd, encoding: "utf8", env });
260
+ const r = runCli(w, null, {
261
+ at: (dir) => {
262
+ const here = join(dir, "here");
263
+ other = join(dir, "other");
264
+ for (const repo of [here, other]) {
265
+ mkdirSync(repo);
266
+ assert.equal(git(repo, gitEnv(), "init", "-q").status, 0);
267
+ assert.equal(git(repo, gitEnv(), "commit", "-q", "--allow-empty", "-m", "x").status, 0);
268
+ }
269
+ const m = join(dir, "docs", "metrics");
270
+ return {
271
+ cwd: here, env: { GIT_DIR: join(other, ".git") },
272
+ args: ["--guard", "--member-outcomes", join(m, "member-outcomes.tsv"), "--tier-outcomes", join(m, "tier-outcomes.tsv"),
273
+ "--ticket-features", join(m, "ticket-features.tsv")],
274
+ };
275
+ },
276
+ });
277
+ assert.equal(r.status, 4, r.stderr);
278
+ // The injection reaches a child: an unscrubbed git, in the same cwd, answers for the other repository.
279
+ const unscrubbed = git(join(r.dir, "here"), { ...process.env, GIT_DIR: join(other, ".git") }, "rev-parse", "--git-common-dir");
280
+ assert.equal(unscrubbed.stdout.trim(), join(other, ".git"));
281
+ assert.equal(existsSync(join(r.dir, "here", ".fleet", "cost-guard.json")), true);
282
+ assert.equal(existsSync(join(other, ".fleet")), false);
283
+ });
284
+
285
+ test("an ambient GIT_DIR does not reach the gh child that reads merged state", () => {
286
+ const w = world();
287
+ addCell(w, "slow-high", 1, 0, 10);
288
+ const r = runCli(w, ["--json"], { at: (dir) => ({ cwd: dir, args: ["--json"], env: { GIT_DIR: join(dir, "elsewhere", ".git") } }) });
289
+ assert.equal(r.status, 0, r.stderr);
290
+ assert.equal(r.ghGitDir, "unset");
291
+ // The stub does echo an injected GIT_DIR back when it is not scrubbed.
292
+ const probe = spawnSync(join(r.dir, "bin", "gh"), ["pr", "list"], { env: { ...process.env, GIT_DIR: "probe", FIXTURE_PRS: join(r.dir, "prs.json") } });
293
+ assert.equal(probe.status, 0);
294
+ assert.equal(readFileSync(join(r.dir, "prs.json.gitdir"), "utf8"), "probe");
295
+ });
296
+
297
+ test("without --guard the report prints and exits 0, writing nothing", () => {
298
+ const w = world();
299
+ addCell(w, "slow-high", MIN_N, 0, 10);
300
+ addCell(w, "task-high", MIN_N, MIN_N, 50);
301
+ const r = runCli(w, []);
302
+ assert.equal(r.status, 0, r.stderr);
303
+ assert.equal(r.guard, null);
304
+ // One read, every state, scoped to PRs created since the window opened.
305
+ assert.deepEqual(r.ghArgs, ["pr", "list", "--state", "all", "--search", `created:>=${WINDOW_START}`,
306
+ "--limit", "1000", "--json", "number,state,mergedAt"]);
307
+ assert.match(r.stdout, /^cell\tn_pulls\tn_merged\tn_pass\tfail_rate\tmean_usd\tmedian_usd\trouter_usd\tusd_diff_ci95\tguard$/m);
308
+ const json = JSON.parse(runCli(w, ["--json"]).stdout);
309
+ assert.deepEqual(json.tripped, ["task-high"]);
310
+ });
311
+
312
+ test("the retire condition holds once every non-default stage-1 cell has tripped, and only then", () => {
313
+ const w = world();
314
+ addCell(w, "slow-high", MIN_N, 0, 10);
315
+ addCell(w, "task-high", MIN_N, 0, 11);
316
+ const one = computeReport(parsed(w));
317
+ assert.deepEqual([one.tripped, one.retire], [["task-high"], false]);
318
+ addCell(w, "smol-high", MIN_N, 5, 1);
319
+ const both = computeReport(parsed(w));
320
+ assert.deepEqual([both.tripped, both.retire], [["smol-high", "task-high"], true]);
321
+ assert.equal(guardFile(both, "t").retire, true);
322
+ // A stratum that already adopted a non-default cell keeps the router.
323
+ assert.equal(computeReport({ ...parsed(w), routerTable: { rows: { "*": "slow-high", light: "task-high" } } }).retire, false);
324
+ });
325
+
326
+ test("trips() takes the 15-point margin in counts, so 3 of 20 is exactly the margin", () => {
327
+ const base = { n: 20, n_pass: 16, mean_usd: 10 };
328
+ assert.equal(trips({ n: 20, n_pass: 13, mean_usd: 1 }, base), true, "7 of 20 failing vs 4 of 20 is +15 points");
329
+ assert.equal(trips({ n: 20, n_pass: 14, mean_usd: 1 }, base), false, "6 of 20 vs 4 of 20 is +10 points");
330
+ assert.equal(trips({ n: 40, n_pass: 26, mean_usd: 1 }, base), true, "35% vs 20% across different n");
331
+ assert.equal(trips({ n: 20, n_pass: 20, mean_usd: 10 }, base), true, "equal $ trips");
332
+ assert.equal(trips({ n: 20, n_pass: 20, mean_usd: 9.99 }, base), false);
333
+ });
334
+
335
+ // ---------------------------------------------------------------------------
336
+ // Booking.
337
+
338
+ test("a PR carries every attempt for its ticket, its PR-named members and their nested members; merge-bot and memory carry nothing", () => {
339
+ const w = world();
340
+ const T = 41, P = 141;
341
+ // A superseded attempt in another cell, then the verdict-carrying one.
342
+ w.features.push(pull({ ticket: String(T), agent: `impl-${T}`, chosen_cell: "task-high" }));
343
+ w.members.push(member({ agent: `impl-${T}`, cost: 2, effort: "high", subagentType: "fleet-implementer-task-high" }));
344
+ w.features.push(pull({ ticket: String(T), agent: `impl-${T}-b`, chosen_cell: "slow-high", router_usd: "0.0002" }));
345
+ w.members.push(member({ agent: `impl-${T}-b`, cost: 3, effort: "high", subagentType: "fleet-implementer-slow-high" }));
346
+ w.members.push(member({ agent: `impl-${T}-b/Probe`, member: `impl-${T}-b/Probe`, cost: 0.5 }));
347
+ for (const [agent, cost] of [[`review-pr-${P}`, 1], [`fix-pr-${P}`, 1.25], [`finisher-pr-${P}`, 0.25],
348
+ [`reviewcorrectnesspr${P}`, 0.75], [`verifycorrectnesspr${P}-2`, 0.125], [`snapshotpr${P}`, 0.0625], [`test-runpr${P}`, 0.0625],
349
+ [`review-pr-${P}/Helper`, 0.25], ["merge-bot-3", 100], ["memory", 100], ["__advisor", 100], ["MergeBot4", 100],
350
+ // Excluded by its own name even where an ancestor is booked — nested as
351
+ // `<parent>/<name>` and as omp writes it, `<parent>/<parent>.<name>`.
352
+ [`impl-${T}-b/__advisor`, 100], [`review-pr-${P}/memory`, 100], [`review-pr-${P}/mergebot-2`, 100],
353
+ [`impl-${T}-b/impl-${T}-b.__advisor`, 100], [`review-pr-${P}/review-pr-${P}.memory`, 100],
354
+ [`review-pr-${P}/review-pr-${P}.merge-bot-2`, 100]]) {
355
+ w.members.push(member({ agent, cost, role: "reviewer" }));
356
+ }
357
+ w.tiers.push(tier({ pr: String(P), ticket: String(T) }));
358
+ w.prs.push({ number: P, state: "MERGED" });
359
+ const rep = computeReport(parsed(w));
360
+ const slow = rep.cells.find((c) => c.cell === "slow-high");
361
+ assert.equal(slow.n_merged, 1);
362
+ assert.equal(slow.mean_usd, 2 + 3 + 0.5 + 1 + 1.25 + 0.25 + 0.75 + 0.125 + 0.0625 + 0.0625 + 0.25);
363
+ assert.equal(slow.router_usd, 0.0002);
364
+ // The superseded attempt counts as a Pull of its own cell, its $ on the PR's.
365
+ assert.equal(rep.cells.find((c) => c.cell === "task-high").n_pulls, 1);
366
+ assert.equal(rep.cells.find((c) => c.cell === "task-high").mean_usd, null);
367
+ });
368
+
369
+ test("unruled and closed spend lands on its cell's mean, an open PR is pending, and a cell mismatch is excluded from every cell", () => {
370
+ const w = world();
371
+ addPr(w, { ticket: 1, pr: 101, cell: "slow-high", usd: 10 });
372
+ addPr(w, { ticket: 2, pr: 102, cell: "slow-high", usd: 4, state: "CLOSED" });
373
+ addPr(w, { ticket: 3, pr: 103, cell: "slow-high", usd: 99, state: "OPEN" });
374
+ // A Pull with no ruling yet, whose implementer opened nothing.
375
+ w.features.push(pull({ ticket: "4", agent: "impl-4", chosen_cell: "slow-high" }));
376
+ w.members.push(member({ agent: "impl-4", cost: 6, subagentType: "fleet-implementer-slow-high" }));
377
+ // A Pull with no ruling whose implementer opened a PR still open.
378
+ w.features.push(pull({ ticket: "5", agent: "impl-5", chosen_cell: "slow-high" }));
379
+ w.members.push(member({ agent: "impl-5", cost: 50, pr: "105", subagentType: "fleet-implementer-slow-high" }));
380
+ w.prs.push({ number: 105, state: "OPEN" });
381
+ // Booked as smol-high, but its member ran a different definition.
382
+ addPr(w, { ticket: 6, pr: 106, cell: "smol-high", usd: 1 });
383
+ w.members.at(-1).subagentType = "fleet-implementer-slow-high";
384
+ // A ticket before the window is not a Pull of this guard.
385
+ w.features.push(pull({ ticket: "7", agent: "impl-7", chosen_cell: "slow-high", run_date: "2026-09-01" }));
386
+ w.members.push(member({ agent: "impl-7", cost: 1000 }));
387
+
388
+ const rep = computeReport(parsed(w));
389
+ const slow = rep.cells.find((c) => c.cell === "slow-high");
390
+ assert.deepEqual({ n_pulls: slow.n_pulls, n_merged: slow.n_merged, mean_usd: slow.mean_usd, median_usd: slow.median_usd },
391
+ { n_pulls: 3, n_merged: 1, mean_usd: 20, median_usd: 10 });
392
+ assert.deepEqual(rep.pending, ["103", "105"]);
393
+ assert.deepEqual(rep.mismatch, [{ pr: "106", cell: "smol-high", subagent_type: "fleet-implementer-slow-high", effort: "high" }]);
394
+ assert.equal(rep.cells.find((c) => c.cell === "smol-high"), undefined);
395
+ assert.match(formatReport(rep), /^# mismatch \(excluded from every cell\): PR#106 smol-high vs fleet-implementer-slow-high\/high$/m);
396
+ });
397
+
398
+ test("a blank cost books 0 and is counted unpriced; Claude rows are outside the instrument", () => {
399
+ const w = world();
400
+ addPr(w, { ticket: 1, pr: 101, cell: "slow-high", usd: "" });
401
+ w.members.push(member({ agent: "review-pr-101", cost: 2 }));
402
+ w.members.push(member({ agent: "fix-pr-101", cost: 7, harness: "claude" }));
403
+ const rep = computeReport(parsed(w));
404
+ assert.equal(rep.unpriced, 1);
405
+ assert.equal(rep.cells[0].mean_usd, 2);
406
+ });
407
+
408
+ test("a Pull with no member row books $0 and is listed as unbooked, so its cell's mean never reads cheap unflagged", () => {
409
+ const w = world();
410
+ addPr(w, { ticket: 1, pr: 101, cell: "slow-high", usd: 10 });
411
+ addPr(w, { ticket: 2, pr: 102, cell: "slow-high", usd: 6 });
412
+ const lost = w.members.pop();
413
+ assert.equal(lost.agent, "impl-2");
414
+ const rep = computeReport(parsed(w));
415
+ assert.deepEqual(rep.unbooked_pulls, ["impl-2"]);
416
+ assert.equal(rep.cells.find((c) => c.cell === "slow-high").mean_usd, 5);
417
+ assert.match(formatReport(rep), /^# unbooked Pulls \(no member row, so \$0 in the mean\): impl-2$/m);
418
+ w.members.push(lost);
419
+ const whole = computeReport(parsed(w));
420
+ assert.deepEqual(whole.unbooked_pulls, []);
421
+ assert.doesNotMatch(formatReport(whole), /unbooked/);
422
+ });
423
+
424
+ test("the A/B report says B has not run until a B Pull carries a sizing_pre", () => {
425
+ const w = world();
426
+ addCell(w, "slow-high", 3, 0, 10);
427
+ assert.deepEqual(computeReport(parsed(w)).ab,
428
+ { status: "not-run", reason: "B not run — the table has no row a better classifier could change" });
429
+ w.features.find((f) => Number(f.ticket) % 2 === 1).sizing_pre = "light";
430
+ assert.equal(computeReport(parsed(w)).ab.status, "insufficient");
431
+ });
432
+
433
+ test("the A/B report is underpowered until B disagrees with the policy on MIN_N Pulls, and not at MIN_N", () => {
434
+ const build = (agreeing) => {
435
+ const w = world();
436
+ addCell(w, "task-high", 2 * MIN_N, 0, 10);
437
+ for (const f of w.features) if (Number(f.ticket) % 2 === 1) f.sizing_pre = "light";
438
+ // `policy_cell` is slow-high on every Pull; make `agreeing` of the B Pulls agree with it.
439
+ for (const f of w.features.filter((f) => Number(f.ticket) % 2 === 1).slice(0, agreeing)) f.policy_cell = f.chosen_cell;
440
+ return computeReport(parsed(w)).ab;
441
+ };
442
+ const under = build(1);
443
+ assert.equal(under.disagreement, (MIN_N - 1) / MIN_N);
444
+ assert.equal(under.status, "underpowered");
445
+ const enough = build(0);
446
+ assert.equal(enough.disagreement, 1);
447
+ assert.notEqual(enough.status, "underpowered");
448
+ });
449
+
450
+ test("trips() draws the 15-point line at counts that are not multiples of 5", () => {
451
+ const base = { n: 20, n_pass: 20, mean_usd: 10 };
452
+ assert.equal(trips({ n: 50, n_pass: 43, mean_usd: 1 }, base), false, "7 of 50 failing is +14 points");
453
+ assert.equal(trips({ n: 100, n_pass: 85, mean_usd: 1 }, base), true, "15 of 100 failing is +15 points");
454
+ });
455
+
456
+ test("the table labels a non-baseline cell under MIN_N as n<MIN_N rather than ok", () => {
457
+ const w = world();
458
+ addCell(w, "slow-high", MIN_N, 0, 10);
459
+ addCell(w, "smol-high", MIN_N - 1, 0, 5);
460
+ assert.match(formatReport(computeReport(parsed(w))), new RegExp(`^smol-high\\t.*\\tn<${MIN_N}$`, "m"));
461
+ });
462
+
463
+ test("the cross-check reports recorded cost against pricing.json as a ratio, never a $", () => {
464
+ const w = world();
465
+ addPr(w, { ticket: 1, pr: 101, cell: "slow-high", usd: 0.03 });
466
+ Object.assign(w.members[0], { tokensIn: 1000, tokensOut: 1000, tokensCacheRead: 0, tokensCacheCreate: 0, tokensCacheWrite1h: 0 });
467
+ const pricing = { models: { "claude-opus-5": { input: 5, output: 25, cache_read: 0.5, cache_write_5m: 6.25, cache_write_1h: 10 } } };
468
+ assert.deepEqual(computeReport({ ...parsed(w), pricing }).cross_check, { rows: 1, skipped: 0, ratio: 1 });
469
+ assert.deepEqual(computeReport({ ...parsed(w), pricing: { models: {} } }).cross_check, { rows: 0, skipped: 1, ratio: null });
470
+ });
471
+
472
+ test("the cross-check skips a blank cost and an unpriced token kind, and prices a 1h cache write apart from the 5m remainder", () => {
473
+ const pricing = { models: { m: { input: 5, output: 25, cache_read: 0.5, cache_write_5m: 6.25, cache_write_1h: 10 } } };
474
+ const row = (o) => parseMemberTsv(formatTsv([member({ model: "m", cost: 0, ...o })]))[0];
475
+ // Blank cost: no figure of record, so nothing to compare.
476
+ assert.deepEqual(crossCheck([row({ cost: "", tokensIn: 1000 })], pricing), { rows: 0, skipped: 1, ratio: null });
477
+ // A token kind the model has no price for cannot be priced as 0.
478
+ assert.deepEqual(crossCheck([row({ cost: 1, tokensCacheRead: 100 })], { models: { m: { input: 5, output: 25 } } }),
479
+ { rows: 0, skipped: 1, ratio: null });
480
+ // A cache write with no recorded 1h split is unpriceable.
481
+ assert.deepEqual(crossCheck([row({ cost: 1, tokensCacheCreate: 100, tokensCacheWrite1h: "" })], pricing),
482
+ { rows: 0, skipped: 1, ratio: null });
483
+ // 100 of the 300 cache-write tokens are 1h; the other 200 are priced 5m, none twice.
484
+ const exact = (100 * 10 + 200 * 6.25) / 1e6;
485
+ assert.deepEqual(crossCheck([row({ cost: exact, tokensCacheCreate: 300, tokensCacheWrite1h: 100 })], pricing),
486
+ { rows: 1, skipped: 0, ratio: 1 });
487
+ });
488
+
489
+ test("a ticket's ruling is its LAST tier row: a re-ruling on a later PR decides the cell and n_pass", () => {
490
+ const w = world();
491
+ addPr(w, { ticket: 1, pr: 101, cell: "slow-high", usd: 10, fail: true });
492
+ w.tiers.push(tier({ pr: "102", ticket: "1" }));
493
+ w.prs.push({ number: 102, state: "MERGED" });
494
+ const slow = computeReport(parsed(w)).cells.find((c) => c.cell === "slow-high");
495
+ assert.deepEqual({ n_merged: slow.n_merged, n_pass: slow.n_pass }, { n_merged: 1, n_pass: 1 });
496
+ });
497
+
498
+ test("two tickets ruled by one PR: the later-dated ruling carries the quality verdict, whichever ticket is read first", () => {
499
+ const w = world();
500
+ for (const t of [1, 2, 3, 4]) {
501
+ w.features.push(pull({ ticket: String(t), agent: `impl-${t}`, chosen_cell: "slow-high" }));
502
+ w.members.push(member({ agent: `impl-${t}`, cost: 1, effort: "high", subagentType: "fleet-implementer-slow-high" }));
503
+ }
504
+ const LATER = "2026-10-04";
505
+ // PR 101: the earlier ticket's ruling is the early failure, the later ticket's the late pass.
506
+ w.tiers.push(tier({ pr: "101", ticket: "1", minted_false_claim: "yes" }), tier({ pr: "101", ticket: "2", run_date: LATER }));
507
+ // PR 102: the earlier ticket's ruling is the late pass, the later ticket's the early failure.
508
+ w.tiers.push(tier({ pr: "102", ticket: "3", run_date: LATER }), tier({ pr: "102", ticket: "4", minted_false_claim: "yes" }));
509
+ w.prs.push({ number: 101, state: "MERGED" }, { number: 102, state: "MERGED" });
510
+ const slow = computeReport(parsed(w)).cells.find((c) => c.cell === "slow-high");
511
+ assert.deepEqual({ n_merged: slow.n_merged, n_pass: slow.n_pass }, { n_merged: 2, n_pass: 2 });
512
+ });
513
+
514
+ test("a Pull whose member ran at another effort than its cell's level is a mismatch even when its subagent_type matches", () => {
515
+ const w = world();
516
+ addPr(w, { ticket: 1, pr: 101, cell: "slow-high", usd: 10 });
517
+ w.members.at(-1).effort = "medium";
518
+ const rep = computeReport(parsed(w));
519
+ assert.deepEqual(rep.mismatch, [{ pr: "101", cell: "slow-high", subagent_type: "fleet-implementer-slow-high", effort: "medium" }]);
520
+ assert.equal(rep.cells.find((c) => c.cell === "slow-high"), undefined);
521
+ });
@@ -0,0 +1,35 @@
1
+ // Every review fan-out dispatch books its cost to the PR it reviews: the label
2
+ // review-core.mjs dispatches under carries the PR, and parseMemberName reads it
3
+ // back, both off the label itself and off the member id omp derives from it.
4
+ import { test } from "node:test";
5
+ import assert from "node:assert/strict";
6
+ import { runReview } from "./review-core.mjs";
7
+ import { parseMemberName } from "./member-record.mjs";
8
+ import { ARGS, SNAP, pipeline, parallel, review, finding, vote, scriptedHost } from "./review-host-fixture.mjs";
9
+
10
+ // omp's member id for a labelled dispatch: every character outside
11
+ // [A-Za-z0-9_-] deleted, capped at 48, and `-<n>` appended from the second
12
+ // dispatch of the same label on.
13
+ const ompId = (label, nth = 1) => {
14
+ const id = label.replace(/[^A-Za-z0-9_-]+/g, "").slice(0, 48);
15
+ return nth === 1 ? id : `${id}-${nth}`;
16
+ };
17
+
18
+ test("every dispatch of one review is labelled with its PR, and each label books to that PR", async () => {
19
+ const { host, labels } = scriptedHost({
20
+ snapshot: [SNAP],
21
+ "review:correctness": [review([finding("critical")])],
22
+ "verify:correctness": [vote(false)],
23
+ });
24
+ await runReview({ ...host, pipeline, parallel }, ARGS);
25
+
26
+ const kinds = new Set(labels.map((l) => l.replace(/:pr\d+$/, "")));
27
+ assert.deepEqual([...kinds].sort(), ["review:correctness", "snapshot", "test-run", "verify:correctness"],
28
+ "the review no longer dispatches every kind this test books");
29
+ for (const label of labels) {
30
+ assert.match(label, new RegExp(`:pr${ARGS.pr}$`), `${label} does not carry its PR`);
31
+ for (const name of [label, ompId(label), ompId(label, 2), `review-pr-${ARGS.pr}/${ompId(label, 3)}`]) {
32
+ assert.deepEqual(parseMemberName(name), { ticket: "", pr: String(ARGS.pr) }, name);
33
+ }
34
+ }
35
+ });
@@ -82,9 +82,9 @@ const refuter = promptRenderer({
82
82
  const snapshot = promptRenderer({
83
83
  file: FILE,
84
84
  start: "`In ${worktree}, cut an immutable review snapshot",
85
- end: '{ label: "snapshot"',
85
+ end: "{ label: `snapshot${forPr}`",
86
86
  scope: ["worktree", "scratch", "runRootParent", "runRootPrefix", "pr", "harness"],
87
- what: 'review-core.mjs\'s snapshot prompt (opening `In ${worktree}, cut an immutable review snapshot`, labelled "snapshot")',
87
+ what: "review-core.mjs's snapshot prompt (opening `In ${worktree}, cut an immutable review snapshot`, labelled `snapshot:pr<N>`)",
88
88
  })("/repo/.worktrees/7-x", "/scr", "/scr/pr7", "/scr/pr7/run-", 7, "omp");
89
89
 
90
90
  // --- Part 1: the inherited cwd is named, as a tree not to write to ---------
@@ -320,7 +320,7 @@ test("cwdAuditFrom reads each of the audit line's states, and flags its own abse
320
320
  function fakeReviewHost(scopeSearched) {
321
321
  return {
322
322
  agent: async (_prompt, opts) => {
323
- if (opts.label === "snapshot")
323
+ if (opts.label === "snapshot:pr7")
324
324
  return {
325
325
  runRoot: "/scr/pr7/run-ab12",
326
326
  path: "/scr/pr7/run-ab12/snapshot-abc123",
@@ -329,8 +329,8 @@ function fakeReviewHost(scopeSearched) {
329
329
  repoVerified: true,
330
330
  testCmd: "node --test",
331
331
  };
332
- if (opts.label === "test-run") return { exitCode: 0, tests: 5, pass: 5, fail: 0 };
333
- if (opts.label === "review:correctness")
332
+ if (opts.label === "test-run:pr7") return { exitCode: 0, tests: 5, pass: 5, fail: 0 };
333
+ if (opts.label === "review:correctness:pr7")
334
334
  return {
335
335
  dimension: "correctness",
336
336
  scope_searched: scopeSearched,
@@ -59,7 +59,7 @@ test("one review with all six dimensions launches the test command exactly once,
59
59
  const { host, calls, prompts } = scriptedHost(script({ snapshot: [snap] }));
60
60
  const scripted = host.agent;
61
61
  host.agent = async (prompt, opts) => {
62
- if (opts.label !== "test-run") return scripted(prompt, opts);
62
+ if (opts.label !== `test-run:pr${ARGS.pr}`) return scripted(prompt, opts);
63
63
  calls["test-run"] = (calls["test-run"] ?? 0) + 1;
64
64
  return obeyTestRun(prompt);
65
65
  };
@@ -204,7 +204,7 @@ test("the snapshot agent is told to derive testCmd AND the schema declares it",
204
204
  // Bounded at both ends: an unbounded slice runs to EOF, where the specialist
205
205
  // and refuter prompts could satisfy the assertions below instead.
206
206
  function testRunPrompt() {
207
- return between(CODE, "`Run this repository's test command ONCE", '{ label: "test-run"', "the shared test-run prompt");
207
+ return between(CODE, "`Run this repository's test command ONCE", "{ label: `test-run${forPr}`", "the shared test-run prompt");
208
208
  }
209
209
 
210
210
  // The prompt is what the test-run agent actually obeys, so the reading rule has
@@ -700,6 +700,10 @@ export async function runReview(host, args) {
700
700
  const scratch = A.scratch || `/tmp/review-pr-${pr}`;
701
701
  const runRootParent = `${scratch}/pr${pr}`;
702
702
  const runRootPrefix = `${runRootParent}/run-`;
703
+ // Every dispatch label ends in `:pr${pr}`: the member id the harness derives
704
+ // from a label is all a transcript keeps of it, and member-record.mjs's
705
+ // parseMemberName reads the PR back out of that id to book its cost.
706
+ const forPr = `:pr${pr}`;
703
707
  const verifiersForRun = verifiersFor(A);
704
708
 
705
709
  if (!pr || !worktree) throw new Error("review-pr: args.pr and args.worktree are required");
@@ -838,7 +842,7 @@ STDOUT, copied verbatim. Only runRoot, path, head, pathVerified and repoVerified
838
842
  are ever required — diffStats, diffPath, diffLines, refHead and prHead are each
839
843
  omitted independently when their command failed, and repoError only accompanies
840
844
  a false repoVerified.`,
841
- { label: "snapshot", phase: "Snapshot", agentType: SNAPSHOT_AGENT_TYPE, schema: SNAPSHOT_SCHEMA },
845
+ { label: `snapshot${forPr}`, phase: "Snapshot", agentType: SNAPSHOT_AGENT_TYPE, schema: SNAPSHOT_SCHEMA },
842
846
  );
843
847
 
844
848
  if (snap) {
@@ -930,7 +934,7 @@ If the command hit its deadline, crashed before printing a summary, or printed
930
934
  no counts at all, omit every count and say what happened in \`error\`. Never
931
935
  write 0 for a count the log does not state: an absent count is how the caller
932
936
  learns the run produced none.`,
933
- { label: "test-run", phase: "Test run", agentType: TEST_RUN_AGENT_TYPE, schema: TEST_RUN_SCHEMA },
937
+ { label: `test-run${forPr}`, phase: "Test run", agentType: TEST_RUN_AGENT_TYPE, schema: TEST_RUN_SCHEMA },
934
938
  ).catch((e) => ({ error: `the test-run dispatch threw: ${e?.message ?? e}` }));
935
939
  const sharedRun = { ...(ran || { error: "the test-run agent returned nothing" }), command: testCmd, logPath };
936
940
  log(`shared test run: ${countsOf(sharedRun) || "no counts"} — exit ${sharedRun.exitCode ?? "(absent)"} — log ${logPath}`);
@@ -1017,7 +1021,7 @@ produce, and an omitted line reads exactly like a check never run. Three PRs
1017
1021
  reviewed from one cell left four files modified in that checkout with nothing in
1018
1022
  any payload saying so (#1433), so a path you cannot account for is still yours
1019
1023
  to name.`,
1020
- { label: `review:${d.key}`, phase: "Review", agentType: d.agentType, schema: FINDINGS_SCHEMA },
1024
+ { label: `review:${d.key}${forPr}`, phase: "Review", agentType: d.agentType, schema: FINDINGS_SCHEMA },
1021
1025
  ),
1022
1026
  (review) => !review,
1023
1027
  ),
@@ -1108,7 +1112,7 @@ git answered \`fatal: not a git repository\` — every run, clean or not: an
1108
1112
  omitted line reads exactly like a check never run, and applying a mutation is
1109
1113
  how three reviews from one cell left four files modified in that checkout
1110
1114
  (#1433).`,
1111
- { label: `verify:${d.key}`, phase: "Verify", agentType: VERIFIER_AGENT_TYPE, schema: VERDICT_SCHEMA },
1115
+ { label: `verify:${d.key}${forPr}`, phase: "Verify", agentType: VERIFIER_AGENT_TYPE, schema: VERDICT_SCHEMA },
1112
1116
  ),
1113
1117
  ),
1114
1118
  ),