@feigi/fleet-ctl 3.19.8 → 3.21.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/scripts/ambient-git-vars-mjs-prose.test.mjs +12 -0
- package/scripts/fleet-tick.mjs +65 -3
- package/scripts/fleet-tick.test.mjs +100 -1
- package/scripts/ledger-dispatch.test.mjs +34 -9
- package/scripts/ledger-grammar.mjs +10 -4
- package/scripts/ledger.mjs +6 -1
- package/scripts/member-outcomes.mjs +21 -12
- package/scripts/member-outcomes.test.mjs +68 -11
- package/scripts/member-record.mjs +43 -13
- package/scripts/member-record.test.mjs +24 -0
- package/scripts/pr-cost.mjs +493 -0
- package/scripts/pr-cost.test.mjs +521 -0
- package/scripts/review-core-booking.test.mjs +35 -0
- package/scripts/review-core-cwd-isolation.test.mjs +5 -5
- package/scripts/review-core-shared-test-run.test.mjs +1 -1
- package/scripts/review-core-testcmd.test.mjs +1 -1
- package/scripts/review-core.mjs +8 -4
- package/scripts/review-host-fixture.mjs +12 -3
- package/scripts/review-in-run-retry.test.mjs +4 -4
- package/scripts/router-table.json +24 -0
- package/scripts/select-dimensions.test.mjs +2 -2
- package/scripts/ticket-router.mjs +543 -0
- package/scripts/ticket-router.test.mjs +981 -0
- package/scripts/within-run-pair-prose.test.mjs +46 -10
- package/skills/run-team/SKILL.md +59 -32
|
@@ -0,0 +1,521 @@
|
|
|
1
|
+
// pr-cost.mjs: booking, the per-cell figures, and the guard's exit contract,
|
|
2
|
+
// over synthetic member-outcomes / tier-outcomes / ticket-features TSVs and a
|
|
3
|
+
// stub `gh` that answers the one `gh pr list` from a fixture file.
|
|
4
|
+
import { test } from "node:test";
|
|
5
|
+
import assert from "node:assert/strict";
|
|
6
|
+
import { spawnSync } from "node:child_process";
|
|
7
|
+
import { existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
|
8
|
+
import { join } from "node:path";
|
|
9
|
+
import { fileURLToPath } from "node:url";
|
|
10
|
+
import { tempDir } from "./temp-dir.mjs";
|
|
11
|
+
import { writeExecStub } from "./exec-stub.mjs";
|
|
12
|
+
import { gitEnv } from "./git-env.mjs";
|
|
13
|
+
import { COLUMNS as MEMBER_COLUMNS, formatTsv, parseTsv as parseMemberTsv } from "./member-outcomes.mjs";
|
|
14
|
+
import { COLUMNS as TIER_COLUMNS, formatRow, parseTierOutcomes } from "./tier-outcomes.mjs";
|
|
15
|
+
import {
|
|
16
|
+
computeReport, parseFeatures, trips, guardFile, formatReport, crossCheck,
|
|
17
|
+
FEATURE_COLUMNS, WINDOW_START, MIN_N,
|
|
18
|
+
} from "./pr-cost.mjs";
|
|
19
|
+
import { readCostGuard, routerRows } from "./fleet-tick.mjs";
|
|
20
|
+
|
|
21
|
+
const SCRIPT = fileURLToPath(new URL("./pr-cost.mjs", import.meta.url));
|
|
22
|
+
const DAY = WINDOW_START;
|
|
23
|
+
const SESSION = "2026-10-03T10-00-00-000Z_01a00000-0000-7000-8000-000000000000";
|
|
24
|
+
|
|
25
|
+
// One world: member rows, tier rows, Pulls and PR states, built up by helpers
|
|
26
|
+
// and rendered into the three files the CLI reads.
|
|
27
|
+
function world() {
|
|
28
|
+
return { members: [], tiers: [], features: [], prs: [] };
|
|
29
|
+
}
|
|
30
|
+
const member = (o) => ({
|
|
31
|
+
session: SESSION, run_date: DAY, role: "implementer", member: o.agent, model: "claude-opus-5",
|
|
32
|
+
effort: "high", ticket: "", pr: "", tokensCacheCreate: 0, tokensOut: 0, wallS: 0, turns: 1,
|
|
33
|
+
harness: "omp", subagentType: "", tokensIn: 0, tokensCacheRead: 0, tokensCacheWrite1h: 0, cost: 0, ...o,
|
|
34
|
+
});
|
|
35
|
+
const pull = (o) => ({
|
|
36
|
+
run_date: DAY, session: SESSION, policy_cell: "slow-high", exploration_draw: "", sizing_src: "rule",
|
|
37
|
+
sizing_pre: "", router_usd: "", brief_chars: "100", criteria: "1", comments: "0", age_days: "0",
|
|
38
|
+
paths: "0", test_paths: "0", xrefs: "0", kind: "enhancement", ...o,
|
|
39
|
+
});
|
|
40
|
+
const tier = (o) => ({
|
|
41
|
+
run_date: DAY, class: "", tier: "", closed_own_ticket: "yes", minted_false_claim: "no", note: "n",
|
|
42
|
+
sizing: "light", profile: "small", loc: "10", files: "1", ...o,
|
|
43
|
+
});
|
|
44
|
+
|
|
45
|
+
// A whole ruled PR: one Pull at `cell` costing `usd`, its verdict, its state.
|
|
46
|
+
function addPr(w, { ticket, pr, cell, usd, fail = false, state = "MERGED", agent = `impl-${ticket}` }) {
|
|
47
|
+
const level = cell.split("-")[1];
|
|
48
|
+
w.features.push(pull({ ticket: String(ticket), agent, chosen_cell: cell }));
|
|
49
|
+
w.members.push(member({ agent, cost: usd, effort: level, subagentType: `fleet-implementer-${cell}`, pr: String(pr) }));
|
|
50
|
+
w.tiers.push(tier({ pr: String(pr), ticket: String(ticket), minted_false_claim: fail ? "yes" : "no" }));
|
|
51
|
+
w.prs.push({ number: pr, state, mergedAt: state === "MERGED" ? `${DAY}T12:00:00Z` : null });
|
|
52
|
+
}
|
|
53
|
+
// `n` merged PRs at `cell`, `failures` of them failing the floor, each costing `usd`.
|
|
54
|
+
let nextTicket = 1000;
|
|
55
|
+
function addCell(w, cell, n, failures, usd) {
|
|
56
|
+
for (let i = 0; i < n; i++) {
|
|
57
|
+
const t = nextTicket++;
|
|
58
|
+
addPr(w, { ticket: t, pr: t + 5000, cell, usd, fail: i < failures });
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
const parsed = (w) => ({
|
|
63
|
+
members: parseMemberTsv(formatTsv(w.members)),
|
|
64
|
+
tiers: parseTierOutcomes(w.tiers.map(formatRow).join("\n")),
|
|
65
|
+
features: parseFeatures(featuresText(w)),
|
|
66
|
+
prs: w.prs,
|
|
67
|
+
});
|
|
68
|
+
const featuresText = (w) => [FEATURE_COLUMNS.join("\t"), ...w.features.map((f) => FEATURE_COLUMNS.map((c) => f[c] ?? "").join("\t"))].join("\n") + "\n";
|
|
69
|
+
|
|
70
|
+
// `at(dir)` may replace the cwd and argv, and add env, for a run that needs the scratch dir's own paths.
|
|
71
|
+
function runCli(w, args = ["--guard"], { features, at } = {}) {
|
|
72
|
+
const dir = tempDir("pr-cost-");
|
|
73
|
+
mkdirSync(join(dir, "docs", "metrics"), { recursive: true });
|
|
74
|
+
mkdirSync(join(dir, "bin"));
|
|
75
|
+
writeFileSync(join(dir, "docs", "metrics", "member-outcomes.tsv"), `# ${MEMBER_COLUMNS.join("\t")}\n${formatTsv(w.members)}`);
|
|
76
|
+
writeFileSync(join(dir, "docs", "metrics", "tier-outcomes.tsv"), `# ${TIER_COLUMNS.join("\t")}\n${w.tiers.map(formatRow).join("\n")}\n`);
|
|
77
|
+
writeFileSync(join(dir, "docs", "metrics", "ticket-features.tsv"), features ?? featuresText(w));
|
|
78
|
+
writeFileSync(join(dir, "prs.json"), JSON.stringify(w.prs));
|
|
79
|
+
writeExecStub(join(dir, "bin", "gh"), `#!/bin/sh\n[ "$1 $2" = "pr list" ] || exit 9\nprintf '%s\\n' "$@" > "$FIXTURE_PRS.args"\nprintf '%s' "\${GIT_DIR-unset}" > "$FIXTURE_PRS.gitdir"\ncat "$FIXTURE_PRS"\n`);
|
|
80
|
+
const run = at?.(dir) ?? { cwd: dir, args };
|
|
81
|
+
const r = spawnSync(process.execPath, [SCRIPT, ...run.args], {
|
|
82
|
+
cwd: run.cwd, encoding: "utf8",
|
|
83
|
+
env: { ...process.env, PATH: `${join(dir, "bin")}:${process.env.PATH}`, FIXTURE_PRS: join(dir, "prs.json"), ...run.env },
|
|
84
|
+
});
|
|
85
|
+
const guardPath = join(dir, ".fleet", "cost-guard.json");
|
|
86
|
+
const argsPath = join(dir, "prs.json.args");
|
|
87
|
+
return {
|
|
88
|
+
...r,
|
|
89
|
+
guard: existsSync(guardPath) ? JSON.parse(readFileSync(guardPath, "utf8")) : null,
|
|
90
|
+
ghArgs: existsSync(argsPath) ? readFileSync(argsPath, "utf8").trim().split("\n") : null,
|
|
91
|
+
ghGitDir: existsSync(join(dir, "prs.json.gitdir")) ? readFileSync(join(dir, "prs.json.gitdir"), "utf8") : null,
|
|
92
|
+
guardPath,
|
|
93
|
+
dir,
|
|
94
|
+
};
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
// ---------------------------------------------------------------------------
|
|
98
|
+
// The guard's exit contract.
|
|
99
|
+
|
|
100
|
+
test("--guard: a cell exactly 15 points worse than the baseline trips, exits 3 and is named in tripped[]", () => {
|
|
101
|
+
const w = world();
|
|
102
|
+
addCell(w, "slow-high", MIN_N, 6, 10); // 30% fail, $10
|
|
103
|
+
addCell(w, "task-high", MIN_N, 9, 5); // 45% fail: +15 points, exactly the margin
|
|
104
|
+
addCell(w, "smol-high", MIN_N, 8, 3); // 40% fail: +10 points, under it
|
|
105
|
+
const r = runCli(w);
|
|
106
|
+
assert.equal(r.status, 3, r.stderr);
|
|
107
|
+
assert.deepEqual(r.guard.tripped, ["task-high"]);
|
|
108
|
+
assert.equal(r.guard.verdict, "tripped");
|
|
109
|
+
assert.deepEqual(r.guard.baseline, { cell: "slow-high", n: 20, mean_usd: 10, fail_rate: 0.3 });
|
|
110
|
+
assert.equal(r.guard.min_n, MIN_N);
|
|
111
|
+
// The file is the router row's input: fleet-tick reads it as a verdict.
|
|
112
|
+
const read = readCostGuard(r.guardPath);
|
|
113
|
+
assert.equal(read.status, "ok");
|
|
114
|
+
assert.deepEqual(routerRows({ router: read }).map((x) => [x.action, x.detail]),
|
|
115
|
+
[["DEFAULT-ONLY", `cost guard: task-high $5.00 vs $10.00, fail 45% vs 30%, n=20/20; guard computed ${r.guard.computed_at}`]]);
|
|
116
|
+
assert.deepEqual(r.guard.cells.find((c) => c.cell === "task-high"), { cell: "task-high", n: 20, mean_usd: 5, fail_rate: 0.45 });
|
|
117
|
+
assert.equal(r.guard.window_start, WINDOW_START);
|
|
118
|
+
assert.ok(!Number.isNaN(Date.parse(r.guard.computed_at)));
|
|
119
|
+
assert.match(r.stdout, /^task-high\t20\t20\t11\t0\.45\t5\t5\t0\t[^\t]*\ttripped$/m);
|
|
120
|
+
assert.match(r.stdout, /^smol-high\t.*\tok$/m);
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
test("--guard: a cell at least as dear as the baseline trips on $ alone, at an equal fail rate", () => {
|
|
124
|
+
const w = world();
|
|
125
|
+
addCell(w, "slow-high", MIN_N, 4, 10);
|
|
126
|
+
addCell(w, "task-high", MIN_N, 4, 10);
|
|
127
|
+
const r = runCli(w);
|
|
128
|
+
assert.equal(r.status, 3, r.stderr);
|
|
129
|
+
assert.deepEqual(r.guard.tripped, ["task-high"]);
|
|
130
|
+
});
|
|
131
|
+
|
|
132
|
+
test("--guard: the $ leg compares unrounded means, so a cell cheaper by under half a cent does not trip on rounding", () => {
|
|
133
|
+
const w = world();
|
|
134
|
+
addCell(w, "slow-high", MIN_N, 4, 10.004);
|
|
135
|
+
addCell(w, "task-high", MIN_N, 4, 10.001);
|
|
136
|
+
const r = runCli(w);
|
|
137
|
+
assert.equal(r.status, 0, r.stderr);
|
|
138
|
+
assert.deepEqual(r.guard.cells.map((c) => c.mean_usd), [10, 10], "both print as $10.00");
|
|
139
|
+
assert.deepEqual(r.guard.tripped, []);
|
|
140
|
+
});
|
|
141
|
+
|
|
142
|
+
test("--guard: every cell within the margin and cheaper exits 0, verdict ok", () => {
|
|
143
|
+
const w = world();
|
|
144
|
+
addCell(w, "slow-high", MIN_N, 6, 10);
|
|
145
|
+
addCell(w, "task-high", MIN_N, 8, 5);
|
|
146
|
+
const r = runCli(w);
|
|
147
|
+
assert.equal(r.status, 0, r.stderr);
|
|
148
|
+
assert.deepEqual(r.guard.tripped, []);
|
|
149
|
+
assert.equal(r.guard.verdict, "ok");
|
|
150
|
+
});
|
|
151
|
+
|
|
152
|
+
test("--guard: a baseline under n=20 is no verdict, exit 4, and judges no cell however bad", () => {
|
|
153
|
+
const w = world();
|
|
154
|
+
addCell(w, "slow-high", MIN_N - 1, 0, 10);
|
|
155
|
+
addCell(w, "task-high", MIN_N, MIN_N, 50);
|
|
156
|
+
const r = runCli(w);
|
|
157
|
+
assert.equal(r.status, 4, r.stderr);
|
|
158
|
+
assert.equal(r.guard.verdict, "none");
|
|
159
|
+
assert.deepEqual(r.guard.tripped, []);
|
|
160
|
+
assert.equal(r.guard.baseline.n, 19);
|
|
161
|
+
assert.match(r.stdout, /^# verdict: none \(baseline slow-high n=19\/20\)/m);
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
test("--guard: a cell under n=20 is never tripped, however bad, while the baseline has a verdict", () => {
|
|
165
|
+
const w = world();
|
|
166
|
+
addCell(w, "slow-high", MIN_N, 0, 10);
|
|
167
|
+
addCell(w, "smol-high", MIN_N - 1, MIN_N - 1, 50);
|
|
168
|
+
const r = runCli(w);
|
|
169
|
+
assert.equal(r.status, 0, r.stderr);
|
|
170
|
+
assert.deepEqual(r.guard.tripped, []);
|
|
171
|
+
});
|
|
172
|
+
|
|
173
|
+
test("--guard: an unreadable input is exit 2 and writes no guard file", () => {
|
|
174
|
+
const w = world();
|
|
175
|
+
addCell(w, "slow-high", MIN_N, 0, 10);
|
|
176
|
+
const bad = runCli(w, ["--guard"], { features: `${FEATURE_COLUMNS.join("\t")}\n${DAY}\tshort-row\n` });
|
|
177
|
+
assert.equal(bad.status, 2);
|
|
178
|
+
assert.match(bad.stderr, /ticket-features\.tsv row 1: 2 fields, expected 18/);
|
|
179
|
+
assert.equal(bad.guard, null);
|
|
180
|
+
const noCell = runCli(w, ["--guard"], { features: featuresText(w).replace("\tslow-high\t\t", "\tfast-high\t\t") });
|
|
181
|
+
assert.equal(noCell.status, 2);
|
|
182
|
+
assert.match(noCell.stderr, /chosen_cell "fast-high" is not a cell/);
|
|
183
|
+
});
|
|
184
|
+
|
|
185
|
+
test("--guard: a missing ticket-features.tsv is exit 2 naming its producer, and writes no guard file", () => {
|
|
186
|
+
const w = world();
|
|
187
|
+
addCell(w, "slow-high", MIN_N, 0, 10);
|
|
188
|
+
const r = runCli(w, null, {
|
|
189
|
+
at: (dir) => {
|
|
190
|
+
rmSync(join(dir, "docs", "metrics", "ticket-features.tsv"));
|
|
191
|
+
return { cwd: dir, args: ["--guard"] };
|
|
192
|
+
},
|
|
193
|
+
});
|
|
194
|
+
assert.equal(r.status, 2);
|
|
195
|
+
assert.match(r.stderr, /cannot read docs\/metrics\/ticket-features\.tsv: ENOENT — the router script writes it at dispatch/);
|
|
196
|
+
assert.equal(r.guard, null);
|
|
197
|
+
});
|
|
198
|
+
|
|
199
|
+
test("--guard: a gh pr list page at its cap is refused rather than read as complete", () => {
|
|
200
|
+
const w = world();
|
|
201
|
+
addCell(w, "slow-high", 1, 0, 10);
|
|
202
|
+
for (let i = w.prs.length; i < 1000; i++) w.prs.push({ number: 90000 + i, state: "CLOSED", mergedAt: null });
|
|
203
|
+
const r = runCli(w);
|
|
204
|
+
assert.equal(r.status, 2);
|
|
205
|
+
assert.match(r.stderr, /returned 1000 PRs, its cap/);
|
|
206
|
+
});
|
|
207
|
+
|
|
208
|
+
test("--guard: an explicit --pricing that does not exist is exit 2, while an absent default is simply no cross-check", () => {
|
|
209
|
+
const w = world();
|
|
210
|
+
addCell(w, "slow-high", 1, 0, 10);
|
|
211
|
+
const named = runCli(w, null, { at: (dir) => ({ cwd: dir, args: ["--guard", "--pricing", join(dir, "nope.json")] }) });
|
|
212
|
+
assert.equal(named.status, 2);
|
|
213
|
+
assert.match(named.stderr, /cannot read .*nope\.json: ENOENT/);
|
|
214
|
+
assert.equal(named.guard, null);
|
|
215
|
+
const absent = runCli(w, ["--json"]);
|
|
216
|
+
assert.equal(absent.status, 0, absent.stderr);
|
|
217
|
+
assert.equal(JSON.parse(absent.stdout).cross_check, null);
|
|
218
|
+
const present = runCli(w, null, {
|
|
219
|
+
at: (dir) => {
|
|
220
|
+
writeFileSync(join(dir, "p.json"), JSON.stringify({ models: {} }));
|
|
221
|
+
return { cwd: dir, args: ["--json", "--pricing", join(dir, "p.json")] };
|
|
222
|
+
},
|
|
223
|
+
});
|
|
224
|
+
assert.deepEqual(JSON.parse(present.stdout).cross_check, { rows: 0, skipped: 1, ratio: null });
|
|
225
|
+
});
|
|
226
|
+
|
|
227
|
+
test("--guard: with no --out the file lands in the main workspace, where fleet-tick reads it, even from a linked worktree", () => {
|
|
228
|
+
const w = world();
|
|
229
|
+
addCell(w, "slow-high", MIN_N - 1, 0, 10);
|
|
230
|
+
const r = runCli(w, null, {
|
|
231
|
+
at: (dir) => {
|
|
232
|
+
const git = (cwd, ...a) => {
|
|
233
|
+
const g = spawnSync("git", ["-c", "user.email=t@t", "-c", "user.name=t", "-c", "commit.gpgsign=false", ...a], { cwd, encoding: "utf8", env: gitEnv() });
|
|
234
|
+
assert.equal(g.status, 0, g.stderr);
|
|
235
|
+
};
|
|
236
|
+
const main = join(dir, "main");
|
|
237
|
+
mkdirSync(main);
|
|
238
|
+
git(main, "init", "-q");
|
|
239
|
+
git(main, "commit", "-q", "--allow-empty", "-m", "x");
|
|
240
|
+
git(main, "worktree", "add", "-q", join(dir, "wt"), "-b", "side");
|
|
241
|
+
const m = join(dir, "docs", "metrics");
|
|
242
|
+
return {
|
|
243
|
+
cwd: join(dir, "wt"),
|
|
244
|
+
args: ["--guard", "--member-outcomes", join(m, "member-outcomes.tsv"), "--tier-outcomes", join(m, "tier-outcomes.tsv"),
|
|
245
|
+
"--ticket-features", join(m, "ticket-features.tsv")],
|
|
246
|
+
};
|
|
247
|
+
},
|
|
248
|
+
});
|
|
249
|
+
assert.equal(r.status, 4, r.stderr);
|
|
250
|
+
const file = join(r.dir, "main", ".fleet", "cost-guard.json");
|
|
251
|
+
assert.equal(readCostGuard(file).status, "ok", "the tick's reader accepts the file where it looks for it");
|
|
252
|
+
assert.equal(existsSync(join(r.dir, "wt", ".fleet")), false, "nothing is written under the worktree's own cwd");
|
|
253
|
+
});
|
|
254
|
+
|
|
255
|
+
test("an ambient GIT_DIR naming another repository cannot move the guard file out of the repository pr-cost runs in", () => {
|
|
256
|
+
const w = world();
|
|
257
|
+
addCell(w, "slow-high", MIN_N - 1, 0, 10);
|
|
258
|
+
let other;
|
|
259
|
+
const git = (cwd, env, ...a) => spawnSync("git", ["-c", "user.email=t@t", "-c", "user.name=t", "-c", "commit.gpgsign=false", ...a], { cwd, encoding: "utf8", env });
|
|
260
|
+
const r = runCli(w, null, {
|
|
261
|
+
at: (dir) => {
|
|
262
|
+
const here = join(dir, "here");
|
|
263
|
+
other = join(dir, "other");
|
|
264
|
+
for (const repo of [here, other]) {
|
|
265
|
+
mkdirSync(repo);
|
|
266
|
+
assert.equal(git(repo, gitEnv(), "init", "-q").status, 0);
|
|
267
|
+
assert.equal(git(repo, gitEnv(), "commit", "-q", "--allow-empty", "-m", "x").status, 0);
|
|
268
|
+
}
|
|
269
|
+
const m = join(dir, "docs", "metrics");
|
|
270
|
+
return {
|
|
271
|
+
cwd: here, env: { GIT_DIR: join(other, ".git") },
|
|
272
|
+
args: ["--guard", "--member-outcomes", join(m, "member-outcomes.tsv"), "--tier-outcomes", join(m, "tier-outcomes.tsv"),
|
|
273
|
+
"--ticket-features", join(m, "ticket-features.tsv")],
|
|
274
|
+
};
|
|
275
|
+
},
|
|
276
|
+
});
|
|
277
|
+
assert.equal(r.status, 4, r.stderr);
|
|
278
|
+
// The injection reaches a child: an unscrubbed git, in the same cwd, answers for the other repository.
|
|
279
|
+
const unscrubbed = git(join(r.dir, "here"), { ...process.env, GIT_DIR: join(other, ".git") }, "rev-parse", "--git-common-dir");
|
|
280
|
+
assert.equal(unscrubbed.stdout.trim(), join(other, ".git"));
|
|
281
|
+
assert.equal(existsSync(join(r.dir, "here", ".fleet", "cost-guard.json")), true);
|
|
282
|
+
assert.equal(existsSync(join(other, ".fleet")), false);
|
|
283
|
+
});
|
|
284
|
+
|
|
285
|
+
test("an ambient GIT_DIR does not reach the gh child that reads merged state", () => {
|
|
286
|
+
const w = world();
|
|
287
|
+
addCell(w, "slow-high", 1, 0, 10);
|
|
288
|
+
const r = runCli(w, ["--json"], { at: (dir) => ({ cwd: dir, args: ["--json"], env: { GIT_DIR: join(dir, "elsewhere", ".git") } }) });
|
|
289
|
+
assert.equal(r.status, 0, r.stderr);
|
|
290
|
+
assert.equal(r.ghGitDir, "unset");
|
|
291
|
+
// The stub does echo an injected GIT_DIR back when it is not scrubbed.
|
|
292
|
+
const probe = spawnSync(join(r.dir, "bin", "gh"), ["pr", "list"], { env: { ...process.env, GIT_DIR: "probe", FIXTURE_PRS: join(r.dir, "prs.json") } });
|
|
293
|
+
assert.equal(probe.status, 0);
|
|
294
|
+
assert.equal(readFileSync(join(r.dir, "prs.json.gitdir"), "utf8"), "probe");
|
|
295
|
+
});
|
|
296
|
+
|
|
297
|
+
test("without --guard the report prints and exits 0, writing nothing", () => {
|
|
298
|
+
const w = world();
|
|
299
|
+
addCell(w, "slow-high", MIN_N, 0, 10);
|
|
300
|
+
addCell(w, "task-high", MIN_N, MIN_N, 50);
|
|
301
|
+
const r = runCli(w, []);
|
|
302
|
+
assert.equal(r.status, 0, r.stderr);
|
|
303
|
+
assert.equal(r.guard, null);
|
|
304
|
+
// One read, every state, scoped to PRs created since the window opened.
|
|
305
|
+
assert.deepEqual(r.ghArgs, ["pr", "list", "--state", "all", "--search", `created:>=${WINDOW_START}`,
|
|
306
|
+
"--limit", "1000", "--json", "number,state,mergedAt"]);
|
|
307
|
+
assert.match(r.stdout, /^cell\tn_pulls\tn_merged\tn_pass\tfail_rate\tmean_usd\tmedian_usd\trouter_usd\tusd_diff_ci95\tguard$/m);
|
|
308
|
+
const json = JSON.parse(runCli(w, ["--json"]).stdout);
|
|
309
|
+
assert.deepEqual(json.tripped, ["task-high"]);
|
|
310
|
+
});
|
|
311
|
+
|
|
312
|
+
test("the retire condition holds once every non-default stage-1 cell has tripped, and only then", () => {
|
|
313
|
+
const w = world();
|
|
314
|
+
addCell(w, "slow-high", MIN_N, 0, 10);
|
|
315
|
+
addCell(w, "task-high", MIN_N, 0, 11);
|
|
316
|
+
const one = computeReport(parsed(w));
|
|
317
|
+
assert.deepEqual([one.tripped, one.retire], [["task-high"], false]);
|
|
318
|
+
addCell(w, "smol-high", MIN_N, 5, 1);
|
|
319
|
+
const both = computeReport(parsed(w));
|
|
320
|
+
assert.deepEqual([both.tripped, both.retire], [["smol-high", "task-high"], true]);
|
|
321
|
+
assert.equal(guardFile(both, "t").retire, true);
|
|
322
|
+
// A stratum that already adopted a non-default cell keeps the router.
|
|
323
|
+
assert.equal(computeReport({ ...parsed(w), routerTable: { rows: { "*": "slow-high", light: "task-high" } } }).retire, false);
|
|
324
|
+
});
|
|
325
|
+
|
|
326
|
+
test("trips() takes the 15-point margin in counts, so 3 of 20 is exactly the margin", () => {
|
|
327
|
+
const base = { n: 20, n_pass: 16, mean_usd: 10 };
|
|
328
|
+
assert.equal(trips({ n: 20, n_pass: 13, mean_usd: 1 }, base), true, "7 of 20 failing vs 4 of 20 is +15 points");
|
|
329
|
+
assert.equal(trips({ n: 20, n_pass: 14, mean_usd: 1 }, base), false, "6 of 20 vs 4 of 20 is +10 points");
|
|
330
|
+
assert.equal(trips({ n: 40, n_pass: 26, mean_usd: 1 }, base), true, "35% vs 20% across different n");
|
|
331
|
+
assert.equal(trips({ n: 20, n_pass: 20, mean_usd: 10 }, base), true, "equal $ trips");
|
|
332
|
+
assert.equal(trips({ n: 20, n_pass: 20, mean_usd: 9.99 }, base), false);
|
|
333
|
+
});
|
|
334
|
+
|
|
335
|
+
// ---------------------------------------------------------------------------
|
|
336
|
+
// Booking.
|
|
337
|
+
|
|
338
|
+
test("a PR carries every attempt for its ticket, its PR-named members and their nested members; merge-bot and memory carry nothing", () => {
|
|
339
|
+
const w = world();
|
|
340
|
+
const T = 41, P = 141;
|
|
341
|
+
// A superseded attempt in another cell, then the verdict-carrying one.
|
|
342
|
+
w.features.push(pull({ ticket: String(T), agent: `impl-${T}`, chosen_cell: "task-high" }));
|
|
343
|
+
w.members.push(member({ agent: `impl-${T}`, cost: 2, effort: "high", subagentType: "fleet-implementer-task-high" }));
|
|
344
|
+
w.features.push(pull({ ticket: String(T), agent: `impl-${T}-b`, chosen_cell: "slow-high", router_usd: "0.0002" }));
|
|
345
|
+
w.members.push(member({ agent: `impl-${T}-b`, cost: 3, effort: "high", subagentType: "fleet-implementer-slow-high" }));
|
|
346
|
+
w.members.push(member({ agent: `impl-${T}-b/Probe`, member: `impl-${T}-b/Probe`, cost: 0.5 }));
|
|
347
|
+
for (const [agent, cost] of [[`review-pr-${P}`, 1], [`fix-pr-${P}`, 1.25], [`finisher-pr-${P}`, 0.25],
|
|
348
|
+
[`reviewcorrectnesspr${P}`, 0.75], [`verifycorrectnesspr${P}-2`, 0.125], [`snapshotpr${P}`, 0.0625], [`test-runpr${P}`, 0.0625],
|
|
349
|
+
[`review-pr-${P}/Helper`, 0.25], ["merge-bot-3", 100], ["memory", 100], ["__advisor", 100], ["MergeBot4", 100],
|
|
350
|
+
// Excluded by its own name even where an ancestor is booked — nested as
|
|
351
|
+
// `<parent>/<name>` and as omp writes it, `<parent>/<parent>.<name>`.
|
|
352
|
+
[`impl-${T}-b/__advisor`, 100], [`review-pr-${P}/memory`, 100], [`review-pr-${P}/mergebot-2`, 100],
|
|
353
|
+
[`impl-${T}-b/impl-${T}-b.__advisor`, 100], [`review-pr-${P}/review-pr-${P}.memory`, 100],
|
|
354
|
+
[`review-pr-${P}/review-pr-${P}.merge-bot-2`, 100]]) {
|
|
355
|
+
w.members.push(member({ agent, cost, role: "reviewer" }));
|
|
356
|
+
}
|
|
357
|
+
w.tiers.push(tier({ pr: String(P), ticket: String(T) }));
|
|
358
|
+
w.prs.push({ number: P, state: "MERGED" });
|
|
359
|
+
const rep = computeReport(parsed(w));
|
|
360
|
+
const slow = rep.cells.find((c) => c.cell === "slow-high");
|
|
361
|
+
assert.equal(slow.n_merged, 1);
|
|
362
|
+
assert.equal(slow.mean_usd, 2 + 3 + 0.5 + 1 + 1.25 + 0.25 + 0.75 + 0.125 + 0.0625 + 0.0625 + 0.25);
|
|
363
|
+
assert.equal(slow.router_usd, 0.0002);
|
|
364
|
+
// The superseded attempt counts as a Pull of its own cell, its $ on the PR's.
|
|
365
|
+
assert.equal(rep.cells.find((c) => c.cell === "task-high").n_pulls, 1);
|
|
366
|
+
assert.equal(rep.cells.find((c) => c.cell === "task-high").mean_usd, null);
|
|
367
|
+
});
|
|
368
|
+
|
|
369
|
+
test("unruled and closed spend lands on its cell's mean, an open PR is pending, and a cell mismatch is excluded from every cell", () => {
|
|
370
|
+
const w = world();
|
|
371
|
+
addPr(w, { ticket: 1, pr: 101, cell: "slow-high", usd: 10 });
|
|
372
|
+
addPr(w, { ticket: 2, pr: 102, cell: "slow-high", usd: 4, state: "CLOSED" });
|
|
373
|
+
addPr(w, { ticket: 3, pr: 103, cell: "slow-high", usd: 99, state: "OPEN" });
|
|
374
|
+
// A Pull with no ruling yet, whose implementer opened nothing.
|
|
375
|
+
w.features.push(pull({ ticket: "4", agent: "impl-4", chosen_cell: "slow-high" }));
|
|
376
|
+
w.members.push(member({ agent: "impl-4", cost: 6, subagentType: "fleet-implementer-slow-high" }));
|
|
377
|
+
// A Pull with no ruling whose implementer opened a PR still open.
|
|
378
|
+
w.features.push(pull({ ticket: "5", agent: "impl-5", chosen_cell: "slow-high" }));
|
|
379
|
+
w.members.push(member({ agent: "impl-5", cost: 50, pr: "105", subagentType: "fleet-implementer-slow-high" }));
|
|
380
|
+
w.prs.push({ number: 105, state: "OPEN" });
|
|
381
|
+
// Booked as smol-high, but its member ran a different definition.
|
|
382
|
+
addPr(w, { ticket: 6, pr: 106, cell: "smol-high", usd: 1 });
|
|
383
|
+
w.members.at(-1).subagentType = "fleet-implementer-slow-high";
|
|
384
|
+
// A ticket before the window is not a Pull of this guard.
|
|
385
|
+
w.features.push(pull({ ticket: "7", agent: "impl-7", chosen_cell: "slow-high", run_date: "2026-09-01" }));
|
|
386
|
+
w.members.push(member({ agent: "impl-7", cost: 1000 }));
|
|
387
|
+
|
|
388
|
+
const rep = computeReport(parsed(w));
|
|
389
|
+
const slow = rep.cells.find((c) => c.cell === "slow-high");
|
|
390
|
+
assert.deepEqual({ n_pulls: slow.n_pulls, n_merged: slow.n_merged, mean_usd: slow.mean_usd, median_usd: slow.median_usd },
|
|
391
|
+
{ n_pulls: 3, n_merged: 1, mean_usd: 20, median_usd: 10 });
|
|
392
|
+
assert.deepEqual(rep.pending, ["103", "105"]);
|
|
393
|
+
assert.deepEqual(rep.mismatch, [{ pr: "106", cell: "smol-high", subagent_type: "fleet-implementer-slow-high", effort: "high" }]);
|
|
394
|
+
assert.equal(rep.cells.find((c) => c.cell === "smol-high"), undefined);
|
|
395
|
+
assert.match(formatReport(rep), /^# mismatch \(excluded from every cell\): PR#106 smol-high vs fleet-implementer-slow-high\/high$/m);
|
|
396
|
+
});
|
|
397
|
+
|
|
398
|
+
test("a blank cost books 0 and is counted unpriced; Claude rows are outside the instrument", () => {
|
|
399
|
+
const w = world();
|
|
400
|
+
addPr(w, { ticket: 1, pr: 101, cell: "slow-high", usd: "" });
|
|
401
|
+
w.members.push(member({ agent: "review-pr-101", cost: 2 }));
|
|
402
|
+
w.members.push(member({ agent: "fix-pr-101", cost: 7, harness: "claude" }));
|
|
403
|
+
const rep = computeReport(parsed(w));
|
|
404
|
+
assert.equal(rep.unpriced, 1);
|
|
405
|
+
assert.equal(rep.cells[0].mean_usd, 2);
|
|
406
|
+
});
|
|
407
|
+
|
|
408
|
+
test("a Pull with no member row books $0 and is listed as unbooked, so its cell's mean never reads cheap unflagged", () => {
|
|
409
|
+
const w = world();
|
|
410
|
+
addPr(w, { ticket: 1, pr: 101, cell: "slow-high", usd: 10 });
|
|
411
|
+
addPr(w, { ticket: 2, pr: 102, cell: "slow-high", usd: 6 });
|
|
412
|
+
const lost = w.members.pop();
|
|
413
|
+
assert.equal(lost.agent, "impl-2");
|
|
414
|
+
const rep = computeReport(parsed(w));
|
|
415
|
+
assert.deepEqual(rep.unbooked_pulls, ["impl-2"]);
|
|
416
|
+
assert.equal(rep.cells.find((c) => c.cell === "slow-high").mean_usd, 5);
|
|
417
|
+
assert.match(formatReport(rep), /^# unbooked Pulls \(no member row, so \$0 in the mean\): impl-2$/m);
|
|
418
|
+
w.members.push(lost);
|
|
419
|
+
const whole = computeReport(parsed(w));
|
|
420
|
+
assert.deepEqual(whole.unbooked_pulls, []);
|
|
421
|
+
assert.doesNotMatch(formatReport(whole), /unbooked/);
|
|
422
|
+
});
|
|
423
|
+
|
|
424
|
+
test("the A/B report says B has not run until a B Pull carries a sizing_pre", () => {
|
|
425
|
+
const w = world();
|
|
426
|
+
addCell(w, "slow-high", 3, 0, 10);
|
|
427
|
+
assert.deepEqual(computeReport(parsed(w)).ab,
|
|
428
|
+
{ status: "not-run", reason: "B not run — the table has no row a better classifier could change" });
|
|
429
|
+
w.features.find((f) => Number(f.ticket) % 2 === 1).sizing_pre = "light";
|
|
430
|
+
assert.equal(computeReport(parsed(w)).ab.status, "insufficient");
|
|
431
|
+
});
|
|
432
|
+
|
|
433
|
+
test("the A/B report is underpowered until B disagrees with the policy on MIN_N Pulls, and not at MIN_N", () => {
|
|
434
|
+
const build = (agreeing) => {
|
|
435
|
+
const w = world();
|
|
436
|
+
addCell(w, "task-high", 2 * MIN_N, 0, 10);
|
|
437
|
+
for (const f of w.features) if (Number(f.ticket) % 2 === 1) f.sizing_pre = "light";
|
|
438
|
+
// `policy_cell` is slow-high on every Pull; make `agreeing` of the B Pulls agree with it.
|
|
439
|
+
for (const f of w.features.filter((f) => Number(f.ticket) % 2 === 1).slice(0, agreeing)) f.policy_cell = f.chosen_cell;
|
|
440
|
+
return computeReport(parsed(w)).ab;
|
|
441
|
+
};
|
|
442
|
+
const under = build(1);
|
|
443
|
+
assert.equal(under.disagreement, (MIN_N - 1) / MIN_N);
|
|
444
|
+
assert.equal(under.status, "underpowered");
|
|
445
|
+
const enough = build(0);
|
|
446
|
+
assert.equal(enough.disagreement, 1);
|
|
447
|
+
assert.notEqual(enough.status, "underpowered");
|
|
448
|
+
});
|
|
449
|
+
|
|
450
|
+
test("trips() draws the 15-point line at counts that are not multiples of 5", () => {
|
|
451
|
+
const base = { n: 20, n_pass: 20, mean_usd: 10 };
|
|
452
|
+
assert.equal(trips({ n: 50, n_pass: 43, mean_usd: 1 }, base), false, "7 of 50 failing is +14 points");
|
|
453
|
+
assert.equal(trips({ n: 100, n_pass: 85, mean_usd: 1 }, base), true, "15 of 100 failing is +15 points");
|
|
454
|
+
});
|
|
455
|
+
|
|
456
|
+
test("the table labels a non-baseline cell under MIN_N as n<MIN_N rather than ok", () => {
|
|
457
|
+
const w = world();
|
|
458
|
+
addCell(w, "slow-high", MIN_N, 0, 10);
|
|
459
|
+
addCell(w, "smol-high", MIN_N - 1, 0, 5);
|
|
460
|
+
assert.match(formatReport(computeReport(parsed(w))), new RegExp(`^smol-high\\t.*\\tn<${MIN_N}$`, "m"));
|
|
461
|
+
});
|
|
462
|
+
|
|
463
|
+
test("the cross-check reports recorded cost against pricing.json as a ratio, never a $", () => {
|
|
464
|
+
const w = world();
|
|
465
|
+
addPr(w, { ticket: 1, pr: 101, cell: "slow-high", usd: 0.03 });
|
|
466
|
+
Object.assign(w.members[0], { tokensIn: 1000, tokensOut: 1000, tokensCacheRead: 0, tokensCacheCreate: 0, tokensCacheWrite1h: 0 });
|
|
467
|
+
const pricing = { models: { "claude-opus-5": { input: 5, output: 25, cache_read: 0.5, cache_write_5m: 6.25, cache_write_1h: 10 } } };
|
|
468
|
+
assert.deepEqual(computeReport({ ...parsed(w), pricing }).cross_check, { rows: 1, skipped: 0, ratio: 1 });
|
|
469
|
+
assert.deepEqual(computeReport({ ...parsed(w), pricing: { models: {} } }).cross_check, { rows: 0, skipped: 1, ratio: null });
|
|
470
|
+
});
|
|
471
|
+
|
|
472
|
+
test("the cross-check skips a blank cost and an unpriced token kind, and prices a 1h cache write apart from the 5m remainder", () => {
|
|
473
|
+
const pricing = { models: { m: { input: 5, output: 25, cache_read: 0.5, cache_write_5m: 6.25, cache_write_1h: 10 } } };
|
|
474
|
+
const row = (o) => parseMemberTsv(formatTsv([member({ model: "m", cost: 0, ...o })]))[0];
|
|
475
|
+
// Blank cost: no figure of record, so nothing to compare.
|
|
476
|
+
assert.deepEqual(crossCheck([row({ cost: "", tokensIn: 1000 })], pricing), { rows: 0, skipped: 1, ratio: null });
|
|
477
|
+
// A token kind the model has no price for cannot be priced as 0.
|
|
478
|
+
assert.deepEqual(crossCheck([row({ cost: 1, tokensCacheRead: 100 })], { models: { m: { input: 5, output: 25 } } }),
|
|
479
|
+
{ rows: 0, skipped: 1, ratio: null });
|
|
480
|
+
// A cache write with no recorded 1h split is unpriceable.
|
|
481
|
+
assert.deepEqual(crossCheck([row({ cost: 1, tokensCacheCreate: 100, tokensCacheWrite1h: "" })], pricing),
|
|
482
|
+
{ rows: 0, skipped: 1, ratio: null });
|
|
483
|
+
// 100 of the 300 cache-write tokens are 1h; the other 200 are priced 5m, none twice.
|
|
484
|
+
const exact = (100 * 10 + 200 * 6.25) / 1e6;
|
|
485
|
+
assert.deepEqual(crossCheck([row({ cost: exact, tokensCacheCreate: 300, tokensCacheWrite1h: 100 })], pricing),
|
|
486
|
+
{ rows: 1, skipped: 0, ratio: 1 });
|
|
487
|
+
});
|
|
488
|
+
|
|
489
|
+
test("a ticket's ruling is its LAST tier row: a re-ruling on a later PR decides the cell and n_pass", () => {
|
|
490
|
+
const w = world();
|
|
491
|
+
addPr(w, { ticket: 1, pr: 101, cell: "slow-high", usd: 10, fail: true });
|
|
492
|
+
w.tiers.push(tier({ pr: "102", ticket: "1" }));
|
|
493
|
+
w.prs.push({ number: 102, state: "MERGED" });
|
|
494
|
+
const slow = computeReport(parsed(w)).cells.find((c) => c.cell === "slow-high");
|
|
495
|
+
assert.deepEqual({ n_merged: slow.n_merged, n_pass: slow.n_pass }, { n_merged: 1, n_pass: 1 });
|
|
496
|
+
});
|
|
497
|
+
|
|
498
|
+
test("two tickets ruled by one PR: the later-dated ruling carries the quality verdict, whichever ticket is read first", () => {
|
|
499
|
+
const w = world();
|
|
500
|
+
for (const t of [1, 2, 3, 4]) {
|
|
501
|
+
w.features.push(pull({ ticket: String(t), agent: `impl-${t}`, chosen_cell: "slow-high" }));
|
|
502
|
+
w.members.push(member({ agent: `impl-${t}`, cost: 1, effort: "high", subagentType: "fleet-implementer-slow-high" }));
|
|
503
|
+
}
|
|
504
|
+
const LATER = "2026-10-04";
|
|
505
|
+
// PR 101: the earlier ticket's ruling is the early failure, the later ticket's the late pass.
|
|
506
|
+
w.tiers.push(tier({ pr: "101", ticket: "1", minted_false_claim: "yes" }), tier({ pr: "101", ticket: "2", run_date: LATER }));
|
|
507
|
+
// PR 102: the earlier ticket's ruling is the late pass, the later ticket's the early failure.
|
|
508
|
+
w.tiers.push(tier({ pr: "102", ticket: "3", run_date: LATER }), tier({ pr: "102", ticket: "4", minted_false_claim: "yes" }));
|
|
509
|
+
w.prs.push({ number: 101, state: "MERGED" }, { number: 102, state: "MERGED" });
|
|
510
|
+
const slow = computeReport(parsed(w)).cells.find((c) => c.cell === "slow-high");
|
|
511
|
+
assert.deepEqual({ n_merged: slow.n_merged, n_pass: slow.n_pass }, { n_merged: 2, n_pass: 2 });
|
|
512
|
+
});
|
|
513
|
+
|
|
514
|
+
test("a Pull whose member ran at another effort than its cell's level is a mismatch even when its subagent_type matches", () => {
|
|
515
|
+
const w = world();
|
|
516
|
+
addPr(w, { ticket: 1, pr: 101, cell: "slow-high", usd: 10 });
|
|
517
|
+
w.members.at(-1).effort = "medium";
|
|
518
|
+
const rep = computeReport(parsed(w));
|
|
519
|
+
assert.deepEqual(rep.mismatch, [{ pr: "101", cell: "slow-high", subagent_type: "fleet-implementer-slow-high", effort: "medium" }]);
|
|
520
|
+
assert.equal(rep.cells.find((c) => c.cell === "slow-high"), undefined);
|
|
521
|
+
});
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
// Every review fan-out dispatch books its cost to the PR it reviews: the label
|
|
2
|
+
// review-core.mjs dispatches under carries the PR, and parseMemberName reads it
|
|
3
|
+
// back, both off the label itself and off the member id omp derives from it.
|
|
4
|
+
import { test } from "node:test";
|
|
5
|
+
import assert from "node:assert/strict";
|
|
6
|
+
import { runReview } from "./review-core.mjs";
|
|
7
|
+
import { parseMemberName } from "./member-record.mjs";
|
|
8
|
+
import { ARGS, SNAP, pipeline, parallel, review, finding, vote, scriptedHost } from "./review-host-fixture.mjs";
|
|
9
|
+
|
|
10
|
+
// omp's member id for a labelled dispatch: every character outside
|
|
11
|
+
// [A-Za-z0-9_-] deleted, capped at 48, and `-<n>` appended from the second
|
|
12
|
+
// dispatch of the same label on.
|
|
13
|
+
const ompId = (label, nth = 1) => {
|
|
14
|
+
const id = label.replace(/[^A-Za-z0-9_-]+/g, "").slice(0, 48);
|
|
15
|
+
return nth === 1 ? id : `${id}-${nth}`;
|
|
16
|
+
};
|
|
17
|
+
|
|
18
|
+
test("every dispatch of one review is labelled with its PR, and each label books to that PR", async () => {
|
|
19
|
+
const { host, labels } = scriptedHost({
|
|
20
|
+
snapshot: [SNAP],
|
|
21
|
+
"review:correctness": [review([finding("critical")])],
|
|
22
|
+
"verify:correctness": [vote(false)],
|
|
23
|
+
});
|
|
24
|
+
await runReview({ ...host, pipeline, parallel }, ARGS);
|
|
25
|
+
|
|
26
|
+
const kinds = new Set(labels.map((l) => l.replace(/:pr\d+$/, "")));
|
|
27
|
+
assert.deepEqual([...kinds].sort(), ["review:correctness", "snapshot", "test-run", "verify:correctness"],
|
|
28
|
+
"the review no longer dispatches every kind this test books");
|
|
29
|
+
for (const label of labels) {
|
|
30
|
+
assert.match(label, new RegExp(`:pr${ARGS.pr}$`), `${label} does not carry its PR`);
|
|
31
|
+
for (const name of [label, ompId(label), ompId(label, 2), `review-pr-${ARGS.pr}/${ompId(label, 3)}`]) {
|
|
32
|
+
assert.deepEqual(parseMemberName(name), { ticket: "", pr: String(ARGS.pr) }, name);
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
});
|
|
@@ -82,9 +82,9 @@ const refuter = promptRenderer({
|
|
|
82
82
|
const snapshot = promptRenderer({
|
|
83
83
|
file: FILE,
|
|
84
84
|
start: "`In ${worktree}, cut an immutable review snapshot",
|
|
85
|
-
end:
|
|
85
|
+
end: "{ label: `snapshot${forPr}`",
|
|
86
86
|
scope: ["worktree", "scratch", "runRootParent", "runRootPrefix", "pr", "harness"],
|
|
87
|
-
what:
|
|
87
|
+
what: "review-core.mjs's snapshot prompt (opening `In ${worktree}, cut an immutable review snapshot`, labelled `snapshot:pr<N>`)",
|
|
88
88
|
})("/repo/.worktrees/7-x", "/scr", "/scr/pr7", "/scr/pr7/run-", 7, "omp");
|
|
89
89
|
|
|
90
90
|
// --- Part 1: the inherited cwd is named, as a tree not to write to ---------
|
|
@@ -320,7 +320,7 @@ test("cwdAuditFrom reads each of the audit line's states, and flags its own abse
|
|
|
320
320
|
function fakeReviewHost(scopeSearched) {
|
|
321
321
|
return {
|
|
322
322
|
agent: async (_prompt, opts) => {
|
|
323
|
-
if (opts.label === "snapshot")
|
|
323
|
+
if (opts.label === "snapshot:pr7")
|
|
324
324
|
return {
|
|
325
325
|
runRoot: "/scr/pr7/run-ab12",
|
|
326
326
|
path: "/scr/pr7/run-ab12/snapshot-abc123",
|
|
@@ -329,8 +329,8 @@ function fakeReviewHost(scopeSearched) {
|
|
|
329
329
|
repoVerified: true,
|
|
330
330
|
testCmd: "node --test",
|
|
331
331
|
};
|
|
332
|
-
if (opts.label === "test-run") return { exitCode: 0, tests: 5, pass: 5, fail: 0 };
|
|
333
|
-
if (opts.label === "review:correctness")
|
|
332
|
+
if (opts.label === "test-run:pr7") return { exitCode: 0, tests: 5, pass: 5, fail: 0 };
|
|
333
|
+
if (opts.label === "review:correctness:pr7")
|
|
334
334
|
return {
|
|
335
335
|
dimension: "correctness",
|
|
336
336
|
scope_searched: scopeSearched,
|
|
@@ -59,7 +59,7 @@ test("one review with all six dimensions launches the test command exactly once,
|
|
|
59
59
|
const { host, calls, prompts } = scriptedHost(script({ snapshot: [snap] }));
|
|
60
60
|
const scripted = host.agent;
|
|
61
61
|
host.agent = async (prompt, opts) => {
|
|
62
|
-
if (opts.label !==
|
|
62
|
+
if (opts.label !== `test-run:pr${ARGS.pr}`) return scripted(prompt, opts);
|
|
63
63
|
calls["test-run"] = (calls["test-run"] ?? 0) + 1;
|
|
64
64
|
return obeyTestRun(prompt);
|
|
65
65
|
};
|
|
@@ -204,7 +204,7 @@ test("the snapshot agent is told to derive testCmd AND the schema declares it",
|
|
|
204
204
|
// Bounded at both ends: an unbounded slice runs to EOF, where the specialist
|
|
205
205
|
// and refuter prompts could satisfy the assertions below instead.
|
|
206
206
|
function testRunPrompt() {
|
|
207
|
-
return between(CODE, "`Run this repository's test command ONCE",
|
|
207
|
+
return between(CODE, "`Run this repository's test command ONCE", "{ label: `test-run${forPr}`", "the shared test-run prompt");
|
|
208
208
|
}
|
|
209
209
|
|
|
210
210
|
// The prompt is what the test-run agent actually obeys, so the reading rule has
|
package/scripts/review-core.mjs
CHANGED
|
@@ -700,6 +700,10 @@ export async function runReview(host, args) {
|
|
|
700
700
|
const scratch = A.scratch || `/tmp/review-pr-${pr}`;
|
|
701
701
|
const runRootParent = `${scratch}/pr${pr}`;
|
|
702
702
|
const runRootPrefix = `${runRootParent}/run-`;
|
|
703
|
+
// Every dispatch label ends in `:pr${pr}`: the member id the harness derives
|
|
704
|
+
// from a label is all a transcript keeps of it, and member-record.mjs's
|
|
705
|
+
// parseMemberName reads the PR back out of that id to book its cost.
|
|
706
|
+
const forPr = `:pr${pr}`;
|
|
703
707
|
const verifiersForRun = verifiersFor(A);
|
|
704
708
|
|
|
705
709
|
if (!pr || !worktree) throw new Error("review-pr: args.pr and args.worktree are required");
|
|
@@ -838,7 +842,7 @@ STDOUT, copied verbatim. Only runRoot, path, head, pathVerified and repoVerified
|
|
|
838
842
|
are ever required — diffStats, diffPath, diffLines, refHead and prHead are each
|
|
839
843
|
omitted independently when their command failed, and repoError only accompanies
|
|
840
844
|
a false repoVerified.`,
|
|
841
|
-
{ label:
|
|
845
|
+
{ label: `snapshot${forPr}`, phase: "Snapshot", agentType: SNAPSHOT_AGENT_TYPE, schema: SNAPSHOT_SCHEMA },
|
|
842
846
|
);
|
|
843
847
|
|
|
844
848
|
if (snap) {
|
|
@@ -930,7 +934,7 @@ If the command hit its deadline, crashed before printing a summary, or printed
|
|
|
930
934
|
no counts at all, omit every count and say what happened in \`error\`. Never
|
|
931
935
|
write 0 for a count the log does not state: an absent count is how the caller
|
|
932
936
|
learns the run produced none.`,
|
|
933
|
-
{ label:
|
|
937
|
+
{ label: `test-run${forPr}`, phase: "Test run", agentType: TEST_RUN_AGENT_TYPE, schema: TEST_RUN_SCHEMA },
|
|
934
938
|
).catch((e) => ({ error: `the test-run dispatch threw: ${e?.message ?? e}` }));
|
|
935
939
|
const sharedRun = { ...(ran || { error: "the test-run agent returned nothing" }), command: testCmd, logPath };
|
|
936
940
|
log(`shared test run: ${countsOf(sharedRun) || "no counts"} — exit ${sharedRun.exitCode ?? "(absent)"} — log ${logPath}`);
|
|
@@ -1017,7 +1021,7 @@ produce, and an omitted line reads exactly like a check never run. Three PRs
|
|
|
1017
1021
|
reviewed from one cell left four files modified in that checkout with nothing in
|
|
1018
1022
|
any payload saying so (#1433), so a path you cannot account for is still yours
|
|
1019
1023
|
to name.`,
|
|
1020
|
-
{ label: `review:${d.key}`, phase: "Review", agentType: d.agentType, schema: FINDINGS_SCHEMA },
|
|
1024
|
+
{ label: `review:${d.key}${forPr}`, phase: "Review", agentType: d.agentType, schema: FINDINGS_SCHEMA },
|
|
1021
1025
|
),
|
|
1022
1026
|
(review) => !review,
|
|
1023
1027
|
),
|
|
@@ -1108,7 +1112,7 @@ git answered \`fatal: not a git repository\` — every run, clean or not: an
|
|
|
1108
1112
|
omitted line reads exactly like a check never run, and applying a mutation is
|
|
1109
1113
|
how three reviews from one cell left four files modified in that checkout
|
|
1110
1114
|
(#1433).`,
|
|
1111
|
-
{ label: `verify:${d.key}`, phase: "Verify", agentType: VERIFIER_AGENT_TYPE, schema: VERDICT_SCHEMA },
|
|
1115
|
+
{ label: `verify:${d.key}${forPr}`, phase: "Verify", agentType: VERIFIER_AGENT_TYPE, schema: VERDICT_SCHEMA },
|
|
1112
1116
|
),
|
|
1113
1117
|
),
|
|
1114
1118
|
),
|