@feigi/fleet-ctl 3.21.13 → 3.22.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@feigi/fleet-ctl",
3
- "version": "3.21.13",
3
+ "version": "3.22.1",
4
4
  "description": "Agent fleet: run-team controller, merge bot, PR reviewer, and the ticket pipeline they share",
5
5
  "license": "Apache-2.0",
6
6
  "repository": {
@@ -201,8 +201,8 @@ const COVERED_MJS = {
201
201
  // children (`node candidates.mjs`, `node ledger.mjs`, `sh inflight.sh`)
202
202
  // name no git.
203
203
  "shortlist.mjs": 2,
204
- // THREE. The first two are the pair shortlist.mjs carries (#1803), the
205
- // third is below (#2485). `shortlistPath()`'s
204
+ // FOUR. The first two are the pair shortlist.mjs carries (#1803), the
205
+ // third is below (#2485), the fourth after it. `shortlistPath()`'s
206
206
  // `--git-common-dir` probe names the `.fleet/shortlist.json` the tick PULLs
207
207
  // from; an ambient GIT_DIR would read another repository's shortlist and
208
208
  // name its tickets. Measured in fleet-tick.test.mjs, "an ambient GIT_DIR
@@ -216,7 +216,11 @@ const COVERED_MJS = {
216
216
  // implementer's ticket is already closed, and resolves the repository the
217
217
  // same way; measured in fleet-tick.test.mjs, "an inherited GIT_DIR cannot
218
218
  // retarget the tier-mismatch closed-ticket probe".
219
- "fleet-tick.mjs": 3,
219
+ // `finishedReviewPrs()`'s `gh pr view` asks whether a PR with a dangling
220
+ // in-flight review is already merged or closed, and resolves the repository
221
+ // the same way; measured in fleet-tick.test.mjs, "an inherited GIT_DIR
222
+ // cannot retarget the merged-PR review probe".
223
+ "fleet-tick.mjs": 4,
220
224
  // ONE spawn primitive, `git(args, cwd)`, behind both of the file's git calls
221
225
  // — `resolveMainRoot()`'s `rev-parse --git-common-dir` and `checkIgnored()`'s
222
226
  // `check-ignore` (#1411). An ambient GIT_DIR would answer the first for
@@ -0,0 +1,148 @@
1
+ #!/usr/bin/env node
2
+ // The per-cell readout: how many within-session comparisons each implementer
3
+ // cell has against the policy cell, and whether that is enough to read it.
4
+ //
5
+ // cell-readout.mjs [--ticket-features <tsv>] [--member-outcomes <tsv>]
6
+ //
7
+ // Inputs default to docs/metrics/ under the working directory. A Pull is a
8
+ // ticket-features.tsv row; what it ran is the member-outcomes.tsv row with the
9
+ // same `session` + `agent`.
10
+ //
11
+ // ADMISSIBLE ROW. A Pull counts only when its member row exists, was
12
+ // dispatched as the drawn cell's own definition (`subagent_type` =
13
+ // `fleet-implementer-<chosen_cell>`), ran at the cell's level (`effort` = the
14
+ // cell's `<level>`) and has a resolved `model` on record. So a generic `task`
15
+ // dispatch, a clamped effort and every pre-cutover `fleet-implementer` /
16
+ // `fleet-implementer-alt` row are never admissible. Whether the resolved model
17
+ // was the role's target at dispatch is recorded in neither file; that half of
18
+ // the tier check is the run ledger's, not this readout's.
19
+ //
20
+ // COMPARISON. For a cell X other than slow-high, a comparison is one session
21
+ // holding at least one admissible X row and at least one admissible slow-high
22
+ // row whose resolved (`model`, `effort`) differ from that X row's. A session
23
+ // counts once however many rows it holds, and its date is its member-outcomes
24
+ // `run_date`; a blank `run_date` means unknown, so that session is still a
25
+ // comparison but adds no date to the distinct-date count.
26
+ //
27
+ // GATE. A cell is read only once it has at least GATE.comparisons comparisons
28
+ // across at least GATE.runDates distinct run_dates. A gated cell prints one
29
+ // line on stdout:
30
+ //
31
+ // <cell> <comparisons> <run_dates> <resolved models>
32
+ //
33
+ // `<resolved models>` is the model of every admissible row at the cell — one
34
+ // model by name, or `mixed (<model-a> n=…, <model-b> n=…)` when they span more
35
+ // than one, because a role's target is operator config that can move under a
36
+ // cell's history. A cell below the gate prints nothing on stdout and one
37
+ // `below the gate` count on stderr. Nothing on either stream names a session,
38
+ // ticket or member: the gate forbids reading a comparison early.
39
+ //
40
+ // Exit 0 whatever the counts; 2 on a usage error or an unreadable or
41
+ // malformed input, with nothing on stdout.
42
+
43
+ import { readFileSync } from "node:fs";
44
+ import { makeDie, defineFlags } from "./arg.mjs";
45
+ import { isCLI } from "./is-cli.mjs";
46
+ import { CELL, POLICY_CELL } from "./ledger-grammar.mjs";
47
+ import { parseTsv as parseMemberTsv } from "./member-outcomes.mjs";
48
+ import { parseFeatures } from "./pr-cost.mjs";
49
+
50
+ const NAME = "cell-readout";
51
+ export const GATE = Object.freeze({ comparisons: 10, runDates: 5 });
52
+
53
+ const key = (session, agent) => `${session}\0${agent}`;
54
+ const levelOf = (cell) => CELL.exec(cell)?.[2] ?? null;
55
+
56
+ /** The join's member row for a Pull when the Pull is admissible, else null. */
57
+ function admissibleMember(pull, members) {
58
+ const m = members.get(key(pull.session, pull.agent));
59
+ if (!m) return null;
60
+ const level = levelOf(pull.chosen_cell);
61
+ if (level === null || m.subagentType !== `fleet-implementer-${pull.chosen_cell}` || m.effort !== level || !m.model) return null;
62
+ return m;
63
+ }
64
+
65
+ /**
66
+ * `cells`: one entry per cell other than the policy cell that has a
67
+ * ticket-features row, `{ cell, comparisons, runDates, dates, models, gated }`,
68
+ * sorted by cell. `dates` is the comparisons' distinct run_dates, sorted;
69
+ * `models` maps each resolved model of the cell's admissible rows to its row
70
+ * count, largest first. `unjoined` counts Pulls with no member row.
71
+ */
72
+ export function readout({ features, members }) {
73
+ const byKey = new Map(members.map((m) => [key(m.session, m.agent), m]));
74
+ // One Pull per session+agent: a re-dispatch lands on the same cell and the
75
+ // same member row.
76
+ const pulls = new Map(features.map((p) => [key(p.session, p.agent), p]));
77
+ const sessions = new Map();
78
+ const cells = new Map();
79
+ let unjoined = 0;
80
+ for (const p of pulls.values()) {
81
+ if (p.chosen_cell !== POLICY_CELL && !cells.has(p.chosen_cell)) cells.set(p.chosen_cell, new Map());
82
+ if (!byKey.has(key(p.session, p.agent))) unjoined++;
83
+ const m = admissibleMember(p, byKey);
84
+ if (!m) continue;
85
+ if (p.chosen_cell !== POLICY_CELL) cells.get(p.chosen_cell).set(m.model, (cells.get(p.chosen_cell).get(m.model) ?? 0) + 1);
86
+ if (!sessions.has(p.session)) sessions.set(p.session, { runDate: m.run_date, rows: [] });
87
+ sessions.get(p.session).rows.push({ cell: p.chosen_cell, model: m.model, effort: m.effort });
88
+ }
89
+
90
+ const out = [...cells].sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0)).map(([cell, modelCounts]) => {
91
+ let comparisons = 0;
92
+ const dates = new Set();
93
+ for (const { runDate, rows } of sessions.values()) {
94
+ const base = rows.filter((r) => r.cell === POLICY_CELL);
95
+ const compared = rows.some((x) => x.cell === cell && base.some((s) => s.model !== x.model || s.effort !== x.effort));
96
+ if (!compared) continue;
97
+ comparisons++;
98
+ if (runDate) dates.add(runDate);
99
+ }
100
+ const models = new Map([...modelCounts].sort(([a, n], [b, k]) => k - n || (a < b ? -1 : a > b ? 1 : 0)));
101
+ const runDates = dates.size;
102
+ return { cell, comparisons, runDates, dates: [...dates].sort(), models, gated: comparisons >= GATE.comparisons && runDates >= GATE.runDates };
103
+ });
104
+ return { cells: out, unjoined };
105
+ }
106
+
107
+ // Only a gated cell is formatted, and a comparison needs an admissible row at
108
+ // the cell, so `models` is never empty here.
109
+ function formatModels(models) {
110
+ if (models.size === 1) return [...models.keys()][0];
111
+ return `mixed (${[...models].map(([m, n]) => `${m} n=${n}`).join(", ")})`;
112
+ }
113
+
114
+ function main() {
115
+ const die = makeDie(NAME);
116
+ const { arg, sweep, stray } = defineFlags(die, {
117
+ flags: { "ticket-features": "value", "member-outcomes": "value" },
118
+ });
119
+ sweep();
120
+ stray();
121
+ const featuresPath = arg("ticket-features") ?? "docs/metrics/ticket-features.tsv";
122
+ const membersPath = arg("member-outcomes") ?? "docs/metrics/member-outcomes.tsv";
123
+ const load = (path, parse) => {
124
+ let text;
125
+ try { text = readFileSync(path, "utf8"); }
126
+ catch (e) { die(`cannot read ${path}: ${e.code ?? e.message}`); }
127
+ try { return parse(text); }
128
+ catch (e) { die(`${path}: ${e.message}`); }
129
+ };
130
+ const features = load(featuresPath, parseFeatures);
131
+ const members = load(membersPath, parseMemberTsv);
132
+
133
+ const { cells, unjoined } = readout({ features, members });
134
+ const lines = [];
135
+ const notes = [];
136
+ if (unjoined > 0) {
137
+ notes.push(`${unjoined} ticket-features row${unjoined === 1 ? " has" : "s have"} no member-outcomes row — never admissible`);
138
+ }
139
+ if (cells.length === 0) notes.push(`no cell other than ${POLICY_CELL} has a ticket-features row`);
140
+ for (const c of cells) {
141
+ if (c.gated) lines.push(`${c.cell} ${c.comparisons} ${c.runDates} ${formatModels(c.models)}`);
142
+ else notes.push(`${c.cell} below the gate: ${c.comparisons} comparisons across ${c.runDates} run_dates (needs ${GATE.comparisons} across ${GATE.runDates})`);
143
+ }
144
+ for (const n of notes) process.stderr.write(`${NAME}: ${n}\n`);
145
+ if (lines.length) process.stdout.write(`${lines.join("\n")}\n`);
146
+ }
147
+
148
+ if (isCLI(import.meta.url)) main();
@@ -0,0 +1,341 @@
1
+ // cell-readout.mjs: the per-cell gate over ticket-features.tsv joined to
2
+ // member-outcomes.tsv, run as the CLI against synthetic fixture files.
3
+ import { test } from "node:test";
4
+ import assert from "node:assert/strict";
5
+ import { spawnSync } from "node:child_process";
6
+ import { existsSync, readFileSync, writeFileSync } from "node:fs";
7
+ import { dirname, join } from "node:path";
8
+ import { fileURLToPath } from "node:url";
9
+ import { tempDir } from "./temp-dir.mjs";
10
+ import { paragraph, phrase, unemphasized } from "./prose-pin.mjs";
11
+ import { formatTsv } from "./member-outcomes.mjs";
12
+ import { FEATURE_COLUMNS } from "./pr-cost.mjs";
13
+ import { readout, GATE } from "./cell-readout.mjs";
14
+
15
+ const SCRIPT = fileURLToPath(new URL("./cell-readout.mjs", import.meta.url));
16
+
17
+ const member = (o) => ({
18
+ run_date: "2026-10-01", role: "implementer", member: o.agent, model: "claude-opus-5", effort: "high",
19
+ ticket: "", pr: "", tokensCacheCreate: 0, tokensOut: 0, wallS: 0, turns: 1, harness: "omp",
20
+ tokensIn: 0, tokensCacheRead: 0, tokensCacheWrite1h: 0, cost: 0, ...o,
21
+ });
22
+ const pull = (o) => ({
23
+ run_date: "2026-10-01", policy_cell: "slow-high", exploration_draw: "", sizing_src: "rule", sizing_pre: "",
24
+ router_usd: "", brief_chars: "100", criteria: "1", comments: "0", age_days: "0", paths: "0",
25
+ test_paths: "0", xrefs: "0", kind: "enhancement", ...o,
26
+ });
27
+
28
+ function world() {
29
+ return { features: [], members: [], ticket: 100 };
30
+ }
31
+ // One implementer Pull at `cell` in `session`, and the member row it left.
32
+ // `model`/`effort`/`subagentType` override what resolved and what was dispatched.
33
+ function addRow(w, { session, date, cell, model, effort, subagentType, member: withMember = true }) {
34
+ const ticket = String(w.ticket++);
35
+ const agent = `impl-${ticket}`;
36
+ const level = cell.split("-").slice(1).join("-");
37
+ w.features.push(pull({ session, agent, ticket, chosen_cell: cell }));
38
+ if (withMember) {
39
+ w.members.push(member({
40
+ session, agent, run_date: date, ticket,
41
+ model: model ?? (cell.startsWith("slow-") ? "claude-opus-5" : "claude-sonnet-5"),
42
+ effort: effort ?? level, subagentType: subagentType ?? `fleet-implementer-${cell}`,
43
+ }));
44
+ }
45
+ }
46
+ // `n` sessions, each one comparison for `cell`: a `cell` row beside a slow-high
47
+ // row, spread round-robin over `dates` distinct run_dates.
48
+ let nextSession = 0;
49
+ function addComparisons(w, cell, n, dates, rowOpts = {}) {
50
+ for (let i = 0; i < n; i++) {
51
+ const session = `2026-10-01T00-00-00-000Z_s${nextSession++}`;
52
+ const date = `2026-10-${String(1 + (i % dates)).padStart(2, "0")}`;
53
+ addRow(w, { session, date, cell, ...rowOpts });
54
+ addRow(w, { session, date, cell: "slow-high" });
55
+ }
56
+ }
57
+
58
+ function files(w) {
59
+ const dir = tempDir("cell-readout-");
60
+ const features = join(dir, "ticket-features.tsv");
61
+ const members = join(dir, "member-outcomes.tsv");
62
+ writeFileSync(features, [FEATURE_COLUMNS.join("\t"), ...w.features.map((r) => FEATURE_COLUMNS.map((c) => r[c] ?? "").join("\t"))].join("\n") + "\n");
63
+ writeFileSync(members, "# header\n" + formatTsv(w.members));
64
+ return { features, members, dir };
65
+ }
66
+ function cli(w, extra = []) {
67
+ const f = files(w);
68
+ return spawnSync(process.execPath, [SCRIPT, "--ticket-features", f.features, "--member-outcomes", f.members, ...extra], { encoding: "utf8", cwd: f.dir });
69
+ }
70
+ const lineFor = (stdout, cell) => stdout.split("\n").find((l) => l.split(" ")[0] === cell);
71
+
72
+ test("the gate is ten comparisons across five run_dates: nine across five does not gate, ten across five does", () => {
73
+ const nine = world();
74
+ addComparisons(nine, "task-high", 9, 5);
75
+ const below = cli(nine);
76
+ assert.equal(below.status, 0, below.stderr);
77
+ assert.equal(lineFor(below.stdout, "task-high"), undefined, "a cell below the gate prints no line");
78
+ assert.match(below.stderr, /task-high below the gate: 9 comparisons across 5 run_dates/);
79
+
80
+ const ten = world();
81
+ addComparisons(ten, "task-high", 10, 5);
82
+ const at = cli(ten);
83
+ assert.equal(at.status, 0, at.stderr);
84
+ assert.equal(lineFor(at.stdout, "task-high"), "task-high 10 5 claude-sonnet-5");
85
+ assert.doesNotMatch(at.stderr, /task-high below the gate/);
86
+ assert.deepEqual(GATE, { comparisons: 10, runDates: 5 });
87
+ });
88
+
89
+ test("ten comparisons across four run_dates does not gate: both halves of the gate bind", () => {
90
+ const w = world();
91
+ addComparisons(w, "task-high", 12, 4);
92
+ const r = cli(w);
93
+ assert.equal(r.status, 0, r.stderr);
94
+ assert.equal(lineFor(r.stdout, "task-high"), undefined);
95
+ assert.match(r.stderr, /task-high below the gate: 12 comparisons across 4 run_dates/);
96
+ });
97
+
98
+ test("two resolved models in one cell's gate window print `mixed (...)` with each model's row count", () => {
99
+ const w = world();
100
+ addComparisons(w, "task-high", 7, 5);
101
+ addComparisons(w, "task-high", 3, 3, { model: "claude-opus-5" });
102
+ const r = cli(w);
103
+ assert.equal(r.status, 0, r.stderr);
104
+ // claude-opus-5 at task-high/high still differs from slow-high only if its
105
+ // effort does — it does not, so those three sessions are no comparison.
106
+ assert.match(r.stderr, /task-high below the gate: 7 comparisons across 5 run_dates/);
107
+
108
+ const gated = world();
109
+ addComparisons(gated, "task-high", 10, 5);
110
+ addComparisons(gated, "task-high", 2, 1, { model: "claude-haiku-4-5" });
111
+ const g = cli(gated);
112
+ assert.equal(g.status, 0, g.stderr);
113
+ assert.equal(lineFor(g.stdout, "task-high"), "task-high 12 5 mixed (claude-sonnet-5 n=10, claude-haiku-4-5 n=2)");
114
+ });
115
+
116
+ test("a comparison is one session, however many rows it holds, and a row on each side is required", () => {
117
+ const w = world();
118
+ // One session, three task-high rows and two slow-high rows: one comparison.
119
+ addRow(w, { session: "sA", date: "2026-10-01", cell: "task-high" });
120
+ addRow(w, { session: "sA", date: "2026-10-01", cell: "task-high" });
121
+ addRow(w, { session: "sA", date: "2026-10-01", cell: "task-high" });
122
+ addRow(w, { session: "sA", date: "2026-10-01", cell: "slow-high" });
123
+ addRow(w, { session: "sA", date: "2026-10-01", cell: "slow-high" });
124
+ // task-high with no slow-high beside it, and slow-high alone: no comparison.
125
+ addRow(w, { session: "sB", date: "2026-10-02", cell: "task-high" });
126
+ addRow(w, { session: "sC", date: "2026-10-03", cell: "slow-high" });
127
+ // A pair split across two sessions is no comparison either.
128
+ addRow(w, { session: "sD", date: "2026-10-04", cell: "smol-high" });
129
+ addRow(w, { session: "sE", date: "2026-10-04", cell: "slow-high" });
130
+ const [task, smol] = ["task-high", "smol-high"].map((c) => readout(parsed(w)).cells.find((x) => x.cell === c));
131
+ assert.deepEqual([task.comparisons, task.runDates], [1, 1]);
132
+ assert.deepEqual([smol.comparisons, smol.runDates], [0, 0]);
133
+ });
134
+
135
+ test("only admissible rows count: dispatched as the drawn cell's definition, at the cell's level, with a model on record", () => {
136
+ const w = world();
137
+ const s = (n) => `s-adm-${n}`;
138
+ // effort off the cell's level (a clamp): inadmissible
139
+ addRow(w, { session: s(1), date: "2026-10-01", cell: "task-high", effort: "medium" });
140
+ addRow(w, { session: s(1), date: "2026-10-01", cell: "slow-high" });
141
+ // dispatched as a generic task, not the cell's definition: inadmissible
142
+ addRow(w, { session: s(2), date: "2026-10-01", cell: "task-high", subagentType: "task" });
143
+ addRow(w, { session: s(2), date: "2026-10-01", cell: "slow-high" });
144
+ // the slow-high side dispatched as the pre-cutover definition: inadmissible
145
+ addRow(w, { session: s(3), date: "2026-10-01", cell: "task-high" });
146
+ addRow(w, { session: s(3), date: "2026-10-01", cell: "slow-high", subagentType: "fleet-implementer" });
147
+ // the slow-high side has no member row at all: no join, inadmissible
148
+ addRow(w, { session: s(4), date: "2026-10-01", cell: "task-high" });
149
+ addRow(w, { session: s(4), date: "2026-10-01", cell: "slow-high", member: false });
150
+ // blank model: the resolved model is unknown, so nothing can differ from it
151
+ addRow(w, { session: s(5), date: "2026-10-01", cell: "task-high", model: "" });
152
+ addRow(w, { session: s(5), date: "2026-10-01", cell: "slow-high" });
153
+ // the control: admissible on both sides
154
+ addRow(w, { session: s(6), date: "2026-10-01", cell: "task-high" });
155
+ addRow(w, { session: s(6), date: "2026-10-01", cell: "slow-high" });
156
+ const task = readout(parsed(w)).cells.find((x) => x.cell === "task-high");
157
+ assert.equal(task.comparisons, 1, "only the admissible session counts");
158
+ assert.deepEqual([...task.models], [["claude-sonnet-5", 3]], "the models column reads admissible rows only — sessions 3, 4 and 6");
159
+ });
160
+
161
+ test("a comparison needs a resolved (model, effort) that differs from slow-high's — the same model at a different level counts", () => {
162
+ const w = world();
163
+ // task-high resolved to slow-high's own model at the same level: empty comparison
164
+ addRow(w, { session: "sSame", date: "2026-10-01", cell: "task-high", model: "claude-opus-5" });
165
+ addRow(w, { session: "sSame", date: "2026-10-01", cell: "slow-high" });
166
+ // slow-medium: same model, a different effort — a real comparison
167
+ addRow(w, { session: "sEff", date: "2026-10-02", cell: "slow-medium" });
168
+ addRow(w, { session: "sEff", date: "2026-10-02", cell: "slow-high" });
169
+ const rows = readout(parsed(w)).cells;
170
+ assert.equal(rows.find((x) => x.cell === "task-high").comparisons, 0);
171
+ assert.equal(rows.find((x) => x.cell === "slow-medium").comparisons, 1);
172
+ assert.equal(rows.find((x) => x.cell === "slow-high"), undefined, "slow-high is the baseline, never a readout line");
173
+ });
174
+
175
+ test("a run_date is the comparison session's member-outcomes run_date, not the ticket-features one", () => {
176
+ const w = world();
177
+ addComparisons(w, "task-high", 10, 5);
178
+ for (const f of w.features) f.run_date = "2026-09-01";
179
+ const r = cli(w);
180
+ assert.equal(lineFor(r.stdout, "task-high"), "task-high 10 5 claude-sonnet-5");
181
+ });
182
+
183
+ test("a re-dispatched Pull is one Pull: two ticket-features rows for one session+agent count the member row once", () => {
184
+ const w = world();
185
+ addRow(w, { session: "sRe", date: "2026-10-01", cell: "task-high" });
186
+ addRow(w, { session: "sRe", date: "2026-10-01", cell: "slow-high" });
187
+ // The re-dispatch: the same Pull's features row again, on the same member row.
188
+ w.features.push({ ...w.features[0] });
189
+ // An unjoined Pull, likewise written twice.
190
+ addRow(w, { session: "sOrphan", date: "2026-10-01", cell: "smol-high", member: false });
191
+ w.features.push({ ...w.features[w.features.length - 1] });
192
+ const { cells, unjoined } = readout(parsed(w));
193
+ const task = cells.find((x) => x.cell === "task-high");
194
+ assert.equal(task.comparisons, 1);
195
+ assert.deepEqual([...task.models], [["claude-sonnet-5", 1]], "the member row behind a re-dispatched Pull is counted once");
196
+ assert.equal(unjoined, 1, "a re-dispatched Pull with no member row is one unjoined Pull");
197
+ });
198
+
199
+ test("a blank member-outcomes run_date is no date: its session is a comparison but adds nothing to the distinct-date count", () => {
200
+ const w = world();
201
+ addComparisons(w, "task-high", 4, 4);
202
+ for (let i = 0; i < 6; i++) {
203
+ const session = `sBlank${i}`;
204
+ addRow(w, { session, date: "", cell: "task-high" });
205
+ addRow(w, { session, date: "", cell: "slow-high" });
206
+ }
207
+ const task = readout(parsed(w)).cells.find((x) => x.cell === "task-high");
208
+ assert.equal(task.comparisons, 10);
209
+ assert.deepEqual([task.runDates, task.gated], [4, false], "ten comparisons over four real dates is below the five-date floor");
210
+ assert.ok(!task.dates.includes(""), "a blank run_date is listed as a date");
211
+ });
212
+
213
+ test("gated cells print sorted by cell name, not in ticket-features order", () => {
214
+ const w = world();
215
+ // Rows are added in reverse alphabetical order: task-high first, smol-high second.
216
+ addComparisons(w, "task-high", 10, 5);
217
+ addComparisons(w, "smol-high", 10, 5);
218
+ const r = cli(w);
219
+ assert.equal(r.status, 0, r.stderr);
220
+ assert.deepEqual(
221
+ r.stdout.split("\n").filter(Boolean),
222
+ ["smol-high 10 5 claude-sonnet-5", "task-high 10 5 claude-sonnet-5"],
223
+ );
224
+ });
225
+
226
+ test("equal per-model counts in a mixed cell list models alphabetically", () => {
227
+ const w = world();
228
+ // Reverse-alphabetical insertion order: sonnet rows first, then haiku, five each.
229
+ addComparisons(w, "task-high", 5, 5, { model: "claude-sonnet-5" });
230
+ addComparisons(w, "task-high", 5, 5, { model: "claude-haiku-4-5" });
231
+ const r = cli(w);
232
+ assert.equal(r.status, 0, r.stderr);
233
+ assert.equal(lineFor(r.stdout, "task-high"), "task-high 10 5 mixed (claude-haiku-4-5 n=5, claude-sonnet-5 n=5)");
234
+ });
235
+
236
+ // run-team/SKILL.md sends the controller here for a cell's comparison count.
237
+ // The prose is the only thing that does, and nothing else reads it: renaming
238
+ // the script or pointing the floor back at another query leaves every test
239
+ // above green. Each slice starts at an anchor that must occur exactly once and
240
+ // ends at the paragraph's blank line, so a restatement elsewhere in the file
241
+ // cannot satisfy a pin on the paragraph that carries the rule.
242
+ const RUN_TEAM = readFileSync(join(import.meta.dirname, "..", "skills", "run-team", "SKILL.md"), "utf8");
243
+
244
+ test("SKILL.md's floor names the script that prints a cell's count, and that script exists", () => {
245
+ const floor = unemphasized(paragraph(RUN_TEAM, "**Report the count the per-cell readout prints", "run-team's per-cell floor"));
246
+ const named = /`~\/\.fleet\/bin\/fleet-run (\S+\.mjs)`/.exec(floor)?.[1];
247
+ assert.ok(named, "the floor no longer tells the controller which script to run");
248
+ assert.equal(named, "cell-readout.mjs");
249
+ assert.ok(existsSync(join(dirname(SCRIPT), named)), `${named} is not a script beside cell-readout.mjs`);
250
+ assert.match(floor, phrase("prints `<cell> <comparisons> <run_dates> <resolved models>` for each cell past that floor"));
251
+ assert.match(floor, phrase("a cell below it only as a count on stderr"));
252
+ });
253
+
254
+ test("the line SKILL.md says the readout prints is the line it prints: four fields, a cell below the floor on stderr only", () => {
255
+ const gated = world();
256
+ addComparisons(gated, "task-high", 10, 5);
257
+ addComparisons(gated, "smol-high", 2, 2);
258
+ const r = cli(gated);
259
+ assert.equal(r.status, 0, r.stderr);
260
+ assert.equal(lineFor(r.stdout, "task-high"), "task-high 10 5 claude-sonnet-5");
261
+ assert.equal(lineFor(r.stdout, "smol-high"), undefined, "a cell below the floor must not print on stdout");
262
+ assert.match(r.stderr, /smol-high below the gate: 2 comparisons/);
263
+ });
264
+
265
+ test("SKILL.md says the readout counts a pair only when the two rows ran a different model or effort, as the readout does", () => {
266
+ const deliberate = unemphasized(paragraph(RUN_TEAM, "**The readout counts DELIBERATE comparisons.**", "run-team's DELIBERATE paragraph"));
267
+ assert.match(deliberate, phrase("the two rows must actually have RUN a different model or effort"));
268
+ // Run, not read: the same model at the same level is no comparison, the same
269
+ // model at a different level is one, a different model at the same level is one.
270
+ const w = world();
271
+ addRow(w, { session: "sNone", date: "2026-10-01", cell: "task-high", model: "claude-opus-5" });
272
+ addRow(w, { session: "sNone", date: "2026-10-01", cell: "slow-high" });
273
+ addRow(w, { session: "sModel", date: "2026-10-02", cell: "task-high" });
274
+ addRow(w, { session: "sModel", date: "2026-10-02", cell: "slow-high" });
275
+ addRow(w, { session: "sEffort", date: "2026-10-03", cell: "slow-medium" });
276
+ addRow(w, { session: "sEffort", date: "2026-10-03", cell: "slow-high" });
277
+ const { cells } = readout(parsed(w));
278
+ const countOf = (cell) => cells.find((c) => c.cell === cell).comparisons;
279
+ assert.deepEqual([countOf("task-high"), countOf("slow-medium")], [1, 1]);
280
+ });
281
+
282
+ test("the output identifies no comparison: no session, ticket or agent appears on either stream", () => {
283
+ const w = world();
284
+ addComparisons(w, "task-high", 10, 5);
285
+ addComparisons(w, "smol-high", 3, 2);
286
+ addRow(w, { session: "sOrphan", date: "2026-10-01", cell: "task-high", member: false });
287
+ const r = cli(w);
288
+ assert.equal(r.status, 0, r.stderr);
289
+ const out = r.stdout + r.stderr;
290
+ for (const f of w.features) {
291
+ assert.ok(!out.includes(f.session), `output names session ${f.session}`);
292
+ assert.ok(!out.includes(f.agent), `output names agent ${f.agent}`);
293
+ }
294
+ assert.match(r.stderr, /1 ticket-features row has no member-outcomes row/);
295
+ });
296
+
297
+ test("an empty ticket-features.tsv prints no line and says why, exit 0", () => {
298
+ const r = cli(world());
299
+ assert.equal(r.status, 0, r.stderr);
300
+ assert.equal(r.stdout, "");
301
+ assert.match(r.stderr, /no cell other than slow-high has a ticket-features row/);
302
+ });
303
+
304
+ test("usage and input errors exit 2 and print nothing on stdout", () => {
305
+ const w = world();
306
+ addComparisons(w, "task-high", 1, 1);
307
+ const f = files(w);
308
+ for (const args of [
309
+ ["--ticket-features", join(f.dir, "absent.tsv"), "--member-outcomes", f.members],
310
+ ["--ticket-features", f.features, "--member-outcomes", join(f.dir, "absent.tsv")],
311
+ ["--ticket-features", f.features, "--member-outcomes", f.members, "--bogus"],
312
+ ]) {
313
+ const r = spawnSync(process.execPath, [SCRIPT, ...args], { encoding: "utf8", cwd: f.dir });
314
+ assert.equal(r.status, 2, `${args.join(" ")}: ${r.stderr}`);
315
+ assert.equal(r.stdout, "");
316
+ }
317
+ writeFileSync(f.members, "# header\nshort\trow\n");
318
+ const bad = spawnSync(process.execPath, [SCRIPT, "--ticket-features", f.features, "--member-outcomes", f.members], { encoding: "utf8", cwd: f.dir });
319
+ assert.equal(bad.status, 2, bad.stderr);
320
+ assert.match(bad.stderr, /malformed row/);
321
+ });
322
+
323
+ test("with no flags it reads docs/metrics/ under the working directory", () => {
324
+ const w = world();
325
+ addComparisons(w, "task-high", 10, 5);
326
+ const f = files(w);
327
+ const metrics = join(f.dir, "docs", "metrics");
328
+ spawnSync("mkdir", ["-p", metrics]);
329
+ spawnSync("cp", [f.features, f.members, metrics]);
330
+ const r = spawnSync(process.execPath, [SCRIPT], { encoding: "utf8", cwd: f.dir });
331
+ assert.equal(r.status, 0, r.stderr);
332
+ assert.equal(lineFor(r.stdout, "task-high"), "task-high 10 5 claude-sonnet-5");
333
+ });
334
+
335
+ // The pure function's inputs, parsed the way the CLI parses them.
336
+ function parsed(w) {
337
+ return {
338
+ features: w.features.map((r) => ({ ...r })),
339
+ members: w.members.map((r) => ({ ...r })),
340
+ };
341
+ }
@@ -39,7 +39,10 @@
39
39
  // merge queue, and which PRs are owed a review. Also, for each ticket
40
40
  // holding the implementer row on a tier mismatch, whether its issue
41
41
  // is CLOSED (`gh issue view`, #2485): a closed ticket's mismatch
42
- // holds nothing.
42
+ // holds nothing. And, for each in-flight `review=` token whose PR
43
+ // the open list does not carry, whether that PR is MERGED or
44
+ // CLOSED (`gh pr view`): a review cannot be running against a
45
+ // finished PR, so its dangling token holds no reviewer slot.
43
46
  // main The main checkout against the run-start baseline
44
47
  // `.fleet/main-checkout.sha`, through main-checkout.mjs (#2210).
45
48
  // Anything but `clean` — dirty, unknown, no baseline — prints one
@@ -496,7 +499,7 @@ export function unlabelledFinishers(ledger, unqueued) {
496
499
  .map((u) => ({ pr: u.pr, labelled: u.attempts.filter((a) => a.outcome === "labelled").map((a) => a.name) }));
497
500
  }
498
501
 
499
- export function deriveRun({ rows, dispatched, drain }, prs, closed = new Set()) {
502
+ export function deriveRun({ rows, dispatched, drain }, prs, closed = new Set(), finished = new Set()) {
500
503
  // One entry per member name across `## Dispatched` and every row. A member
501
504
  // settled ANYWHERE is settled: `settle` is the only writer of an outcome, and
502
505
  // a bare copy beside it is what a whole-line `row` rewrite leaves behind.
@@ -752,11 +755,23 @@ export function deriveRun({ rows, dispatched, drain }, prs, closed = new Set())
752
755
  // head that review read, and there is nothing more to review.
753
756
  const pastPinDue = (st, head) => st !== undefined && st.pastPinHalt && !st.inFlight && st.reviewedHead !== null
754
757
  && !String(head).toLowerCase().startsWith(st.reviewedHead);
758
+ // A `review=` token with no `reviewed=` after it reads as in flight, and the
759
+ // ledger alone never says otherwise: a review whose own `reviewed=` write
760
+ // never landed stays in flight after its PR merges, holding a reviewer slot
761
+ // for good. `finished` is the PR numbers the caller has found MERGED or
762
+ // CLOSED, and a finished PR has no review running against it. A PR on the
763
+ // open list is open whatever `finished` says, and none is the default, so a
764
+ // PR nobody has asked about keeps its review live.
765
+ const reviewing = [...byPr.entries()].filter(([, st]) => st.inFlight);
766
+ const liveReviews = reviewing.filter(([n]) => open.has(n) || !finished.has(n));
755
767
  return {
756
768
  implLive: live("impl"),
757
769
  fixLive: live("fix-pr"),
758
770
  mergeBotLive: live("merge-bot"),
759
- reviewsLive: [...byPr.values()].filter((st) => st.inFlight).length,
771
+ reviewsLive: liveReviews.length,
772
+ // The in-flight reviews of PRs the open list does not carry: the ones
773
+ // `finished` could retire, so the caller asks gh about these and no others.
774
+ reviewsOffList: reviewing.filter(([n]) => !open.has(n)).map(([n]) => n).sort(asc),
760
775
  // A returned review whose survivors no review fix-applier has answered,
761
776
  // a conflict hold no fix-applier has cleared, or a dispositions mismatch
762
777
  // no retry has answered — on a PR still open, with no fix-applier working
@@ -824,7 +839,7 @@ export function deriveRun({ rows, dispatched, drain }, prs, closed = new Set())
824
839
  // run carries no member token of its own.
825
840
  live: [
826
841
  ...all.filter((m) => m.outcome === null).map((m) => m.name),
827
- ...[...byPr.entries()].filter(([, st]) => st.inFlight).map(([n]) => `review:PR#${n}`),
842
+ ...liveReviews.map(([n]) => `review:PR#${n}`),
828
843
  ],
829
844
  };
830
845
  }
@@ -1106,6 +1121,31 @@ function closedTickets(mismatched) {
1106
1121
  return closed;
1107
1122
  }
1108
1123
 
1124
+ // The PRs, of those whose review the ledger reads as in flight yet the open
1125
+ // list does not carry, that gh says are MERGED or CLOSED. A review cannot be
1126
+ // running against a finished PR, so the ledger's dangling `review=` token must
1127
+ // not hold a reviewer slot for good. A probe that cannot answer — a nonzero
1128
+ // exit, a reply carrying no state, a state that is none of the three — is
1129
+ // disclosed and the review keeps its slot: an unreadable state is never read
1130
+ // as finished. A PR gh calls OPEN keeps it quietly.
1131
+ function finishedReviewPrs(offList) {
1132
+ const finished = new Set();
1133
+ for (const number of offList) {
1134
+ const r = spawnSync("gh", ["pr", "view", String(number), "--json", "state"], { encoding: "utf8", env: gitEnv({ GH_REPO: "" }) });
1135
+ if (r.error || r.status !== 0) {
1136
+ console.error(`${NAME}: ${failure(r, `gh pr view ${number}`)} — review on PR#${number} unconfirmed finished, its slot stands`);
1137
+ continue;
1138
+ }
1139
+ let st = null;
1140
+ try { st = JSON.parse(r.stdout).state; } catch { /* no state: disclosed below */ }
1141
+ if (st === "MERGED" || st === "CLOSED") finished.add(number);
1142
+ else if (st !== "OPEN") {
1143
+ console.error(`${NAME}: gh pr view ${number} printed no PR state — review on PR#${number} unconfirmed finished, its slot stands`);
1144
+ }
1145
+ }
1146
+ return finished;
1147
+ }
1148
+
1109
1149
  // shortlist.mjs, run by the tick itself (§ 6 §3). Its per-ticket stderr is
1110
1150
  // captured and dropped rather than billed to the controller's context; only a
1111
1151
  // failure's last line rides along.
@@ -1166,7 +1206,8 @@ function main() {
1166
1206
  const ledger = readLedger();
1167
1207
  run = deriveRun(ledger, prs);
1168
1208
  const closed = closedTickets(run.tierMismatch);
1169
- if (closed.size) run = deriveRun(ledger, prs, closed);
1209
+ const finished = finishedReviewPrs(run.reviewsOffList);
1210
+ if (closed.size || finished.size) run = deriveRun(ledger, prs, closed, finished);
1170
1211
  } catch (e) {
1171
1212
  if (e instanceof LedgerError) die(e.message);
1172
1213
  throw e;