@feigi/fleet-ctl 3.21.13 → 3.22.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/scripts/ambient-git-vars-mjs-prose.test.mjs +7 -3
- package/scripts/cell-readout.mjs +148 -0
- package/scripts/cell-readout.test.mjs +341 -0
- package/scripts/fleet-tick.mjs +46 -5
- package/scripts/fleet-tick.test.mjs +132 -4
- package/scripts/member-outcomes-header.test.mjs +45 -85
- package/scripts/member-outcomes.mjs +2 -2
- package/scripts/member-outcomes.test.mjs +3 -3
- package/scripts/tier-outcomes-header.test.mjs +37 -62
- package/skills/run-team/SKILL.md +39 -34
package/package.json
CHANGED
|
@@ -201,8 +201,8 @@ const COVERED_MJS = {
|
|
|
201
201
|
// children (`node candidates.mjs`, `node ledger.mjs`, `sh inflight.sh`)
|
|
202
202
|
// name no git.
|
|
203
203
|
"shortlist.mjs": 2,
|
|
204
|
-
//
|
|
205
|
-
// third is below (#2485). `shortlistPath()`'s
|
|
204
|
+
// FOUR. The first two are the pair shortlist.mjs carries (#1803), the
|
|
205
|
+
// third is below (#2485), the fourth after it. `shortlistPath()`'s
|
|
206
206
|
// `--git-common-dir` probe names the `.fleet/shortlist.json` the tick PULLs
|
|
207
207
|
// from; an ambient GIT_DIR would read another repository's shortlist and
|
|
208
208
|
// name its tickets. Measured in fleet-tick.test.mjs, "an ambient GIT_DIR
|
|
@@ -216,7 +216,11 @@ const COVERED_MJS = {
|
|
|
216
216
|
// implementer's ticket is already closed, and resolves the repository the
|
|
217
217
|
// same way; measured in fleet-tick.test.mjs, "an inherited GIT_DIR cannot
|
|
218
218
|
// retarget the tier-mismatch closed-ticket probe".
|
|
219
|
-
|
|
219
|
+
// `finishedReviewPrs()`'s `gh pr view` asks whether a PR with a dangling
|
|
220
|
+
// in-flight review is already merged or closed, and resolves the repository
|
|
221
|
+
// the same way; measured in fleet-tick.test.mjs, "an inherited GIT_DIR
|
|
222
|
+
// cannot retarget the merged-PR review probe".
|
|
223
|
+
"fleet-tick.mjs": 4,
|
|
220
224
|
// ONE spawn primitive, `git(args, cwd)`, behind both of the file's git calls
|
|
221
225
|
// — `resolveMainRoot()`'s `rev-parse --git-common-dir` and `checkIgnored()`'s
|
|
222
226
|
// `check-ignore` (#1411). An ambient GIT_DIR would answer the first for
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// The per-cell readout: how many within-session comparisons each implementer
|
|
3
|
+
// cell has against the policy cell, and whether that is enough to read it.
|
|
4
|
+
//
|
|
5
|
+
// cell-readout.mjs [--ticket-features <tsv>] [--member-outcomes <tsv>]
|
|
6
|
+
//
|
|
7
|
+
// Inputs default to docs/metrics/ under the working directory. A Pull is a
|
|
8
|
+
// ticket-features.tsv row; what it ran is the member-outcomes.tsv row with the
|
|
9
|
+
// same `session` + `agent`.
|
|
10
|
+
//
|
|
11
|
+
// ADMISSIBLE ROW. A Pull counts only when its member row exists, was
|
|
12
|
+
// dispatched as the drawn cell's own definition (`subagent_type` =
|
|
13
|
+
// `fleet-implementer-<chosen_cell>`), ran at the cell's level (`effort` = the
|
|
14
|
+
// cell's `<level>`) and has a resolved `model` on record. So a generic `task`
|
|
15
|
+
// dispatch, a clamped effort and every pre-cutover `fleet-implementer` /
|
|
16
|
+
// `fleet-implementer-alt` row are never admissible. Whether the resolved model
|
|
17
|
+
// was the role's target at dispatch is recorded in neither file; that half of
|
|
18
|
+
// the tier check is the run ledger's, not this readout's.
|
|
19
|
+
//
|
|
20
|
+
// COMPARISON. For a cell X other than slow-high, a comparison is one session
|
|
21
|
+
// holding at least one admissible X row and at least one admissible slow-high
|
|
22
|
+
// row whose resolved (`model`, `effort`) differ from that X row's. A session
|
|
23
|
+
// counts once however many rows it holds, and its date is its member-outcomes
|
|
24
|
+
// `run_date`; a blank `run_date` means unknown, so that session is still a
|
|
25
|
+
// comparison but adds no date to the distinct-date count.
|
|
26
|
+
//
|
|
27
|
+
// GATE. A cell is read only once it has at least GATE.comparisons comparisons
|
|
28
|
+
// across at least GATE.runDates distinct run_dates. A gated cell prints one
|
|
29
|
+
// line on stdout:
|
|
30
|
+
//
|
|
31
|
+
// <cell> <comparisons> <run_dates> <resolved models>
|
|
32
|
+
//
|
|
33
|
+
// `<resolved models>` is the model of every admissible row at the cell — one
|
|
34
|
+
// model by name, or `mixed (<model-a> n=…, <model-b> n=…)` when they span more
|
|
35
|
+
// than one, because a role's target is operator config that can move under a
|
|
36
|
+
// cell's history. A cell below the gate prints nothing on stdout and one
|
|
37
|
+
// `below the gate` count on stderr. Nothing on either stream names a session,
|
|
38
|
+
// ticket or member: the gate forbids reading a comparison early.
|
|
39
|
+
//
|
|
40
|
+
// Exit 0 whatever the counts; 2 on a usage error or an unreadable or
|
|
41
|
+
// malformed input, with nothing on stdout.
|
|
42
|
+
|
|
43
|
+
import { readFileSync } from "node:fs";
|
|
44
|
+
import { makeDie, defineFlags } from "./arg.mjs";
|
|
45
|
+
import { isCLI } from "./is-cli.mjs";
|
|
46
|
+
import { CELL, POLICY_CELL } from "./ledger-grammar.mjs";
|
|
47
|
+
import { parseTsv as parseMemberTsv } from "./member-outcomes.mjs";
|
|
48
|
+
import { parseFeatures } from "./pr-cost.mjs";
|
|
49
|
+
|
|
50
|
+
const NAME = "cell-readout";
|
|
51
|
+
export const GATE = Object.freeze({ comparisons: 10, runDates: 5 });
|
|
52
|
+
|
|
53
|
+
const key = (session, agent) => `${session}\0${agent}`;
|
|
54
|
+
const levelOf = (cell) => CELL.exec(cell)?.[2] ?? null;
|
|
55
|
+
|
|
56
|
+
/** The join's member row for a Pull when the Pull is admissible, else null. */
|
|
57
|
+
function admissibleMember(pull, members) {
|
|
58
|
+
const m = members.get(key(pull.session, pull.agent));
|
|
59
|
+
if (!m) return null;
|
|
60
|
+
const level = levelOf(pull.chosen_cell);
|
|
61
|
+
if (level === null || m.subagentType !== `fleet-implementer-${pull.chosen_cell}` || m.effort !== level || !m.model) return null;
|
|
62
|
+
return m;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* `cells`: one entry per cell other than the policy cell that has a
|
|
67
|
+
* ticket-features row, `{ cell, comparisons, runDates, dates, models, gated }`,
|
|
68
|
+
* sorted by cell. `dates` is the comparisons' distinct run_dates, sorted;
|
|
69
|
+
* `models` maps each resolved model of the cell's admissible rows to its row
|
|
70
|
+
* count, largest first. `unjoined` counts Pulls with no member row.
|
|
71
|
+
*/
|
|
72
|
+
export function readout({ features, members }) {
|
|
73
|
+
const byKey = new Map(members.map((m) => [key(m.session, m.agent), m]));
|
|
74
|
+
// One Pull per session+agent: a re-dispatch lands on the same cell and the
|
|
75
|
+
// same member row.
|
|
76
|
+
const pulls = new Map(features.map((p) => [key(p.session, p.agent), p]));
|
|
77
|
+
const sessions = new Map();
|
|
78
|
+
const cells = new Map();
|
|
79
|
+
let unjoined = 0;
|
|
80
|
+
for (const p of pulls.values()) {
|
|
81
|
+
if (p.chosen_cell !== POLICY_CELL && !cells.has(p.chosen_cell)) cells.set(p.chosen_cell, new Map());
|
|
82
|
+
if (!byKey.has(key(p.session, p.agent))) unjoined++;
|
|
83
|
+
const m = admissibleMember(p, byKey);
|
|
84
|
+
if (!m) continue;
|
|
85
|
+
if (p.chosen_cell !== POLICY_CELL) cells.get(p.chosen_cell).set(m.model, (cells.get(p.chosen_cell).get(m.model) ?? 0) + 1);
|
|
86
|
+
if (!sessions.has(p.session)) sessions.set(p.session, { runDate: m.run_date, rows: [] });
|
|
87
|
+
sessions.get(p.session).rows.push({ cell: p.chosen_cell, model: m.model, effort: m.effort });
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
const out = [...cells].sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0)).map(([cell, modelCounts]) => {
|
|
91
|
+
let comparisons = 0;
|
|
92
|
+
const dates = new Set();
|
|
93
|
+
for (const { runDate, rows } of sessions.values()) {
|
|
94
|
+
const base = rows.filter((r) => r.cell === POLICY_CELL);
|
|
95
|
+
const compared = rows.some((x) => x.cell === cell && base.some((s) => s.model !== x.model || s.effort !== x.effort));
|
|
96
|
+
if (!compared) continue;
|
|
97
|
+
comparisons++;
|
|
98
|
+
if (runDate) dates.add(runDate);
|
|
99
|
+
}
|
|
100
|
+
const models = new Map([...modelCounts].sort(([a, n], [b, k]) => k - n || (a < b ? -1 : a > b ? 1 : 0)));
|
|
101
|
+
const runDates = dates.size;
|
|
102
|
+
return { cell, comparisons, runDates, dates: [...dates].sort(), models, gated: comparisons >= GATE.comparisons && runDates >= GATE.runDates };
|
|
103
|
+
});
|
|
104
|
+
return { cells: out, unjoined };
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
// Only a gated cell is formatted, and a comparison needs an admissible row at
|
|
108
|
+
// the cell, so `models` is never empty here.
|
|
109
|
+
function formatModels(models) {
|
|
110
|
+
if (models.size === 1) return [...models.keys()][0];
|
|
111
|
+
return `mixed (${[...models].map(([m, n]) => `${m} n=${n}`).join(", ")})`;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
function main() {
|
|
115
|
+
const die = makeDie(NAME);
|
|
116
|
+
const { arg, sweep, stray } = defineFlags(die, {
|
|
117
|
+
flags: { "ticket-features": "value", "member-outcomes": "value" },
|
|
118
|
+
});
|
|
119
|
+
sweep();
|
|
120
|
+
stray();
|
|
121
|
+
const featuresPath = arg("ticket-features") ?? "docs/metrics/ticket-features.tsv";
|
|
122
|
+
const membersPath = arg("member-outcomes") ?? "docs/metrics/member-outcomes.tsv";
|
|
123
|
+
const load = (path, parse) => {
|
|
124
|
+
let text;
|
|
125
|
+
try { text = readFileSync(path, "utf8"); }
|
|
126
|
+
catch (e) { die(`cannot read ${path}: ${e.code ?? e.message}`); }
|
|
127
|
+
try { return parse(text); }
|
|
128
|
+
catch (e) { die(`${path}: ${e.message}`); }
|
|
129
|
+
};
|
|
130
|
+
const features = load(featuresPath, parseFeatures);
|
|
131
|
+
const members = load(membersPath, parseMemberTsv);
|
|
132
|
+
|
|
133
|
+
const { cells, unjoined } = readout({ features, members });
|
|
134
|
+
const lines = [];
|
|
135
|
+
const notes = [];
|
|
136
|
+
if (unjoined > 0) {
|
|
137
|
+
notes.push(`${unjoined} ticket-features row${unjoined === 1 ? " has" : "s have"} no member-outcomes row — never admissible`);
|
|
138
|
+
}
|
|
139
|
+
if (cells.length === 0) notes.push(`no cell other than ${POLICY_CELL} has a ticket-features row`);
|
|
140
|
+
for (const c of cells) {
|
|
141
|
+
if (c.gated) lines.push(`${c.cell} ${c.comparisons} ${c.runDates} ${formatModels(c.models)}`);
|
|
142
|
+
else notes.push(`${c.cell} below the gate: ${c.comparisons} comparisons across ${c.runDates} run_dates (needs ${GATE.comparisons} across ${GATE.runDates})`);
|
|
143
|
+
}
|
|
144
|
+
for (const n of notes) process.stderr.write(`${NAME}: ${n}\n`);
|
|
145
|
+
if (lines.length) process.stdout.write(`${lines.join("\n")}\n`);
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
if (isCLI(import.meta.url)) main();
|
|
@@ -0,0 +1,341 @@
|
|
|
1
|
+
// cell-readout.mjs: the per-cell gate over ticket-features.tsv joined to
|
|
2
|
+
// member-outcomes.tsv, run as the CLI against synthetic fixture files.
|
|
3
|
+
import { test } from "node:test";
|
|
4
|
+
import assert from "node:assert/strict";
|
|
5
|
+
import { spawnSync } from "node:child_process";
|
|
6
|
+
import { existsSync, readFileSync, writeFileSync } from "node:fs";
|
|
7
|
+
import { dirname, join } from "node:path";
|
|
8
|
+
import { fileURLToPath } from "node:url";
|
|
9
|
+
import { tempDir } from "./temp-dir.mjs";
|
|
10
|
+
import { paragraph, phrase, unemphasized } from "./prose-pin.mjs";
|
|
11
|
+
import { formatTsv } from "./member-outcomes.mjs";
|
|
12
|
+
import { FEATURE_COLUMNS } from "./pr-cost.mjs";
|
|
13
|
+
import { readout, GATE } from "./cell-readout.mjs";
|
|
14
|
+
|
|
15
|
+
const SCRIPT = fileURLToPath(new URL("./cell-readout.mjs", import.meta.url));
|
|
16
|
+
|
|
17
|
+
const member = (o) => ({
|
|
18
|
+
run_date: "2026-10-01", role: "implementer", member: o.agent, model: "claude-opus-5", effort: "high",
|
|
19
|
+
ticket: "", pr: "", tokensCacheCreate: 0, tokensOut: 0, wallS: 0, turns: 1, harness: "omp",
|
|
20
|
+
tokensIn: 0, tokensCacheRead: 0, tokensCacheWrite1h: 0, cost: 0, ...o,
|
|
21
|
+
});
|
|
22
|
+
const pull = (o) => ({
|
|
23
|
+
run_date: "2026-10-01", policy_cell: "slow-high", exploration_draw: "", sizing_src: "rule", sizing_pre: "",
|
|
24
|
+
router_usd: "", brief_chars: "100", criteria: "1", comments: "0", age_days: "0", paths: "0",
|
|
25
|
+
test_paths: "0", xrefs: "0", kind: "enhancement", ...o,
|
|
26
|
+
});
|
|
27
|
+
|
|
28
|
+
function world() {
|
|
29
|
+
return { features: [], members: [], ticket: 100 };
|
|
30
|
+
}
|
|
31
|
+
// One implementer Pull at `cell` in `session`, and the member row it left.
|
|
32
|
+
// `model`/`effort`/`subagentType` override what resolved and what was dispatched.
|
|
33
|
+
function addRow(w, { session, date, cell, model, effort, subagentType, member: withMember = true }) {
|
|
34
|
+
const ticket = String(w.ticket++);
|
|
35
|
+
const agent = `impl-${ticket}`;
|
|
36
|
+
const level = cell.split("-").slice(1).join("-");
|
|
37
|
+
w.features.push(pull({ session, agent, ticket, chosen_cell: cell }));
|
|
38
|
+
if (withMember) {
|
|
39
|
+
w.members.push(member({
|
|
40
|
+
session, agent, run_date: date, ticket,
|
|
41
|
+
model: model ?? (cell.startsWith("slow-") ? "claude-opus-5" : "claude-sonnet-5"),
|
|
42
|
+
effort: effort ?? level, subagentType: subagentType ?? `fleet-implementer-${cell}`,
|
|
43
|
+
}));
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
// `n` sessions, each one comparison for `cell`: a `cell` row beside a slow-high
|
|
47
|
+
// row, spread round-robin over `dates` distinct run_dates.
|
|
48
|
+
let nextSession = 0;
|
|
49
|
+
function addComparisons(w, cell, n, dates, rowOpts = {}) {
|
|
50
|
+
for (let i = 0; i < n; i++) {
|
|
51
|
+
const session = `2026-10-01T00-00-00-000Z_s${nextSession++}`;
|
|
52
|
+
const date = `2026-10-${String(1 + (i % dates)).padStart(2, "0")}`;
|
|
53
|
+
addRow(w, { session, date, cell, ...rowOpts });
|
|
54
|
+
addRow(w, { session, date, cell: "slow-high" });
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function files(w) {
|
|
59
|
+
const dir = tempDir("cell-readout-");
|
|
60
|
+
const features = join(dir, "ticket-features.tsv");
|
|
61
|
+
const members = join(dir, "member-outcomes.tsv");
|
|
62
|
+
writeFileSync(features, [FEATURE_COLUMNS.join("\t"), ...w.features.map((r) => FEATURE_COLUMNS.map((c) => r[c] ?? "").join("\t"))].join("\n") + "\n");
|
|
63
|
+
writeFileSync(members, "# header\n" + formatTsv(w.members));
|
|
64
|
+
return { features, members, dir };
|
|
65
|
+
}
|
|
66
|
+
function cli(w, extra = []) {
|
|
67
|
+
const f = files(w);
|
|
68
|
+
return spawnSync(process.execPath, [SCRIPT, "--ticket-features", f.features, "--member-outcomes", f.members, ...extra], { encoding: "utf8", cwd: f.dir });
|
|
69
|
+
}
|
|
70
|
+
const lineFor = (stdout, cell) => stdout.split("\n").find((l) => l.split(" ")[0] === cell);
|
|
71
|
+
|
|
72
|
+
test("the gate is ten comparisons across five run_dates: nine across five does not gate, ten across five does", () => {
|
|
73
|
+
const nine = world();
|
|
74
|
+
addComparisons(nine, "task-high", 9, 5);
|
|
75
|
+
const below = cli(nine);
|
|
76
|
+
assert.equal(below.status, 0, below.stderr);
|
|
77
|
+
assert.equal(lineFor(below.stdout, "task-high"), undefined, "a cell below the gate prints no line");
|
|
78
|
+
assert.match(below.stderr, /task-high below the gate: 9 comparisons across 5 run_dates/);
|
|
79
|
+
|
|
80
|
+
const ten = world();
|
|
81
|
+
addComparisons(ten, "task-high", 10, 5);
|
|
82
|
+
const at = cli(ten);
|
|
83
|
+
assert.equal(at.status, 0, at.stderr);
|
|
84
|
+
assert.equal(lineFor(at.stdout, "task-high"), "task-high 10 5 claude-sonnet-5");
|
|
85
|
+
assert.doesNotMatch(at.stderr, /task-high below the gate/);
|
|
86
|
+
assert.deepEqual(GATE, { comparisons: 10, runDates: 5 });
|
|
87
|
+
});
|
|
88
|
+
|
|
89
|
+
test("ten comparisons across four run_dates does not gate: both halves of the gate bind", () => {
|
|
90
|
+
const w = world();
|
|
91
|
+
addComparisons(w, "task-high", 12, 4);
|
|
92
|
+
const r = cli(w);
|
|
93
|
+
assert.equal(r.status, 0, r.stderr);
|
|
94
|
+
assert.equal(lineFor(r.stdout, "task-high"), undefined);
|
|
95
|
+
assert.match(r.stderr, /task-high below the gate: 12 comparisons across 4 run_dates/);
|
|
96
|
+
});
|
|
97
|
+
|
|
98
|
+
test("two resolved models in one cell's gate window print `mixed (...)` with each model's row count", () => {
|
|
99
|
+
const w = world();
|
|
100
|
+
addComparisons(w, "task-high", 7, 5);
|
|
101
|
+
addComparisons(w, "task-high", 3, 3, { model: "claude-opus-5" });
|
|
102
|
+
const r = cli(w);
|
|
103
|
+
assert.equal(r.status, 0, r.stderr);
|
|
104
|
+
// claude-opus-5 at task-high/high still differs from slow-high only if its
|
|
105
|
+
// effort does — it does not, so those three sessions are no comparison.
|
|
106
|
+
assert.match(r.stderr, /task-high below the gate: 7 comparisons across 5 run_dates/);
|
|
107
|
+
|
|
108
|
+
const gated = world();
|
|
109
|
+
addComparisons(gated, "task-high", 10, 5);
|
|
110
|
+
addComparisons(gated, "task-high", 2, 1, { model: "claude-haiku-4-5" });
|
|
111
|
+
const g = cli(gated);
|
|
112
|
+
assert.equal(g.status, 0, g.stderr);
|
|
113
|
+
assert.equal(lineFor(g.stdout, "task-high"), "task-high 12 5 mixed (claude-sonnet-5 n=10, claude-haiku-4-5 n=2)");
|
|
114
|
+
});
|
|
115
|
+
|
|
116
|
+
test("a comparison is one session, however many rows it holds, and a row on each side is required", () => {
|
|
117
|
+
const w = world();
|
|
118
|
+
// One session, three task-high rows and two slow-high rows: one comparison.
|
|
119
|
+
addRow(w, { session: "sA", date: "2026-10-01", cell: "task-high" });
|
|
120
|
+
addRow(w, { session: "sA", date: "2026-10-01", cell: "task-high" });
|
|
121
|
+
addRow(w, { session: "sA", date: "2026-10-01", cell: "task-high" });
|
|
122
|
+
addRow(w, { session: "sA", date: "2026-10-01", cell: "slow-high" });
|
|
123
|
+
addRow(w, { session: "sA", date: "2026-10-01", cell: "slow-high" });
|
|
124
|
+
// task-high with no slow-high beside it, and slow-high alone: no comparison.
|
|
125
|
+
addRow(w, { session: "sB", date: "2026-10-02", cell: "task-high" });
|
|
126
|
+
addRow(w, { session: "sC", date: "2026-10-03", cell: "slow-high" });
|
|
127
|
+
// A pair split across two sessions is no comparison either.
|
|
128
|
+
addRow(w, { session: "sD", date: "2026-10-04", cell: "smol-high" });
|
|
129
|
+
addRow(w, { session: "sE", date: "2026-10-04", cell: "slow-high" });
|
|
130
|
+
const [task, smol] = ["task-high", "smol-high"].map((c) => readout(parsed(w)).cells.find((x) => x.cell === c));
|
|
131
|
+
assert.deepEqual([task.comparisons, task.runDates], [1, 1]);
|
|
132
|
+
assert.deepEqual([smol.comparisons, smol.runDates], [0, 0]);
|
|
133
|
+
});
|
|
134
|
+
|
|
135
|
+
test("only admissible rows count: dispatched as the drawn cell's definition, at the cell's level, with a model on record", () => {
|
|
136
|
+
const w = world();
|
|
137
|
+
const s = (n) => `s-adm-${n}`;
|
|
138
|
+
// effort off the cell's level (a clamp): inadmissible
|
|
139
|
+
addRow(w, { session: s(1), date: "2026-10-01", cell: "task-high", effort: "medium" });
|
|
140
|
+
addRow(w, { session: s(1), date: "2026-10-01", cell: "slow-high" });
|
|
141
|
+
// dispatched as a generic task, not the cell's definition: inadmissible
|
|
142
|
+
addRow(w, { session: s(2), date: "2026-10-01", cell: "task-high", subagentType: "task" });
|
|
143
|
+
addRow(w, { session: s(2), date: "2026-10-01", cell: "slow-high" });
|
|
144
|
+
// the slow-high side dispatched as the pre-cutover definition: inadmissible
|
|
145
|
+
addRow(w, { session: s(3), date: "2026-10-01", cell: "task-high" });
|
|
146
|
+
addRow(w, { session: s(3), date: "2026-10-01", cell: "slow-high", subagentType: "fleet-implementer" });
|
|
147
|
+
// the slow-high side has no member row at all: no join, inadmissible
|
|
148
|
+
addRow(w, { session: s(4), date: "2026-10-01", cell: "task-high" });
|
|
149
|
+
addRow(w, { session: s(4), date: "2026-10-01", cell: "slow-high", member: false });
|
|
150
|
+
// blank model: the resolved model is unknown, so nothing can differ from it
|
|
151
|
+
addRow(w, { session: s(5), date: "2026-10-01", cell: "task-high", model: "" });
|
|
152
|
+
addRow(w, { session: s(5), date: "2026-10-01", cell: "slow-high" });
|
|
153
|
+
// the control: admissible on both sides
|
|
154
|
+
addRow(w, { session: s(6), date: "2026-10-01", cell: "task-high" });
|
|
155
|
+
addRow(w, { session: s(6), date: "2026-10-01", cell: "slow-high" });
|
|
156
|
+
const task = readout(parsed(w)).cells.find((x) => x.cell === "task-high");
|
|
157
|
+
assert.equal(task.comparisons, 1, "only the admissible session counts");
|
|
158
|
+
assert.deepEqual([...task.models], [["claude-sonnet-5", 3]], "the models column reads admissible rows only — sessions 3, 4 and 6");
|
|
159
|
+
});
|
|
160
|
+
|
|
161
|
+
test("a comparison needs a resolved (model, effort) that differs from slow-high's — the same model at a different level counts", () => {
|
|
162
|
+
const w = world();
|
|
163
|
+
// task-high resolved to slow-high's own model at the same level: empty comparison
|
|
164
|
+
addRow(w, { session: "sSame", date: "2026-10-01", cell: "task-high", model: "claude-opus-5" });
|
|
165
|
+
addRow(w, { session: "sSame", date: "2026-10-01", cell: "slow-high" });
|
|
166
|
+
// slow-medium: same model, a different effort — a real comparison
|
|
167
|
+
addRow(w, { session: "sEff", date: "2026-10-02", cell: "slow-medium" });
|
|
168
|
+
addRow(w, { session: "sEff", date: "2026-10-02", cell: "slow-high" });
|
|
169
|
+
const rows = readout(parsed(w)).cells;
|
|
170
|
+
assert.equal(rows.find((x) => x.cell === "task-high").comparisons, 0);
|
|
171
|
+
assert.equal(rows.find((x) => x.cell === "slow-medium").comparisons, 1);
|
|
172
|
+
assert.equal(rows.find((x) => x.cell === "slow-high"), undefined, "slow-high is the baseline, never a readout line");
|
|
173
|
+
});
|
|
174
|
+
|
|
175
|
+
test("a run_date is the comparison session's member-outcomes run_date, not the ticket-features one", () => {
|
|
176
|
+
const w = world();
|
|
177
|
+
addComparisons(w, "task-high", 10, 5);
|
|
178
|
+
for (const f of w.features) f.run_date = "2026-09-01";
|
|
179
|
+
const r = cli(w);
|
|
180
|
+
assert.equal(lineFor(r.stdout, "task-high"), "task-high 10 5 claude-sonnet-5");
|
|
181
|
+
});
|
|
182
|
+
|
|
183
|
+
test("a re-dispatched Pull is one Pull: two ticket-features rows for one session+agent count the member row once", () => {
|
|
184
|
+
const w = world();
|
|
185
|
+
addRow(w, { session: "sRe", date: "2026-10-01", cell: "task-high" });
|
|
186
|
+
addRow(w, { session: "sRe", date: "2026-10-01", cell: "slow-high" });
|
|
187
|
+
// The re-dispatch: the same Pull's features row again, on the same member row.
|
|
188
|
+
w.features.push({ ...w.features[0] });
|
|
189
|
+
// An unjoined Pull, likewise written twice.
|
|
190
|
+
addRow(w, { session: "sOrphan", date: "2026-10-01", cell: "smol-high", member: false });
|
|
191
|
+
w.features.push({ ...w.features[w.features.length - 1] });
|
|
192
|
+
const { cells, unjoined } = readout(parsed(w));
|
|
193
|
+
const task = cells.find((x) => x.cell === "task-high");
|
|
194
|
+
assert.equal(task.comparisons, 1);
|
|
195
|
+
assert.deepEqual([...task.models], [["claude-sonnet-5", 1]], "the member row behind a re-dispatched Pull is counted once");
|
|
196
|
+
assert.equal(unjoined, 1, "a re-dispatched Pull with no member row is one unjoined Pull");
|
|
197
|
+
});
|
|
198
|
+
|
|
199
|
+
test("a blank member-outcomes run_date is no date: its session is a comparison but adds nothing to the distinct-date count", () => {
|
|
200
|
+
const w = world();
|
|
201
|
+
addComparisons(w, "task-high", 4, 4);
|
|
202
|
+
for (let i = 0; i < 6; i++) {
|
|
203
|
+
const session = `sBlank${i}`;
|
|
204
|
+
addRow(w, { session, date: "", cell: "task-high" });
|
|
205
|
+
addRow(w, { session, date: "", cell: "slow-high" });
|
|
206
|
+
}
|
|
207
|
+
const task = readout(parsed(w)).cells.find((x) => x.cell === "task-high");
|
|
208
|
+
assert.equal(task.comparisons, 10);
|
|
209
|
+
assert.deepEqual([task.runDates, task.gated], [4, false], "ten comparisons over four real dates is below the five-date floor");
|
|
210
|
+
assert.ok(!task.dates.includes(""), "a blank run_date is listed as a date");
|
|
211
|
+
});
|
|
212
|
+
|
|
213
|
+
test("gated cells print sorted by cell name, not in ticket-features order", () => {
|
|
214
|
+
const w = world();
|
|
215
|
+
// Rows are added in reverse alphabetical order: task-high first, smol-high second.
|
|
216
|
+
addComparisons(w, "task-high", 10, 5);
|
|
217
|
+
addComparisons(w, "smol-high", 10, 5);
|
|
218
|
+
const r = cli(w);
|
|
219
|
+
assert.equal(r.status, 0, r.stderr);
|
|
220
|
+
assert.deepEqual(
|
|
221
|
+
r.stdout.split("\n").filter(Boolean),
|
|
222
|
+
["smol-high 10 5 claude-sonnet-5", "task-high 10 5 claude-sonnet-5"],
|
|
223
|
+
);
|
|
224
|
+
});
|
|
225
|
+
|
|
226
|
+
test("equal per-model counts in a mixed cell list models alphabetically", () => {
|
|
227
|
+
const w = world();
|
|
228
|
+
// Reverse-alphabetical insertion order: sonnet rows first, then haiku, five each.
|
|
229
|
+
addComparisons(w, "task-high", 5, 5, { model: "claude-sonnet-5" });
|
|
230
|
+
addComparisons(w, "task-high", 5, 5, { model: "claude-haiku-4-5" });
|
|
231
|
+
const r = cli(w);
|
|
232
|
+
assert.equal(r.status, 0, r.stderr);
|
|
233
|
+
assert.equal(lineFor(r.stdout, "task-high"), "task-high 10 5 mixed (claude-haiku-4-5 n=5, claude-sonnet-5 n=5)");
|
|
234
|
+
});
|
|
235
|
+
|
|
236
|
+
// run-team/SKILL.md sends the controller here for a cell's comparison count.
|
|
237
|
+
// The prose is the only thing that does, and nothing else reads it: renaming
|
|
238
|
+
// the script or pointing the floor back at another query leaves every test
|
|
239
|
+
// above green. Each slice starts at an anchor that must occur exactly once and
|
|
240
|
+
// ends at the paragraph's blank line, so a restatement elsewhere in the file
|
|
241
|
+
// cannot satisfy a pin on the paragraph that carries the rule.
|
|
242
|
+
const RUN_TEAM = readFileSync(join(import.meta.dirname, "..", "skills", "run-team", "SKILL.md"), "utf8");
|
|
243
|
+
|
|
244
|
+
test("SKILL.md's floor names the script that prints a cell's count, and that script exists", () => {
|
|
245
|
+
const floor = unemphasized(paragraph(RUN_TEAM, "**Report the count the per-cell readout prints", "run-team's per-cell floor"));
|
|
246
|
+
const named = /`~\/\.fleet\/bin\/fleet-run (\S+\.mjs)`/.exec(floor)?.[1];
|
|
247
|
+
assert.ok(named, "the floor no longer tells the controller which script to run");
|
|
248
|
+
assert.equal(named, "cell-readout.mjs");
|
|
249
|
+
assert.ok(existsSync(join(dirname(SCRIPT), named)), `${named} is not a script beside cell-readout.mjs`);
|
|
250
|
+
assert.match(floor, phrase("prints `<cell> <comparisons> <run_dates> <resolved models>` for each cell past that floor"));
|
|
251
|
+
assert.match(floor, phrase("a cell below it only as a count on stderr"));
|
|
252
|
+
});
|
|
253
|
+
|
|
254
|
+
test("the line SKILL.md says the readout prints is the line it prints: four fields, a cell below the floor on stderr only", () => {
|
|
255
|
+
const gated = world();
|
|
256
|
+
addComparisons(gated, "task-high", 10, 5);
|
|
257
|
+
addComparisons(gated, "smol-high", 2, 2);
|
|
258
|
+
const r = cli(gated);
|
|
259
|
+
assert.equal(r.status, 0, r.stderr);
|
|
260
|
+
assert.equal(lineFor(r.stdout, "task-high"), "task-high 10 5 claude-sonnet-5");
|
|
261
|
+
assert.equal(lineFor(r.stdout, "smol-high"), undefined, "a cell below the floor must not print on stdout");
|
|
262
|
+
assert.match(r.stderr, /smol-high below the gate: 2 comparisons/);
|
|
263
|
+
});
|
|
264
|
+
|
|
265
|
+
test("SKILL.md says the readout counts a pair only when the two rows ran a different model or effort, as the readout does", () => {
|
|
266
|
+
const deliberate = unemphasized(paragraph(RUN_TEAM, "**The readout counts DELIBERATE comparisons.**", "run-team's DELIBERATE paragraph"));
|
|
267
|
+
assert.match(deliberate, phrase("the two rows must actually have RUN a different model or effort"));
|
|
268
|
+
// Run, not read: the same model at the same level is no comparison, the same
|
|
269
|
+
// model at a different level is one, a different model at the same level is one.
|
|
270
|
+
const w = world();
|
|
271
|
+
addRow(w, { session: "sNone", date: "2026-10-01", cell: "task-high", model: "claude-opus-5" });
|
|
272
|
+
addRow(w, { session: "sNone", date: "2026-10-01", cell: "slow-high" });
|
|
273
|
+
addRow(w, { session: "sModel", date: "2026-10-02", cell: "task-high" });
|
|
274
|
+
addRow(w, { session: "sModel", date: "2026-10-02", cell: "slow-high" });
|
|
275
|
+
addRow(w, { session: "sEffort", date: "2026-10-03", cell: "slow-medium" });
|
|
276
|
+
addRow(w, { session: "sEffort", date: "2026-10-03", cell: "slow-high" });
|
|
277
|
+
const { cells } = readout(parsed(w));
|
|
278
|
+
const countOf = (cell) => cells.find((c) => c.cell === cell).comparisons;
|
|
279
|
+
assert.deepEqual([countOf("task-high"), countOf("slow-medium")], [1, 1]);
|
|
280
|
+
});
|
|
281
|
+
|
|
282
|
+
test("the output identifies no comparison: no session, ticket or agent appears on either stream", () => {
|
|
283
|
+
const w = world();
|
|
284
|
+
addComparisons(w, "task-high", 10, 5);
|
|
285
|
+
addComparisons(w, "smol-high", 3, 2);
|
|
286
|
+
addRow(w, { session: "sOrphan", date: "2026-10-01", cell: "task-high", member: false });
|
|
287
|
+
const r = cli(w);
|
|
288
|
+
assert.equal(r.status, 0, r.stderr);
|
|
289
|
+
const out = r.stdout + r.stderr;
|
|
290
|
+
for (const f of w.features) {
|
|
291
|
+
assert.ok(!out.includes(f.session), `output names session ${f.session}`);
|
|
292
|
+
assert.ok(!out.includes(f.agent), `output names agent ${f.agent}`);
|
|
293
|
+
}
|
|
294
|
+
assert.match(r.stderr, /1 ticket-features row has no member-outcomes row/);
|
|
295
|
+
});
|
|
296
|
+
|
|
297
|
+
test("an empty ticket-features.tsv prints no line and says why, exit 0", () => {
|
|
298
|
+
const r = cli(world());
|
|
299
|
+
assert.equal(r.status, 0, r.stderr);
|
|
300
|
+
assert.equal(r.stdout, "");
|
|
301
|
+
assert.match(r.stderr, /no cell other than slow-high has a ticket-features row/);
|
|
302
|
+
});
|
|
303
|
+
|
|
304
|
+
test("usage and input errors exit 2 and print nothing on stdout", () => {
|
|
305
|
+
const w = world();
|
|
306
|
+
addComparisons(w, "task-high", 1, 1);
|
|
307
|
+
const f = files(w);
|
|
308
|
+
for (const args of [
|
|
309
|
+
["--ticket-features", join(f.dir, "absent.tsv"), "--member-outcomes", f.members],
|
|
310
|
+
["--ticket-features", f.features, "--member-outcomes", join(f.dir, "absent.tsv")],
|
|
311
|
+
["--ticket-features", f.features, "--member-outcomes", f.members, "--bogus"],
|
|
312
|
+
]) {
|
|
313
|
+
const r = spawnSync(process.execPath, [SCRIPT, ...args], { encoding: "utf8", cwd: f.dir });
|
|
314
|
+
assert.equal(r.status, 2, `${args.join(" ")}: ${r.stderr}`);
|
|
315
|
+
assert.equal(r.stdout, "");
|
|
316
|
+
}
|
|
317
|
+
writeFileSync(f.members, "# header\nshort\trow\n");
|
|
318
|
+
const bad = spawnSync(process.execPath, [SCRIPT, "--ticket-features", f.features, "--member-outcomes", f.members], { encoding: "utf8", cwd: f.dir });
|
|
319
|
+
assert.equal(bad.status, 2, bad.stderr);
|
|
320
|
+
assert.match(bad.stderr, /malformed row/);
|
|
321
|
+
});
|
|
322
|
+
|
|
323
|
+
test("with no flags it reads docs/metrics/ under the working directory", () => {
|
|
324
|
+
const w = world();
|
|
325
|
+
addComparisons(w, "task-high", 10, 5);
|
|
326
|
+
const f = files(w);
|
|
327
|
+
const metrics = join(f.dir, "docs", "metrics");
|
|
328
|
+
spawnSync("mkdir", ["-p", metrics]);
|
|
329
|
+
spawnSync("cp", [f.features, f.members, metrics]);
|
|
330
|
+
const r = spawnSync(process.execPath, [SCRIPT], { encoding: "utf8", cwd: f.dir });
|
|
331
|
+
assert.equal(r.status, 0, r.stderr);
|
|
332
|
+
assert.equal(lineFor(r.stdout, "task-high"), "task-high 10 5 claude-sonnet-5");
|
|
333
|
+
});
|
|
334
|
+
|
|
335
|
+
// The pure function's inputs, parsed the way the CLI parses them.
|
|
336
|
+
function parsed(w) {
|
|
337
|
+
return {
|
|
338
|
+
features: w.features.map((r) => ({ ...r })),
|
|
339
|
+
members: w.members.map((r) => ({ ...r })),
|
|
340
|
+
};
|
|
341
|
+
}
|
package/scripts/fleet-tick.mjs
CHANGED
|
@@ -39,7 +39,10 @@
|
|
|
39
39
|
// merge queue, and which PRs are owed a review. Also, for each ticket
|
|
40
40
|
// holding the implementer row on a tier mismatch, whether its issue
|
|
41
41
|
// is CLOSED (`gh issue view`, #2485): a closed ticket's mismatch
|
|
42
|
-
// holds nothing.
|
|
42
|
+
// holds nothing. And, for each in-flight `review=` token whose PR
|
|
43
|
+
// the open list does not carry, whether that PR is MERGED or
|
|
44
|
+
// CLOSED (`gh pr view`): a review cannot be running against a
|
|
45
|
+
// finished PR, so its dangling token holds no reviewer slot.
|
|
43
46
|
// main The main checkout against the run-start baseline
|
|
44
47
|
// `.fleet/main-checkout.sha`, through main-checkout.mjs (#2210).
|
|
45
48
|
// Anything but `clean` — dirty, unknown, no baseline — prints one
|
|
@@ -496,7 +499,7 @@ export function unlabelledFinishers(ledger, unqueued) {
|
|
|
496
499
|
.map((u) => ({ pr: u.pr, labelled: u.attempts.filter((a) => a.outcome === "labelled").map((a) => a.name) }));
|
|
497
500
|
}
|
|
498
501
|
|
|
499
|
-
export function deriveRun({ rows, dispatched, drain }, prs, closed = new Set()) {
|
|
502
|
+
export function deriveRun({ rows, dispatched, drain }, prs, closed = new Set(), finished = new Set()) {
|
|
500
503
|
// One entry per member name across `## Dispatched` and every row. A member
|
|
501
504
|
// settled ANYWHERE is settled: `settle` is the only writer of an outcome, and
|
|
502
505
|
// a bare copy beside it is what a whole-line `row` rewrite leaves behind.
|
|
@@ -752,11 +755,23 @@ export function deriveRun({ rows, dispatched, drain }, prs, closed = new Set())
|
|
|
752
755
|
// head that review read, and there is nothing more to review.
|
|
753
756
|
const pastPinDue = (st, head) => st !== undefined && st.pastPinHalt && !st.inFlight && st.reviewedHead !== null
|
|
754
757
|
&& !String(head).toLowerCase().startsWith(st.reviewedHead);
|
|
758
|
+
// A `review=` token with no `reviewed=` after it reads as in flight, and the
|
|
759
|
+
// ledger alone never says otherwise: a review whose own `reviewed=` write
|
|
760
|
+
// never landed stays in flight after its PR merges, holding a reviewer slot
|
|
761
|
+
// for good. `finished` is the PR numbers the caller has found MERGED or
|
|
762
|
+
// CLOSED, and a finished PR has no review running against it. A PR on the
|
|
763
|
+
// open list is open whatever `finished` says, and none is the default, so a
|
|
764
|
+
// PR nobody has asked about keeps its review live.
|
|
765
|
+
const reviewing = [...byPr.entries()].filter(([, st]) => st.inFlight);
|
|
766
|
+
const liveReviews = reviewing.filter(([n]) => open.has(n) || !finished.has(n));
|
|
755
767
|
return {
|
|
756
768
|
implLive: live("impl"),
|
|
757
769
|
fixLive: live("fix-pr"),
|
|
758
770
|
mergeBotLive: live("merge-bot"),
|
|
759
|
-
reviewsLive:
|
|
771
|
+
reviewsLive: liveReviews.length,
|
|
772
|
+
// The in-flight reviews of PRs the open list does not carry: the ones
|
|
773
|
+
// `finished` could retire, so the caller asks gh about these and no others.
|
|
774
|
+
reviewsOffList: reviewing.filter(([n]) => !open.has(n)).map(([n]) => n).sort(asc),
|
|
760
775
|
// A returned review whose survivors no review fix-applier has answered,
|
|
761
776
|
// a conflict hold no fix-applier has cleared, or a dispositions mismatch
|
|
762
777
|
// no retry has answered — on a PR still open, with no fix-applier working
|
|
@@ -824,7 +839,7 @@ export function deriveRun({ rows, dispatched, drain }, prs, closed = new Set())
|
|
|
824
839
|
// run carries no member token of its own.
|
|
825
840
|
live: [
|
|
826
841
|
...all.filter((m) => m.outcome === null).map((m) => m.name),
|
|
827
|
-
...
|
|
842
|
+
...liveReviews.map(([n]) => `review:PR#${n}`),
|
|
828
843
|
],
|
|
829
844
|
};
|
|
830
845
|
}
|
|
@@ -1106,6 +1121,31 @@ function closedTickets(mismatched) {
|
|
|
1106
1121
|
return closed;
|
|
1107
1122
|
}
|
|
1108
1123
|
|
|
1124
|
+
// The PRs, of those whose review the ledger reads as in flight yet the open
|
|
1125
|
+
// list does not carry, that gh says are MERGED or CLOSED. A review cannot be
|
|
1126
|
+
// running against a finished PR, so the ledger's dangling `review=` token must
|
|
1127
|
+
// not hold a reviewer slot for good. A probe that cannot answer — a nonzero
|
|
1128
|
+
// exit, a reply carrying no state, a state that is none of the three — is
|
|
1129
|
+
// disclosed and the review keeps its slot: an unreadable state is never read
|
|
1130
|
+
// as finished. A PR gh calls OPEN keeps it quietly.
|
|
1131
|
+
function finishedReviewPrs(offList) {
|
|
1132
|
+
const finished = new Set();
|
|
1133
|
+
for (const number of offList) {
|
|
1134
|
+
const r = spawnSync("gh", ["pr", "view", String(number), "--json", "state"], { encoding: "utf8", env: gitEnv({ GH_REPO: "" }) });
|
|
1135
|
+
if (r.error || r.status !== 0) {
|
|
1136
|
+
console.error(`${NAME}: ${failure(r, `gh pr view ${number}`)} — review on PR#${number} unconfirmed finished, its slot stands`);
|
|
1137
|
+
continue;
|
|
1138
|
+
}
|
|
1139
|
+
let st = null;
|
|
1140
|
+
try { st = JSON.parse(r.stdout).state; } catch { /* no state: disclosed below */ }
|
|
1141
|
+
if (st === "MERGED" || st === "CLOSED") finished.add(number);
|
|
1142
|
+
else if (st !== "OPEN") {
|
|
1143
|
+
console.error(`${NAME}: gh pr view ${number} printed no PR state — review on PR#${number} unconfirmed finished, its slot stands`);
|
|
1144
|
+
}
|
|
1145
|
+
}
|
|
1146
|
+
return finished;
|
|
1147
|
+
}
|
|
1148
|
+
|
|
1109
1149
|
// shortlist.mjs, run by the tick itself (§ 6 §3). Its per-ticket stderr is
|
|
1110
1150
|
// captured and dropped rather than billed to the controller's context; only a
|
|
1111
1151
|
// failure's last line rides along.
|
|
@@ -1166,7 +1206,8 @@ function main() {
|
|
|
1166
1206
|
const ledger = readLedger();
|
|
1167
1207
|
run = deriveRun(ledger, prs);
|
|
1168
1208
|
const closed = closedTickets(run.tierMismatch);
|
|
1169
|
-
|
|
1209
|
+
const finished = finishedReviewPrs(run.reviewsOffList);
|
|
1210
|
+
if (closed.size || finished.size) run = deriveRun(ledger, prs, closed, finished);
|
|
1170
1211
|
} catch (e) {
|
|
1171
1212
|
if (e instanceof LedgerError) die(e.message);
|
|
1172
1213
|
throw e;
|