@dzhechkov/harness-core 0.8.36 → 0.8.37
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +195 -75
- package/README.md +235 -8
- package/dist/agentdb-index.d.ts +87 -7
- package/dist/agentdb-index.d.ts.map +1 -1
- package/dist/agentdb-index.js +416 -57
- package/dist/agentdb-index.js.map +1 -1
- package/dist/apply-leg.d.ts +19 -1
- package/dist/apply-leg.d.ts.map +1 -1
- package/dist/apply-leg.js +187 -36
- package/dist/apply-leg.js.map +1 -1
- package/dist/codex-rollouts.d.ts +118 -0
- package/dist/codex-rollouts.d.ts.map +1 -0
- package/dist/codex-rollouts.js +297 -0
- package/dist/codex-rollouts.js.map +1 -0
- package/dist/cost-ledger.d.ts +56 -4
- package/dist/cost-ledger.d.ts.map +1 -1
- package/dist/cost-ledger.js +176 -20
- package/dist/cost-ledger.js.map +1 -1
- package/dist/cross-family-control.d.ts +345 -0
- package/dist/cross-family-control.d.ts.map +1 -0
- package/dist/cross-family-control.js +802 -0
- package/dist/cross-family-control.js.map +1 -0
- package/dist/debt-ratchet.d.ts +53 -0
- package/dist/debt-ratchet.d.ts.map +1 -0
- package/dist/debt-ratchet.js +107 -0
- package/dist/debt-ratchet.js.map +1 -0
- package/dist/embedding-config.d.ts +42 -0
- package/dist/embedding-config.d.ts.map +1 -1
- package/dist/embedding-config.js +106 -10
- package/dist/embedding-config.js.map +1 -1
- package/dist/feature-adr-checkpoints.d.ts +6 -0
- package/dist/feature-adr-checkpoints.d.ts.map +1 -1
- package/dist/feature-adr-checkpoints.js +29 -0
- package/dist/feature-adr-checkpoints.js.map +1 -1
- package/dist/feature-adr-decision-recall.d.ts +2 -2
- package/dist/feature-adr-decision-recall.d.ts.map +1 -1
- package/dist/feature-adr-decision-recall.js +5 -3
- package/dist/feature-adr-decision-recall.js.map +1 -1
- package/dist/feature-adr-envelope.d.ts +96 -0
- package/dist/feature-adr-envelope.d.ts.map +1 -0
- package/dist/feature-adr-envelope.js +183 -0
- package/dist/feature-adr-envelope.js.map +1 -0
- package/dist/feature-adr-routing.d.ts +64 -0
- package/dist/feature-adr-routing.d.ts.map +1 -1
- package/dist/feature-adr-routing.js +122 -2
- package/dist/feature-adr-routing.js.map +1 -1
- package/dist/feature-adr-stage-canon.d.ts +79 -0
- package/dist/feature-adr-stage-canon.d.ts.map +1 -0
- package/dist/feature-adr-stage-canon.js +117 -0
- package/dist/feature-adr-stage-canon.js.map +1 -0
- package/dist/index.d.ts +19 -9
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +13 -5
- package/dist/index.js.map +1 -1
- package/dist/loop-blobs.generated.js +4 -4
- package/dist/loop-blobs.generated.js.map +1 -1
- package/dist/mutation-gate.d.ts +51 -0
- package/dist/mutation-gate.d.ts.map +1 -1
- package/dist/mutation-gate.js +295 -0
- package/dist/mutation-gate.js.map +1 -1
- package/dist/qe-bridge.d.ts.map +1 -1
- package/dist/qe-bridge.js +4 -2
- package/dist/qe-bridge.js.map +1 -1
- package/dist/qe-findings.d.ts +107 -0
- package/dist/qe-findings.d.ts.map +1 -0
- package/dist/qe-findings.js +417 -0
- package/dist/qe-findings.js.map +1 -0
- package/dist/recap.d.ts +1 -1
- package/dist/recap.d.ts.map +1 -1
- package/dist/recap.js +4 -2
- package/dist/recap.js.map +1 -1
- package/dist/round.d.ts +74 -1
- package/dist/round.d.ts.map +1 -1
- package/dist/round.js +112 -4
- package/dist/round.js.map +1 -1
- package/dist/run-records.d.ts +60 -0
- package/dist/run-records.d.ts.map +1 -1
- package/dist/run-records.js +244 -2
- package/dist/run-records.js.map +1 -1
- package/dist/score.d.ts +44 -1
- package/dist/score.d.ts.map +1 -1
- package/dist/score.js +78 -5
- package/dist/score.js.map +1 -1
- package/package.json +1 -1
- package/sbom.json +374 -74
- package/src/agentdb-index.ts +423 -60
- package/src/apply-leg.ts +187 -36
- package/src/codex-rollouts.ts +374 -0
- package/src/cost-ledger.ts +232 -24
- package/src/cross-family-control.ts +960 -0
- package/src/debt-ratchet.ts +143 -0
- package/src/embedding-config.ts +131 -10
- package/src/feature-adr-checkpoints.ts +29 -0
- package/src/feature-adr-decision-recall.ts +6 -4
- package/src/feature-adr-envelope.ts +242 -0
- package/src/feature-adr-routing.ts +139 -2
- package/src/feature-adr-stage-canon.ts +141 -0
- package/src/index.ts +60 -6
- package/src/loop-blobs.generated.ts +4 -4
- package/src/mutation-gate.ts +316 -0
- package/src/qe-bridge.ts +4 -2
- package/src/qe-findings.ts +463 -0
- package/src/recap.ts +10 -3
- package/src/round.ts +165 -6
- package/src/run-records.ts +282 -2
- package/src/score.ts +115 -6
|
@@ -0,0 +1,802 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* cross-family-control-branch (ADR-001, tier M): the pure core of `dz control-review` and
|
|
3
|
+
* `dz score --by-family` — normalization, matching, diffing and ledger aggregation for the
|
|
4
|
+
* measurement of foreign-unique findings between two INDEPENDENT reviews of the same tree.
|
|
5
|
+
*
|
|
6
|
+
* PURE, deliberately: no node:fs / node:child_process import here (NFR-1, guarded by
|
|
7
|
+
* test/core-boundary.test.ts). Every file read, subprocess spawn or hash computation belongs to
|
|
8
|
+
* the CLI (`dz control-review`), same as every other core module (qe-findings.ts, qe-bridge.ts).
|
|
9
|
+
*
|
|
10
|
+
* Context (ADR-001): every review in the run-cost ledger and every qe-bridge signoff is ONE
|
|
11
|
+
* direction over one tree — there was no observation of whether the OTHER family finds what the
|
|
12
|
+
* coder's own family misses. This module answers that by diffing two closed-vocabulary finding
|
|
13
|
+
* lists (Codex's `## Findings ledger` table, Claude's qe-bridge signoff) over the SAME scope.
|
|
14
|
+
*
|
|
15
|
+
* ── Fix round 1 (codex-r1-verdict.txt, Grade C, 17 findings) — the vocabulary shift ──────────────
|
|
16
|
+
* Round 1 called an automatic title/location match "matched" — an OVERLAP claim. Codex r1 finding 1
|
|
17
|
+
* proved that claim false with two real counter-examples (a 4/6-token false pair; two unrelated
|
|
18
|
+
* findings in the same file three lines apart). The fix is not a smarter matcher — a smarter matcher
|
|
19
|
+
* still guesses — it is an honest vocabulary: automatic pairs are **candidates**, never overlap.
|
|
20
|
+
* Only a human adjudication produces a **confirmed** pair. Every output (`ControlDiff`, the ledger
|
|
21
|
+
* row, `dz score --by-family`) now keeps FOUR buckets apart: `confirmed`, `candidate`, `onlyCodex`,
|
|
22
|
+
* `onlyClaude` — and the word "overlap"/"matched" is reserved for `confirmed` alone.
|
|
23
|
+
*/
|
|
24
|
+
import { QE_SEVERITIES } from './qe-findings.js';
|
|
25
|
+
/* ── D1: severity normalization (A4) ─────────────────────────────────────────────────────────── */
|
|
26
|
+
/**
|
|
27
|
+
* The Codex control brief and the Claude qe-bridge signoff both speak the informal
|
|
28
|
+
* critical/major/minor vocabulary (`buildBridgePrompt`'s own brief text: "a severity
|
|
29
|
+
* (critical/major/minor)"). This maps that vocabulary onto the CLOSED `QeSeverity` dictionary
|
|
30
|
+
* qe-findings.ts already defines — `critical`→`CRITICAL`, `major`→`HIGH`, `minor`→`LOW` — and
|
|
31
|
+
* refuses (returns `null`) anything else, case/whitespace-insensitive. The CALLER decides what a
|
|
32
|
+
* refusal means (A4: the finding is written into a `refused` bucket with a named reason, never
|
|
33
|
+
* coerced to the nearest known value — coercion here would make the resulting severity tally
|
|
34
|
+
* unprovable, the same argument qe-findings.ts already makes for its own closed dictionaries).
|
|
35
|
+
*/
|
|
36
|
+
export function normalizeBridgeSeverity(s) {
|
|
37
|
+
const t = String(s ?? '').trim().toLowerCase();
|
|
38
|
+
if (t === 'critical')
|
|
39
|
+
return 'CRITICAL';
|
|
40
|
+
if (t === 'major')
|
|
41
|
+
return 'HIGH';
|
|
42
|
+
if (t === 'minor')
|
|
43
|
+
return 'LOW';
|
|
44
|
+
return null;
|
|
45
|
+
}
|
|
46
|
+
/* ── D2: title normalization + matching (A5, A7) ─────────────────────────────────────────────── */
|
|
47
|
+
/**
|
|
48
|
+
* Lowercase, strip punctuation/backticks (anything that is not a Unicode letter or digit becomes
|
|
49
|
+
* a separator), split on whitespace, keep tokens of length >= 3, deduplicate, sort. The resulting
|
|
50
|
+
* token SET is what `matchFindings`/`dedupeWithinFamily` compare with Jaccard similarity — a
|
|
51
|
+
* bag-of-words match, not a substring one, so word order never matters.
|
|
52
|
+
*/
|
|
53
|
+
export function normalizeFindingTitle(t) {
|
|
54
|
+
const cleaned = String(t ?? '').toLowerCase().replace(/[^\p{L}\p{N}]+/gu, ' ');
|
|
55
|
+
const tokens = cleaned.split(/\s+/).filter((w) => w.length >= 3);
|
|
56
|
+
return [...new Set(tokens)].sort();
|
|
57
|
+
}
|
|
58
|
+
function jaccard(a, b) {
|
|
59
|
+
if (a.length === 0 || b.length === 0)
|
|
60
|
+
return 0;
|
|
61
|
+
const sa = new Set(a);
|
|
62
|
+
const sb = new Set(b);
|
|
63
|
+
let inter = 0;
|
|
64
|
+
for (const w of sa)
|
|
65
|
+
if (sb.has(w))
|
|
66
|
+
inter++;
|
|
67
|
+
const union = new Set([...sa, ...sb]).size;
|
|
68
|
+
return union === 0 ? 0 : inter / union;
|
|
69
|
+
}
|
|
70
|
+
/** Two findings are a compatible location for matching purposes when at least one names no file
|
|
71
|
+
* (nothing to contradict), or both name the SAME file. Two findings that each name a DIFFERENT
|
|
72
|
+
* file are never compatible — r1-1/r1-3's shared premise: a file is corroborating evidence only
|
|
73
|
+
* when it agrees; disagreeing file names are disqualifying, not merely uninformative. */
|
|
74
|
+
function filesCompatible(fa, fb) {
|
|
75
|
+
return fa === undefined || fb === undefined || fa === fb;
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* Solves the small assignment problem (Kuhn–Munkres / Hungarian algorithm, O(rows²·cols)) that
|
|
79
|
+
* finds the MINIMUM total cost perfect assignment of every row to a distinct column, `rows <=
|
|
80
|
+
* cols`. Every row/column is real — including a "column" a row is assigned to on the cost-matrix
|
|
81
|
+
* even where there is no genuine candidate — so the CALLER decides which assignments are real
|
|
82
|
+
* matches (this function has no notion of "no match", only of cost).
|
|
83
|
+
*/
|
|
84
|
+
function hungarianMinCost(cost) {
|
|
85
|
+
const rows = cost.length;
|
|
86
|
+
const cols = rows === 0 ? 0 : cost[0].length;
|
|
87
|
+
if (rows === 0 || cols === 0)
|
|
88
|
+
return [];
|
|
89
|
+
const INF = Number.POSITIVE_INFINITY;
|
|
90
|
+
const u = new Array(rows + 1).fill(0);
|
|
91
|
+
const v = new Array(cols + 1).fill(0);
|
|
92
|
+
const p = new Array(cols + 1).fill(0); // p[j] = 1-based row assigned to column j (0 = none)
|
|
93
|
+
const way = new Array(cols + 1).fill(0);
|
|
94
|
+
for (let i = 1; i <= rows; i++) {
|
|
95
|
+
p[0] = i;
|
|
96
|
+
let j0 = 0;
|
|
97
|
+
const minv = new Array(cols + 1).fill(INF);
|
|
98
|
+
const used = new Array(cols + 1).fill(false);
|
|
99
|
+
do {
|
|
100
|
+
used[j0] = true;
|
|
101
|
+
const i0 = p[j0];
|
|
102
|
+
let delta = INF;
|
|
103
|
+
let j1 = -1;
|
|
104
|
+
for (let j = 1; j <= cols; j++) {
|
|
105
|
+
if (used[j])
|
|
106
|
+
continue;
|
|
107
|
+
const cur = cost[i0 - 1][j - 1] - u[i0] - v[j];
|
|
108
|
+
if (cur < minv[j]) {
|
|
109
|
+
minv[j] = cur;
|
|
110
|
+
way[j] = j0;
|
|
111
|
+
}
|
|
112
|
+
if (minv[j] < delta) {
|
|
113
|
+
delta = minv[j];
|
|
114
|
+
j1 = j;
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
for (let j = 0; j <= cols; j++) {
|
|
118
|
+
if (used[j]) {
|
|
119
|
+
u[p[j]] = u[p[j]] + delta;
|
|
120
|
+
v[j] = v[j] - delta;
|
|
121
|
+
}
|
|
122
|
+
else
|
|
123
|
+
minv[j] = minv[j] - delta;
|
|
124
|
+
}
|
|
125
|
+
j0 = j1;
|
|
126
|
+
} while (p[j0] !== 0);
|
|
127
|
+
do {
|
|
128
|
+
const j1 = way[j0];
|
|
129
|
+
p[j0] = p[j1];
|
|
130
|
+
j0 = j1;
|
|
131
|
+
} while (j0 !== 0);
|
|
132
|
+
}
|
|
133
|
+
const result = new Array(rows).fill(-1);
|
|
134
|
+
for (let j = 1; j <= cols; j++) {
|
|
135
|
+
const r = p[j];
|
|
136
|
+
if (r > 0)
|
|
137
|
+
result[r - 1] = j - 1;
|
|
138
|
+
}
|
|
139
|
+
return result;
|
|
140
|
+
}
|
|
141
|
+
/**
|
|
142
|
+
* Deterministic MAXIMUM-CARDINALITY matching between two finding lists, sum-of-score as the
|
|
143
|
+
* secondary objective (ADR-001, amended after Codex r1 finding 2 — greedy-by-score needlessly
|
|
144
|
+
* drops valid pairs: titles `A1/B1="alpha beta gamma delta"`, `A2="gamma delta"`,
|
|
145
|
+
* `B2="alpha beta"` greedily keep only A1–B1 and lose two genuine pairs A1–B2/A2–B1).
|
|
146
|
+
*
|
|
147
|
+
* Solved as a weighted bipartite ASSIGNMENT (Hungarian): a real candidate edge costs
|
|
148
|
+
* `-(BONUS + score)` with `BONUS = 1 + min(rows, cols)`; a non-candidate edge costs `0`. Lead delta
|
|
149
|
+
* after Codex r2 (new HIGH #1): a bonus of `1` did NOT make cardinality dominant — two score-1 edges
|
|
150
|
+
* (weight 4) beat three score-0.25 edges (weight 3.75). Since every score is <= 1, the total score of
|
|
151
|
+
* ANY matching is < min(rows, cols) + 1 = BONUS, so one extra edge always outweighs any score
|
|
152
|
+
* difference: cardinality first, total score second, in one assignment.
|
|
153
|
+
*
|
|
154
|
+
* Two rules propose CANDIDATES (never confirmed overlap — Codex r1 finding 1: an automatic pair,
|
|
155
|
+
* however matched, is a candidate for lead adjudication, printed as such everywhere it travels):
|
|
156
|
+
* - title Jaccard >= `opts.jaccard` (default 0.5) on tokens of length >= 3, AND a compatible file
|
|
157
|
+
* (r1-1: a file that DISAGREES between the two findings disqualifies an otherwise-good title
|
|
158
|
+
* match — two findings about the "same" defect in two different files are two defects);
|
|
159
|
+
* - same file with `|line delta| <= opts.lineSlack` (default 3) AND title Jaccard >=
|
|
160
|
+
* `opts.fileLineJaccard` (default 0.2) — r1-1's second counter-example: same file, adjacent
|
|
161
|
+
* lines, ZERO shared vocabulary used to pair for free; a location match now needs SOME
|
|
162
|
+
* corroborating text, not just proximity.
|
|
163
|
+
* A pair meeting neither rule's threshold is never proposed — "below-threshold titles stay
|
|
164
|
+
* unique" remains the load-bearing property this module protects.
|
|
165
|
+
*/
|
|
166
|
+
export function matchFindings(a, b, opts) {
|
|
167
|
+
if (a.length === 0 || b.length === 0)
|
|
168
|
+
return { pairs: [] };
|
|
169
|
+
const jaccardThreshold = opts?.jaccard ?? 0.5;
|
|
170
|
+
const lineSlack = opts?.lineSlack ?? 3;
|
|
171
|
+
const fileLineJaccardThreshold = opts?.fileLineJaccard ?? 0.2;
|
|
172
|
+
// Canonical id order: any residual tie in the assignment algorithm then resolves the same way
|
|
173
|
+
// every run (determinism), favoring the earliest-scanned column for a tied minimum delta.
|
|
174
|
+
const sa = [...a].sort((x, y) => (x.id < y.id ? -1 : x.id > y.id ? 1 : 0));
|
|
175
|
+
const sb = [...b].sort((x, y) => (x.id < y.id ? -1 : x.id > y.id ? 1 : 0));
|
|
176
|
+
const titleA = sa.map((f) => normalizeFindingTitle(f.title));
|
|
177
|
+
const titleB = sb.map((f) => normalizeFindingTitle(f.title));
|
|
178
|
+
const best = [];
|
|
179
|
+
for (let i = 0; i < sa.length; i++) {
|
|
180
|
+
const row = [];
|
|
181
|
+
const fa = sa[i];
|
|
182
|
+
for (let j = 0; j < sb.length; j++) {
|
|
183
|
+
const fb = sb[j];
|
|
184
|
+
const jscore = jaccard(titleA[i], titleB[j]);
|
|
185
|
+
let candidate = null;
|
|
186
|
+
if (jscore >= jaccardThreshold && filesCompatible(fa.file, fb.file)) {
|
|
187
|
+
candidate = { rule: 'title-jaccard', score: jscore };
|
|
188
|
+
}
|
|
189
|
+
if (fa.file !== undefined && fb.file !== undefined && fa.file === fb.file && fa.line !== undefined && fb.line !== undefined) {
|
|
190
|
+
const dl = Math.abs(fa.line - fb.line);
|
|
191
|
+
if (dl <= lineSlack && jscore >= fileLineJaccardThreshold) {
|
|
192
|
+
const flScore = 1 - dl / (lineSlack + 1);
|
|
193
|
+
if (candidate === null || flScore > candidate.score)
|
|
194
|
+
candidate = { rule: 'file-line', score: flScore };
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
row.push(candidate);
|
|
198
|
+
}
|
|
199
|
+
best.push(row);
|
|
200
|
+
}
|
|
201
|
+
const n = sa.length;
|
|
202
|
+
const m = sb.length;
|
|
203
|
+
const rowsAreA = n <= m;
|
|
204
|
+
const rows = rowsAreA ? n : m;
|
|
205
|
+
const cols = rowsAreA ? m : n;
|
|
206
|
+
const cardinalityBonus = 1 + Math.min(rows, cols);
|
|
207
|
+
const cost = [];
|
|
208
|
+
for (let r = 0; r < rows; r++) {
|
|
209
|
+
const costRow = [];
|
|
210
|
+
for (let c = 0; c < cols; c++) {
|
|
211
|
+
const cand = rowsAreA ? best[r][c] : best[c][r];
|
|
212
|
+
costRow.push(cand ? -(cardinalityBonus + cand.score) : 0);
|
|
213
|
+
}
|
|
214
|
+
cost.push(costRow);
|
|
215
|
+
}
|
|
216
|
+
const assignment = hungarianMinCost(cost);
|
|
217
|
+
const pairs = [];
|
|
218
|
+
for (let r = 0; r < rows; r++) {
|
|
219
|
+
const c = assignment[r];
|
|
220
|
+
if (c < 0)
|
|
221
|
+
continue;
|
|
222
|
+
const cand = rowsAreA ? best[r][c] : best[c][r];
|
|
223
|
+
if (cand === null || cand === undefined)
|
|
224
|
+
continue; // assigned to a zero-cost filler — no real match
|
|
225
|
+
const ai = rowsAreA ? r : c;
|
|
226
|
+
const bi = rowsAreA ? c : r;
|
|
227
|
+
pairs.push({ a: sa[ai].id, b: sb[bi].id, rule: cand.rule, score: cand.score });
|
|
228
|
+
}
|
|
229
|
+
pairs.sort((x, y) => (x.a < y.a ? -1 : x.a > y.a ? 1 : x.b < y.b ? -1 : x.b > y.b ? 1 : 0));
|
|
230
|
+
return { pairs };
|
|
231
|
+
}
|
|
232
|
+
/**
|
|
233
|
+
* A5: collapse near-duplicate findings WITHIN one reviewer's own list (Jaccard >= 0.8 on titles,
|
|
234
|
+
* a tighter threshold than cross-family matching since these are the SAME reviewer restating
|
|
235
|
+
* itself, e.g. across multiple table rows). The first occurrence in list order is kept; every
|
|
236
|
+
* later near-duplicate is recorded in `collapsed` rather than silently dropped.
|
|
237
|
+
*
|
|
238
|
+
* r1-3 (Codex r1 finding 3): collapsing on title tokens ALONE let two DISTINCT defects with the
|
|
239
|
+
* same generic title ("missing null check in parser") in two different files collapse into one —
|
|
240
|
+
* a real defect silently lost. A location disagreement now blocks the collapse: two findings only
|
|
241
|
+
* collapse when their files are COMPATIBLE (both absent, or identical) — same rule `matchFindings`
|
|
242
|
+
* applies across families, applied here within one.
|
|
243
|
+
*/
|
|
244
|
+
export function dedupeWithinFamily(list, jaccardThreshold = 0.8) {
|
|
245
|
+
const kept = [];
|
|
246
|
+
const collapsed = [];
|
|
247
|
+
for (const f of list) {
|
|
248
|
+
const ft = normalizeFindingTitle(f.title);
|
|
249
|
+
let dupOf = null;
|
|
250
|
+
for (const k of kept) {
|
|
251
|
+
if (jaccard(ft, normalizeFindingTitle(k.title)) >= jaccardThreshold && filesCompatible(f.file, k.file)) {
|
|
252
|
+
dupOf = k;
|
|
253
|
+
break;
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
if (dupOf !== null)
|
|
257
|
+
collapsed.push({ kept: dupOf.id, dropped: f.id });
|
|
258
|
+
else
|
|
259
|
+
kept.push(f);
|
|
260
|
+
}
|
|
261
|
+
return { kept, collapsed };
|
|
262
|
+
}
|
|
263
|
+
const MATCH_RULE_LABEL = 'CANDIDATE only (confirmed requires adjudication): title-jaccard>=0.5 with compatible file | ' +
|
|
264
|
+
'same file+line±3 AND jaccard>=0.2';
|
|
265
|
+
function bySeverityCounts(list) {
|
|
266
|
+
const out = {};
|
|
267
|
+
for (const f of list)
|
|
268
|
+
out[f.severity] = (out[f.severity] ?? 0) + 1;
|
|
269
|
+
return out;
|
|
270
|
+
}
|
|
271
|
+
/** Duplicate ids WITHIN one family's raw list are an identity error, not a matching problem
|
|
272
|
+
* (Codex r1 finding 4: `new Map(...)` used to silently keep only the LATER of two same-id
|
|
273
|
+
* findings, discarding the first without a trace). Checked before dedupe, on the raw list. */
|
|
274
|
+
function findDuplicateId(list) {
|
|
275
|
+
const seen = new Set();
|
|
276
|
+
for (const f of list) {
|
|
277
|
+
if (seen.has(f.id))
|
|
278
|
+
return f.id;
|
|
279
|
+
seen.add(f.id);
|
|
280
|
+
}
|
|
281
|
+
return null;
|
|
282
|
+
}
|
|
283
|
+
const NONE_ENTRY_RE = /^(codex|claude):(.+)$/;
|
|
284
|
+
export function diffFamilyFindings(codex, claude, adjudication) {
|
|
285
|
+
const dupCodex = findDuplicateId(codex);
|
|
286
|
+
if (dupCodex !== null)
|
|
287
|
+
return { ok: false, reason: `duplicate finding id codex:${dupCodex}` };
|
|
288
|
+
const dupClaude = findDuplicateId(claude);
|
|
289
|
+
if (dupClaude !== null)
|
|
290
|
+
return { ok: false, reason: `duplicate finding id claude:${dupClaude}` };
|
|
291
|
+
const dedupCodex = dedupeWithinFamily(codex);
|
|
292
|
+
const dedupClaude = dedupeWithinFamily(claude);
|
|
293
|
+
const codexById = new Map(dedupCodex.kept.map((f) => [f.id, f]));
|
|
294
|
+
const claudeById = new Map(dedupClaude.kept.map((f) => [f.id, f]));
|
|
295
|
+
const confirmed = [];
|
|
296
|
+
const excludedCodex = new Set();
|
|
297
|
+
const excludedClaude = new Set();
|
|
298
|
+
let adjudicated = false;
|
|
299
|
+
if (adjudication !== undefined) {
|
|
300
|
+
adjudicated = true;
|
|
301
|
+
// r1-5: every codex/claude endpoint may be named in `pairs` AT MOST ONCE — a repeated endpoint
|
|
302
|
+
// produced one-to-many matches (Codex r1 finding 5's `{codex:"c1",claude:"l1"}` +
|
|
303
|
+
// `{codex:"c1",claude:"l2"}`).
|
|
304
|
+
const usedCodexInPairs = new Set();
|
|
305
|
+
const usedClaudeInPairs = new Set();
|
|
306
|
+
for (const p of adjudication.pairs) {
|
|
307
|
+
const cId = String(p.codex);
|
|
308
|
+
const clId = String(p.claude);
|
|
309
|
+
if (usedCodexInPairs.has(cId))
|
|
310
|
+
return { ok: false, reason: `adjudication reuses codex finding id ${cId} in more than one pair` };
|
|
311
|
+
if (usedClaudeInPairs.has(clId))
|
|
312
|
+
return { ok: false, reason: `adjudication reuses claude finding id ${clId} in more than one pair` };
|
|
313
|
+
usedCodexInPairs.add(cId);
|
|
314
|
+
usedClaudeInPairs.add(clId);
|
|
315
|
+
if (!codexById.has(cId))
|
|
316
|
+
return { ok: false, reason: `unknown finding id codex:${cId}` };
|
|
317
|
+
if (!claudeById.has(clId))
|
|
318
|
+
return { ok: false, reason: `unknown finding id ${clId}` };
|
|
319
|
+
confirmed.push({ codex: cId, claude: clId });
|
|
320
|
+
excludedCodex.add(cId);
|
|
321
|
+
excludedClaude.add(clId);
|
|
322
|
+
}
|
|
323
|
+
for (const raw of adjudication.none) {
|
|
324
|
+
const m = NONE_ENTRY_RE.exec(raw);
|
|
325
|
+
if (m === null) {
|
|
326
|
+
return { ok: false, reason: `adjudication "none" entry ${JSON.stringify(raw)} must be family-qualified as codex:<id> or claude:<id>` };
|
|
327
|
+
}
|
|
328
|
+
const fam = m[1];
|
|
329
|
+
const id = m[2];
|
|
330
|
+
if (fam === 'codex') {
|
|
331
|
+
if (usedCodexInPairs.has(id))
|
|
332
|
+
return { ok: false, reason: `adjudication pair/none conflict for codex:${id}` };
|
|
333
|
+
if (!codexById.has(id))
|
|
334
|
+
return { ok: false, reason: `unknown finding id codex:${id}` };
|
|
335
|
+
excludedCodex.add(id);
|
|
336
|
+
}
|
|
337
|
+
else {
|
|
338
|
+
if (usedClaudeInPairs.has(id))
|
|
339
|
+
return { ok: false, reason: `adjudication pair/none conflict for claude:${id}` };
|
|
340
|
+
if (!claudeById.has(id))
|
|
341
|
+
return { ok: false, reason: `unknown finding id claude:${id}` };
|
|
342
|
+
excludedClaude.add(id);
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
}
|
|
346
|
+
// The automatic rule runs ONLY over what adjudication left unclaimed — a named pair or a named
|
|
347
|
+
// "none" always wins over the heuristic (A5/A7's "adjudication overrides the automatic match").
|
|
348
|
+
const remainingCodex = [...codexById.values()].filter((f) => !excludedCodex.has(f.id));
|
|
349
|
+
const remainingClaude = [...claudeById.values()].filter((f) => !excludedClaude.has(f.id));
|
|
350
|
+
const auto = matchFindings(remainingCodex, remainingClaude);
|
|
351
|
+
const candidate = auto.pairs.map((p) => ({ codex: p.a, claude: p.b, rule: p.rule, score: p.score }));
|
|
352
|
+
const confirmedCodexIds = new Set(confirmed.map((p) => p.codex));
|
|
353
|
+
const confirmedClaudeIds = new Set(confirmed.map((p) => p.claude));
|
|
354
|
+
const candidateCodexIds = new Set(candidate.map((p) => p.codex));
|
|
355
|
+
const candidateClaudeIds = new Set(candidate.map((p) => p.claude));
|
|
356
|
+
const onlyCodex = [...codexById.values()].filter((f) => !confirmedCodexIds.has(f.id) && !candidateCodexIds.has(f.id)).map((f) => f.id);
|
|
357
|
+
const onlyClaude = [...claudeById.values()].filter((f) => !confirmedClaudeIds.has(f.id) && !candidateClaudeIds.has(f.id)).map((f) => f.id);
|
|
358
|
+
const bySeverity = {
|
|
359
|
+
confirmed: bySeverityCounts(confirmed.map((p) => codexById.get(p.codex))),
|
|
360
|
+
candidate: bySeverityCounts(candidate.map((p) => codexById.get(p.codex))),
|
|
361
|
+
onlyCodex: bySeverityCounts(onlyCodex.map((id) => codexById.get(id))),
|
|
362
|
+
onlyClaude: bySeverityCounts(onlyClaude.map((id) => claudeById.get(id))),
|
|
363
|
+
};
|
|
364
|
+
return {
|
|
365
|
+
ok: true,
|
|
366
|
+
confirmed,
|
|
367
|
+
candidate,
|
|
368
|
+
onlyCodex,
|
|
369
|
+
onlyClaude,
|
|
370
|
+
bySeverity,
|
|
371
|
+
matchRule: MATCH_RULE_LABEL,
|
|
372
|
+
adjudicated,
|
|
373
|
+
collapsed: { codex: dedupCodex.collapsed.length, claude: dedupClaude.collapsed.length },
|
|
374
|
+
};
|
|
375
|
+
}
|
|
376
|
+
/** A named field present but EMPTY (`{}`, an object with no keys) is not the same as absent — the
|
|
377
|
+
* lesson this guards: a presence-only check on a required object field lets an empty stand-in
|
|
378
|
+
* through. Every required nested object below is checked for at least one key, not merely typeof. */
|
|
379
|
+
function isNonEmptyObject(v) {
|
|
380
|
+
return v !== null && typeof v === 'object' && !Array.isArray(v) && Object.keys(v).length > 0;
|
|
381
|
+
}
|
|
382
|
+
/**
|
|
383
|
+
* Refuses (never throws) on:
|
|
384
|
+
* - an empty scope (nothing was reviewed);
|
|
385
|
+
* - either required half missing, malformed, or carrying an EMPTY object where a real result was
|
|
386
|
+
* required (the empty-required-object lesson);
|
|
387
|
+
* - either half lacking an accepted findings table (A1) — an accepted-HOLLOW half (a genuine
|
|
388
|
+
* zero-findings verdict) is fine; a half whose only table was rejected, or that has none, is not;
|
|
389
|
+
* - a tree-hash drift across EITHER half (A2) — the claude half compares `before` to
|
|
390
|
+
* `afterClaude`, the codex half compares `afterClaude` to `afterCodex`; a control whose halves
|
|
391
|
+
* did not see the identical tree writes NO ledger row.
|
|
392
|
+
* Unnormalizable-severity / out-of-scope findings do NOT refuse the row outright (they are a
|
|
393
|
+
* partial-measurement fact, not a total failure) — they land in `refused`/`complete:false` instead.
|
|
394
|
+
*/
|
|
395
|
+
export function buildControlRow(input) {
|
|
396
|
+
if (typeof input.slug !== 'string' || input.slug.trim() === '')
|
|
397
|
+
return { ok: false, reason: 'slug is required' };
|
|
398
|
+
if (typeof input.runId !== 'string' || input.runId.trim() === '')
|
|
399
|
+
return { ok: false, reason: 'runId is required' };
|
|
400
|
+
if (input.coderFamily !== 'codex' && input.coderFamily !== 'claude') {
|
|
401
|
+
return { ok: false, reason: 'coderFamily must be codex or claude' };
|
|
402
|
+
}
|
|
403
|
+
if (!Array.isArray(input.scope) || input.scope.length === 0 || input.scope.some((s) => typeof s !== 'string' || s.trim() === '')) {
|
|
404
|
+
return { ok: false, reason: 'scope must be a non-empty list of files — nothing was reviewed' };
|
|
405
|
+
}
|
|
406
|
+
if (!isNonEmptyObject(input.tree) ||
|
|
407
|
+
typeof input.tree.before !== 'string' || input.tree.before.trim() === '' ||
|
|
408
|
+
typeof input.tree.afterClaude !== 'string' || input.tree.afterClaude.trim() === '' ||
|
|
409
|
+
typeof input.tree.afterCodex !== 'string' || input.tree.afterCodex.trim() === '') {
|
|
410
|
+
return { ok: false, reason: 'tree snapshot hashes (before/afterClaude/afterCodex) are required' };
|
|
411
|
+
}
|
|
412
|
+
if (input.tree.before !== input.tree.afterClaude) {
|
|
413
|
+
return { ok: false, reason: `treeSha drift after claude half: ${input.tree.before} -> ${input.tree.afterClaude}` };
|
|
414
|
+
}
|
|
415
|
+
if (input.tree.afterClaude !== input.tree.afterCodex) {
|
|
416
|
+
return { ok: false, reason: `treeSha drift after codex half: ${input.tree.afterClaude} -> ${input.tree.afterCodex}` };
|
|
417
|
+
}
|
|
418
|
+
if (!isNonEmptyObject(input.claude) || typeof input.claude.accepted !== 'boolean') {
|
|
419
|
+
return { ok: false, reason: 'claude half result is required' };
|
|
420
|
+
}
|
|
421
|
+
if (!input.claude.accepted)
|
|
422
|
+
return { ok: false, reason: 'claude half has no accepted findings table' };
|
|
423
|
+
if (!isNonEmptyObject(input.codex) || typeof input.codex.accepted !== 'boolean') {
|
|
424
|
+
return { ok: false, reason: 'codex half result is required' };
|
|
425
|
+
}
|
|
426
|
+
if (!input.codex.accepted)
|
|
427
|
+
return { ok: false, reason: 'codex half has no accepted findings table' };
|
|
428
|
+
if (!isNonEmptyObject(input.diff))
|
|
429
|
+
return { ok: false, reason: 'diff is required' };
|
|
430
|
+
if (!isNonEmptyObject(input.diff.bySeverity))
|
|
431
|
+
return { ok: false, reason: 'diff.bySeverity is required' };
|
|
432
|
+
if (!Array.isArray(input.diff.confirmed) || !Array.isArray(input.diff.candidate) ||
|
|
433
|
+
!Array.isArray(input.diff.onlyCodex) || !Array.isArray(input.diff.onlyClaude)) {
|
|
434
|
+
return { ok: false, reason: 'diff.confirmed/candidate/onlyCodex/onlyClaude must be arrays' };
|
|
435
|
+
}
|
|
436
|
+
const refusedClaude = input.refused?.claude ?? 0;
|
|
437
|
+
const refusedCodex = input.refused?.codex ?? 0;
|
|
438
|
+
const refusedReasons = input.refused?.reasons ?? [];
|
|
439
|
+
return {
|
|
440
|
+
ok: true,
|
|
441
|
+
row: {
|
|
442
|
+
slug: input.slug,
|
|
443
|
+
stage: 'control',
|
|
444
|
+
runId: input.runId,
|
|
445
|
+
coderFamily: input.coderFamily,
|
|
446
|
+
scope: input.scope,
|
|
447
|
+
treeShaBefore: input.tree.before,
|
|
448
|
+
treeShaAfterClaude: input.tree.afterClaude,
|
|
449
|
+
treeShaAfterCodex: input.tree.afterCodex,
|
|
450
|
+
codexGrade: input.codex.grade,
|
|
451
|
+
claudeGrade: input.claude.grade,
|
|
452
|
+
confirmed: input.diff.confirmed.length,
|
|
453
|
+
candidate: input.diff.candidate.length,
|
|
454
|
+
onlyCodex: input.diff.onlyCodex.length,
|
|
455
|
+
onlyClaude: input.diff.onlyClaude.length,
|
|
456
|
+
bySeverity: input.diff.bySeverity,
|
|
457
|
+
matchRule: input.diff.matchRule,
|
|
458
|
+
adjudicated: input.diff.adjudicated,
|
|
459
|
+
collapsed: input.diff.collapsed,
|
|
460
|
+
refused: { claude: refusedClaude, codex: refusedCodex, reasons: refusedReasons },
|
|
461
|
+
complete: refusedClaude === 0 && refusedCodex === 0,
|
|
462
|
+
tokens: input.tokens ?? null,
|
|
463
|
+
minutes: input.minutes ?? null,
|
|
464
|
+
},
|
|
465
|
+
};
|
|
466
|
+
}
|
|
467
|
+
const CONTROL_BY_SEVERITY_KEYS = ['confirmed', 'candidate', 'onlyCodex', 'onlyClaude'];
|
|
468
|
+
const QE_SEVERITY_SET = new Set(QE_SEVERITIES);
|
|
469
|
+
function isNonNegInt(v) {
|
|
470
|
+
return typeof v === 'number' && Number.isFinite(v) && Number.isInteger(v) && v >= 0;
|
|
471
|
+
}
|
|
472
|
+
/** A severity-count map: any object whose keys are all in the closed `QE_SEVERITIES` dictionary
|
|
473
|
+
* and whose values are all non-negative integers. An EMPTY map (`{}`) is valid — a genuine
|
|
474
|
+
* zero-findings bucket is not the same defect as the presence-only-check lesson guards against
|
|
475
|
+
* (that lesson is about a REQUIRED object being empty, not a legitimately-empty COUNT map). */
|
|
476
|
+
function isValidSeverityCounts(v) {
|
|
477
|
+
if (v === null || typeof v !== 'object' || Array.isArray(v))
|
|
478
|
+
return false;
|
|
479
|
+
for (const [k, val] of Object.entries(v)) {
|
|
480
|
+
if (!QE_SEVERITY_SET.has(k))
|
|
481
|
+
return false;
|
|
482
|
+
if (!isNonNegInt(val))
|
|
483
|
+
return false;
|
|
484
|
+
}
|
|
485
|
+
return true;
|
|
486
|
+
}
|
|
487
|
+
/** r1-12 (Codex r1 finding 12): `bySeverity:{bogus:1}` used to pass a presence-only check and then
|
|
488
|
+
* crash `aggregateByFamily`'s `Object.entries(undefined)` on the missing `onlyCodex` key. Every
|
|
489
|
+
* one of the four named buckets is now required to be PRESENT and individually valid. */
|
|
490
|
+
function isValidControlBySeverity(v) {
|
|
491
|
+
if (v === null || typeof v !== 'object' || Array.isArray(v))
|
|
492
|
+
return false;
|
|
493
|
+
const obj = v;
|
|
494
|
+
for (const k of Object.keys(obj))
|
|
495
|
+
if (!CONTROL_BY_SEVERITY_KEYS.includes(k))
|
|
496
|
+
return false;
|
|
497
|
+
for (const bucket of CONTROL_BY_SEVERITY_KEYS) {
|
|
498
|
+
if (!(bucket in obj))
|
|
499
|
+
return false;
|
|
500
|
+
if (!isValidSeverityCounts(obj[bucket]))
|
|
501
|
+
return false;
|
|
502
|
+
}
|
|
503
|
+
return true;
|
|
504
|
+
}
|
|
505
|
+
/** Full structural validation of one `stage:'control'` ledger row (r1-12) — every field named in
|
|
506
|
+
* the fix-round brief, checked for TYPE and SHAPE, never merely presence. A row that fails any of
|
|
507
|
+
* these is `unreadable` (folded into `parsed.unreadable`, INCOMPLETE), never a thrown exception. */
|
|
508
|
+
function isValidControlRow(obj) {
|
|
509
|
+
if (typeof obj['slug'] !== 'string' || obj['slug'].trim() === '')
|
|
510
|
+
return false;
|
|
511
|
+
if (typeof obj['runId'] !== 'string' || obj['runId'].trim() === '')
|
|
512
|
+
return false;
|
|
513
|
+
if (obj['coderFamily'] !== 'codex' && obj['coderFamily'] !== 'claude')
|
|
514
|
+
return false;
|
|
515
|
+
const scope = obj['scope'];
|
|
516
|
+
if (!Array.isArray(scope) || scope.length === 0 || scope.some((s) => typeof s !== 'string' || s.trim() === ''))
|
|
517
|
+
return false;
|
|
518
|
+
for (const key of ['treeShaBefore', 'treeShaAfterClaude', 'treeShaAfterCodex']) {
|
|
519
|
+
const v = obj[key];
|
|
520
|
+
if (typeof v !== 'string' || v.trim() === '')
|
|
521
|
+
return false;
|
|
522
|
+
}
|
|
523
|
+
if (!(obj['codexGrade'] === null || typeof obj['codexGrade'] === 'string'))
|
|
524
|
+
return false;
|
|
525
|
+
if (!(obj['claudeGrade'] === null || typeof obj['claudeGrade'] === 'string'))
|
|
526
|
+
return false;
|
|
527
|
+
for (const key of ['confirmed', 'candidate', 'onlyCodex', 'onlyClaude']) {
|
|
528
|
+
if (!isNonNegInt(obj[key]))
|
|
529
|
+
return false;
|
|
530
|
+
}
|
|
531
|
+
if (!isValidControlBySeverity(obj['bySeverity']))
|
|
532
|
+
return false;
|
|
533
|
+
if (typeof obj['matchRule'] !== 'string' || obj['matchRule'].trim() === '')
|
|
534
|
+
return false;
|
|
535
|
+
if (typeof obj['adjudicated'] !== 'boolean')
|
|
536
|
+
return false;
|
|
537
|
+
const collapsed = obj['collapsed'];
|
|
538
|
+
if (!isNonEmptyObject(collapsed) || !isNonNegInt(collapsed['codex']) || !isNonNegInt(collapsed['claude']))
|
|
539
|
+
return false;
|
|
540
|
+
if (typeof obj['complete'] !== 'boolean')
|
|
541
|
+
return false;
|
|
542
|
+
const refused = obj['refused'];
|
|
543
|
+
if (!isNonEmptyObject(refused) || !isNonNegInt(refused['claude']) || !isNonNegInt(refused['codex']))
|
|
544
|
+
return false;
|
|
545
|
+
if (!Array.isArray(refused['reasons']) || refused['reasons'].some((r) => typeof r !== 'string'))
|
|
546
|
+
return false;
|
|
547
|
+
if (!(obj['tokens'] === null || typeof obj['tokens'] === 'number'))
|
|
548
|
+
return false;
|
|
549
|
+
if (!(obj['minutes'] === null || typeof obj['minutes'] === 'number'))
|
|
550
|
+
return false;
|
|
551
|
+
// Lead delta after Codex r2 (new MEDIUM #2): RELATIONAL invariants, not only field shapes — a row
|
|
552
|
+
// claiming `complete:true` with refused entries, a top-level count that disagrees with its own
|
|
553
|
+
// severity map, or unequal tree hashes is a self-contradicting row and is unreadable.
|
|
554
|
+
const refusedTotal = refused['claude'] + refused['codex'];
|
|
555
|
+
if (obj['complete'] !== (refusedTotal === 0))
|
|
556
|
+
return false;
|
|
557
|
+
const bySev = obj['bySeverity'];
|
|
558
|
+
for (const key of ['confirmed', 'candidate', 'onlyCodex', 'onlyClaude']) {
|
|
559
|
+
const sum = Object.values(bySev[key]).reduce((a, b) => a + b, 0);
|
|
560
|
+
if (sum !== obj[key])
|
|
561
|
+
return false;
|
|
562
|
+
}
|
|
563
|
+
if (obj['treeShaBefore'] !== obj['treeShaAfterClaude'] || obj['treeShaAfterClaude'] !== obj['treeShaAfterCodex'])
|
|
564
|
+
return false;
|
|
565
|
+
return true;
|
|
566
|
+
}
|
|
567
|
+
/**
|
|
568
|
+
* Reads every line of a run-cost-ledger.jsonl body, classifying `stage:'control'` rows (this
|
|
569
|
+
* feature's own, fully schema-validated — r1-12), `stage:'round'` and `stage:'full'` rows (the two
|
|
570
|
+
* existing per-review stages `aggregateByFamily` reads for its per-pair table), and counting
|
|
571
|
+
* everything unreadable (A8).
|
|
572
|
+
*/
|
|
573
|
+
export function parseControlRows(lines) {
|
|
574
|
+
const rows = [];
|
|
575
|
+
const roundRows = [];
|
|
576
|
+
const fullRows = [];
|
|
577
|
+
let unreadable = 0;
|
|
578
|
+
for (const raw of lines) {
|
|
579
|
+
const line = raw.trim();
|
|
580
|
+
if (line === '')
|
|
581
|
+
continue;
|
|
582
|
+
let parsed;
|
|
583
|
+
try {
|
|
584
|
+
parsed = JSON.parse(line);
|
|
585
|
+
}
|
|
586
|
+
catch {
|
|
587
|
+
unreadable++;
|
|
588
|
+
continue;
|
|
589
|
+
}
|
|
590
|
+
if (parsed === null || typeof parsed !== 'object' || Array.isArray(parsed)) {
|
|
591
|
+
unreadable++;
|
|
592
|
+
continue;
|
|
593
|
+
}
|
|
594
|
+
const obj = parsed;
|
|
595
|
+
const stage = obj['stage'];
|
|
596
|
+
if (stage === 'control') {
|
|
597
|
+
if (!isValidControlRow(obj)) {
|
|
598
|
+
unreadable++;
|
|
599
|
+
continue;
|
|
600
|
+
}
|
|
601
|
+
rows.push(obj);
|
|
602
|
+
}
|
|
603
|
+
else if (stage === 'round') {
|
|
604
|
+
roundRows.push(obj);
|
|
605
|
+
}
|
|
606
|
+
else if (stage === 'full') {
|
|
607
|
+
fullRows.push(obj);
|
|
608
|
+
}
|
|
609
|
+
// Every other stage (plan/impl/fix/loop-run/round-exec/…) and the header/comment row (no
|
|
610
|
+
// string `stage`) are readable, just not addressed by this module.
|
|
611
|
+
}
|
|
612
|
+
return { rows, roundRows, fullRows, unreadable };
|
|
613
|
+
}
|
|
614
|
+
/**
|
|
615
|
+
* r1-15 (Codex r1 finding 15): a provider-qualified spec like `anthropic/claude-sonnet-4` or
|
|
616
|
+
* `codex:gpt-5.6-sol:high` used to normalize to `other` because the WHOLE string never started
|
|
617
|
+
* with a bare family keyword. The spec is now split on BOTH `/` and `:` into segments, and any
|
|
618
|
+
* segment matching the closed vocabulary decides the family — order-independent, so a leading
|
|
619
|
+
* provider qualifier (`anthropic/…`) or a trailing modifier (`…:high`) no longer hides the model.
|
|
620
|
+
*/
|
|
621
|
+
function normalizeSpecFamily(spec) {
|
|
622
|
+
if (typeof spec !== 'string')
|
|
623
|
+
return 'other';
|
|
624
|
+
const s = spec.trim().toLowerCase();
|
|
625
|
+
if (s === '')
|
|
626
|
+
return 'other';
|
|
627
|
+
const segments = s.split(/[/:]+/).filter((seg) => seg !== '');
|
|
628
|
+
const CLAUDE_SEGMENTS = new Set(['claude', 'sonnet', 'opus', 'fable', 'haiku', 'anthropic']);
|
|
629
|
+
const CODEX_SEGMENTS = new Set(['codex', 'openai']);
|
|
630
|
+
for (const seg of segments)
|
|
631
|
+
if (CLAUDE_SEGMENTS.has(seg) || seg.startsWith('claude'))
|
|
632
|
+
return 'claude';
|
|
633
|
+
for (const seg of segments)
|
|
634
|
+
if (CODEX_SEGMENTS.has(seg) || seg.startsWith('gpt'))
|
|
635
|
+
return 'codex';
|
|
636
|
+
return 'other';
|
|
637
|
+
}
|
|
638
|
+
function newPair() {
|
|
639
|
+
return {
|
|
640
|
+
n: 0, grades: {}, shipped: 0, shippedTotal: 0, fixRoundsList: [],
|
|
641
|
+
foreignN: 0, foreignBySeverity: {}, foreignAuto: 0, foreignAdjudicated: 0, foreignAutoRuns: 0, foreignAdjudicatedRuns: 0, foreignIncomplete: 0,
|
|
642
|
+
refutedN: 0, refutedSum: 0, costN: 0, costSum: 0, drafts: [],
|
|
643
|
+
};
|
|
644
|
+
}
|
|
645
|
+
/**
|
|
646
|
+
* `dz score --by-family`'s aggregate: per (coder family, reviewer family) pair, how many rounds
|
|
647
|
+
* ran, their grade distribution, `shippedShare`/`notShipped`, mean `fixRounds`, the FOREIGN-unique
|
|
648
|
+
* findings measured by `control` rows for that pair's cross-family direction (split `auto` vs
|
|
649
|
+
* `adjudicated`, r1-1/r1-13), `refutedShare` and `costPerConfirmed` from `full` rows' findings
|
|
650
|
+
* tables, and `draftToShipped` — the earliest known verdict for a slug (a qe-bridge signoff, or
|
|
651
|
+
* this run's own claude-half grade) next to its final SHIPPED `round`/`full` grade for that exact
|
|
652
|
+
* (slug, family-pair) key (r1-14). Every ratio is `'unknown'`, never a fabricated `0`, when its
|
|
653
|
+
* denominator is zero (NFR-4: absent data is reported as absent).
|
|
654
|
+
*/
|
|
655
|
+
export function aggregateByFamily(parsed, signoffs) {
|
|
656
|
+
const pairs = new Map();
|
|
657
|
+
const bucket = (coder, reviewer) => {
|
|
658
|
+
const key = `${normalizeSpecFamily(coder)}:${normalizeSpecFamily(reviewer)}`;
|
|
659
|
+
let b = pairs.get(key);
|
|
660
|
+
if (b === undefined) {
|
|
661
|
+
b = newPair();
|
|
662
|
+
pairs.set(key, b);
|
|
663
|
+
}
|
|
664
|
+
return b;
|
|
665
|
+
};
|
|
666
|
+
for (const row of parsed.roundRows) {
|
|
667
|
+
const b = bucket(row['coder'], row['reviewer']);
|
|
668
|
+
b.n++;
|
|
669
|
+
const grade = typeof row['grade'] === 'string' ? row['grade'] : null;
|
|
670
|
+
if (grade !== null)
|
|
671
|
+
b.grades[grade] = (b.grades[grade] ?? 0) + 1;
|
|
672
|
+
const outcome = row['outcome'];
|
|
673
|
+
if (outcome === 'shipped') {
|
|
674
|
+
b.shipped++;
|
|
675
|
+
b.shippedTotal++;
|
|
676
|
+
}
|
|
677
|
+
else if (outcome === 'refuted' || outcome === 'blocked' || outcome === 'abandoned') {
|
|
678
|
+
b.shippedTotal++;
|
|
679
|
+
}
|
|
680
|
+
}
|
|
681
|
+
for (const row of parsed.fullRows) {
|
|
682
|
+
const b = bucket(row['coder'], row['reviewer'] ?? null);
|
|
683
|
+
b.n++;
|
|
684
|
+
const grade = typeof row['grade'] === 'string' ? row['grade'] : null;
|
|
685
|
+
if (grade !== null)
|
|
686
|
+
b.grades[grade] = (b.grades[grade] ?? 0) + 1;
|
|
687
|
+
if (typeof row['fixRounds'] === 'number' && Number.isFinite(row['fixRounds']))
|
|
688
|
+
b.fixRoundsList.push(row['fixRounds']);
|
|
689
|
+
const findings = row['findings'];
|
|
690
|
+
if (findings !== undefined && findings.status === 'present' && isNonEmptyObject(findings.summary?.byStatus)) {
|
|
691
|
+
const byStatus = findings.summary.byStatus;
|
|
692
|
+
const asNum = (v) => (typeof v === 'number' && Number.isFinite(v) ? v : 0);
|
|
693
|
+
const total = Object.values(byStatus).reduce((s, v) => s + asNum(v), 0);
|
|
694
|
+
if (total > 0) {
|
|
695
|
+
b.refutedN++;
|
|
696
|
+
b.refutedSum += asNum(byStatus['refuted']) / total;
|
|
697
|
+
}
|
|
698
|
+
const denom = asNum(byStatus['fixed']) + asNum(byStatus['confirmed']);
|
|
699
|
+
const tokens = typeof row['tokens'] === 'number' && Number.isFinite(row['tokens']) ? row['tokens'] : null;
|
|
700
|
+
if (tokens !== null && tokens > 0 && denom > 0) {
|
|
701
|
+
b.costN++;
|
|
702
|
+
b.costSum += tokens / denom;
|
|
703
|
+
}
|
|
704
|
+
}
|
|
705
|
+
}
|
|
706
|
+
for (const cr of parsed.rows) {
|
|
707
|
+
const reviewerOfInterest = cr.coderFamily === 'codex' ? 'claude' : 'codex';
|
|
708
|
+
const b = bucket(cr.coderFamily, reviewerOfInterest);
|
|
709
|
+
if (!cr.complete) {
|
|
710
|
+
b.foreignIncomplete++;
|
|
711
|
+
continue;
|
|
712
|
+
} // excluded from every measured figure
|
|
713
|
+
b.foreignN++;
|
|
714
|
+
const foreign = cr.coderFamily === 'codex' ? cr.bySeverity.onlyClaude : cr.bySeverity.onlyCodex;
|
|
715
|
+
let foreignTotal = 0;
|
|
716
|
+
for (const [sev, n] of Object.entries(foreign)) {
|
|
717
|
+
b.foreignBySeverity[sev] = (b.foreignBySeverity[sev] ?? 0) + n;
|
|
718
|
+
foreignTotal += n;
|
|
719
|
+
}
|
|
720
|
+
if (cr.adjudicated) {
|
|
721
|
+
b.foreignAdjudicatedRuns++;
|
|
722
|
+
b.foreignAdjudicated += foreignTotal;
|
|
723
|
+
}
|
|
724
|
+
else {
|
|
725
|
+
b.foreignAutoRuns++;
|
|
726
|
+
b.foreignAuto += foreignTotal;
|
|
727
|
+
}
|
|
728
|
+
}
|
|
729
|
+
// draftToShipped: earliest known verdict per slug (a qe-bridge signoff, or this control row's
|
|
730
|
+
// own claude-half grade when no signoff was given) against the slug's final grade — keyed by
|
|
731
|
+
// slug PLUS the normalized (coder,reviewer) family pair (r1-14: two final rows for the same slug
|
|
732
|
+
// but opposite family pairs must never overwrite each other), counting only round/full rows
|
|
733
|
+
// whose `outcome` is EXPLICITLY `'shipped'` (a refuted/blocked/abandoned row is never a "final"
|
|
734
|
+
// — its count is folded into `notShipped` above via `shippedTotal - shipped`). Several shipped
|
|
735
|
+
// candidates for the same key resolve to the LATEST by `ts`, and the discarded count survives as
|
|
736
|
+
// `finals` on the winning entry.
|
|
737
|
+
// Lead delta after Codex r2 (new HIGH #4): the FIRST grade is keyed by slug PLUS the family pair,
|
|
738
|
+
// exactly like the final side — a qe-bridge signoff is always a Claude review of `coderFamily`
|
|
739
|
+
// code, so its key is `<slug>::<coderFamily>:claude`; a control row's Claude half likewise.
|
|
740
|
+
const bySlugFirst = new Map();
|
|
741
|
+
if (signoffs !== undefined) {
|
|
742
|
+
const sorted = [...signoffs].sort((x, y) => (x.emittedAt < y.emittedAt ? -1 : x.emittedAt > y.emittedAt ? 1 : 0));
|
|
743
|
+
for (const s of sorted) {
|
|
744
|
+
const key = `${s.slug}::${normalizeSpecFamily(s.coderFamily)}:claude`;
|
|
745
|
+
if (!bySlugFirst.has(key))
|
|
746
|
+
bySlugFirst.set(key, s.grade);
|
|
747
|
+
}
|
|
748
|
+
}
|
|
749
|
+
for (const cr of parsed.rows) {
|
|
750
|
+
const key = `${cr.slug}::${cr.coderFamily}:claude`;
|
|
751
|
+
if (!bySlugFirst.has(key) && cr.claudeGrade !== null)
|
|
752
|
+
bySlugFirst.set(key, cr.claudeGrade);
|
|
753
|
+
}
|
|
754
|
+
const finalCandidatesByKey = new Map();
|
|
755
|
+
for (const row of [...parsed.roundRows, ...parsed.fullRows]) {
|
|
756
|
+
const slug = typeof row['slug'] === 'string' ? row['slug'] : null;
|
|
757
|
+
const grade = typeof row['grade'] === 'string' ? row['grade'] : null;
|
|
758
|
+
if (slug === null || grade === null)
|
|
759
|
+
continue;
|
|
760
|
+
if (row['outcome'] !== 'shipped')
|
|
761
|
+
continue;
|
|
762
|
+
const key = `${slug}::${normalizeSpecFamily(row['coder'])}:${normalizeSpecFamily(row['reviewer'] ?? null)}`;
|
|
763
|
+
const ts = typeof row['ts'] === 'string' ? row['ts'] : '';
|
|
764
|
+
const list = finalCandidatesByKey.get(key) ?? [];
|
|
765
|
+
list.push({ slug, grade, ts, coder: row['coder'], reviewer: row['reviewer'], key });
|
|
766
|
+
finalCandidatesByKey.set(key, list);
|
|
767
|
+
}
|
|
768
|
+
for (const candidates of finalCandidatesByKey.values()) {
|
|
769
|
+
const sorted = [...candidates].sort((x, y) => (x.ts < y.ts ? -1 : x.ts > y.ts ? 1 : 0));
|
|
770
|
+
const final = sorted[sorted.length - 1];
|
|
771
|
+
const b = bucket(final.coder, final.reviewer);
|
|
772
|
+
b.drafts.push({ slug: final.slug, first: bySlugFirst.get(final.key) ?? 'unknown', final: final.grade, finals: candidates.length });
|
|
773
|
+
}
|
|
774
|
+
const out = {};
|
|
775
|
+
for (const [key, b] of pairs) {
|
|
776
|
+
out[key] = {
|
|
777
|
+
n: b.n,
|
|
778
|
+
grades: b.grades,
|
|
779
|
+
shippedShare: b.shippedTotal > 0 ? b.shipped / b.shippedTotal : 'unknown',
|
|
780
|
+
notShipped: b.shippedTotal - b.shipped,
|
|
781
|
+
fixRounds: {
|
|
782
|
+
n: b.fixRoundsList.length,
|
|
783
|
+
mean: b.fixRoundsList.length > 0 ? b.fixRoundsList.reduce((s, n) => s + n, 0) / b.fixRoundsList.length : 'unknown',
|
|
784
|
+
},
|
|
785
|
+
foreignUnique: {
|
|
786
|
+
n: b.foreignN,
|
|
787
|
+
incompleteRuns: b.foreignIncomplete,
|
|
788
|
+
bySeverity: b.foreignN > 0 ? b.foreignBySeverity : 'unknown',
|
|
789
|
+
auto: b.foreignN > 0 ? b.foreignAuto : 'unknown',
|
|
790
|
+
adjudicated: b.foreignN > 0 ? b.foreignAdjudicated : 'unknown',
|
|
791
|
+
autoRuns: b.foreignN > 0 ? b.foreignAutoRuns : 'unknown',
|
|
792
|
+
adjudicatedRuns: b.foreignN > 0 ? b.foreignAdjudicatedRuns : 'unknown',
|
|
793
|
+
},
|
|
794
|
+
refutedShare: { n: b.refutedN, value: b.refutedN > 0 ? b.refutedSum / b.refutedN : 'unknown' },
|
|
795
|
+
costPerConfirmed: { n: b.costN, value: b.costN > 0 ? b.costSum / b.costN : 'unknown' },
|
|
796
|
+
draftToShipped: b.drafts,
|
|
797
|
+
};
|
|
798
|
+
}
|
|
799
|
+
const incompleteControlRows = parsed.rows.filter((r) => !r.complete).length;
|
|
800
|
+
return { pairs: out, incomplete: parsed.unreadable > 0 || incompleteControlRows > 0, incompleteControlRows, controlRows: parsed.rows.length };
|
|
801
|
+
}
|
|
802
|
+
//# sourceMappingURL=cross-family-control.js.map
|