@dzhechkov/harness-core 0.8.36 → 0.8.38

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/.dz-manifest.json +216 -76
  2. package/README.md +349 -8
  3. package/dist/agentdb-index.d.ts +87 -7
  4. package/dist/agentdb-index.d.ts.map +1 -1
  5. package/dist/agentdb-index.js +416 -57
  6. package/dist/agentdb-index.js.map +1 -1
  7. package/dist/apply-leg.d.ts +19 -1
  8. package/dist/apply-leg.d.ts.map +1 -1
  9. package/dist/apply-leg.js +187 -36
  10. package/dist/apply-leg.js.map +1 -1
  11. package/dist/codex-rollouts.d.ts +118 -0
  12. package/dist/codex-rollouts.d.ts.map +1 -0
  13. package/dist/codex-rollouts.js +297 -0
  14. package/dist/codex-rollouts.js.map +1 -0
  15. package/dist/cost-ledger.d.ts +56 -4
  16. package/dist/cost-ledger.d.ts.map +1 -1
  17. package/dist/cost-ledger.js +176 -20
  18. package/dist/cost-ledger.js.map +1 -1
  19. package/dist/cross-family-control.d.ts +380 -0
  20. package/dist/cross-family-control.d.ts.map +1 -0
  21. package/dist/cross-family-control.js +848 -0
  22. package/dist/cross-family-control.js.map +1 -0
  23. package/dist/debt-ratchet.d.ts +53 -0
  24. package/dist/debt-ratchet.d.ts.map +1 -0
  25. package/dist/debt-ratchet.js +107 -0
  26. package/dist/debt-ratchet.js.map +1 -0
  27. package/dist/embedding-config.d.ts +42 -0
  28. package/dist/embedding-config.d.ts.map +1 -1
  29. package/dist/embedding-config.js +106 -10
  30. package/dist/embedding-config.js.map +1 -1
  31. package/dist/feature-adr-checkpoints.d.ts +6 -0
  32. package/dist/feature-adr-checkpoints.d.ts.map +1 -1
  33. package/dist/feature-adr-checkpoints.js +29 -0
  34. package/dist/feature-adr-checkpoints.js.map +1 -1
  35. package/dist/feature-adr-decision-recall.d.ts +2 -2
  36. package/dist/feature-adr-decision-recall.d.ts.map +1 -1
  37. package/dist/feature-adr-decision-recall.js +5 -3
  38. package/dist/feature-adr-decision-recall.js.map +1 -1
  39. package/dist/feature-adr-envelope.d.ts +96 -0
  40. package/dist/feature-adr-envelope.d.ts.map +1 -0
  41. package/dist/feature-adr-envelope.js +183 -0
  42. package/dist/feature-adr-envelope.js.map +1 -0
  43. package/dist/feature-adr-routing.d.ts +64 -0
  44. package/dist/feature-adr-routing.d.ts.map +1 -1
  45. package/dist/feature-adr-routing.js +133 -3
  46. package/dist/feature-adr-routing.js.map +1 -1
  47. package/dist/feature-adr-stage-canon.d.ts +79 -0
  48. package/dist/feature-adr-stage-canon.d.ts.map +1 -0
  49. package/dist/feature-adr-stage-canon.js +117 -0
  50. package/dist/feature-adr-stage-canon.js.map +1 -0
  51. package/dist/index.d.ts +21 -9
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +15 -5
  54. package/dist/index.js.map +1 -1
  55. package/dist/loop-blobs.generated.js +4 -4
  56. package/dist/loop-blobs.generated.js.map +1 -1
  57. package/dist/mutation-gate.d.ts +51 -0
  58. package/dist/mutation-gate.d.ts.map +1 -1
  59. package/dist/mutation-gate.js +295 -0
  60. package/dist/mutation-gate.js.map +1 -1
  61. package/dist/qe-bridge.d.ts +8 -0
  62. package/dist/qe-bridge.d.ts.map +1 -1
  63. package/dist/qe-bridge.js +4 -2
  64. package/dist/qe-bridge.js.map +1 -1
  65. package/dist/qe-findings.d.ts +107 -0
  66. package/dist/qe-findings.d.ts.map +1 -0
  67. package/dist/qe-findings.js +417 -0
  68. package/dist/qe-findings.js.map +1 -0
  69. package/dist/recap.d.ts +1 -1
  70. package/dist/recap.d.ts.map +1 -1
  71. package/dist/recap.js +4 -2
  72. package/dist/recap.js.map +1 -1
  73. package/dist/review-cost.d.ts +51 -0
  74. package/dist/review-cost.d.ts.map +1 -0
  75. package/dist/review-cost.js +110 -0
  76. package/dist/review-cost.js.map +1 -0
  77. package/dist/round.d.ts +207 -1
  78. package/dist/round.d.ts.map +1 -1
  79. package/dist/round.js +321 -4
  80. package/dist/round.js.map +1 -1
  81. package/dist/run-records.d.ts +97 -0
  82. package/dist/run-records.d.ts.map +1 -1
  83. package/dist/run-records.js +336 -2
  84. package/dist/run-records.js.map +1 -1
  85. package/dist/score.d.ts +44 -1
  86. package/dist/score.d.ts.map +1 -1
  87. package/dist/score.js +78 -5
  88. package/dist/score.js.map +1 -1
  89. package/package.json +1 -1
  90. package/sbom.json +425 -75
  91. package/src/agentdb-index.ts +423 -60
  92. package/src/apply-leg.ts +187 -36
  93. package/src/codex-rollouts.ts +374 -0
  94. package/src/cost-ledger.ts +232 -24
  95. package/src/cross-family-control.ts +1038 -0
  96. package/src/debt-ratchet.ts +143 -0
  97. package/src/embedding-config.ts +131 -10
  98. package/src/feature-adr-checkpoints.ts +29 -0
  99. package/src/feature-adr-decision-recall.ts +6 -4
  100. package/src/feature-adr-envelope.ts +242 -0
  101. package/src/feature-adr-routing.ts +150 -3
  102. package/src/feature-adr-stage-canon.ts +141 -0
  103. package/src/index.ts +65 -6
  104. package/src/loop-blobs.generated.ts +4 -4
  105. package/src/mutation-gate.ts +316 -0
  106. package/src/qe-bridge.ts +12 -2
  107. package/src/qe-findings.ts +463 -0
  108. package/src/recap.ts +10 -3
  109. package/src/review-cost.ts +139 -0
  110. package/src/round.ts +481 -6
  111. package/src/run-records.ts +388 -2
  112. package/src/score.ts +115 -6
@@ -0,0 +1,848 @@
1
+ /**
2
+ * cross-family-control-branch (ADR-001, tier M): the pure core of `dz control-review` and
3
+ * `dz score --by-family` — normalization, matching, diffing and ledger aggregation for the
4
+ * measurement of foreign-unique findings between two INDEPENDENT reviews of the same tree.
5
+ *
6
+ * PURE, deliberately: no node:fs / node:child_process import here (NFR-1, guarded by
7
+ * test/core-boundary.test.ts). Every file read, subprocess spawn or hash computation belongs to
8
+ * the CLI (`dz control-review`), same as every other core module (qe-findings.ts, qe-bridge.ts).
9
+ *
10
+ * Context (ADR-001): every review in the run-cost ledger and every qe-bridge signoff is ONE
11
+ * direction over one tree — there was no observation of whether the OTHER family finds what the
12
+ * coder's own family misses. This module answers that by diffing two closed-vocabulary finding
13
+ * lists (Codex's `## Findings ledger` table, Claude's qe-bridge signoff) over the SAME scope.
14
+ *
15
+ * ── Fix round 1 (codex-r1-verdict.txt, Grade C, 17 findings) — the vocabulary shift ──────────────
16
+ * Round 1 called an automatic title/location match "matched" — an OVERLAP claim. Codex r1 finding 1
17
+ * proved that claim false with two real counter-examples (a 4/6-token false pair; two unrelated
18
+ * findings in the same file three lines apart). The fix is not a smarter matcher — a smarter matcher
19
+ * still guesses — it is an honest vocabulary: automatic pairs are **candidates**, never overlap.
20
+ * Only a human adjudication produces a **confirmed** pair. Every output (`ControlDiff`, the ledger
21
+ * row, `dz score --by-family`) now keeps FOUR buckets apart: `confirmed`, `candidate`, `onlyCodex`,
22
+ * `onlyClaude` — and the word "overlap"/"matched" is reserved for `confirmed` alone.
23
+ */
24
+ import { QE_SEVERITIES } from './qe-findings.js';
25
+ /* ── D1: severity normalization (A4) ─────────────────────────────────────────────────────────── */
26
+ /**
27
+ * The Codex control brief and the Claude qe-bridge signoff both speak the informal
28
+ * critical/major/minor vocabulary (`buildBridgePrompt`'s own brief text: "a severity
29
+ * (critical/major/minor)"). This maps that vocabulary onto the CLOSED `QeSeverity` dictionary
30
+ * qe-findings.ts already defines — `critical`→`CRITICAL`, `major`→`HIGH`, `minor`→`LOW` — and
31
+ * refuses (returns `null`) anything else, case/whitespace-insensitive. The CALLER decides what a
32
+ * refusal means (A4: the finding is written into a `refused` bucket with a named reason, never
33
+ * coerced to the nearest known value — coercion here would make the resulting severity tally
34
+ * unprovable, the same argument qe-findings.ts already makes for its own closed dictionaries).
35
+ */
36
+ export function normalizeBridgeSeverity(s) {
37
+ const t = String(s ?? '').trim().toLowerCase();
38
+ if (t === 'critical')
39
+ return 'CRITICAL';
40
+ if (t === 'major')
41
+ return 'HIGH';
42
+ if (t === 'minor')
43
+ return 'LOW';
44
+ return null;
45
+ }
46
+ /* ── D2: title normalization + matching (A5, A7) ─────────────────────────────────────────────── */
47
+ /**
48
+ * Lowercase, strip punctuation/backticks (anything that is not a Unicode letter or digit becomes
49
+ * a separator), split on whitespace, keep tokens of length >= 3, deduplicate, sort. The resulting
50
+ * token SET is what `matchFindings`/`dedupeWithinFamily` compare with Jaccard similarity — a
51
+ * bag-of-words match, not a substring one, so word order never matters.
52
+ */
53
+ export function normalizeFindingTitle(t) {
54
+ const cleaned = String(t ?? '').toLowerCase().replace(/[^\p{L}\p{N}]+/gu, ' ');
55
+ const tokens = cleaned.split(/\s+/).filter((w) => w.length >= 3);
56
+ return [...new Set(tokens)].sort();
57
+ }
58
+ function jaccard(a, b) {
59
+ if (a.length === 0 || b.length === 0)
60
+ return 0;
61
+ const sa = new Set(a);
62
+ const sb = new Set(b);
63
+ let inter = 0;
64
+ for (const w of sa)
65
+ if (sb.has(w))
66
+ inter++;
67
+ const union = new Set([...sa, ...sb]).size;
68
+ return union === 0 ? 0 : inter / union;
69
+ }
70
+ /** Two findings are a compatible location for matching purposes when at least one names no file
71
+ * (nothing to contradict), or both name the SAME file. Two findings that each name a DIFFERENT
72
+ * file are never compatible — r1-1/r1-3's shared premise: a file is corroborating evidence only
73
+ * when it agrees; disagreeing file names are disqualifying, not merely uninformative. */
74
+ function filesCompatible(fa, fb) {
75
+ return fa === undefined || fb === undefined || fa === fb;
76
+ }
77
+ /**
78
+ * Solves the small assignment problem (Kuhn–Munkres / Hungarian algorithm, O(rows²·cols)) that
79
+ * finds the MINIMUM total cost perfect assignment of every row to a distinct column, `rows <=
80
+ * cols`. Every row/column is real — including a "column" a row is assigned to on the cost-matrix
81
+ * even where there is no genuine candidate — so the CALLER decides which assignments are real
82
+ * matches (this function has no notion of "no match", only of cost).
83
+ */
84
+ function hungarianMinCost(cost) {
85
+ const rows = cost.length;
86
+ const cols = rows === 0 ? 0 : cost[0].length;
87
+ if (rows === 0 || cols === 0)
88
+ return [];
89
+ const INF = Number.POSITIVE_INFINITY;
90
+ const u = new Array(rows + 1).fill(0);
91
+ const v = new Array(cols + 1).fill(0);
92
+ const p = new Array(cols + 1).fill(0); // p[j] = 1-based row assigned to column j (0 = none)
93
+ const way = new Array(cols + 1).fill(0);
94
+ for (let i = 1; i <= rows; i++) {
95
+ p[0] = i;
96
+ let j0 = 0;
97
+ const minv = new Array(cols + 1).fill(INF);
98
+ const used = new Array(cols + 1).fill(false);
99
+ do {
100
+ used[j0] = true;
101
+ const i0 = p[j0];
102
+ let delta = INF;
103
+ let j1 = -1;
104
+ for (let j = 1; j <= cols; j++) {
105
+ if (used[j])
106
+ continue;
107
+ const cur = cost[i0 - 1][j - 1] - u[i0] - v[j];
108
+ if (cur < minv[j]) {
109
+ minv[j] = cur;
110
+ way[j] = j0;
111
+ }
112
+ if (minv[j] < delta) {
113
+ delta = minv[j];
114
+ j1 = j;
115
+ }
116
+ }
117
+ for (let j = 0; j <= cols; j++) {
118
+ if (used[j]) {
119
+ u[p[j]] = u[p[j]] + delta;
120
+ v[j] = v[j] - delta;
121
+ }
122
+ else
123
+ minv[j] = minv[j] - delta;
124
+ }
125
+ j0 = j1;
126
+ } while (p[j0] !== 0);
127
+ do {
128
+ const j1 = way[j0];
129
+ p[j0] = p[j1];
130
+ j0 = j1;
131
+ } while (j0 !== 0);
132
+ }
133
+ const result = new Array(rows).fill(-1);
134
+ for (let j = 1; j <= cols; j++) {
135
+ const r = p[j];
136
+ if (r > 0)
137
+ result[r - 1] = j - 1;
138
+ }
139
+ return result;
140
+ }
141
+ /**
142
+ * Deterministic MAXIMUM-CARDINALITY matching between two finding lists, sum-of-score as the
143
+ * secondary objective (ADR-001, amended after Codex r1 finding 2 — greedy-by-score needlessly
144
+ * drops valid pairs: titles `A1/B1="alpha beta gamma delta"`, `A2="gamma delta"`,
145
+ * `B2="alpha beta"` greedily keep only A1–B1 and lose two genuine pairs A1–B2/A2–B1).
146
+ *
147
+ * Solved as a weighted bipartite ASSIGNMENT (Hungarian): a real candidate edge costs
148
+ * `-(BONUS + score)` with `BONUS = 1 + min(rows, cols)`; a non-candidate edge costs `0`. Lead delta
149
+ * after Codex r2 (new HIGH #1): a bonus of `1` did NOT make cardinality dominant — two score-1 edges
150
+ * (weight 4) beat three score-0.25 edges (weight 3.75). Since every score is <= 1, the total score of
151
+ * ANY matching is < min(rows, cols) + 1 = BONUS, so one extra edge always outweighs any score
152
+ * difference: cardinality first, total score second, in one assignment.
153
+ *
154
+ * Two rules propose CANDIDATES (never confirmed overlap — Codex r1 finding 1: an automatic pair,
155
+ * however matched, is a candidate for lead adjudication, printed as such everywhere it travels):
156
+ * - title Jaccard >= `opts.jaccard` (default 0.5) on tokens of length >= 3, AND a compatible file
157
+ * (r1-1: a file that DISAGREES between the two findings disqualifies an otherwise-good title
158
+ * match — two findings about the "same" defect in two different files are two defects);
159
+ * - same file with `|line delta| <= opts.lineSlack` (default 3) AND title Jaccard >=
160
+ * `opts.fileLineJaccard` (default 0.2) — r1-1's second counter-example: same file, adjacent
161
+ * lines, ZERO shared vocabulary used to pair for free; a location match now needs SOME
162
+ * corroborating text, not just proximity.
163
+ * A pair meeting neither rule's threshold is never proposed — "below-threshold titles stay
164
+ * unique" remains the load-bearing property this module protects.
165
+ */
166
+ export function matchFindings(a, b, opts) {
167
+ if (a.length === 0 || b.length === 0)
168
+ return { pairs: [] };
169
+ const jaccardThreshold = opts?.jaccard ?? 0.5;
170
+ const lineSlack = opts?.lineSlack ?? 3;
171
+ const fileLineJaccardThreshold = opts?.fileLineJaccard ?? 0.2;
172
+ // Canonical id order: any residual tie in the assignment algorithm then resolves the same way
173
+ // every run (determinism), favoring the earliest-scanned column for a tied minimum delta.
174
+ const sa = [...a].sort((x, y) => (x.id < y.id ? -1 : x.id > y.id ? 1 : 0));
175
+ const sb = [...b].sort((x, y) => (x.id < y.id ? -1 : x.id > y.id ? 1 : 0));
176
+ const titleA = sa.map((f) => normalizeFindingTitle(f.title));
177
+ const titleB = sb.map((f) => normalizeFindingTitle(f.title));
178
+ const best = [];
179
+ for (let i = 0; i < sa.length; i++) {
180
+ const row = [];
181
+ const fa = sa[i];
182
+ for (let j = 0; j < sb.length; j++) {
183
+ const fb = sb[j];
184
+ const jscore = jaccard(titleA[i], titleB[j]);
185
+ let candidate = null;
186
+ if (jscore >= jaccardThreshold && filesCompatible(fa.file, fb.file)) {
187
+ candidate = { rule: 'title-jaccard', score: jscore };
188
+ }
189
+ if (fa.file !== undefined && fb.file !== undefined && fa.file === fb.file && fa.line !== undefined && fb.line !== undefined) {
190
+ const dl = Math.abs(fa.line - fb.line);
191
+ if (dl <= lineSlack && jscore >= fileLineJaccardThreshold) {
192
+ const flScore = 1 - dl / (lineSlack + 1);
193
+ if (candidate === null || flScore > candidate.score)
194
+ candidate = { rule: 'file-line', score: flScore };
195
+ }
196
+ }
197
+ row.push(candidate);
198
+ }
199
+ best.push(row);
200
+ }
201
+ const n = sa.length;
202
+ const m = sb.length;
203
+ const rowsAreA = n <= m;
204
+ const rows = rowsAreA ? n : m;
205
+ const cols = rowsAreA ? m : n;
206
+ const cardinalityBonus = 1 + Math.min(rows, cols);
207
+ const cost = [];
208
+ for (let r = 0; r < rows; r++) {
209
+ const costRow = [];
210
+ for (let c = 0; c < cols; c++) {
211
+ const cand = rowsAreA ? best[r][c] : best[c][r];
212
+ costRow.push(cand ? -(cardinalityBonus + cand.score) : 0);
213
+ }
214
+ cost.push(costRow);
215
+ }
216
+ const assignment = hungarianMinCost(cost);
217
+ const pairs = [];
218
+ for (let r = 0; r < rows; r++) {
219
+ const c = assignment[r];
220
+ if (c < 0)
221
+ continue;
222
+ const cand = rowsAreA ? best[r][c] : best[c][r];
223
+ if (cand === null || cand === undefined)
224
+ continue; // assigned to a zero-cost filler — no real match
225
+ const ai = rowsAreA ? r : c;
226
+ const bi = rowsAreA ? c : r;
227
+ pairs.push({ a: sa[ai].id, b: sb[bi].id, rule: cand.rule, score: cand.score });
228
+ }
229
+ pairs.sort((x, y) => (x.a < y.a ? -1 : x.a > y.a ? 1 : x.b < y.b ? -1 : x.b > y.b ? 1 : 0));
230
+ return { pairs };
231
+ }
232
+ /**
233
+ * A5: collapse near-duplicate findings WITHIN one reviewer's own list (Jaccard >= 0.8 on titles,
234
+ * a tighter threshold than cross-family matching since these are the SAME reviewer restating
235
+ * itself, e.g. across multiple table rows). The first occurrence in list order is kept; every
236
+ * later near-duplicate is recorded in `collapsed` rather than silently dropped.
237
+ *
238
+ * r1-3 (Codex r1 finding 3): collapsing on title tokens ALONE let two DISTINCT defects with the
239
+ * same generic title ("missing null check in parser") in two different files collapse into one —
240
+ * a real defect silently lost. A location disagreement now blocks the collapse: two findings only
241
+ * collapse when their files are COMPATIBLE (both absent, or identical) — same rule `matchFindings`
242
+ * applies across families, applied here within one.
243
+ */
244
+ export function dedupeWithinFamily(list, jaccardThreshold = 0.8) {
245
+ const kept = [];
246
+ const collapsed = [];
247
+ for (const f of list) {
248
+ const ft = normalizeFindingTitle(f.title);
249
+ let dupOf = null;
250
+ for (const k of kept) {
251
+ if (jaccard(ft, normalizeFindingTitle(k.title)) >= jaccardThreshold && filesCompatible(f.file, k.file)) {
252
+ dupOf = k;
253
+ break;
254
+ }
255
+ }
256
+ if (dupOf !== null)
257
+ collapsed.push({ kept: dupOf.id, dropped: f.id });
258
+ else
259
+ kept.push(f);
260
+ }
261
+ return { kept, collapsed };
262
+ }
263
+ const MATCH_RULE_LABEL = 'CANDIDATE only (confirmed requires adjudication): title-jaccard>=0.5 with compatible file | ' +
264
+ 'same file+line±3 AND jaccard>=0.2';
265
+ function bySeverityCounts(list) {
266
+ const out = {};
267
+ for (const f of list)
268
+ out[f.severity] = (out[f.severity] ?? 0) + 1;
269
+ return out;
270
+ }
271
+ /** Duplicate ids WITHIN one family's raw list are an identity error, not a matching problem
272
+ * (Codex r1 finding 4: `new Map(...)` used to silently keep only the LATER of two same-id
273
+ * findings, discarding the first without a trace). Checked before dedupe, on the raw list. */
274
+ function findDuplicateId(list) {
275
+ const seen = new Set();
276
+ for (const f of list) {
277
+ if (seen.has(f.id))
278
+ return f.id;
279
+ seen.add(f.id);
280
+ }
281
+ return null;
282
+ }
283
+ const NONE_ENTRY_RE = /^(codex|claude):(.+)$/;
284
+ export function diffFamilyFindings(codex, claude, adjudication) {
285
+ const dupCodex = findDuplicateId(codex);
286
+ if (dupCodex !== null)
287
+ return { ok: false, reason: `duplicate finding id codex:${dupCodex}` };
288
+ const dupClaude = findDuplicateId(claude);
289
+ if (dupClaude !== null)
290
+ return { ok: false, reason: `duplicate finding id claude:${dupClaude}` };
291
+ const dedupCodex = dedupeWithinFamily(codex);
292
+ const dedupClaude = dedupeWithinFamily(claude);
293
+ const codexById = new Map(dedupCodex.kept.map((f) => [f.id, f]));
294
+ const claudeById = new Map(dedupClaude.kept.map((f) => [f.id, f]));
295
+ const confirmed = [];
296
+ const excludedCodex = new Set();
297
+ const excludedClaude = new Set();
298
+ let adjudicated = false;
299
+ if (adjudication !== undefined) {
300
+ adjudicated = true;
301
+ // r1-5: every codex/claude endpoint may be named in `pairs` AT MOST ONCE — a repeated endpoint
302
+ // produced one-to-many matches (Codex r1 finding 5's `{codex:"c1",claude:"l1"}` +
303
+ // `{codex:"c1",claude:"l2"}`).
304
+ const usedCodexInPairs = new Set();
305
+ const usedClaudeInPairs = new Set();
306
+ for (const p of adjudication.pairs) {
307
+ const cId = String(p.codex);
308
+ const clId = String(p.claude);
309
+ if (usedCodexInPairs.has(cId))
310
+ return { ok: false, reason: `adjudication reuses codex finding id ${cId} in more than one pair` };
311
+ if (usedClaudeInPairs.has(clId))
312
+ return { ok: false, reason: `adjudication reuses claude finding id ${clId} in more than one pair` };
313
+ usedCodexInPairs.add(cId);
314
+ usedClaudeInPairs.add(clId);
315
+ if (!codexById.has(cId))
316
+ return { ok: false, reason: `unknown finding id codex:${cId}` };
317
+ if (!claudeById.has(clId))
318
+ return { ok: false, reason: `unknown finding id ${clId}` };
319
+ confirmed.push({ codex: cId, claude: clId });
320
+ excludedCodex.add(cId);
321
+ excludedClaude.add(clId);
322
+ }
323
+ for (const raw of adjudication.none) {
324
+ const m = NONE_ENTRY_RE.exec(raw);
325
+ if (m === null) {
326
+ return { ok: false, reason: `adjudication "none" entry ${JSON.stringify(raw)} must be family-qualified as codex:<id> or claude:<id>` };
327
+ }
328
+ const fam = m[1];
329
+ const id = m[2];
330
+ if (fam === 'codex') {
331
+ if (usedCodexInPairs.has(id))
332
+ return { ok: false, reason: `adjudication pair/none conflict for codex:${id}` };
333
+ if (!codexById.has(id))
334
+ return { ok: false, reason: `unknown finding id codex:${id}` };
335
+ excludedCodex.add(id);
336
+ }
337
+ else {
338
+ if (usedClaudeInPairs.has(id))
339
+ return { ok: false, reason: `adjudication pair/none conflict for claude:${id}` };
340
+ if (!claudeById.has(id))
341
+ return { ok: false, reason: `unknown finding id claude:${id}` };
342
+ excludedClaude.add(id);
343
+ }
344
+ }
345
+ }
346
+ // The automatic rule runs ONLY over what adjudication left unclaimed — a named pair or a named
347
+ // "none" always wins over the heuristic (A5/A7's "adjudication overrides the automatic match").
348
+ const remainingCodex = [...codexById.values()].filter((f) => !excludedCodex.has(f.id));
349
+ const remainingClaude = [...claudeById.values()].filter((f) => !excludedClaude.has(f.id));
350
+ const auto = matchFindings(remainingCodex, remainingClaude);
351
+ const candidate = auto.pairs.map((p) => ({ codex: p.a, claude: p.b, rule: p.rule, score: p.score }));
352
+ const confirmedCodexIds = new Set(confirmed.map((p) => p.codex));
353
+ const confirmedClaudeIds = new Set(confirmed.map((p) => p.claude));
354
+ const candidateCodexIds = new Set(candidate.map((p) => p.codex));
355
+ const candidateClaudeIds = new Set(candidate.map((p) => p.claude));
356
+ const onlyCodex = [...codexById.values()].filter((f) => !confirmedCodexIds.has(f.id) && !candidateCodexIds.has(f.id)).map((f) => f.id);
357
+ const onlyClaude = [...claudeById.values()].filter((f) => !confirmedClaudeIds.has(f.id) && !candidateClaudeIds.has(f.id)).map((f) => f.id);
358
+ const bySeverity = {
359
+ confirmed: bySeverityCounts(confirmed.map((p) => codexById.get(p.codex))),
360
+ candidate: bySeverityCounts(candidate.map((p) => codexById.get(p.codex))),
361
+ onlyCodex: bySeverityCounts(onlyCodex.map((id) => codexById.get(id))),
362
+ onlyClaude: bySeverityCounts(onlyClaude.map((id) => claudeById.get(id))),
363
+ };
364
+ return {
365
+ ok: true,
366
+ confirmed,
367
+ candidate,
368
+ onlyCodex,
369
+ onlyClaude,
370
+ bySeverity,
371
+ matchRule: MATCH_RULE_LABEL,
372
+ adjudicated,
373
+ collapsed: { codex: dedupCodex.collapsed.length, claude: dedupClaude.collapsed.length },
374
+ };
375
+ }
376
+ /** A named field present but EMPTY (`{}`, an object with no keys) is not the same as absent — the
377
+ * lesson this guards: a presence-only check on a required object field lets an empty stand-in
378
+ * through. Every required nested object below is checked for at least one key, not merely typeof. */
379
+ function isNonEmptyObject(v) {
380
+ return v !== null && typeof v === 'object' && !Array.isArray(v) && Object.keys(v).length > 0;
381
+ }
382
+ /**
383
+ * Refuses (never throws) on:
384
+ * - an empty scope (nothing was reviewed);
385
+ * - either required half missing, malformed, or carrying an EMPTY object where a real result was
386
+ * required (the empty-required-object lesson);
387
+ * - either half lacking an accepted findings table (A1) — an accepted-HOLLOW half (a genuine
388
+ * zero-findings verdict) is fine; a half whose only table was rejected, or that has none, is not;
389
+ * - a tree-hash drift across EITHER half (A2) — the claude half compares `before` to
390
+ * `afterClaude`, the codex half compares `afterClaude` to `afterCodex`; a control whose halves
391
+ * did not see the identical tree writes NO ledger row.
392
+ * Unnormalizable-severity / out-of-scope findings do NOT refuse the row outright (they are a
393
+ * partial-measurement fact, not a total failure) — they land in `refused`/`complete:false` instead.
394
+ */
395
+ export function buildControlRow(input) {
396
+ if (typeof input.slug !== 'string' || input.slug.trim() === '')
397
+ return { ok: false, reason: 'slug is required' };
398
+ if (typeof input.runId !== 'string' || input.runId.trim() === '')
399
+ return { ok: false, reason: 'runId is required' };
400
+ if (input.coderFamily !== 'codex' && input.coderFamily !== 'claude') {
401
+ return { ok: false, reason: 'coderFamily must be codex or claude' };
402
+ }
403
+ if (!Array.isArray(input.scope) || input.scope.length === 0 || input.scope.some((s) => typeof s !== 'string' || s.trim() === '')) {
404
+ return { ok: false, reason: 'scope must be a non-empty list of files — nothing was reviewed' };
405
+ }
406
+ if (!isNonEmptyObject(input.tree) ||
407
+ typeof input.tree.before !== 'string' || input.tree.before.trim() === '' ||
408
+ typeof input.tree.afterClaude !== 'string' || input.tree.afterClaude.trim() === '' ||
409
+ typeof input.tree.afterCodex !== 'string' || input.tree.afterCodex.trim() === '') {
410
+ return { ok: false, reason: 'tree snapshot hashes (before/afterClaude/afterCodex) are required' };
411
+ }
412
+ if (input.tree.before !== input.tree.afterClaude) {
413
+ return { ok: false, reason: `treeSha drift after claude half: ${input.tree.before} -> ${input.tree.afterClaude}` };
414
+ }
415
+ if (input.tree.afterClaude !== input.tree.afterCodex) {
416
+ return { ok: false, reason: `treeSha drift after codex half: ${input.tree.afterClaude} -> ${input.tree.afterCodex}` };
417
+ }
418
+ if (!isNonEmptyObject(input.claude) || typeof input.claude.accepted !== 'boolean') {
419
+ return { ok: false, reason: 'claude half result is required' };
420
+ }
421
+ if (!input.claude.accepted)
422
+ return { ok: false, reason: 'claude half has no accepted findings table' };
423
+ if (!isNonEmptyObject(input.codex) || typeof input.codex.accepted !== 'boolean') {
424
+ return { ok: false, reason: 'codex half result is required' };
425
+ }
426
+ if (!input.codex.accepted)
427
+ return { ok: false, reason: 'codex half has no accepted findings table' };
428
+ if (!isNonEmptyObject(input.diff))
429
+ return { ok: false, reason: 'diff is required' };
430
+ if (!isNonEmptyObject(input.diff.bySeverity))
431
+ return { ok: false, reason: 'diff.bySeverity is required' };
432
+ if (!Array.isArray(input.diff.confirmed) || !Array.isArray(input.diff.candidate) ||
433
+ !Array.isArray(input.diff.onlyCodex) || !Array.isArray(input.diff.onlyClaude)) {
434
+ return { ok: false, reason: 'diff.confirmed/candidate/onlyCodex/onlyClaude must be arrays' };
435
+ }
436
+ const refusedClaude = input.refused?.claude ?? 0;
437
+ const refusedCodex = input.refused?.codex ?? 0;
438
+ const refusedReasons = input.refused?.reasons ?? [];
439
+ return {
440
+ ok: true,
441
+ row: {
442
+ slug: input.slug,
443
+ stage: 'control',
444
+ runId: input.runId,
445
+ coderFamily: input.coderFamily,
446
+ scope: input.scope,
447
+ treeShaBefore: input.tree.before,
448
+ treeShaAfterClaude: input.tree.afterClaude,
449
+ treeShaAfterCodex: input.tree.afterCodex,
450
+ codexGrade: input.codex.grade,
451
+ claudeGrade: input.claude.grade,
452
+ confirmed: input.diff.confirmed.length,
453
+ candidate: input.diff.candidate.length,
454
+ onlyCodex: input.diff.onlyCodex.length,
455
+ onlyClaude: input.diff.onlyClaude.length,
456
+ bySeverity: input.diff.bySeverity,
457
+ matchRule: input.diff.matchRule,
458
+ adjudicated: input.diff.adjudicated,
459
+ collapsed: input.diff.collapsed,
460
+ refused: { claude: refusedClaude, codex: refusedCodex, reasons: refusedReasons },
461
+ complete: refusedClaude === 0 && refusedCodex === 0,
462
+ tokens: input.tokens ?? null,
463
+ minutes: input.minutes ?? null,
464
+ },
465
+ };
466
+ }
467
+ function isValidControlRefusedRow(obj) {
468
+ if (typeof obj['slug'] !== 'string' || obj['slug'].trim() === '')
469
+ return false;
470
+ if (typeof obj['runId'] !== 'string' || obj['runId'].trim() === '')
471
+ return false;
472
+ if (obj['half'] !== 'claude' && obj['half'] !== 'codex' && obj['half'] !== 'setup')
473
+ return false;
474
+ if (typeof obj['reason'] !== 'string' || obj['reason'].trim() === '')
475
+ return false;
476
+ if (!(obj['minutes'] === null || (typeof obj['minutes'] === 'number' && Number.isFinite(obj['minutes']))))
477
+ return false;
478
+ if (obj['coderFamily'] !== undefined && obj['coderFamily'] !== 'codex' && obj['coderFamily'] !== 'claude')
479
+ return false;
480
+ return true;
481
+ }
482
+ const CONTROL_BY_SEVERITY_KEYS = ['confirmed', 'candidate', 'onlyCodex', 'onlyClaude'];
483
+ const QE_SEVERITY_SET = new Set(QE_SEVERITIES);
484
+ function isNonNegInt(v) {
485
+ return typeof v === 'number' && Number.isFinite(v) && Number.isInteger(v) && v >= 0;
486
+ }
487
+ /** A severity-count map: any object whose keys are all in the closed `QE_SEVERITIES` dictionary
488
+ * and whose values are all non-negative integers. An EMPTY map (`{}`) is valid — a genuine
489
+ * zero-findings bucket is not the same defect as the presence-only-check lesson guards against
490
+ * (that lesson is about a REQUIRED object being empty, not a legitimately-empty COUNT map). */
491
+ function isValidSeverityCounts(v) {
492
+ if (v === null || typeof v !== 'object' || Array.isArray(v))
493
+ return false;
494
+ for (const [k, val] of Object.entries(v)) {
495
+ if (!QE_SEVERITY_SET.has(k))
496
+ return false;
497
+ if (!isNonNegInt(val))
498
+ return false;
499
+ }
500
+ return true;
501
+ }
502
+ /** r1-12 (Codex r1 finding 12): `bySeverity:{bogus:1}` used to pass a presence-only check and then
503
+ * crash `aggregateByFamily`'s `Object.entries(undefined)` on the missing `onlyCodex` key. Every
504
+ * one of the four named buckets is now required to be PRESENT and individually valid. */
505
+ function isValidControlBySeverity(v) {
506
+ if (v === null || typeof v !== 'object' || Array.isArray(v))
507
+ return false;
508
+ const obj = v;
509
+ for (const k of Object.keys(obj))
510
+ if (!CONTROL_BY_SEVERITY_KEYS.includes(k))
511
+ return false;
512
+ for (const bucket of CONTROL_BY_SEVERITY_KEYS) {
513
+ if (!(bucket in obj))
514
+ return false;
515
+ if (!isValidSeverityCounts(obj[bucket]))
516
+ return false;
517
+ }
518
+ return true;
519
+ }
520
+ /** Full structural validation of one `stage:'control'` ledger row (r1-12) — every field named in
521
+ * the fix-round brief, checked for TYPE and SHAPE, never merely presence. A row that fails any of
522
+ * these is `unreadable` (folded into `parsed.unreadable`, INCOMPLETE), never a thrown exception. */
523
+ function isValidControlRow(obj) {
524
+ if (typeof obj['slug'] !== 'string' || obj['slug'].trim() === '')
525
+ return false;
526
+ if (typeof obj['runId'] !== 'string' || obj['runId'].trim() === '')
527
+ return false;
528
+ if (obj['coderFamily'] !== 'codex' && obj['coderFamily'] !== 'claude')
529
+ return false;
530
+ const scope = obj['scope'];
531
+ if (!Array.isArray(scope) || scope.length === 0 || scope.some((s) => typeof s !== 'string' || s.trim() === ''))
532
+ return false;
533
+ for (const key of ['treeShaBefore', 'treeShaAfterClaude', 'treeShaAfterCodex']) {
534
+ const v = obj[key];
535
+ if (typeof v !== 'string' || v.trim() === '')
536
+ return false;
537
+ }
538
+ if (!(obj['codexGrade'] === null || typeof obj['codexGrade'] === 'string'))
539
+ return false;
540
+ if (!(obj['claudeGrade'] === null || typeof obj['claudeGrade'] === 'string'))
541
+ return false;
542
+ for (const key of ['confirmed', 'candidate', 'onlyCodex', 'onlyClaude']) {
543
+ if (!isNonNegInt(obj[key]))
544
+ return false;
545
+ }
546
+ if (!isValidControlBySeverity(obj['bySeverity']))
547
+ return false;
548
+ if (typeof obj['matchRule'] !== 'string' || obj['matchRule'].trim() === '')
549
+ return false;
550
+ if (typeof obj['adjudicated'] !== 'boolean')
551
+ return false;
552
+ const collapsed = obj['collapsed'];
553
+ if (!isNonEmptyObject(collapsed) || !isNonNegInt(collapsed['codex']) || !isNonNegInt(collapsed['claude']))
554
+ return false;
555
+ if (typeof obj['complete'] !== 'boolean')
556
+ return false;
557
+ const refused = obj['refused'];
558
+ if (!isNonEmptyObject(refused) || !isNonNegInt(refused['claude']) || !isNonNegInt(refused['codex']))
559
+ return false;
560
+ if (!Array.isArray(refused['reasons']) || refused['reasons'].some((r) => typeof r !== 'string'))
561
+ return false;
562
+ if (!(obj['tokens'] === null || typeof obj['tokens'] === 'number'))
563
+ return false;
564
+ if (!(obj['minutes'] === null || typeof obj['minutes'] === 'number'))
565
+ return false;
566
+ // Lead delta after Codex r2 (new MEDIUM #2): RELATIONAL invariants, not only field shapes — a row
567
+ // claiming `complete:true` with refused entries, a top-level count that disagrees with its own
568
+ // severity map, or unequal tree hashes is a self-contradicting row and is unreadable.
569
+ const refusedTotal = refused['claude'] + refused['codex'];
570
+ if (obj['complete'] !== (refusedTotal === 0))
571
+ return false;
572
+ const bySev = obj['bySeverity'];
573
+ for (const key of ['confirmed', 'candidate', 'onlyCodex', 'onlyClaude']) {
574
+ const sum = Object.values(bySev[key]).reduce((a, b) => a + b, 0);
575
+ if (sum !== obj[key])
576
+ return false;
577
+ }
578
+ if (obj['treeShaBefore'] !== obj['treeShaAfterClaude'] || obj['treeShaAfterClaude'] !== obj['treeShaAfterCodex'])
579
+ return false;
580
+ return true;
581
+ }
582
+ /**
583
+ * Reads every line of a run-cost-ledger.jsonl body, classifying `stage:'control'` rows (this
584
+ * feature's own, fully schema-validated — r1-12), `stage:'round'` and `stage:'full'` rows (the two
585
+ * existing per-review stages `aggregateByFamily` reads for its per-pair table), and counting
586
+ * everything unreadable (A8).
587
+ */
588
+ export function parseControlRows(lines) {
589
+ const rows = [];
590
+ const refusedRows = [];
591
+ const roundRows = [];
592
+ const fullRows = [];
593
+ let unreadable = 0;
594
+ for (const raw of lines) {
595
+ const line = raw.trim();
596
+ if (line === '')
597
+ continue;
598
+ let parsed;
599
+ try {
600
+ parsed = JSON.parse(line);
601
+ }
602
+ catch {
603
+ unreadable++;
604
+ continue;
605
+ }
606
+ if (parsed === null || typeof parsed !== 'object' || Array.isArray(parsed)) {
607
+ unreadable++;
608
+ continue;
609
+ }
610
+ const obj = parsed;
611
+ const stage = obj['stage'];
612
+ if (stage === 'control') {
613
+ // experiment-instrument FR-4/A6: `outcome:'refused'` is a DIFFERENT schema from a successful
614
+ // control row — checked FIRST, so a refused row is never mistaken for a malformed successful
615
+ // one (which would count it `unreadable`, losing exactly the receipt this feature adds).
616
+ if (obj['outcome'] === 'refused') {
617
+ if (!isValidControlRefusedRow(obj)) {
618
+ unreadable++;
619
+ continue;
620
+ }
621
+ refusedRows.push(obj);
622
+ continue;
623
+ }
624
+ if (!isValidControlRow(obj)) {
625
+ unreadable++;
626
+ continue;
627
+ }
628
+ rows.push(obj);
629
+ }
630
+ else if (stage === 'round') {
631
+ roundRows.push(obj);
632
+ }
633
+ else if (stage === 'full') {
634
+ fullRows.push(obj);
635
+ }
636
+ // Every other stage (plan/impl/fix/loop-run/round-exec/…) and the header/comment row (no
637
+ // string `stage`) are readable, just not addressed by this module.
638
+ }
639
+ return { rows, refusedRows, roundRows, fullRows, unreadable };
640
+ }
641
+ /**
642
+ * r1-15 (Codex r1 finding 15): a provider-qualified spec like `anthropic/claude-sonnet-4` or
643
+ * `codex:gpt-5.6-sol:high` used to normalize to `other` because the WHOLE string never started
644
+ * with a bare family keyword. The spec is now split on BOTH `/` and `:` into segments, and any
645
+ * segment matching the closed vocabulary decides the family — order-independent, so a leading
646
+ * provider qualifier (`anthropic/…`) or a trailing modifier (`…:high`) no longer hides the model.
647
+ */
648
+ function normalizeSpecFamily(spec) {
649
+ if (typeof spec !== 'string')
650
+ return 'other';
651
+ const s = spec.trim().toLowerCase();
652
+ if (s === '')
653
+ return 'other';
654
+ const segments = s.split(/[/:]+/).filter((seg) => seg !== '');
655
+ const CLAUDE_SEGMENTS = new Set(['claude', 'sonnet', 'opus', 'fable', 'haiku', 'anthropic']);
656
+ const CODEX_SEGMENTS = new Set(['codex', 'openai']);
657
+ for (const seg of segments)
658
+ if (CLAUDE_SEGMENTS.has(seg) || seg.startsWith('claude'))
659
+ return 'claude';
660
+ for (const seg of segments)
661
+ if (CODEX_SEGMENTS.has(seg) || seg.startsWith('gpt'))
662
+ return 'codex';
663
+ return 'other';
664
+ }
665
+ function newPair() {
666
+ return {
667
+ n: 0, grades: {}, shipped: 0, shippedTotal: 0, fixRoundsList: [],
668
+ foreignN: 0, foreignBySeverity: {}, foreignAuto: 0, foreignAdjudicated: 0, foreignAutoRuns: 0, foreignAdjudicatedRuns: 0, foreignIncomplete: 0,
669
+ refutedN: 0, refutedSum: 0, costN: 0, costSum: 0, drafts: [], refusedRuns: 0,
670
+ };
671
+ }
672
+ /**
673
+ * `dz score --by-family`'s aggregate: per (coder family, reviewer family) pair, how many rounds
674
+ * ran, their grade distribution, `shippedShare`/`notShipped`, mean `fixRounds`, the FOREIGN-unique
675
+ * findings measured by `control` rows for that pair's cross-family direction (split `auto` vs
676
+ * `adjudicated`, r1-1/r1-13), `refutedShare` and `costPerConfirmed` from `full` rows' findings
677
+ * tables, and `draftToShipped` — the earliest known verdict for a slug (a qe-bridge signoff, or
678
+ * this run's own claude-half grade) next to its final SHIPPED `round`/`full` grade for that exact
679
+ * (slug, family-pair) key (r1-14). Every ratio is `'unknown'`, never a fabricated `0`, when its
680
+ * denominator is zero (NFR-4: absent data is reported as absent).
681
+ */
682
+ export function aggregateByFamily(parsed, signoffs) {
683
+ const pairs = new Map();
684
+ const bucket = (coder, reviewer) => {
685
+ const key = `${normalizeSpecFamily(coder)}:${normalizeSpecFamily(reviewer)}`;
686
+ let b = pairs.get(key);
687
+ if (b === undefined) {
688
+ b = newPair();
689
+ pairs.set(key, b);
690
+ }
691
+ return b;
692
+ };
693
+ for (const row of parsed.roundRows) {
694
+ const b = bucket(row['coder'], row['reviewer']);
695
+ b.n++;
696
+ const grade = typeof row['grade'] === 'string' ? row['grade'] : null;
697
+ if (grade !== null)
698
+ b.grades[grade] = (b.grades[grade] ?? 0) + 1;
699
+ const outcome = row['outcome'];
700
+ if (outcome === 'shipped') {
701
+ b.shipped++;
702
+ b.shippedTotal++;
703
+ }
704
+ else if (outcome === 'refuted' || outcome === 'blocked' || outcome === 'abandoned') {
705
+ b.shippedTotal++;
706
+ }
707
+ }
708
+ for (const row of parsed.fullRows) {
709
+ const b = bucket(row['coder'], row['reviewer'] ?? null);
710
+ b.n++;
711
+ const grade = typeof row['grade'] === 'string' ? row['grade'] : null;
712
+ if (grade !== null)
713
+ b.grades[grade] = (b.grades[grade] ?? 0) + 1;
714
+ if (typeof row['fixRounds'] === 'number' && Number.isFinite(row['fixRounds']))
715
+ b.fixRoundsList.push(row['fixRounds']);
716
+ const findings = row['findings'];
717
+ if (findings !== undefined && findings.status === 'present' && isNonEmptyObject(findings.summary?.byStatus)) {
718
+ const byStatus = findings.summary.byStatus;
719
+ const asNum = (v) => (typeof v === 'number' && Number.isFinite(v) ? v : 0);
720
+ const total = Object.values(byStatus).reduce((s, v) => s + asNum(v), 0);
721
+ if (total > 0) {
722
+ b.refutedN++;
723
+ b.refutedSum += asNum(byStatus['refuted']) / total;
724
+ }
725
+ const denom = asNum(byStatus['fixed']) + asNum(byStatus['confirmed']);
726
+ const tokens = typeof row['tokens'] === 'number' && Number.isFinite(row['tokens']) ? row['tokens'] : null;
727
+ if (tokens !== null && tokens > 0 && denom > 0) {
728
+ b.costN++;
729
+ b.costSum += tokens / denom;
730
+ }
731
+ }
732
+ }
733
+ for (const cr of parsed.rows) {
734
+ const reviewerOfInterest = cr.coderFamily === 'codex' ? 'claude' : 'codex';
735
+ const b = bucket(cr.coderFamily, reviewerOfInterest);
736
+ if (!cr.complete) {
737
+ b.foreignIncomplete++;
738
+ continue;
739
+ } // excluded from every measured figure
740
+ b.foreignN++;
741
+ const foreign = cr.coderFamily === 'codex' ? cr.bySeverity.onlyClaude : cr.bySeverity.onlyCodex;
742
+ let foreignTotal = 0;
743
+ for (const [sev, n] of Object.entries(foreign)) {
744
+ b.foreignBySeverity[sev] = (b.foreignBySeverity[sev] ?? 0) + n;
745
+ foreignTotal += n;
746
+ }
747
+ if (cr.adjudicated) {
748
+ b.foreignAdjudicatedRuns++;
749
+ b.foreignAdjudicated += foreignTotal;
750
+ }
751
+ else {
752
+ b.foreignAutoRuns++;
753
+ b.foreignAuto += foreignTotal;
754
+ }
755
+ }
756
+ // experiment-instrument FR-4/A6: a refused row is bucketed the SAME way a successful control row
757
+ // is — by its own coderFamily and the complementary reviewer — but ONLY when coderFamily is known
758
+ // (the earliest failures, before the claude half reports which family it reviewed, cannot be
759
+ // attributed to a pair; they still count toward `refusedControlRows` at the top level below,
760
+ // never silently dropped). `n` is deliberately untouched: `n` measures COMPLETED reviews.
761
+ for (const rr of parsed.refusedRows) {
762
+ if (rr.coderFamily === undefined)
763
+ continue;
764
+ const reviewerOfInterest = rr.coderFamily === 'codex' ? 'claude' : 'codex';
765
+ const b = bucket(rr.coderFamily, reviewerOfInterest);
766
+ b.refusedRuns++;
767
+ }
768
+ // draftToShipped: earliest known verdict per slug (a qe-bridge signoff, or this control row's
769
+ // own claude-half grade when no signoff was given) against the slug's final grade — keyed by
770
+ // slug PLUS the normalized (coder,reviewer) family pair (r1-14: two final rows for the same slug
771
+ // but opposite family pairs must never overwrite each other), counting only round/full rows
772
+ // whose `outcome` is EXPLICITLY `'shipped'` (a refuted/blocked/abandoned row is never a "final"
773
+ // — its count is folded into `notShipped` above via `shippedTotal - shipped`). Several shipped
774
+ // candidates for the same key resolve to the LATEST by `ts`, and the discarded count survives as
775
+ // `finals` on the winning entry.
776
+ // Lead delta after Codex r2 (new HIGH #4): the FIRST grade is keyed by slug PLUS the family pair,
777
+ // exactly like the final side — a qe-bridge signoff is always a Claude review of `coderFamily`
778
+ // code, so its key is `<slug>::<coderFamily>:claude`; a control row's Claude half likewise.
779
+ const bySlugFirst = new Map();
780
+ if (signoffs !== undefined) {
781
+ const sorted = [...signoffs].sort((x, y) => (x.emittedAt < y.emittedAt ? -1 : x.emittedAt > y.emittedAt ? 1 : 0));
782
+ for (const s of sorted) {
783
+ const key = `${s.slug}::${normalizeSpecFamily(s.coderFamily)}:claude`;
784
+ if (!bySlugFirst.has(key))
785
+ bySlugFirst.set(key, s.grade);
786
+ }
787
+ }
788
+ for (const cr of parsed.rows) {
789
+ const key = `${cr.slug}::${cr.coderFamily}:claude`;
790
+ if (!bySlugFirst.has(key) && cr.claudeGrade !== null)
791
+ bySlugFirst.set(key, cr.claudeGrade);
792
+ }
793
+ const finalCandidatesByKey = new Map();
794
+ for (const row of [...parsed.roundRows, ...parsed.fullRows]) {
795
+ const slug = typeof row['slug'] === 'string' ? row['slug'] : null;
796
+ const grade = typeof row['grade'] === 'string' ? row['grade'] : null;
797
+ if (slug === null || grade === null)
798
+ continue;
799
+ if (row['outcome'] !== 'shipped')
800
+ continue;
801
+ const key = `${slug}::${normalizeSpecFamily(row['coder'])}:${normalizeSpecFamily(row['reviewer'] ?? null)}`;
802
+ const ts = typeof row['ts'] === 'string' ? row['ts'] : '';
803
+ const list = finalCandidatesByKey.get(key) ?? [];
804
+ list.push({ slug, grade, ts, coder: row['coder'], reviewer: row['reviewer'], key });
805
+ finalCandidatesByKey.set(key, list);
806
+ }
807
+ for (const candidates of finalCandidatesByKey.values()) {
808
+ const sorted = [...candidates].sort((x, y) => (x.ts < y.ts ? -1 : x.ts > y.ts ? 1 : 0));
809
+ const final = sorted[sorted.length - 1];
810
+ const b = bucket(final.coder, final.reviewer);
811
+ b.drafts.push({ slug: final.slug, first: bySlugFirst.get(final.key) ?? 'unknown', final: final.grade, finals: candidates.length });
812
+ }
813
+ const out = {};
814
+ for (const [key, b] of pairs) {
815
+ out[key] = {
816
+ n: b.n,
817
+ grades: b.grades,
818
+ shippedShare: b.shippedTotal > 0 ? b.shipped / b.shippedTotal : 'unknown',
819
+ notShipped: b.shippedTotal - b.shipped,
820
+ fixRounds: {
821
+ n: b.fixRoundsList.length,
822
+ mean: b.fixRoundsList.length > 0 ? b.fixRoundsList.reduce((s, n) => s + n, 0) / b.fixRoundsList.length : 'unknown',
823
+ },
824
+ foreignUnique: {
825
+ n: b.foreignN,
826
+ incompleteRuns: b.foreignIncomplete,
827
+ bySeverity: b.foreignN > 0 ? b.foreignBySeverity : 'unknown',
828
+ auto: b.foreignN > 0 ? b.foreignAuto : 'unknown',
829
+ adjudicated: b.foreignN > 0 ? b.foreignAdjudicated : 'unknown',
830
+ autoRuns: b.foreignN > 0 ? b.foreignAutoRuns : 'unknown',
831
+ adjudicatedRuns: b.foreignN > 0 ? b.foreignAdjudicatedRuns : 'unknown',
832
+ },
833
+ refutedShare: { n: b.refutedN, value: b.refutedN > 0 ? b.refutedSum / b.refutedN : 'unknown' },
834
+ costPerConfirmed: { n: b.costN, value: b.costN > 0 ? b.costSum / b.costN : 'unknown' },
835
+ draftToShipped: b.drafts,
836
+ refusedRuns: b.refusedRuns,
837
+ };
838
+ }
839
+ const incompleteControlRows = parsed.rows.filter((r) => !r.complete).length;
840
+ return {
841
+ pairs: out,
842
+ incomplete: parsed.unreadable > 0 || incompleteControlRows > 0,
843
+ incompleteControlRows,
844
+ controlRows: parsed.rows.length,
845
+ refusedControlRows: parsed.refusedRows.length,
846
+ };
847
+ }
848
+ //# sourceMappingURL=cross-family-control.js.map