@dzhechkov/harness-core 0.8.36 → 0.8.38

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/.dz-manifest.json +216 -76
  2. package/README.md +349 -8
  3. package/dist/agentdb-index.d.ts +87 -7
  4. package/dist/agentdb-index.d.ts.map +1 -1
  5. package/dist/agentdb-index.js +416 -57
  6. package/dist/agentdb-index.js.map +1 -1
  7. package/dist/apply-leg.d.ts +19 -1
  8. package/dist/apply-leg.d.ts.map +1 -1
  9. package/dist/apply-leg.js +187 -36
  10. package/dist/apply-leg.js.map +1 -1
  11. package/dist/codex-rollouts.d.ts +118 -0
  12. package/dist/codex-rollouts.d.ts.map +1 -0
  13. package/dist/codex-rollouts.js +297 -0
  14. package/dist/codex-rollouts.js.map +1 -0
  15. package/dist/cost-ledger.d.ts +56 -4
  16. package/dist/cost-ledger.d.ts.map +1 -1
  17. package/dist/cost-ledger.js +176 -20
  18. package/dist/cost-ledger.js.map +1 -1
  19. package/dist/cross-family-control.d.ts +380 -0
  20. package/dist/cross-family-control.d.ts.map +1 -0
  21. package/dist/cross-family-control.js +848 -0
  22. package/dist/cross-family-control.js.map +1 -0
  23. package/dist/debt-ratchet.d.ts +53 -0
  24. package/dist/debt-ratchet.d.ts.map +1 -0
  25. package/dist/debt-ratchet.js +107 -0
  26. package/dist/debt-ratchet.js.map +1 -0
  27. package/dist/embedding-config.d.ts +42 -0
  28. package/dist/embedding-config.d.ts.map +1 -1
  29. package/dist/embedding-config.js +106 -10
  30. package/dist/embedding-config.js.map +1 -1
  31. package/dist/feature-adr-checkpoints.d.ts +6 -0
  32. package/dist/feature-adr-checkpoints.d.ts.map +1 -1
  33. package/dist/feature-adr-checkpoints.js +29 -0
  34. package/dist/feature-adr-checkpoints.js.map +1 -1
  35. package/dist/feature-adr-decision-recall.d.ts +2 -2
  36. package/dist/feature-adr-decision-recall.d.ts.map +1 -1
  37. package/dist/feature-adr-decision-recall.js +5 -3
  38. package/dist/feature-adr-decision-recall.js.map +1 -1
  39. package/dist/feature-adr-envelope.d.ts +96 -0
  40. package/dist/feature-adr-envelope.d.ts.map +1 -0
  41. package/dist/feature-adr-envelope.js +183 -0
  42. package/dist/feature-adr-envelope.js.map +1 -0
  43. package/dist/feature-adr-routing.d.ts +64 -0
  44. package/dist/feature-adr-routing.d.ts.map +1 -1
  45. package/dist/feature-adr-routing.js +133 -3
  46. package/dist/feature-adr-routing.js.map +1 -1
  47. package/dist/feature-adr-stage-canon.d.ts +79 -0
  48. package/dist/feature-adr-stage-canon.d.ts.map +1 -0
  49. package/dist/feature-adr-stage-canon.js +117 -0
  50. package/dist/feature-adr-stage-canon.js.map +1 -0
  51. package/dist/index.d.ts +21 -9
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +15 -5
  54. package/dist/index.js.map +1 -1
  55. package/dist/loop-blobs.generated.js +4 -4
  56. package/dist/loop-blobs.generated.js.map +1 -1
  57. package/dist/mutation-gate.d.ts +51 -0
  58. package/dist/mutation-gate.d.ts.map +1 -1
  59. package/dist/mutation-gate.js +295 -0
  60. package/dist/mutation-gate.js.map +1 -1
  61. package/dist/qe-bridge.d.ts +8 -0
  62. package/dist/qe-bridge.d.ts.map +1 -1
  63. package/dist/qe-bridge.js +4 -2
  64. package/dist/qe-bridge.js.map +1 -1
  65. package/dist/qe-findings.d.ts +107 -0
  66. package/dist/qe-findings.d.ts.map +1 -0
  67. package/dist/qe-findings.js +417 -0
  68. package/dist/qe-findings.js.map +1 -0
  69. package/dist/recap.d.ts +1 -1
  70. package/dist/recap.d.ts.map +1 -1
  71. package/dist/recap.js +4 -2
  72. package/dist/recap.js.map +1 -1
  73. package/dist/review-cost.d.ts +51 -0
  74. package/dist/review-cost.d.ts.map +1 -0
  75. package/dist/review-cost.js +110 -0
  76. package/dist/review-cost.js.map +1 -0
  77. package/dist/round.d.ts +207 -1
  78. package/dist/round.d.ts.map +1 -1
  79. package/dist/round.js +321 -4
  80. package/dist/round.js.map +1 -1
  81. package/dist/run-records.d.ts +97 -0
  82. package/dist/run-records.d.ts.map +1 -1
  83. package/dist/run-records.js +336 -2
  84. package/dist/run-records.js.map +1 -1
  85. package/dist/score.d.ts +44 -1
  86. package/dist/score.d.ts.map +1 -1
  87. package/dist/score.js +78 -5
  88. package/dist/score.js.map +1 -1
  89. package/package.json +1 -1
  90. package/sbom.json +425 -75
  91. package/src/agentdb-index.ts +423 -60
  92. package/src/apply-leg.ts +187 -36
  93. package/src/codex-rollouts.ts +374 -0
  94. package/src/cost-ledger.ts +232 -24
  95. package/src/cross-family-control.ts +1038 -0
  96. package/src/debt-ratchet.ts +143 -0
  97. package/src/embedding-config.ts +131 -10
  98. package/src/feature-adr-checkpoints.ts +29 -0
  99. package/src/feature-adr-decision-recall.ts +6 -4
  100. package/src/feature-adr-envelope.ts +242 -0
  101. package/src/feature-adr-routing.ts +150 -3
  102. package/src/feature-adr-stage-canon.ts +141 -0
  103. package/src/index.ts +65 -6
  104. package/src/loop-blobs.generated.ts +4 -4
  105. package/src/mutation-gate.ts +316 -0
  106. package/src/qe-bridge.ts +12 -2
  107. package/src/qe-findings.ts +463 -0
  108. package/src/recap.ts +10 -3
  109. package/src/review-cost.ts +139 -0
  110. package/src/round.ts +481 -6
  111. package/src/run-records.ts +388 -2
  112. package/src/score.ts +115 -6
@@ -13,7 +13,133 @@
13
13
  * Pure: payload in, verdict out. The CLI owns paths, the append, the read-back and the exit code.
14
14
  */
15
15
 
16
+ import { matchCodexRollouts } from './codex-rollouts.js';
17
+ import type { CodexRollout } from './codex-rollouts.js';
16
18
  import { redactTrainingPayload } from './feature-adr-checkpoints.js';
19
+ import { validateExperimentEnvelope } from './feature-adr-envelope.js';
20
+
21
+ /** Structural — a caller passes `cost-scoring.ts`'s `ModelPricing`; kept local so `run-records.ts`
22
+ * does not have to import `cost-scoring.ts` just to name a type.
23
+ *
24
+ * measurement-integrity fix-round-1/F6 (Codex r1 HIGH #6): `cacheCreation` used to be dropped here —
25
+ * the snapshot silently lacked the ONE rate a cache-WRITE-heavy row needs to reproduce its own cost
26
+ * later, even though the caller's own `ModelPricing` carries it. Now carried through verbatim. */
27
+ export interface LedgerPriceEntry {
28
+ readonly prompt: number;
29
+ readonly completion: number;
30
+ readonly cachedInput: number;
31
+ readonly cacheCreation: number;
32
+ }
33
+
34
+ /** measurement-integrity fix-round-1/F4 (Codex r1 HIGH #4): a resolvable executor spec, split into
35
+ * its three parts. Never invents a model: {@link parseModelSpec} returns `null` for anything it
36
+ * cannot resolve to exactly one model, rather than guessing. */
37
+ export interface ParsedModelSpec {
38
+ readonly family: 'claude' | 'codex';
39
+ readonly model: string;
40
+ readonly effort: string | null;
41
+ }
42
+
43
+ /** Bare Claude model names the pipeline actually emits with no `claude:` prefix (`coder: 'sonnet'`,
44
+ * `'opus'`, `'fable'`, real values observed in `.dz/feature-adr/run-cost-ledger.jsonl`). */
45
+ const BARE_CLAUDE_NAMES = new Set(['sonnet', 'opus', 'haiku', 'fable']);
46
+ /** id/effort alphabet a model spec component may use (lead delta after Codex r2, new MEDIUM #3). */
47
+ const SPEC_PART = /^[A-Za-z0-9._-]+$/;
48
+
49
+ /**
50
+ * Parse an executor spec — the shapes actually recorded in `coder`/`reviewer` fields
51
+ * (`codex:gpt-5.6-sol:high`, `claude:sonnet`, bare `sonnet`/`opus`, bare `codex`, bare `claude`) —
52
+ * into `{family, model, effort}`. Pure, never throws.
53
+ *
54
+ * Returns `null` (never a guess) for anything that cannot be resolved to exactly ONE model:
55
+ * - a bare `'codex'` or `'claude'` (family named, no model at all);
56
+ * - an annotated/aggregate field such as `'claude:sonnet x2'` or `'qe-bridge:claude x2 + lead'` (real
57
+ * values this ledger carries for a MULTI-reviewer round) — any embedded whitespace means the field
58
+ * names more than one resolvable spec, and picking one would misattribute to the others;
59
+ * - a bare model id with no family marker that is not one of the known bare Claude names (e.g. a full
60
+ * `'claude-sonnet-5'` — that shape is handled by the OLDER vendor-prefix path in {@link priceLookup}
61
+ * for backward compatibility, not by this parser).
62
+ */
63
+ export function parseModelSpec(spec: unknown): ParsedModelSpec | null {
64
+ if (typeof spec !== 'string') return null;
65
+ const trimmed = spec.trim();
66
+ if (trimmed === '' || /\s/.test(trimmed)) return null;
67
+ const parts = trimmed.split(':');
68
+ const head = parts[0] ?? '';
69
+ if (head === 'codex') {
70
+ const model = parts[1];
71
+ if (model === undefined || model === '') return null; // bare 'codex' — no reliable model
72
+ // Lead delta after Codex r2 (new MEDIUM #3): exactly 2 or 3 non-empty components, id alphabet
73
+ // only — `codex:gpt-5.6-sol:high:garbage` is a corrupt/aggregate spec, never a reliable model.
74
+ if (parts.length > 3 || !SPEC_PART.test(model) || (parts.length === 3 && (parts[2] === '' || !SPEC_PART.test(parts[2]!)))) return null;
75
+ const effort = parts[2] ?? null;
76
+ return { family: 'codex', model, effort: effort === '' ? null : effort };
77
+ }
78
+ if (head === 'claude') {
79
+ const model = parts[1];
80
+ if (model === undefined || model === '') return null; // bare 'claude' — no reliable model
81
+ return { family: 'claude', model, effort: null };
82
+ }
83
+ if (parts.length === 1 && BARE_CLAUDE_NAMES.has(head)) {
84
+ return { family: 'claude', model: head, effort: null };
85
+ }
86
+ return null;
87
+ }
88
+
89
+ /** measurement-integrity FR-5/FR-6: enrichment the WRITER supplies at write time — the rollout logs
90
+ * it already read (I/O lives in the CLI; this stays pure) and the price table snapshot. Absent
91
+ * entirely ⇒ zero behavior change from before this feature (NFR-1). */
92
+ export interface LedgerEnrichInput {
93
+ /** Parsed Codex rollout logs for the window the CLI read — usually every rollout from the days the
94
+ * window spans. Pure data; the CLI is the one that walked `~/.codex/sessions`. */
95
+ readonly rollouts?: readonly CodexRollout[];
96
+ /** The stage's own time window — usually [the previous ledger row's `ts`, this write's `ts`], or
97
+ * an explicit `--window-from/--window-to`. Omitted ⇒ no rollout match is even attempted. */
98
+ readonly window?: { readonly from: string; readonly to: string };
99
+ /** Narrows an otherwise-ambiguous match — usually the repo root the stage ran in. */
100
+ readonly cwd?: string;
101
+ /** A model-pricing table SNAPSHOT (FR-6) — the CALLER's table, captured at write time, never the
102
+ * ledger's own idea of "current" pricing (ADR-001 D4 rejects re-pricing after the fact). */
103
+ readonly prices?: Readonly<Record<string, LedgerPriceEntry>>;
104
+ }
105
+
106
+ function isCodexFamily(v: unknown): boolean {
107
+ return typeof v === 'string' && /codex/i.test(v);
108
+ }
109
+
110
+ function isRecord(v: unknown): v is Record<string, unknown> {
111
+ return typeof v === 'object' && v !== null && !Array.isArray(v);
112
+ }
113
+
114
+ /** Longest-prefix match against the CALLER'S OWN table (never `cost-scoring.ts`'s internal
115
+ * constant) — the whole point of a price SNAPSHOT is that it answers only from what the caller
116
+ * handed in at write time, mirroring `pricingFor`'s matching rule without importing it.
117
+ *
118
+ * measurement-integrity fix-round-1/F7 (Codex r1 HIGH #7): the OLD version stripped only a
119
+ * `vendor/`-shaped prefix, so the REAL recorded shape `codex:gpt-5.6-sol:high` (or `claude:sonnet`)
120
+ * never matched anything and always landed in `prices.unknown[]`. This now runs the id through the
121
+ * SAME {@link parseModelSpec} FR-4's matcher uses, then reconstructs the normalized id the way
122
+ * `cost-scoring.ts`'s `pricingFor`/`hasKnownPricing` key their table (`claude-<model>` for the
123
+ * Claude family; the bare model for Codex — its ids carry no vendor prefix). A spec this parser
124
+ * cannot resolve falls back to the OLD vendor-prefix strip, so an already-working bare id
125
+ * (`claude-sonnet-5`, `gpt-4o`) keeps matching exactly as before (NFR-1). */
126
+ function priceLookup(modelId: string, table: Readonly<Record<string, LedgerPriceEntry>>): LedgerPriceEntry | null {
127
+ if (typeof modelId !== 'string' || modelId.length === 0) return null;
128
+ const parsed = parseModelSpec(modelId);
129
+ const id = parsed !== null
130
+ ? (parsed.family === 'claude' ? `claude-${parsed.model}` : parsed.model).toLowerCase()
131
+ : modelId.toLowerCase().replace(/^[a-z0-9-]+\//, '');
132
+ let best: LedgerPriceEntry | null = null;
133
+ let bestLen = 0;
134
+ for (const [key, price] of Object.entries(table)) {
135
+ const k = key.toLowerCase();
136
+ if (id.startsWith(k) && k.length > bestLen) {
137
+ best = price;
138
+ bestLen = k.length;
139
+ }
140
+ }
141
+ return best;
142
+ }
17
143
 
18
144
  export type RecordKind = 'ledger' | 'training-pair';
19
145
 
@@ -65,7 +191,7 @@ const noop = (verdict: 'duplicate' | 'skipped', reason: string): RecordDecision
65
191
  line: null,
66
192
  });
67
193
 
68
- function shapeMismatch(kind: RecordKind, payload: Record<string, unknown>): string | null {
194
+ function shapeMismatch(kind: RecordKind, payload: Record<string, unknown>, autoFlag: boolean): string | null {
69
195
  // AM-1 FIRST, before the required-field sweep. A ledger row offered as a training pair fails BOTH
70
196
  // checks, and the wrong-kind reason is the one that tells the caller what actually happened —
71
197
  // "missing field `output`" sends them looking for a field they never meant to send.
@@ -78,6 +204,39 @@ function shapeMismatch(kind: RecordKind, payload: Record<string, unknown>): stri
78
204
  if (kind === 'training-pair' && 'tokens' in payload && !('input' in payload)) {
79
205
  return 'this payload looks like a ledger row (`tokens` without `input`), not a training pair';
80
206
  }
207
+ // experiment-envelope FR-5 / ADR-001 D2: `auto:true` marks a pipeline-written row, and the pipeline
208
+ // is obligated to carry the envelope built once after the Step-0 router — a row missing it is
209
+ // useless for learning and would silently corrupt the sample (the "warn but write" alternative was
210
+ // rejected in the ADR: a warning nobody reads left `tokens=null` unnoticed for years). A MANUAL row
211
+ // (no `auto`) stays compatible: no envelope required, but one that IS present is still validated —
212
+ // never trusted just because a human typed it.
213
+ //
214
+ // fix-round-1/F2 (cross-family review, HIGH #2): the PAYLOAD's own `auto` field used to be the
215
+ // ONLY signal — an automatic producer that forgot it, or sent `"true"`/`false`, was silently
216
+ // accepted as a manual row and skipped the envelope requirement entirely. `autoFlag` is the CLI's
217
+ // own trusted `--auto` argument (never JSON a caller could typo): it is a SECOND, independent
218
+ // trust source, unioned with the payload field rather than replacing it — nothing that used to be
219
+ // gated stops being gated, and a `--auto`-dispatched caller is now gated even if its hand-built
220
+ // payload forgot the field. Independently of either source, a PRESENT `auto` field is checked for
221
+ // shape: only the literal `true` is a legal value — anything else (a string, `false`, a number) is
222
+ // refused outright, because a field whose only sane value is `true` holding something else is a
223
+ // caller bug worth surfacing, not silently downgrading to "manual".
224
+ if (kind === 'ledger') {
225
+ const rawAuto = payload['auto'];
226
+ if (rawAuto !== undefined && rawAuto !== true) {
227
+ return 'a ledger record\'s `auto` field must be `true` or absent';
228
+ }
229
+ const envelope = payload['envelope'];
230
+ const envelopePresent = envelope !== undefined && envelope !== null;
231
+ const isAuto = autoFlag === true || rawAuto === true;
232
+ if (isAuto && !envelopePresent) {
233
+ return 'an auto ledger row must carry `envelope` (experiment-envelope FR-5)';
234
+ }
235
+ if (envelopePresent) {
236
+ const v = validateExperimentEnvelope(envelope);
237
+ if (!v.ok) return `envelope invalid — ${v.reason}`;
238
+ }
239
+ }
81
240
  const required = kind === 'ledger' ? LEDGER_REQUIRED : PAIR_REQUIRED;
82
241
  for (const field of required) {
83
242
  const v = payload[field];
@@ -125,7 +284,16 @@ export function decideRecordWrite(input: {
125
284
  /** Who ran it. Supplied by the CALLER, which lives outside the workflow sandbox and can see the
126
285
  * host; absent stays absent (see the stamping comment below). */
127
286
  runnerId?: string | null;
287
+ /** fix-round-1/F2: the CLI's own trusted `--auto` flag — see the comment inside `shapeMismatch`. */
288
+ auto?: boolean;
128
289
  maxChars?: number;
290
+ /** measurement-integrity FR-5/FR-6: rollout-log + price enrichment for a ledger row. Absent ⇒ zero
291
+ * behavior change (NFR-1). */
292
+ enrich?: LedgerEnrichInput;
293
+ /** experiment-instrument FR-2/A3 (ADR-001): when `true`, an AUTO ledger row that would be written
294
+ * `complete:false` is refused instead (exit 2, before any write) — a круг-B default candidate, opt-
295
+ * in today so nothing that already writes incomplete auto rows starts failing underfoot (NFR-1). */
296
+ strict?: boolean;
129
297
  }): RecordDecision {
130
298
  const { kind, payloadRaw, stage } = input;
131
299
  if (kind !== 'ledger' && kind !== 'training-pair') {
@@ -153,7 +321,7 @@ export function decideRecordWrite(input: {
153
321
  // the read-back, the refusal texts — ever sees a byte of the profile block.
154
322
  const obj = (kind === 'training-pair' ? redactTrainingPayload(payload) : payload) as Record<string, unknown>;
155
323
 
156
- const mismatch = shapeMismatch(kind, obj);
324
+ const mismatch = shapeMismatch(kind, obj, input.auto === true);
157
325
  if (mismatch !== null) return refuse(mismatch);
158
326
 
159
327
  // The record is filed under `--stage`, and the payload carries its own. A disagreement means the
@@ -182,6 +350,11 @@ export function decideRecordWrite(input: {
182
350
  // inside an already-serialised document — text surgery on a structured value, and the exact place
183
351
  // a payload containing that literal token could corrupt itself.
184
352
  const stamped: Record<string, unknown> = { ...obj };
353
+ // fix-round-1/F2: the CLI's own `--auto` flag is authoritative — when set, the written row MUST
354
+ // carry `auto:true` too (not just gate on it transiently), so every downstream reader of the
355
+ // PERSISTED line keeps seeing the same signal `shapeMismatch` already gated on above. A no-op when
356
+ // the payload already said `auto:true` (shapeMismatch already refused any OTHER value).
357
+ if (kind === 'ledger' && input.auto === true) stamped['auto'] = true;
185
358
  const isGap = (v: unknown): boolean => v === null || v === undefined || (typeof v === 'string' && v.trim() === '');
186
359
  if (input.timestamp != null && input.timestamp !== '') {
187
360
  // An EMPTY STRING is a gap, not a value. Stamping only over null/undefined let
@@ -273,6 +446,168 @@ export function decideRecordWrite(input: {
273
446
  }
274
447
  }
275
448
 
449
+ // measurement-integrity FR-5 (rollout enrichment) + FR-6 (price snapshot). Both are ADDITIVE and
450
+ // OPT-IN on `input.enrich` — a caller that never passes it gets byte-identical output to before
451
+ // this feature (NFR-1).
452
+ if (kind === 'ledger' && input.enrich !== undefined) {
453
+ const enrich = input.enrich;
454
+
455
+ // FR-5: only a codex-family coder/reviewer with `tokens: null` is a candidate — a Claude row, or
456
+ // one that already has a token figure, is left untouched. The loose `/codex/i` check below only
457
+ // decides whether this row is WORTH TRYING at all.
458
+ const tokensIsNull = stamped['tokens'] === null;
459
+ const looksCodexFamily = isCodexFamily(stamped['coder']) || isCodexFamily(stamped['reviewer']);
460
+ if (tokensIsNull && looksCodexFamily) {
461
+ // measurement-integrity fix-round-1/F4 (Codex r1 HIGH #4): the matcher REQUIRES a reliable
462
+ // model, parsed the same way FR-7's price lookup parses one — never `/codex/i` alone. If
463
+ // `coder`/`reviewer` do not resolve to exactly ONE codex model between them (a bare `'codex'`
464
+ // with no model at all, or the two fields naming DIFFERENT codex models), the matcher is never
465
+ // even called with an unreliable/omitted model filter — a lone rollout in the window would
466
+ // otherwise be accepted as `'one'` on time+cwd alone and its tokens misattributed to the wrong
467
+ // model's stage.
468
+ const codexModels = new Set(
469
+ [parseModelSpec(stamped['coder']), parseModelSpec(stamped['reviewer'])]
470
+ .filter((s): s is ParsedModelSpec => s !== null && s.family === 'codex')
471
+ .map((s) => s.model),
472
+ );
473
+ if (codexModels.size !== 1) {
474
+ stamped['tokensSource'] = 'codex-rollout:no-model';
475
+ } else if (enrich.window !== undefined) {
476
+ const model = [...codexModels][0]!;
477
+ const match = matchCodexRollouts(enrich.rollouts ?? [], {
478
+ from: enrich.window.from,
479
+ to: enrich.window.to,
480
+ model,
481
+ ...(enrich.cwd !== undefined ? { cwd: enrich.cwd } : {}),
482
+ });
483
+ if (match.status === 'one') {
484
+ stamped['tokens'] = match.rollout.totals.total;
485
+ const startMs = match.rollout.startedAt !== null ? Date.parse(match.rollout.startedAt) : NaN;
486
+ const endMs = match.rollout.endedAt !== null ? Date.parse(match.rollout.endedAt) : NaN;
487
+ // measurement-integrity fix-round-1/F6 (Codex r1 HIGH #6): fill-ONLY-null — an existing
488
+ // `minutes` figure (a manually recorded one, say) must never be silently overwritten by a
489
+ // derived rollout duration.
490
+ if (stamped['minutes'] === null && Number.isFinite(startMs) && Number.isFinite(endMs) && endMs >= startMs) {
491
+ stamped['minutes'] = Math.round(((endMs - startMs) / 60000) * 10) / 10;
492
+ }
493
+ stamped['tokensSource'] = 'codex-rollout';
494
+ stamped['rolloutId'] = match.rollout.id;
495
+ } else {
496
+ // `none` or `ambiguous` — NFR-3: an explicit status, never a guessed number.
497
+ stamped['tokensSource'] = `codex-rollout:${match.status}`;
498
+ }
499
+ } else {
500
+ // Eligible in principle (codex family, one reliable model, tokens null) but no window was
501
+ // supplied — AC-4's "old row without a window": no enrichment is even attempted, and that
502
+ // fact is itself recorded rather than left silently absent.
503
+ stamped['tokensSource'] = 'unavailable';
504
+ }
505
+ }
506
+
507
+ // FR-6: the price snapshot, for every model this row names — independent of the FR-5 branch
508
+ // above (a Claude row gets priced too; only tokens enrichment is codex-specific).
509
+ if (enrich.prices !== undefined) {
510
+ const modelIds = new Set<string>();
511
+ for (const v of [stamped['coder'], stamped['reviewer']]) {
512
+ if (typeof v === 'string' && v.trim() !== '') modelIds.add(v.trim());
513
+ }
514
+ const envelope = stamped['envelope'];
515
+ if (isRecord(envelope)) {
516
+ const chosen = envelope['chosen'];
517
+ if (isRecord(chosen)) {
518
+ const stages = chosen['stages'];
519
+ if (isRecord(stages)) {
520
+ for (const v of Object.values(stages)) {
521
+ if (typeof v === 'string' && v.trim() !== '') modelIds.add(v.trim());
522
+ }
523
+ }
524
+ }
525
+ }
526
+ const table: Record<string, LedgerPriceEntry> = {};
527
+ const unknown: string[] = [];
528
+ for (const modelId of modelIds) {
529
+ const price = priceLookup(modelId, enrich.prices);
530
+ if (price === null) unknown.push(modelId);
531
+ else table[modelId] = { prompt: price.prompt, completion: price.completion, cachedInput: price.cachedInput, cacheCreation: price.cacheCreation };
532
+ }
533
+ const computedPrices = {
534
+ snapshotAt: input.timestamp ?? null,
535
+ table,
536
+ ...(unknown.length > 0 ? { unknown } : {}),
537
+ };
538
+ // measurement-integrity fix-round-1/F6 (Codex r1 HIGH #6): fill-ONLY-null for `prices` too — an
539
+ // existing snapshot (a previous write already priced this row) is never unconditionally
540
+ // replaced. Equal → left alone (idempotent re-enrichment, common on a retried write). Different
541
+ // → a named `pricesConflict`, never a silent re-price (ADR-001 D4 forbids re-pricing after the
542
+ // fact — a DIFFERING recomputation is exactly that, so it is surfaced, not applied).
543
+ const existingPrices = stamped['prices'];
544
+ if (existingPrices === undefined || existingPrices === null) {
545
+ stamped['prices'] = computedPrices;
546
+ } else if (isRecord(existingPrices) && isRecord(existingPrices['table']) && (existingPrices['unknown'] === undefined || Array.isArray(existingPrices['unknown']))) {
547
+ // Lead delta after Codex r2 (new MEDIUM #4): compare the WHOLE snapshot canonically (table +
548
+ // sorted unknown), not the table alone.
549
+ const canon = (t: unknown, u: unknown): string => JSON.stringify({ table: t, unknown: Array.isArray(u) ? [...u].map(String).sort() : [] });
550
+ if (canon(existingPrices['table'], existingPrices['unknown']) !== canon(table, unknown)) {
551
+ stamped['pricesConflict'] = { existing: existingPrices, recomputed: computedPrices };
552
+ }
553
+ } else {
554
+ // A malformed existing snapshot (no table / bad unknown) is a CONFLICT, never silently trusted.
555
+ stamped['pricesConflict'] = { existing: existingPrices, recomputed: computedPrices, reason: 'existing prices snapshot is malformed' };
556
+ }
557
+ }
558
+ }
559
+
560
+ // experiment-instrument FR-2/A3 (ADR-001): completeness of an AUTO ledger row. `minutes` is
561
+ // fill-only-null from a `wallSec` the payload carries (the workflow sandbox has a clock delta even
562
+ // when it has no wall clock of its own — FR-2's `wallSec` field, distinct from `minutesSincePrev`
563
+ // above, which needs a PREVIOUS row and a runId neither of which every auto row has). `tokens` is
564
+ // judged complete when it is a real number OR the row already NAMES why it is not (`tokensSource`,
565
+ // set above by the FR-5 rollout match, or supplied by the caller) — an unexplained non-number is the
566
+ // one shape that is actually incomplete. Gated on `auto` only: a MANUAL row never gains any of these
567
+ // three keys, so it stays byte-identical to before this feature (NFR-1).
568
+ if (kind === 'ledger' && stamped['auto'] === true) {
569
+ const incompleteReasons: string[] = [];
570
+ if (stamped['minutes'] === null || stamped['minutes'] === undefined) {
571
+ const wallSec = stamped['wallSec'];
572
+ if (typeof wallSec === 'number' && Number.isFinite(wallSec) && wallSec >= 0) {
573
+ stamped['minutes'] = Math.round((wallSec / 60) * 10) / 10;
574
+ stamped['minutesSource'] = 'wallSec';
575
+ } else {
576
+ incompleteReasons.push('minutes');
577
+ }
578
+ }
579
+ // r1-10 (Codex r1 HIGH #10): completeness requires a real, finite NUMBER of tokens. Naming why
580
+ // tokens are missing (`tokensSource:'unavailable'`, or any other provenance) is diagnostic, never
581
+ // a substitute for the number itself — ADR-001 says missing tokens makes the row incomplete, full
582
+ // stop. The old check (`tokensSource === undefined`) treated a NAMED absence as if it were data.
583
+ if (typeof stamped['tokens'] !== 'number' || !Number.isFinite(stamped['tokens'])) {
584
+ incompleteReasons.push('tokens');
585
+ }
586
+ // r1-11 (Codex r1 MEDIUM #11): a payload that ALREADY declared itself incomplete (its own
587
+ // `complete:false` + `incompleteReasons`, e.g. a workflow-side `'artifact'` reason this module
588
+ // knows nothing about) must never be overwritten back to `complete:true` just because THIS
589
+ // module's own minutes/tokens checks both passed — that erases a true fact and leaves a
590
+ // contradictory row (`complete:true` alongside a stale `incompleteReasons`). Preserve and MERGE.
591
+ const existingCompleteRaw = stamped['complete'];
592
+ const existingWasIncomplete = existingCompleteRaw === false;
593
+ const existingReasonsRaw = stamped['incompleteReasons'];
594
+ const existingReasons = Array.isArray(existingReasonsRaw)
595
+ ? existingReasonsRaw.filter((r): r is string => typeof r === 'string')
596
+ : [];
597
+ const mergedReasons = existingWasIncomplete
598
+ ? [...new Set([...existingReasons, ...incompleteReasons])]
599
+ : incompleteReasons;
600
+ const complete = mergedReasons.length === 0 && !existingWasIncomplete;
601
+ // A3: under `--strict`, incompleteness is a REFUSAL — before any write, the target untouched —
602
+ // rather than a loudly-marked write. Without `--strict` (the default today; круг-B may flip it),
603
+ // the row is still written, just honestly marked `complete:false` with its reasons.
604
+ if (input.strict === true && !complete) {
605
+ return refuse(`auto ledger row is incomplete (${mergedReasons.join(', ') || 'previously marked incomplete'}) — refused under --strict before any write`);
606
+ }
607
+ stamped['complete'] = complete;
608
+ if (!complete) stamped['incompleteReasons'] = mergedReasons.length > 0 ? mergedReasons : existingReasons;
609
+ }
610
+
276
611
  let line: string;
277
612
  try {
278
613
  line = JSON.stringify(stamped);
@@ -324,3 +659,54 @@ export function decideReadBack(appended: string, lastLineOnDisk: string | null):
324
659
  export function recordVerdictLine(kind: RecordKind, stage: string, d: RecordDecision): string {
325
660
  return `feature-adr record (${kind}/${stage}): ${d.verdict.toUpperCase()} — ${d.reason}`;
326
661
  }
662
+
663
+ /** experiment-instrument FR-1/FR-3 (ADR-001): what `round.ts`'s `readOpenRoundTaskId` returns — the
664
+ * single source `applyTaskId` fills from. Duplicated here rather than imported so this pure module
665
+ * never depends on `round.ts`'s own shape; the CLI is the one holding both and wiring them together.
666
+ * r1-1/r1-2 (Codex r1 #1/#2): extended with `'derived-legacy'` and `'unavailable'` to stay in
667
+ * lockstep with `round.ts`'s own `readOpenRoundTaskId` return type. */
668
+ export interface TaskIdLookup {
669
+ readonly taskId: string | null;
670
+ readonly source: 'open-round' | 'derived-legacy' | 'no-open-round' | 'ambiguous' | 'unavailable';
671
+ }
672
+
673
+ /**
674
+ * experiment-instrument FR-1/FR-3/A8 (ADR-001): propagate `taskId` onto a ledger/training-pair payload
675
+ * BEFORE it reaches {@link decideRecordWrite} — fill-only-null, never overwritten.
676
+ *
677
+ * - The payload already names a non-empty `taskId` string ⇒ it is authoritative. When it DISAGREES
678
+ * with the round's own current taskId, that disagreement is a real fact worth keeping — recorded as
679
+ * `taskIdConflict: {payload, round}` — never silently resolved either way (A8).
680
+ * - The payload's `taskId` key is absent, or explicitly `null`/`undefined` ⇒ filled from `lookup`,
681
+ * INCLUDING the honest `null` case: no open round (A4) or two of them (A5) still stamps `taskId:
682
+ * null` + `taskIdSource` naming why, rather than leaving the field silently absent — absence with a
683
+ * named reason beats absence with none.
684
+ * - r1-3 (Codex r1 HIGH #3): the payload's `taskId` key is PRESENT with a value that is neither a
685
+ * non-empty string nor null/undefined (a number, a boolean, an object, or a blank/whitespace-only
686
+ * string) ⇒ that is a present-but-INVALID value, a THIRD case distinct from both of the above. It
687
+ * used to be treated exactly like "absent" (`typeof !== 'string'` fell through to the fill branch),
688
+ * silently replacing the caller's own (malformed) value with the round's — violating both
689
+ * fill-only-null and "a present payload value always wins". Now: the row's own value is preserved
690
+ * UNTOUCHED (never replaced with a guess about what the caller meant), and the problem is named in
691
+ * `taskIdInvalid` so a reader can see the row was neither filled nor trusted blindly.
692
+ *
693
+ * Pure: no filesystem, no clock. The CALLER (the cli) is the one that read `.dz/rounds/` to build
694
+ * `lookup` in the first place.
695
+ */
696
+ export function applyTaskId(row: Record<string, unknown>, lookup: TaskIdLookup): Record<string, unknown> {
697
+ const hasTaskIdKey = Object.prototype.hasOwnProperty.call(row, 'taskId');
698
+ const rawPayloadTaskId = row['taskId'];
699
+ if (hasTaskIdKey && rawPayloadTaskId !== null && rawPayloadTaskId !== undefined) {
700
+ if (typeof rawPayloadTaskId === 'string' && rawPayloadTaskId.trim() !== '') {
701
+ const payloadTaskId = rawPayloadTaskId.trim();
702
+ if (lookup.taskId !== null && lookup.taskId !== payloadTaskId) {
703
+ return { ...row, taskIdConflict: { payload: payloadTaskId, round: lookup.taskId } };
704
+ }
705
+ return { ...row };
706
+ }
707
+ // r1-3: present but not a usable identity (non-string, or blank after trim) — refuse to replace
708
+ // it with a lookup guess; preserve it verbatim and name the problem.
709
+ return { ...row, taskIdInvalid: { value: rawPayloadTaskId, reason: 'taskId present but not a non-empty string' } };
710
+ }
711
+ return { ...row, taskId: lookup.taskId, taskIdSource: lookup.source };
712
+ }
package/src/score.ts CHANGED
@@ -21,6 +21,7 @@
21
21
  */
22
22
 
23
23
  import { amendmentIdsIn } from './amendment-trace.js';
24
+ import { QE_SEVERITIES, findInvalidQeVerdictLines, parseQeFindings, readQeVerdictLines, type QeFindingsRefusedRow, type QeFindingsSummary } from './qe-findings.js';
24
25
 
25
26
  export type DisciplineVerdict = 'pass' | 'partial' | 'absent';
26
27
 
@@ -45,11 +46,29 @@ export interface RunScorecard {
45
46
  * stay distinguishable.
46
47
  */
47
48
  readonly mutationEvidence?: MutationEvidence;
49
+ /**
50
+ * qe-findings-record FR-4: where `qeGrade` came from — a machine-written `QE-VERDICT:` line, the
51
+ * pre-existing prose scan, or neither. ADDITIVE for the same reason as `mutationEvidence`: every
52
+ * scorecard built before this field existed still satisfies `RunScorecard`.
53
+ */
54
+ readonly gradeSource?: GradeSource;
55
+ /**
56
+ * qe-findings-record FR-4: the report's Findings ledger, when one exists. `absent` for the 406
57
+ * pre-existing reports that carry neither the table nor the heading (NFR-1) — never omitted, for
58
+ * the same "gate never ran vs a field an older scorer never wrote" reason `mutationEvidence` gives.
59
+ */
60
+ readonly findings?: QeFindingsScoreView;
48
61
  readonly passed: number;
49
62
  readonly total: number;
50
63
  readonly summary: string;
51
64
  }
52
65
 
66
+ /** A lighter projection of `QeFindingsResult` for the scorecard — summary + refused + hollow, never
67
+ * the full row list (that stays in `parseQeFindings`'s own return value for a caller that wants it). */
68
+ export type QeFindingsScoreView =
69
+ | { readonly status: 'absent' }
70
+ | { readonly status: 'present'; readonly hollow: boolean; readonly summary: QeFindingsSummary; readonly refused: readonly QeFindingsRefusedRow[] };
71
+
53
72
  /** The artifact texts of one run, keyed by RELATIVE path under `features/<slug>/`. */
54
73
  /**
55
74
  * The exact heading Step 5 asks for, and the exact heading the check looks for — ONE constant, so
@@ -306,7 +325,22 @@ function normaliseGradeSign(grade: string): string {
306
325
  return grade.replace('\u2212', '-');
307
326
  }
308
327
 
309
- export type GradeReadStatus = 'unique' | 'ambiguous' | 'none';
328
+ /**
329
+ * fix-round-1 (codex-r1-verdict finding 2): `'invalid'` is a report that ATTEMPTED a machine verdict
330
+ * (a line starting `QE-VERDICT`, any case/spacing) and got the grammar wrong — `QE-VERDICT: B – final`
331
+ * (en dash + trailing prose), wrong case, no colon, a grade outside A-D. That is a DIFFERENT fact from
332
+ * `'none'` (no attempt at all, legacy prose scan still applies): a malformed attempt must never fall
333
+ * back to guessing a grade from prose — the malformed line is itself evidence the report is unreliable
334
+ * here, and guessing past it would silently launder that unreliability into a confident number.
335
+ */
336
+ export type GradeReadStatus = 'unique' | 'ambiguous' | 'none' | 'invalid';
337
+
338
+ /**
339
+ * qe-findings-record (ADR-001 D1): where the grade came from. `'verdict-line'` — a machine-written
340
+ * `QE-VERDICT:` line, the source of truth when present. `'prose'` — the pre-existing GRADE_RE scan,
341
+ * unchanged, used only when NO verdict line exists. `'none'` — neither surface names a grade.
342
+ */
343
+ export type GradeSource = 'verdict-line' | 'prose' | 'none';
310
344
 
311
345
  export interface GradeReading {
312
346
  readonly status: GradeReadStatus;
@@ -314,6 +348,7 @@ export interface GradeReading {
314
348
  readonly grade: string | null;
315
349
  /** Every distinct grade found, normalised — what makes an `ambiguous` verdict inspectable. */
316
350
  readonly found: readonly string[];
351
+ readonly source: GradeSource;
317
352
  }
318
353
 
319
354
  /**
@@ -329,6 +364,44 @@ export interface GradeReading {
329
364
  * report is ambiguous, which is a fact about the report, not a missing number.
330
365
  */
331
366
  export function readQeGrade(qeText: string): GradeReading {
367
+ // ADR-001 D1: a machine-written verdict line is the source of truth WHEN it exists — checked
368
+ // first, and the prose scan below never runs when it does. Exactly one -> unique; more than one
369
+ // -> ambiguous ("две строки — ambiguous, никогда «последняя побеждает»", even if both name the
370
+ // SAME grade — two lines is a fact about the report, not a number to reconcile); zero -> the
371
+ // prose scan runs exactly as it always has (NFR-1: 406 pre-existing reports are unaffected) —
372
+ // UNLESS a malformed attempt exists (fix-round-1 finding 2, checked next): the report is `invalid`
373
+ // and the prose fallback is refused, never silently reached.
374
+ const verdictLines = readQeVerdictLines(qeText);
375
+ // Lead delta after Codex r2 (new HIGH #1): a malformed declaration next to a valid one is NOT a
376
+ // unique verdict — `QE-VERDICT: A` + `qe-verdict: D` used to read as unique A. Invalid lines are
377
+ // checked FIRST, whatever the count of valid ones.
378
+ const invalidFirst = findInvalidQeVerdictLines(qeText);
379
+ if (invalidFirst.length > 0) {
380
+ const reasons = invalidFirst.map(
381
+ (l) => `line ${l.line}: ${JSON.stringify(l.text.trim())} does not match "QE-VERDICT: <A|B|C|D><+|-> "`,
382
+ );
383
+ return { status: 'invalid', grade: null, found: [...verdictLines, ...reasons], source: 'verdict-line' };
384
+ }
385
+ if (verdictLines.length === 1) {
386
+ return { status: 'unique', grade: verdictLines[0] as string, found: verdictLines, source: 'verdict-line' };
387
+ }
388
+ if (verdictLines.length > 1) {
389
+ const distinct: string[] = [];
390
+ for (const g of verdictLines) if (!distinct.includes(g)) distinct.push(g);
391
+ return { status: 'ambiguous', grade: null, found: distinct, source: 'verdict-line' };
392
+ }
393
+
394
+ // fix-round-1 finding 2: zero GRAMMATICALLY VALID verdict lines is not yet "no attempt" — a line
395
+ // that clearly opens a verdict declaration but botches the grammar (wrong dash, wrong case, no
396
+ // colon, a grade outside A-D) is `invalid`, and the legacy prose scan below is FORBIDDEN for it.
397
+ // The reason lives inside `found` (GradeReading keeps the same 4-field shape for every status).
398
+ // Lead delta after Codex r2 (#2 PARTIAL): the prose fallback is for LEGACY reports only. A report
399
+ // that already carries the new-format ledger heading but no verdict line is a NEW-format report
400
+ // missing its verdict — `invalid`, never a prose guess.
401
+ if (/^## Findings ledger\s*$/m.test(qeText)) {
402
+ return { status: 'invalid', grade: null, found: ['new-format report (has "## Findings ledger") without a QE-VERDICT line'], source: 'verdict-line' };
403
+ }
404
+
332
405
  const found: string[] = [];
333
406
  const lines = qeText.split('\n');
334
407
  // Line offsets once, so each match maps to ITS line for the negation screen (67d7883d: the
@@ -352,9 +425,9 @@ export function readQeGrade(qeText: string): GradeReading {
352
425
  const g = normaliseGradeSign(m[1] as string);
353
426
  if (!found.includes(g)) found.push(g);
354
427
  }
355
- if (found.length === 0) return { status: 'none', grade: null, found: [] };
356
- if (found.length === 1) return { status: 'unique', grade: found[0] as string, found };
357
- return { status: 'ambiguous', grade: null, found };
428
+ if (found.length === 0) return { status: 'none', grade: null, found: [], source: 'none' };
429
+ if (found.length === 1) return { status: 'unique', grade: found[0] as string, found, source: 'prose' };
430
+ return { status: 'ambiguous', grade: null, found, source: 'prose' };
358
431
  }
359
432
 
360
433
  export function extractQeGrade(qeText: string): string | null {
@@ -451,7 +524,13 @@ export function scoreRun(slug: string, artifacts: RunArtifacts): RunScorecard {
451
524
  );
452
525
 
453
526
  // 3. Cross-model QE — an independent family reviewed it, and a grade exists.
454
- const grade = extractQeGrade(qeText);
527
+ const gradeReading = readQeGrade(qeText);
528
+ const grade = gradeReading.grade;
529
+ const findingsParsed = parseQeFindings(qeText);
530
+ const findings: QeFindingsScoreView =
531
+ findingsParsed.status === 'absent'
532
+ ? { status: 'absent' }
533
+ : { status: 'present', hollow: findingsParsed.hollow, summary: findingsParsed.summary, refused: findingsParsed.refused };
455
534
  if (qeText === '') {
456
535
  add('cross-model-qe', 'independent cross-model review with a grade', 'absent', 'no 08_qe_report.md artifact');
457
536
  } else {
@@ -559,11 +638,36 @@ export function scoreRun(slug: string, artifacts: RunArtifacts): RunScorecard {
559
638
  (grade !== null ? ` · QE grade ${grade}` : ' · no QE grade') +
560
639
  (worst.length > 0 ? ` · absent: ${worst.join(', ')}` : '');
561
640
 
562
- return { slug, disciplines, qeGrade: grade, mutationEvidence, passed, total, summary };
641
+ return { slug, disciplines, qeGrade: grade, gradeSource: gradeReading.source, findings, mutationEvidence, passed, total, summary };
563
642
  }
564
643
 
565
644
  const MARK: Record<DisciplineVerdict, string> = { pass: '✓', partial: '◐', absent: '✗' };
566
645
 
646
+ /** qe-findings-record FR-4: the one-line summary `dz score` prints for a report's Findings ledger —
647
+ * "findings: 3 HIGH / 2 MEDIUM; 1 refused (line 84: severity "Major" not in dictionary)". Ordered by
648
+ * QE_SEVERITIES (BLOCKER first) so the worst finding always reads first. `null` when there is no
649
+ * table at all — the common case, which earns no noise (same rule as mutationEvidence above). */
650
+ export function renderFindingsLine(findings: QeFindingsScoreView | undefined): string | null {
651
+ if (findings === undefined || findings.status === 'absent') return null;
652
+ // fix-round-1 finding 6: hollow used to short-circuit BEFORE the refused check, so a report whose
653
+ // real ledger table (with CRITICAL rows) was refused as duplicate/outside-section alongside an
654
+ // empty accepted table read as pure "EMPTY" — the refused rows vanished from the printed line, not
655
+ // merely from the tally. Both facts are ALWAYS reported together now; neither hides the other.
656
+ const refusedSuffix = ((): string => {
657
+ if (findings.refused.length === 0) return '';
658
+ const first = findings.refused[0] as QeFindingsRefusedRow;
659
+ const more = findings.refused.length > 1 ? `, +${findings.refused.length - 1} more` : '';
660
+ return `; ${findings.refused.length} refused (line ${first.line}: ${first.reason})${more}`;
661
+ })();
662
+ if (findings.hollow) {
663
+ return `findings: table present but EMPTY (hollow) — worse than no table at all${refusedSuffix}`;
664
+ }
665
+ const bySev = findings.summary.bySeverity;
666
+ const parts = QE_SEVERITIES.filter((s) => (bySev[s] ?? 0) > 0).map((s) => `${bySev[s]} ${s}`);
667
+ const head = parts.length > 0 ? parts.join(' / ') : 'no rows';
668
+ return `findings: ${head}${refusedSuffix}`;
669
+ }
670
+
567
671
  export function renderScorecard(card: RunScorecard): string {
568
672
  const out: string[] = [];
569
673
  out.push(`dz score — ${card.slug} (process scorecard; descriptive-only, never a gate)`);
@@ -581,6 +685,11 @@ export function renderScorecard(card: RunScorecard): string {
581
685
  out.push('');
582
686
  out.push(` ${card.mutationEvidence.evidence}`);
583
687
  }
688
+ const findingsLine = renderFindingsLine(card.findings);
689
+ if (findingsLine !== null) {
690
+ out.push('');
691
+ out.push(` ${findingsLine}${card.gradeSource !== undefined && card.gradeSource !== 'none' ? ` (grade source: ${card.gradeSource})` : ''}`);
692
+ }
584
693
  out.push('');
585
694
  out.push(` ${card.summary}`);
586
695
  return out.join('\n');