@dzhechkov/harness-core 0.8.35 → 0.8.37

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. package/.dz-manifest.json +224 -104
  2. package/README.md +335 -10
  3. package/dist/agentdb-index.d.ts +87 -7
  4. package/dist/agentdb-index.d.ts.map +1 -1
  5. package/dist/agentdb-index.js +416 -57
  6. package/dist/agentdb-index.js.map +1 -1
  7. package/dist/apply-leg.d.ts +57 -1
  8. package/dist/apply-leg.d.ts.map +1 -1
  9. package/dist/apply-leg.js +450 -52
  10. package/dist/apply-leg.js.map +1 -1
  11. package/dist/codex-hooks-assets.d.ts.map +1 -1
  12. package/dist/codex-hooks-assets.js +67 -5
  13. package/dist/codex-hooks-assets.js.map +1 -1
  14. package/dist/codex-hooks.d.ts +13 -1
  15. package/dist/codex-hooks.d.ts.map +1 -1
  16. package/dist/codex-hooks.js +13 -1
  17. package/dist/codex-hooks.js.map +1 -1
  18. package/dist/codex-rollouts.d.ts +118 -0
  19. package/dist/codex-rollouts.d.ts.map +1 -0
  20. package/dist/codex-rollouts.js +297 -0
  21. package/dist/codex-rollouts.js.map +1 -0
  22. package/dist/cost-ledger.d.ts +56 -4
  23. package/dist/cost-ledger.d.ts.map +1 -1
  24. package/dist/cost-ledger.js +176 -20
  25. package/dist/cost-ledger.js.map +1 -1
  26. package/dist/cross-family-control.d.ts +345 -0
  27. package/dist/cross-family-control.d.ts.map +1 -0
  28. package/dist/cross-family-control.js +802 -0
  29. package/dist/cross-family-control.js.map +1 -0
  30. package/dist/debt-ratchet.d.ts +53 -0
  31. package/dist/debt-ratchet.d.ts.map +1 -0
  32. package/dist/debt-ratchet.js +107 -0
  33. package/dist/debt-ratchet.js.map +1 -0
  34. package/dist/embedding-config.d.ts +42 -0
  35. package/dist/embedding-config.d.ts.map +1 -1
  36. package/dist/embedding-config.js +106 -10
  37. package/dist/embedding-config.js.map +1 -1
  38. package/dist/feature-adr-checkpoints.d.ts +6 -0
  39. package/dist/feature-adr-checkpoints.d.ts.map +1 -1
  40. package/dist/feature-adr-checkpoints.js +29 -0
  41. package/dist/feature-adr-checkpoints.js.map +1 -1
  42. package/dist/feature-adr-decision-recall.d.ts +2 -2
  43. package/dist/feature-adr-decision-recall.d.ts.map +1 -1
  44. package/dist/feature-adr-decision-recall.js +5 -3
  45. package/dist/feature-adr-decision-recall.js.map +1 -1
  46. package/dist/feature-adr-envelope.d.ts +96 -0
  47. package/dist/feature-adr-envelope.d.ts.map +1 -0
  48. package/dist/feature-adr-envelope.js +183 -0
  49. package/dist/feature-adr-envelope.js.map +1 -0
  50. package/dist/feature-adr-routing.d.ts +64 -0
  51. package/dist/feature-adr-routing.d.ts.map +1 -1
  52. package/dist/feature-adr-routing.js +122 -2
  53. package/dist/feature-adr-routing.js.map +1 -1
  54. package/dist/feature-adr-stage-canon.d.ts +79 -0
  55. package/dist/feature-adr-stage-canon.d.ts.map +1 -0
  56. package/dist/feature-adr-stage-canon.js +117 -0
  57. package/dist/feature-adr-stage-canon.js.map +1 -0
  58. package/dist/index.d.ts +23 -12
  59. package/dist/index.d.ts.map +1 -1
  60. package/dist/index.js +15 -7
  61. package/dist/index.js.map +1 -1
  62. package/dist/loop-blobs.generated.js +4 -4
  63. package/dist/loop-blobs.generated.js.map +1 -1
  64. package/dist/mutation-gate.d.ts +51 -0
  65. package/dist/mutation-gate.d.ts.map +1 -1
  66. package/dist/mutation-gate.js +295 -0
  67. package/dist/mutation-gate.js.map +1 -1
  68. package/dist/operations.d.ts +1 -0
  69. package/dist/operations.d.ts.map +1 -1
  70. package/dist/operations.js +18 -2
  71. package/dist/operations.js.map +1 -1
  72. package/dist/publish.d.ts +59 -7
  73. package/dist/publish.d.ts.map +1 -1
  74. package/dist/publish.js +205 -32
  75. package/dist/publish.js.map +1 -1
  76. package/dist/qe-bridge.d.ts.map +1 -1
  77. package/dist/qe-bridge.js +4 -2
  78. package/dist/qe-bridge.js.map +1 -1
  79. package/dist/qe-findings.d.ts +107 -0
  80. package/dist/qe-findings.d.ts.map +1 -0
  81. package/dist/qe-findings.js +417 -0
  82. package/dist/qe-findings.js.map +1 -0
  83. package/dist/recap.d.ts +1 -1
  84. package/dist/recap.d.ts.map +1 -1
  85. package/dist/recap.js +4 -2
  86. package/dist/recap.js.map +1 -1
  87. package/dist/release-line.d.ts +16 -0
  88. package/dist/release-line.d.ts.map +1 -1
  89. package/dist/release-line.js +31 -0
  90. package/dist/release-line.js.map +1 -1
  91. package/dist/round.d.ts +74 -1
  92. package/dist/round.d.ts.map +1 -1
  93. package/dist/round.js +112 -4
  94. package/dist/round.js.map +1 -1
  95. package/dist/run-records.d.ts +60 -0
  96. package/dist/run-records.d.ts.map +1 -1
  97. package/dist/run-records.js +244 -2
  98. package/dist/run-records.js.map +1 -1
  99. package/dist/score.d.ts +44 -1
  100. package/dist/score.d.ts.map +1 -1
  101. package/dist/score.js +78 -5
  102. package/dist/score.js.map +1 -1
  103. package/dist/vector-tier.d.ts +34 -3
  104. package/dist/vector-tier.d.ts.map +1 -1
  105. package/dist/vector-tier.js +105 -14
  106. package/dist/vector-tier.js.map +1 -1
  107. package/package.json +2 -2
  108. package/sbom.json +403 -103
  109. package/src/agentdb-index.ts +423 -60
  110. package/src/apply-leg.ts +469 -50
  111. package/src/codex-hooks-assets.ts +67 -5
  112. package/src/codex-hooks.ts +13 -1
  113. package/src/codex-rollouts.ts +374 -0
  114. package/src/cost-ledger.ts +232 -24
  115. package/src/cross-family-control.ts +960 -0
  116. package/src/debt-ratchet.ts +143 -0
  117. package/src/embedding-config.ts +131 -10
  118. package/src/feature-adr-checkpoints.ts +29 -0
  119. package/src/feature-adr-decision-recall.ts +6 -4
  120. package/src/feature-adr-envelope.ts +242 -0
  121. package/src/feature-adr-routing.ts +139 -2
  122. package/src/feature-adr-stage-canon.ts +141 -0
  123. package/src/index.ts +66 -7
  124. package/src/loop-blobs.generated.ts +4 -4
  125. package/src/mutation-gate.ts +316 -0
  126. package/src/operations.ts +18 -3
  127. package/src/publish.ts +247 -30
  128. package/src/qe-bridge.ts +4 -2
  129. package/src/qe-findings.ts +463 -0
  130. package/src/recap.ts +10 -3
  131. package/src/release-line.ts +32 -0
  132. package/src/round.ts +165 -6
  133. package/src/run-records.ts +282 -2
  134. package/src/score.ts +115 -6
  135. package/src/vector-tier.ts +127 -14
package/src/round.ts CHANGED
@@ -4,6 +4,8 @@
4
4
  * filesystem or process API; the CLI owns `.dz/rounds/` and the witnessed ledger writer.
5
5
  */
6
6
 
7
+ import { validateExperimentEnvelope } from './feature-adr-envelope.js';
8
+
7
9
  const ROUND_OUTCOMES = ['shipped', 'refuted', 'blocked', 'abandoned'] as const;
8
10
  type RoundOutcome = typeof ROUND_OUTCOMES[number];
9
11
 
@@ -18,6 +20,10 @@ export interface RoundState {
18
20
  readonly run?: string;
19
21
  readonly recalled: readonly string[];
20
22
  readonly execs?: readonly RoundExecState[];
23
+ /** experiment-envelope FR-3(в): the envelope the pipeline built after its Step-0 router, carried
24
+ * unchanged through `round open --envelope` into `closeRound`'s ledger row. Opaque here (never
25
+ * interpreted by round.ts) — validated once, at `openRound`, and trusted from then on. */
26
+ readonly envelope?: unknown;
21
27
  }
22
28
 
23
29
  export interface RoundExecState {
@@ -39,7 +45,10 @@ export interface RoundLedgerRow {
39
45
  readonly minutes: number | null;
40
46
  readonly agents: number | null;
41
47
  readonly tokens: number | null;
42
- readonly grade: null;
48
+ /** measurement-integrity FR-7: was always `null` before this feature — `shipped|refuted` now
49
+ * requires a real grade; `blocked|abandoned` still writes `null` (a review that never finished has
50
+ * nothing to grade). */
51
+ readonly grade: string | null;
43
52
  readonly outcome: RoundOutcome;
44
53
  readonly reason: string | null;
45
54
  readonly round: number;
@@ -48,9 +57,43 @@ export interface RoundLedgerRow {
48
57
  readonly note: string;
49
58
  readonly date: null;
50
59
  readonly costIn?: 'stages';
60
+ /** experiment-envelope FR-3(в): copied verbatim from the state that closed this round, when present. */
61
+ readonly envelope?: unknown;
51
62
  /** round-state-lock (lead edit after Codex re-review): identity of the state instance this row
52
63
  * closes — lets a retried `close` detect its own earlier row regardless of the clock. */
53
64
  readonly stateId?: string;
65
+ /** measurement-integrity FR-7: present ONLY when `reviewer` was filled from the qe-bridge sidecar
66
+ * (no explicit `--reviewer`) — `elapsedMs / 60000`, rounded to 1 decimal. */
67
+ readonly reviewMinutes?: number;
68
+ /** measurement-integrity FR-7: present ONLY alongside `reviewMinutes` — names where `reviewer` and
69
+ * `reviewMinutes` came from, so a reader never confuses a sidecar-sourced figure for a flag. */
70
+ readonly reviewSource?: 'qe-bridge';
71
+ }
72
+
73
+ /**
74
+ * measurement-integrity FR-7: the qe-bridge cross-model-review signoff for THIS round — read by the
75
+ * CLI from `features/<slug>/.fa-state/qe-bridge/signoff-*.json`, selected by matching `slug` AND
76
+ * falling inside THIS round's own `[startedAt, closedAt]` interval (fix-round-1/F9, Codex r1 HIGH
77
+ * #9 — the CLI-side lookup, `findQeBridgeSignoffForRound`, is what enforces that; multiple qualifying
78
+ * signoffs there are an AMBIGUITY refusal, never "the latest wins"), passed in here as DATA.
79
+ * `closeRound` never opens a file.
80
+ */
81
+ export interface RoundReviewSidecar {
82
+ /** Who graded it — family + model, e.g. `codex:gpt-5.6-sol`. */
83
+ readonly gradedBy: string;
84
+ readonly elapsedMs: number;
85
+ /** The sidecar's OWN verdict, when it carries one — checked against `--grade` for a conflict. */
86
+ readonly grade?: string | null;
87
+ /** measurement-integrity fix-round-1/F9: the slug this signoff was written FOR — carried through so
88
+ * a reader of the eventual ledger row can independently confirm it was not another slug's file. */
89
+ readonly slug?: string;
90
+ /** measurement-integrity fix-round-1/F9: the qe-bridge review's OWN internal run id (from the
91
+ * signoff filename, `signoff-<runId>.json`) — a DIFFERENT id space from `RoundState.run` (which
92
+ * names a feature-adr pipeline run); carried for traceability/audit, never used as a join key. */
93
+ readonly runId?: string | null;
94
+ /** measurement-integrity fix-round-1/F9: when this signoff was emitted — the timestamp
95
+ * `findQeBridgeSignoffForRound` uses to confirm it falls inside THIS round's own interval. */
96
+ readonly emittedAt?: string;
54
97
  }
55
98
 
56
99
  type RoundRefusal = { readonly ok: false; readonly exit: 1 | 2; readonly reason: string };
@@ -81,10 +124,18 @@ export function openRound(input: {
81
124
  readonly force: boolean;
82
125
  readonly existingOwnerAlive: boolean | null;
83
126
  readonly isRunAlive: (runId: string) => boolean | null;
127
+ /** experiment-envelope FR-3(в)/AC-5: when present, validated BEFORE anything else — an invalid
128
+ * envelope refuses the open (exit 2) with the validator's own reason, same as any other malformed
129
+ * input to this command. Absent stays absent (round open without --envelope is unaffected). */
130
+ readonly envelope?: unknown;
84
131
  }): { readonly ok: true; readonly state: RoundState; readonly archiveExisting: boolean } | RoundRefusal {
85
132
  if (!validSlug(input.slug) || !Number.isInteger(input.round) || input.round < 1 || !nonEmpty(input.topic)) {
86
133
  return { ok: false, exit: 2, reason: 'нужны безопасный --slug, положительный --round и непустой --topic' };
87
134
  }
135
+ if (input.envelope !== undefined) {
136
+ const v = validateExperimentEnvelope(input.envelope);
137
+ if (!v.ok) return { ok: false, exit: 2, reason: `--envelope invalid — ${v.reason}` };
138
+ }
88
139
  const runOwner = input.ownerKind === 'run';
89
140
  if (!Number.isFinite(Date.parse(input.startedAt)) || !Number.isInteger(input.ownerPid)
90
141
  || (runOwner ? input.ownerPid !== 0 || !nonEmpty(input.ownerRun) : input.ownerPid < 1)
@@ -101,6 +152,7 @@ export function openRound(input: {
101
152
  ...(runOwner ? { ownerRun: input.ownerRun!.trim() } : {}),
102
153
  ...(nonEmpty(input.run) ? { run: input.run.trim() } : {}),
103
154
  recalled: [...input.recalled],
155
+ ...(input.envelope !== undefined ? { envelope: input.envelope } : {}),
104
156
  };
105
157
  let existingOwnerAlive = input.existingOwnerAlive;
106
158
  if (input.existing?.ownerKind === 'run' && input.force) {
@@ -138,13 +190,26 @@ export function closeRound(input: {
138
190
  readonly closedAt: string;
139
191
  readonly knownLessonIds: readonly string[];
140
192
  readonly stateId?: string | undefined;
193
+ /** measurement-integrity FR-7: `A`, `A-`, `B+`, … — mandatory for `shipped|refuted`, a
194
+ * warned-and-dropped no-op for `blocked|abandoned`. */
195
+ readonly grade?: string | undefined;
196
+ /** measurement-integrity FR-7: the qe-bridge signoff for this slug, read by the CALLER. */
197
+ readonly reviewSidecar?: RoundReviewSidecar | undefined;
141
198
  }, io: {
142
199
  readonly writeLedger: (row: RoundLedgerRow) => unknown;
143
200
  readonly readLedgerTail: () => string;
144
- }): { readonly ok: true; readonly row: RoundLedgerRow; readonly marker: string } | RoundRefusal {
201
+ }): { readonly ok: true; readonly row: RoundLedgerRow; readonly marker: string; readonly warnings: readonly string[] } | RoundRefusal {
145
202
  if (!(ROUND_OUTCOMES as readonly string[]).includes(input.outcome)) {
146
203
  return { ok: false, exit: 2, reason: '--outcome: shipped | refuted | blocked | abandoned' };
147
204
  }
205
+ // fix-round-1/F6: `openRound` validated the envelope once, at open time, then trusted it verbatim
206
+ // out of the state FILE from then on. A state file is mutable disk state between open and close —
207
+ // corrupted or hand-edited in that window, it would ride an invalid/tampered envelope straight
208
+ // into the ledger row. Re-validating here, right before the row is built, closes that window.
209
+ if (input.state.envelope !== undefined) {
210
+ const v = validateExperimentEnvelope(input.state.envelope);
211
+ if (!v.ok) return { ok: false, exit: 2, reason: `envelope in round state invalid — ${v.reason}` };
212
+ }
148
213
  if (!validCount(input.tokens) || !validCount(input.agents)) {
149
214
  return { ok: false, exit: 2, reason: '--tokens и --agents должны быть целыми числами не меньше нуля' };
150
215
  }
@@ -160,6 +225,58 @@ export function closeRound(input: {
160
225
  const missing = lessons.find((id) => !/^teach:[a-z0-9]+$/i.test(id) || !known.has(id));
161
226
  if (missing !== undefined) return { ok: false, exit: 1, reason: `урок не найден: ${missing}` };
162
227
 
228
+ // measurement-integrity FR-7 (ADR-001 D5): grade is mandatory for a FINISHED review
229
+ // (shipped|refuted), a warned-and-dropped no-op for one that never finished (blocked|abandoned —
230
+ // "Отвергнуто: --grade всегда обязателен — заблокированный круг оценки не имеет"). Placed AFTER
231
+ // the outcome/envelope/tokens/lesson checks above (unchanged ordering, unchanged refusal reasons
232
+ // for those) and BEFORE the duration/row-build below.
233
+ const outcome = input.outcome as RoundOutcome;
234
+ const finishedOutcome = outcome === 'shipped' || outcome === 'refuted';
235
+ const gradeFlagRaw = nonEmpty(input.grade) ? input.grade.trim() : null;
236
+ if (gradeFlagRaw !== null && !/^[A-F][+-]?$/.test(gradeFlagRaw)) {
237
+ return { ok: false, exit: 2, reason: `--grade "${gradeFlagRaw}" не распознан — ожидается вид A|A-|B+|C…F` };
238
+ }
239
+ const warnings: string[] = [];
240
+ // measurement-integrity fix-round-1/F10 (Codex r1 MEDIUM #10): for an UNFINISHED outcome, --grade
241
+ // is dropped-with-warning HERE, BEFORE it is ever compared against the sidecar. The OLD ordering
242
+ // ran the conflict check first — so `--outcome blocked --grade B` against a STALE sidecar grade
243
+ // `A` refused outright instead of the promised warning-and-drop, because a value that was about to
244
+ // be discarded still had to survive a conflict check on its way to being discarded. `gradeFlag` is
245
+ // `null` for an unfinished outcome from this point on, exactly as if `--grade` had never been
246
+ // passed — the conflict check below therefore never sees it.
247
+ const gradeFlag = finishedOutcome ? gradeFlagRaw : null;
248
+ if (!finishedOutcome && gradeFlagRaw !== null) {
249
+ warnings.push(`--grade "${gradeFlagRaw}" проигнорирован: outcome=${outcome} не является завершённым ревью, оценка не пишется`);
250
+ }
251
+ const sidecarGrade = input.reviewSidecar !== undefined && nonEmpty(input.reviewSidecar.grade ?? undefined)
252
+ ? (input.reviewSidecar.grade as string).trim()
253
+ : null;
254
+ if (gradeFlag !== null && sidecarGrade !== null && gradeFlag !== sidecarGrade) {
255
+ return {
256
+ ok: false,
257
+ exit: 2,
258
+ reason: `--grade "${gradeFlag}" конфликтует с оценкой сайдкара qe-bridge "${sidecarGrade}" для slug ${input.state.slug}`,
259
+ };
260
+ }
261
+ if (finishedOutcome && gradeFlag === null) {
262
+ return { ok: false, exit: 2, reason: `grade required for a finished review: outcome=${outcome} требует --grade <A|A-|B+|…>` };
263
+ }
264
+ const gradeForRow = gradeFlag;
265
+
266
+ // reviewer/reviewMinutes/reviewSource: an explicit --reviewer always wins; the sidecar fills the
267
+ // gap only, and only its OWN two derived fields travel with it (a flag-supplied reviewer never
268
+ // carries a sidecar-sourced `reviewMinutes`/`reviewSource` — that would misattribute where the
269
+ // minutes figure came from).
270
+ let reviewer = nonEmpty(input.reviewer) ? input.reviewer.trim() : null;
271
+ let reviewMinutes: number | null = null;
272
+ let reviewSource: 'qe-bridge' | null = null;
273
+ if (reviewer === null && input.reviewSidecar !== undefined && nonEmpty(input.reviewSidecar.gradedBy)) {
274
+ reviewer = input.reviewSidecar.gradedBy.trim();
275
+ const elapsedMs = input.reviewSidecar.elapsedMs;
276
+ reviewMinutes = Number.isFinite(elapsedMs) && elapsedMs >= 0 ? Math.round((elapsedMs / 60_000) * 10) / 10 : null;
277
+ reviewSource = 'qe-bridge';
278
+ }
279
+
163
280
  const startedMs = Date.parse(input.state.startedAt);
164
281
  const closedMs = Date.parse(input.closedAt);
165
282
  if (!Number.isFinite(startedMs) || !Number.isFinite(closedMs) || closedMs < startedMs) {
@@ -173,13 +290,13 @@ export function closeRound(input: {
173
290
  stage: 'round',
174
291
  tier: null,
175
292
  coder: nonEmpty(input.coder) ? input.coder.trim() : null,
176
- reviewer: nonEmpty(input.reviewer) ? input.reviewer.trim() : null,
293
+ reviewer,
177
294
  lead: null,
178
295
  minutes: input.noCost === true ? null : Math.floor((closedMs - startedMs) / 60_000),
179
296
  agents: input.noCost === true ? null : input.agents ?? null,
180
297
  tokens: input.noCost === true ? null : input.tokens ?? null,
181
- grade: null,
182
- outcome: input.outcome as RoundOutcome,
298
+ grade: gradeForRow,
299
+ outcome,
183
300
  reason: nonEmpty(input.reason) ? input.reason.trim() : null,
184
301
  round: input.state.round,
185
302
  lessons,
@@ -188,6 +305,9 @@ export function closeRound(input: {
188
305
  date: null,
189
306
  ...(input.noCost === true ? { costIn: 'stages' as const } : {}),
190
307
  ...(nonEmpty(input.stateId) ? { stateId: input.stateId } : {}),
308
+ ...(input.state.envelope !== undefined ? { envelope: input.state.envelope } : {}),
309
+ ...(reviewSource !== null ? { reviewSource } : {}),
310
+ ...(reviewMinutes !== null ? { reviewMinutes } : {}),
191
311
  };
192
312
 
193
313
  try {
@@ -200,7 +320,46 @@ export function closeRound(input: {
200
320
  if (!tail.includes(marker)) {
201
321
  return { ok: false, exit: 1, reason: 'строка не найдена — круг НЕ закрыт' };
202
322
  }
203
- return { ok: true, row, marker };
323
+ return { ok: true, row, marker, warnings };
324
+ }
325
+
326
+ /**
327
+ * measurement-integrity fix-round-1/F8 (Codex r1 CRITICAL #8): validate an ALREADY-WRITTEN ledger
328
+ * row against the CURRENT schema's `shipped|refuted require a grade` rule (ADR-001 D5 / FR-7).
329
+ *
330
+ * This exists for exactly one caller: `dz round close`'s idempotent-retry path. When the predicted
331
+ * marker (or `stateId`) is already found in the ledger tail, the CLI used to skip `closeRound`
332
+ * ENTIRELY and delete the round's state file — so an OLD row written before FR-7 shipped (`shipped`
333
+ * with `grade: null`, the exact defect this feature exists to close) could be "recognised as already
334
+ * closed" and the state removed without ever being checked against the rule that is supposed to be
335
+ * mandatory. This function is that missing check, run on the ALREADY-FOUND row before the CLI is
336
+ * allowed to treat the retry as a success.
337
+ *
338
+ * Deliberately NARROW: it re-checks only the ONE FR-7 invariant (a schema rule with a proving test),
339
+ * not every field `closeRound` validates on the FIRST write (grade format, envelope shape, …) — those
340
+ * were already enforced when the row was ORIGINALLY written; re-validating them here would either
341
+ * duplicate that logic or silently drift from it. Pure, never throws.
342
+ */
343
+ export function validateClosedRoundLedgerRow(row: unknown): { readonly ok: true } | { readonly ok: false; readonly reason: string } {
344
+ if (typeof row !== 'object' || row === null || Array.isArray(row)) {
345
+ return { ok: false, reason: 'найденная строка леджера не JSON-объект — не удаётся проверить оценку' };
346
+ }
347
+ const r = row as Record<string, unknown>;
348
+ const outcome = r['outcome'];
349
+ if (typeof outcome !== 'string' || !(ROUND_OUTCOMES as readonly string[]).includes(outcome)) {
350
+ return { ok: false, reason: `найденная строка леджера несёт неизвестный outcome ${JSON.stringify(outcome)}` };
351
+ }
352
+ const finishedOutcome = outcome === 'shipped' || outcome === 'refuted';
353
+ if (!finishedOutcome) return { ok: true }; // blocked|abandoned carry no grade requirement (FR-7)
354
+ const grade = r['grade'];
355
+ if (typeof grade !== 'string' || !/^[A-F][+-]?$/.test(grade)) {
356
+ return {
357
+ ok: false,
358
+ reason: `найденная строка леджера — outcome=${outcome} без валидной оценки (grade=${JSON.stringify(grade)}); ` +
359
+ 'закрытие отказано (measurement-integrity ADR-001 D5 / FR-7 требует непустую оценку для завершённого ревью)',
360
+ };
361
+ }
362
+ return { ok: true };
204
363
  }
205
364
 
206
365
  export function listRounds(states: readonly RoundState[], input: {
@@ -13,7 +13,133 @@
13
13
  * Pure: payload in, verdict out. The CLI owns paths, the append, the read-back and the exit code.
14
14
  */
15
15
 
16
+ import { matchCodexRollouts } from './codex-rollouts.js';
17
+ import type { CodexRollout } from './codex-rollouts.js';
16
18
  import { redactTrainingPayload } from './feature-adr-checkpoints.js';
19
+ import { validateExperimentEnvelope } from './feature-adr-envelope.js';
20
+
21
+ /** Structural — a caller passes `cost-scoring.ts`'s `ModelPricing`; kept local so `run-records.ts`
22
+ * does not have to import `cost-scoring.ts` just to name a type.
23
+ *
24
+ * measurement-integrity fix-round-1/F6 (Codex r1 HIGH #6): `cacheCreation` used to be dropped here —
25
+ * the snapshot silently lacked the ONE rate a cache-WRITE-heavy row needs to reproduce its own cost
26
+ * later, even though the caller's own `ModelPricing` carries it. Now carried through verbatim. */
27
+ export interface LedgerPriceEntry {
28
+ readonly prompt: number;
29
+ readonly completion: number;
30
+ readonly cachedInput: number;
31
+ readonly cacheCreation: number;
32
+ }
33
+
34
+ /** measurement-integrity fix-round-1/F4 (Codex r1 HIGH #4): a resolvable executor spec, split into
35
+ * its three parts. Never invents a model: {@link parseModelSpec} returns `null` for anything it
36
+ * cannot resolve to exactly one model, rather than guessing. */
37
+ export interface ParsedModelSpec {
38
+ readonly family: 'claude' | 'codex';
39
+ readonly model: string;
40
+ readonly effort: string | null;
41
+ }
42
+
43
+ /** Bare Claude model names the pipeline actually emits with no `claude:` prefix (`coder: 'sonnet'`,
44
+ * `'opus'`, `'fable'`, real values observed in `.dz/feature-adr/run-cost-ledger.jsonl`). */
45
+ const BARE_CLAUDE_NAMES = new Set(['sonnet', 'opus', 'haiku', 'fable']);
46
+ /** id/effort alphabet a model spec component may use (lead delta after Codex r2, new MEDIUM #3). */
47
+ const SPEC_PART = /^[A-Za-z0-9._-]+$/;
48
+
49
+ /**
50
+ * Parse an executor spec — the shapes actually recorded in `coder`/`reviewer` fields
51
+ * (`codex:gpt-5.6-sol:high`, `claude:sonnet`, bare `sonnet`/`opus`, bare `codex`, bare `claude`) —
52
+ * into `{family, model, effort}`. Pure, never throws.
53
+ *
54
+ * Returns `null` (never a guess) for anything that cannot be resolved to exactly ONE model:
55
+ * - a bare `'codex'` or `'claude'` (family named, no model at all);
56
+ * - an annotated/aggregate field such as `'claude:sonnet x2'` or `'qe-bridge:claude x2 + lead'` (real
57
+ * values this ledger carries for a MULTI-reviewer round) — any embedded whitespace means the field
58
+ * names more than one resolvable spec, and picking one would misattribute to the others;
59
+ * - a bare model id with no family marker that is not one of the known bare Claude names (e.g. a full
60
+ * `'claude-sonnet-5'` — that shape is handled by the OLDER vendor-prefix path in {@link priceLookup}
61
+ * for backward compatibility, not by this parser).
62
+ */
63
+ export function parseModelSpec(spec: unknown): ParsedModelSpec | null {
64
+ if (typeof spec !== 'string') return null;
65
+ const trimmed = spec.trim();
66
+ if (trimmed === '' || /\s/.test(trimmed)) return null;
67
+ const parts = trimmed.split(':');
68
+ const head = parts[0] ?? '';
69
+ if (head === 'codex') {
70
+ const model = parts[1];
71
+ if (model === undefined || model === '') return null; // bare 'codex' — no reliable model
72
+ // Lead delta after Codex r2 (new MEDIUM #3): exactly 2 or 3 non-empty components, id alphabet
73
+ // only — `codex:gpt-5.6-sol:high:garbage` is a corrupt/aggregate spec, never a reliable model.
74
+ if (parts.length > 3 || !SPEC_PART.test(model) || (parts.length === 3 && (parts[2] === '' || !SPEC_PART.test(parts[2]!)))) return null;
75
+ const effort = parts[2] ?? null;
76
+ return { family: 'codex', model, effort: effort === '' ? null : effort };
77
+ }
78
+ if (head === 'claude') {
79
+ const model = parts[1];
80
+ if (model === undefined || model === '') return null; // bare 'claude' — no reliable model
81
+ return { family: 'claude', model, effort: null };
82
+ }
83
+ if (parts.length === 1 && BARE_CLAUDE_NAMES.has(head)) {
84
+ return { family: 'claude', model: head, effort: null };
85
+ }
86
+ return null;
87
+ }
88
+
89
+ /** measurement-integrity FR-5/FR-6: enrichment the WRITER supplies at write time — the rollout logs
90
+ * it already read (I/O lives in the CLI; this stays pure) and the price table snapshot. Absent
91
+ * entirely ⇒ zero behavior change from before this feature (NFR-1). */
92
+ export interface LedgerEnrichInput {
93
+ /** Parsed Codex rollout logs for the window the CLI read — usually every rollout from the days the
94
+ * window spans. Pure data; the CLI is the one that walked `~/.codex/sessions`. */
95
+ readonly rollouts?: readonly CodexRollout[];
96
+ /** The stage's own time window — usually [the previous ledger row's `ts`, this write's `ts`], or
97
+ * an explicit `--window-from/--window-to`. Omitted ⇒ no rollout match is even attempted. */
98
+ readonly window?: { readonly from: string; readonly to: string };
99
+ /** Narrows an otherwise-ambiguous match — usually the repo root the stage ran in. */
100
+ readonly cwd?: string;
101
+ /** A model-pricing table SNAPSHOT (FR-6) — the CALLER's table, captured at write time, never the
102
+ * ledger's own idea of "current" pricing (ADR-001 D4 rejects re-pricing after the fact). */
103
+ readonly prices?: Readonly<Record<string, LedgerPriceEntry>>;
104
+ }
105
+
106
+ function isCodexFamily(v: unknown): boolean {
107
+ return typeof v === 'string' && /codex/i.test(v);
108
+ }
109
+
110
+ function isRecord(v: unknown): v is Record<string, unknown> {
111
+ return typeof v === 'object' && v !== null && !Array.isArray(v);
112
+ }
113
+
114
+ /** Longest-prefix match against the CALLER'S OWN table (never `cost-scoring.ts`'s internal
115
+ * constant) — the whole point of a price SNAPSHOT is that it answers only from what the caller
116
+ * handed in at write time, mirroring `pricingFor`'s matching rule without importing it.
117
+ *
118
+ * measurement-integrity fix-round-1/F7 (Codex r1 HIGH #7): the OLD version stripped only a
119
+ * `vendor/`-shaped prefix, so the REAL recorded shape `codex:gpt-5.6-sol:high` (or `claude:sonnet`)
120
+ * never matched anything and always landed in `prices.unknown[]`. This now runs the id through the
121
+ * SAME {@link parseModelSpec} FR-4's matcher uses, then reconstructs the normalized id the way
122
+ * `cost-scoring.ts`'s `pricingFor`/`hasKnownPricing` key their table (`claude-<model>` for the
123
+ * Claude family; the bare model for Codex — its ids carry no vendor prefix). A spec this parser
124
+ * cannot resolve falls back to the OLD vendor-prefix strip, so an already-working bare id
125
+ * (`claude-sonnet-5`, `gpt-4o`) keeps matching exactly as before (NFR-1). */
126
+ function priceLookup(modelId: string, table: Readonly<Record<string, LedgerPriceEntry>>): LedgerPriceEntry | null {
127
+ if (typeof modelId !== 'string' || modelId.length === 0) return null;
128
+ const parsed = parseModelSpec(modelId);
129
+ const id = parsed !== null
130
+ ? (parsed.family === 'claude' ? `claude-${parsed.model}` : parsed.model).toLowerCase()
131
+ : modelId.toLowerCase().replace(/^[a-z0-9-]+\//, '');
132
+ let best: LedgerPriceEntry | null = null;
133
+ let bestLen = 0;
134
+ for (const [key, price] of Object.entries(table)) {
135
+ const k = key.toLowerCase();
136
+ if (id.startsWith(k) && k.length > bestLen) {
137
+ best = price;
138
+ bestLen = k.length;
139
+ }
140
+ }
141
+ return best;
142
+ }
17
143
 
18
144
  export type RecordKind = 'ledger' | 'training-pair';
19
145
 
@@ -65,7 +191,7 @@ const noop = (verdict: 'duplicate' | 'skipped', reason: string): RecordDecision
65
191
  line: null,
66
192
  });
67
193
 
68
- function shapeMismatch(kind: RecordKind, payload: Record<string, unknown>): string | null {
194
+ function shapeMismatch(kind: RecordKind, payload: Record<string, unknown>, autoFlag: boolean): string | null {
69
195
  // AM-1 FIRST, before the required-field sweep. A ledger row offered as a training pair fails BOTH
70
196
  // checks, and the wrong-kind reason is the one that tells the caller what actually happened —
71
197
  // "missing field `output`" sends them looking for a field they never meant to send.
@@ -78,6 +204,39 @@ function shapeMismatch(kind: RecordKind, payload: Record<string, unknown>): stri
78
204
  if (kind === 'training-pair' && 'tokens' in payload && !('input' in payload)) {
79
205
  return 'this payload looks like a ledger row (`tokens` without `input`), not a training pair';
80
206
  }
207
+ // experiment-envelope FR-5 / ADR-001 D2: `auto:true` marks a pipeline-written row, and the pipeline
208
+ // is obligated to carry the envelope built once after the Step-0 router — a row missing it is
209
+ // useless for learning and would silently corrupt the sample (the "warn but write" alternative was
210
+ // rejected in the ADR: a warning nobody reads left `tokens=null` unnoticed for years). A MANUAL row
211
+ // (no `auto`) stays compatible: no envelope required, but one that IS present is still validated —
212
+ // never trusted just because a human typed it.
213
+ //
214
+ // fix-round-1/F2 (cross-family review, HIGH #2): the PAYLOAD's own `auto` field used to be the
215
+ // ONLY signal — an automatic producer that forgot it, or sent `"true"`/`false`, was silently
216
+ // accepted as a manual row and skipped the envelope requirement entirely. `autoFlag` is the CLI's
217
+ // own trusted `--auto` argument (never JSON a caller could typo): it is a SECOND, independent
218
+ // trust source, unioned with the payload field rather than replacing it — nothing that used to be
219
+ // gated stops being gated, and a `--auto`-dispatched caller is now gated even if its hand-built
220
+ // payload forgot the field. Independently of either source, a PRESENT `auto` field is checked for
221
+ // shape: only the literal `true` is a legal value — anything else (a string, `false`, a number) is
222
+ // refused outright, because a field whose only sane value is `true` holding something else is a
223
+ // caller bug worth surfacing, not silently downgrading to "manual".
224
+ if (kind === 'ledger') {
225
+ const rawAuto = payload['auto'];
226
+ if (rawAuto !== undefined && rawAuto !== true) {
227
+ return 'a ledger record\'s `auto` field must be `true` or absent';
228
+ }
229
+ const envelope = payload['envelope'];
230
+ const envelopePresent = envelope !== undefined && envelope !== null;
231
+ const isAuto = autoFlag === true || rawAuto === true;
232
+ if (isAuto && !envelopePresent) {
233
+ return 'an auto ledger row must carry `envelope` (experiment-envelope FR-5)';
234
+ }
235
+ if (envelopePresent) {
236
+ const v = validateExperimentEnvelope(envelope);
237
+ if (!v.ok) return `envelope invalid — ${v.reason}`;
238
+ }
239
+ }
81
240
  const required = kind === 'ledger' ? LEDGER_REQUIRED : PAIR_REQUIRED;
82
241
  for (const field of required) {
83
242
  const v = payload[field];
@@ -125,7 +284,12 @@ export function decideRecordWrite(input: {
125
284
  /** Who ran it. Supplied by the CALLER, which lives outside the workflow sandbox and can see the
126
285
  * host; absent stays absent (see the stamping comment below). */
127
286
  runnerId?: string | null;
287
+ /** fix-round-1/F2: the CLI's own trusted `--auto` flag — see the comment inside `shapeMismatch`. */
288
+ auto?: boolean;
128
289
  maxChars?: number;
290
+ /** measurement-integrity FR-5/FR-6: rollout-log + price enrichment for a ledger row. Absent ⇒ zero
291
+ * behavior change (NFR-1). */
292
+ enrich?: LedgerEnrichInput;
129
293
  }): RecordDecision {
130
294
  const { kind, payloadRaw, stage } = input;
131
295
  if (kind !== 'ledger' && kind !== 'training-pair') {
@@ -153,7 +317,7 @@ export function decideRecordWrite(input: {
153
317
  // the read-back, the refusal texts — ever sees a byte of the profile block.
154
318
  const obj = (kind === 'training-pair' ? redactTrainingPayload(payload) : payload) as Record<string, unknown>;
155
319
 
156
- const mismatch = shapeMismatch(kind, obj);
320
+ const mismatch = shapeMismatch(kind, obj, input.auto === true);
157
321
  if (mismatch !== null) return refuse(mismatch);
158
322
 
159
323
  // The record is filed under `--stage`, and the payload carries its own. A disagreement means the
@@ -182,6 +346,11 @@ export function decideRecordWrite(input: {
182
346
  // inside an already-serialised document — text surgery on a structured value, and the exact place
183
347
  // a payload containing that literal token could corrupt itself.
184
348
  const stamped: Record<string, unknown> = { ...obj };
349
+ // fix-round-1/F2: the CLI's own `--auto` flag is authoritative — when set, the written row MUST
350
+ // carry `auto:true` too (not just gate on it transiently), so every downstream reader of the
351
+ // PERSISTED line keeps seeing the same signal `shapeMismatch` already gated on above. A no-op when
352
+ // the payload already said `auto:true` (shapeMismatch already refused any OTHER value).
353
+ if (kind === 'ledger' && input.auto === true) stamped['auto'] = true;
185
354
  const isGap = (v: unknown): boolean => v === null || v === undefined || (typeof v === 'string' && v.trim() === '');
186
355
  if (input.timestamp != null && input.timestamp !== '') {
187
356
  // An EMPTY STRING is a gap, not a value. Stamping only over null/undefined let
@@ -273,6 +442,117 @@ export function decideRecordWrite(input: {
273
442
  }
274
443
  }
275
444
 
445
+ // measurement-integrity FR-5 (rollout enrichment) + FR-6 (price snapshot). Both are ADDITIVE and
446
+ // OPT-IN on `input.enrich` — a caller that never passes it gets byte-identical output to before
447
+ // this feature (NFR-1).
448
+ if (kind === 'ledger' && input.enrich !== undefined) {
449
+ const enrich = input.enrich;
450
+
451
+ // FR-5: only a codex-family coder/reviewer with `tokens: null` is a candidate — a Claude row, or
452
+ // one that already has a token figure, is left untouched. The loose `/codex/i` check below only
453
+ // decides whether this row is WORTH TRYING at all.
454
+ const tokensIsNull = stamped['tokens'] === null;
455
+ const looksCodexFamily = isCodexFamily(stamped['coder']) || isCodexFamily(stamped['reviewer']);
456
+ if (tokensIsNull && looksCodexFamily) {
457
+ // measurement-integrity fix-round-1/F4 (Codex r1 HIGH #4): the matcher REQUIRES a reliable
458
+ // model, parsed the same way FR-7's price lookup parses one — never `/codex/i` alone. If
459
+ // `coder`/`reviewer` do not resolve to exactly ONE codex model between them (a bare `'codex'`
460
+ // with no model at all, or the two fields naming DIFFERENT codex models), the matcher is never
461
+ // even called with an unreliable/omitted model filter — a lone rollout in the window would
462
+ // otherwise be accepted as `'one'` on time+cwd alone and its tokens misattributed to the wrong
463
+ // model's stage.
464
+ const codexModels = new Set(
465
+ [parseModelSpec(stamped['coder']), parseModelSpec(stamped['reviewer'])]
466
+ .filter((s): s is ParsedModelSpec => s !== null && s.family === 'codex')
467
+ .map((s) => s.model),
468
+ );
469
+ if (codexModels.size !== 1) {
470
+ stamped['tokensSource'] = 'codex-rollout:no-model';
471
+ } else if (enrich.window !== undefined) {
472
+ const model = [...codexModels][0]!;
473
+ const match = matchCodexRollouts(enrich.rollouts ?? [], {
474
+ from: enrich.window.from,
475
+ to: enrich.window.to,
476
+ model,
477
+ ...(enrich.cwd !== undefined ? { cwd: enrich.cwd } : {}),
478
+ });
479
+ if (match.status === 'one') {
480
+ stamped['tokens'] = match.rollout.totals.total;
481
+ const startMs = match.rollout.startedAt !== null ? Date.parse(match.rollout.startedAt) : NaN;
482
+ const endMs = match.rollout.endedAt !== null ? Date.parse(match.rollout.endedAt) : NaN;
483
+ // measurement-integrity fix-round-1/F6 (Codex r1 HIGH #6): fill-ONLY-null — an existing
484
+ // `minutes` figure (a manually recorded one, say) must never be silently overwritten by a
485
+ // derived rollout duration.
486
+ if (stamped['minutes'] === null && Number.isFinite(startMs) && Number.isFinite(endMs) && endMs >= startMs) {
487
+ stamped['minutes'] = Math.round(((endMs - startMs) / 60000) * 10) / 10;
488
+ }
489
+ stamped['tokensSource'] = 'codex-rollout';
490
+ stamped['rolloutId'] = match.rollout.id;
491
+ } else {
492
+ // `none` or `ambiguous` — NFR-3: an explicit status, never a guessed number.
493
+ stamped['tokensSource'] = `codex-rollout:${match.status}`;
494
+ }
495
+ } else {
496
+ // Eligible in principle (codex family, one reliable model, tokens null) but no window was
497
+ // supplied — AC-4's "old row without a window": no enrichment is even attempted, and that
498
+ // fact is itself recorded rather than left silently absent.
499
+ stamped['tokensSource'] = 'unavailable';
500
+ }
501
+ }
502
+
503
+ // FR-6: the price snapshot, for every model this row names — independent of the FR-5 branch
504
+ // above (a Claude row gets priced too; only tokens enrichment is codex-specific).
505
+ if (enrich.prices !== undefined) {
506
+ const modelIds = new Set<string>();
507
+ for (const v of [stamped['coder'], stamped['reviewer']]) {
508
+ if (typeof v === 'string' && v.trim() !== '') modelIds.add(v.trim());
509
+ }
510
+ const envelope = stamped['envelope'];
511
+ if (isRecord(envelope)) {
512
+ const chosen = envelope['chosen'];
513
+ if (isRecord(chosen)) {
514
+ const stages = chosen['stages'];
515
+ if (isRecord(stages)) {
516
+ for (const v of Object.values(stages)) {
517
+ if (typeof v === 'string' && v.trim() !== '') modelIds.add(v.trim());
518
+ }
519
+ }
520
+ }
521
+ }
522
+ const table: Record<string, LedgerPriceEntry> = {};
523
+ const unknown: string[] = [];
524
+ for (const modelId of modelIds) {
525
+ const price = priceLookup(modelId, enrich.prices);
526
+ if (price === null) unknown.push(modelId);
527
+ else table[modelId] = { prompt: price.prompt, completion: price.completion, cachedInput: price.cachedInput, cacheCreation: price.cacheCreation };
528
+ }
529
+ const computedPrices = {
530
+ snapshotAt: input.timestamp ?? null,
531
+ table,
532
+ ...(unknown.length > 0 ? { unknown } : {}),
533
+ };
534
+ // measurement-integrity fix-round-1/F6 (Codex r1 HIGH #6): fill-ONLY-null for `prices` too — an
535
+ // existing snapshot (a previous write already priced this row) is never unconditionally
536
+ // replaced. Equal → left alone (idempotent re-enrichment, common on a retried write). Different
537
+ // → a named `pricesConflict`, never a silent re-price (ADR-001 D4 forbids re-pricing after the
538
+ // fact — a DIFFERING recomputation is exactly that, so it is surfaced, not applied).
539
+ const existingPrices = stamped['prices'];
540
+ if (existingPrices === undefined || existingPrices === null) {
541
+ stamped['prices'] = computedPrices;
542
+ } else if (isRecord(existingPrices) && isRecord(existingPrices['table']) && (existingPrices['unknown'] === undefined || Array.isArray(existingPrices['unknown']))) {
543
+ // Lead delta after Codex r2 (new MEDIUM #4): compare the WHOLE snapshot canonically (table +
544
+ // sorted unknown), not the table alone.
545
+ const canon = (t: unknown, u: unknown): string => JSON.stringify({ table: t, unknown: Array.isArray(u) ? [...u].map(String).sort() : [] });
546
+ if (canon(existingPrices['table'], existingPrices['unknown']) !== canon(table, unknown)) {
547
+ stamped['pricesConflict'] = { existing: existingPrices, recomputed: computedPrices };
548
+ }
549
+ } else {
550
+ // A malformed existing snapshot (no table / bad unknown) is a CONFLICT, never silently trusted.
551
+ stamped['pricesConflict'] = { existing: existingPrices, recomputed: computedPrices, reason: 'existing prices snapshot is malformed' };
552
+ }
553
+ }
554
+ }
555
+
276
556
  let line: string;
277
557
  try {
278
558
  line = JSON.stringify(stamped);