@dzhechkov/harness-core 0.8.35 → 0.8.37
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +224 -104
- package/README.md +335 -10
- package/dist/agentdb-index.d.ts +87 -7
- package/dist/agentdb-index.d.ts.map +1 -1
- package/dist/agentdb-index.js +416 -57
- package/dist/agentdb-index.js.map +1 -1
- package/dist/apply-leg.d.ts +57 -1
- package/dist/apply-leg.d.ts.map +1 -1
- package/dist/apply-leg.js +450 -52
- package/dist/apply-leg.js.map +1 -1
- package/dist/codex-hooks-assets.d.ts.map +1 -1
- package/dist/codex-hooks-assets.js +67 -5
- package/dist/codex-hooks-assets.js.map +1 -1
- package/dist/codex-hooks.d.ts +13 -1
- package/dist/codex-hooks.d.ts.map +1 -1
- package/dist/codex-hooks.js +13 -1
- package/dist/codex-hooks.js.map +1 -1
- package/dist/codex-rollouts.d.ts +118 -0
- package/dist/codex-rollouts.d.ts.map +1 -0
- package/dist/codex-rollouts.js +297 -0
- package/dist/codex-rollouts.js.map +1 -0
- package/dist/cost-ledger.d.ts +56 -4
- package/dist/cost-ledger.d.ts.map +1 -1
- package/dist/cost-ledger.js +176 -20
- package/dist/cost-ledger.js.map +1 -1
- package/dist/cross-family-control.d.ts +345 -0
- package/dist/cross-family-control.d.ts.map +1 -0
- package/dist/cross-family-control.js +802 -0
- package/dist/cross-family-control.js.map +1 -0
- package/dist/debt-ratchet.d.ts +53 -0
- package/dist/debt-ratchet.d.ts.map +1 -0
- package/dist/debt-ratchet.js +107 -0
- package/dist/debt-ratchet.js.map +1 -0
- package/dist/embedding-config.d.ts +42 -0
- package/dist/embedding-config.d.ts.map +1 -1
- package/dist/embedding-config.js +106 -10
- package/dist/embedding-config.js.map +1 -1
- package/dist/feature-adr-checkpoints.d.ts +6 -0
- package/dist/feature-adr-checkpoints.d.ts.map +1 -1
- package/dist/feature-adr-checkpoints.js +29 -0
- package/dist/feature-adr-checkpoints.js.map +1 -1
- package/dist/feature-adr-decision-recall.d.ts +2 -2
- package/dist/feature-adr-decision-recall.d.ts.map +1 -1
- package/dist/feature-adr-decision-recall.js +5 -3
- package/dist/feature-adr-decision-recall.js.map +1 -1
- package/dist/feature-adr-envelope.d.ts +96 -0
- package/dist/feature-adr-envelope.d.ts.map +1 -0
- package/dist/feature-adr-envelope.js +183 -0
- package/dist/feature-adr-envelope.js.map +1 -0
- package/dist/feature-adr-routing.d.ts +64 -0
- package/dist/feature-adr-routing.d.ts.map +1 -1
- package/dist/feature-adr-routing.js +122 -2
- package/dist/feature-adr-routing.js.map +1 -1
- package/dist/feature-adr-stage-canon.d.ts +79 -0
- package/dist/feature-adr-stage-canon.d.ts.map +1 -0
- package/dist/feature-adr-stage-canon.js +117 -0
- package/dist/feature-adr-stage-canon.js.map +1 -0
- package/dist/index.d.ts +23 -12
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +15 -7
- package/dist/index.js.map +1 -1
- package/dist/loop-blobs.generated.js +4 -4
- package/dist/loop-blobs.generated.js.map +1 -1
- package/dist/mutation-gate.d.ts +51 -0
- package/dist/mutation-gate.d.ts.map +1 -1
- package/dist/mutation-gate.js +295 -0
- package/dist/mutation-gate.js.map +1 -1
- package/dist/operations.d.ts +1 -0
- package/dist/operations.d.ts.map +1 -1
- package/dist/operations.js +18 -2
- package/dist/operations.js.map +1 -1
- package/dist/publish.d.ts +59 -7
- package/dist/publish.d.ts.map +1 -1
- package/dist/publish.js +205 -32
- package/dist/publish.js.map +1 -1
- package/dist/qe-bridge.d.ts.map +1 -1
- package/dist/qe-bridge.js +4 -2
- package/dist/qe-bridge.js.map +1 -1
- package/dist/qe-findings.d.ts +107 -0
- package/dist/qe-findings.d.ts.map +1 -0
- package/dist/qe-findings.js +417 -0
- package/dist/qe-findings.js.map +1 -0
- package/dist/recap.d.ts +1 -1
- package/dist/recap.d.ts.map +1 -1
- package/dist/recap.js +4 -2
- package/dist/recap.js.map +1 -1
- package/dist/release-line.d.ts +16 -0
- package/dist/release-line.d.ts.map +1 -1
- package/dist/release-line.js +31 -0
- package/dist/release-line.js.map +1 -1
- package/dist/round.d.ts +74 -1
- package/dist/round.d.ts.map +1 -1
- package/dist/round.js +112 -4
- package/dist/round.js.map +1 -1
- package/dist/run-records.d.ts +60 -0
- package/dist/run-records.d.ts.map +1 -1
- package/dist/run-records.js +244 -2
- package/dist/run-records.js.map +1 -1
- package/dist/score.d.ts +44 -1
- package/dist/score.d.ts.map +1 -1
- package/dist/score.js +78 -5
- package/dist/score.js.map +1 -1
- package/dist/vector-tier.d.ts +34 -3
- package/dist/vector-tier.d.ts.map +1 -1
- package/dist/vector-tier.js +105 -14
- package/dist/vector-tier.js.map +1 -1
- package/package.json +2 -2
- package/sbom.json +403 -103
- package/src/agentdb-index.ts +423 -60
- package/src/apply-leg.ts +469 -50
- package/src/codex-hooks-assets.ts +67 -5
- package/src/codex-hooks.ts +13 -1
- package/src/codex-rollouts.ts +374 -0
- package/src/cost-ledger.ts +232 -24
- package/src/cross-family-control.ts +960 -0
- package/src/debt-ratchet.ts +143 -0
- package/src/embedding-config.ts +131 -10
- package/src/feature-adr-checkpoints.ts +29 -0
- package/src/feature-adr-decision-recall.ts +6 -4
- package/src/feature-adr-envelope.ts +242 -0
- package/src/feature-adr-routing.ts +139 -2
- package/src/feature-adr-stage-canon.ts +141 -0
- package/src/index.ts +66 -7
- package/src/loop-blobs.generated.ts +4 -4
- package/src/mutation-gate.ts +316 -0
- package/src/operations.ts +18 -3
- package/src/publish.ts +247 -30
- package/src/qe-bridge.ts +4 -2
- package/src/qe-findings.ts +463 -0
- package/src/recap.ts +10 -3
- package/src/release-line.ts +32 -0
- package/src/round.ts +165 -6
- package/src/run-records.ts +282 -2
- package/src/score.ts +115 -6
- package/src/vector-tier.ts +127 -14
package/src/round.ts
CHANGED
|
@@ -4,6 +4,8 @@
|
|
|
4
4
|
* filesystem or process API; the CLI owns `.dz/rounds/` and the witnessed ledger writer.
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
|
+
import { validateExperimentEnvelope } from './feature-adr-envelope.js';
|
|
8
|
+
|
|
7
9
|
const ROUND_OUTCOMES = ['shipped', 'refuted', 'blocked', 'abandoned'] as const;
|
|
8
10
|
type RoundOutcome = typeof ROUND_OUTCOMES[number];
|
|
9
11
|
|
|
@@ -18,6 +20,10 @@ export interface RoundState {
|
|
|
18
20
|
readonly run?: string;
|
|
19
21
|
readonly recalled: readonly string[];
|
|
20
22
|
readonly execs?: readonly RoundExecState[];
|
|
23
|
+
/** experiment-envelope FR-3(в): the envelope the pipeline built after its Step-0 router, carried
|
|
24
|
+
* unchanged through `round open --envelope` into `closeRound`'s ledger row. Opaque here (never
|
|
25
|
+
* interpreted by round.ts) — validated once, at `openRound`, and trusted from then on. */
|
|
26
|
+
readonly envelope?: unknown;
|
|
21
27
|
}
|
|
22
28
|
|
|
23
29
|
export interface RoundExecState {
|
|
@@ -39,7 +45,10 @@ export interface RoundLedgerRow {
|
|
|
39
45
|
readonly minutes: number | null;
|
|
40
46
|
readonly agents: number | null;
|
|
41
47
|
readonly tokens: number | null;
|
|
42
|
-
|
|
48
|
+
/** measurement-integrity FR-7: was always `null` before this feature — `shipped|refuted` now
|
|
49
|
+
* requires a real grade; `blocked|abandoned` still writes `null` (a review that never finished has
|
|
50
|
+
* nothing to grade). */
|
|
51
|
+
readonly grade: string | null;
|
|
43
52
|
readonly outcome: RoundOutcome;
|
|
44
53
|
readonly reason: string | null;
|
|
45
54
|
readonly round: number;
|
|
@@ -48,9 +57,43 @@ export interface RoundLedgerRow {
|
|
|
48
57
|
readonly note: string;
|
|
49
58
|
readonly date: null;
|
|
50
59
|
readonly costIn?: 'stages';
|
|
60
|
+
/** experiment-envelope FR-3(в): copied verbatim from the state that closed this round, when present. */
|
|
61
|
+
readonly envelope?: unknown;
|
|
51
62
|
/** round-state-lock (lead edit after Codex re-review): identity of the state instance this row
|
|
52
63
|
* closes — lets a retried `close` detect its own earlier row regardless of the clock. */
|
|
53
64
|
readonly stateId?: string;
|
|
65
|
+
/** measurement-integrity FR-7: present ONLY when `reviewer` was filled from the qe-bridge sidecar
|
|
66
|
+
* (no explicit `--reviewer`) — `elapsedMs / 60000`, rounded to 1 decimal. */
|
|
67
|
+
readonly reviewMinutes?: number;
|
|
68
|
+
/** measurement-integrity FR-7: present ONLY alongside `reviewMinutes` — names where `reviewer` and
|
|
69
|
+
* `reviewMinutes` came from, so a reader never confuses a sidecar-sourced figure for a flag. */
|
|
70
|
+
readonly reviewSource?: 'qe-bridge';
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* measurement-integrity FR-7: the qe-bridge cross-model-review signoff for THIS round — read by the
|
|
75
|
+
* CLI from `features/<slug>/.fa-state/qe-bridge/signoff-*.json`, selected by matching `slug` AND
|
|
76
|
+
* falling inside THIS round's own `[startedAt, closedAt]` interval (fix-round-1/F9, Codex r1 HIGH
|
|
77
|
+
* #9 — the CLI-side lookup, `findQeBridgeSignoffForRound`, is what enforces that; multiple qualifying
|
|
78
|
+
* signoffs there are an AMBIGUITY refusal, never "the latest wins"), passed in here as DATA.
|
|
79
|
+
* `closeRound` never opens a file.
|
|
80
|
+
*/
|
|
81
|
+
export interface RoundReviewSidecar {
|
|
82
|
+
/** Who graded it — family + model, e.g. `codex:gpt-5.6-sol`. */
|
|
83
|
+
readonly gradedBy: string;
|
|
84
|
+
readonly elapsedMs: number;
|
|
85
|
+
/** The sidecar's OWN verdict, when it carries one — checked against `--grade` for a conflict. */
|
|
86
|
+
readonly grade?: string | null;
|
|
87
|
+
/** measurement-integrity fix-round-1/F9: the slug this signoff was written FOR — carried through so
|
|
88
|
+
* a reader of the eventual ledger row can independently confirm it was not another slug's file. */
|
|
89
|
+
readonly slug?: string;
|
|
90
|
+
/** measurement-integrity fix-round-1/F9: the qe-bridge review's OWN internal run id (from the
|
|
91
|
+
* signoff filename, `signoff-<runId>.json`) — a DIFFERENT id space from `RoundState.run` (which
|
|
92
|
+
* names a feature-adr pipeline run); carried for traceability/audit, never used as a join key. */
|
|
93
|
+
readonly runId?: string | null;
|
|
94
|
+
/** measurement-integrity fix-round-1/F9: when this signoff was emitted — the timestamp
|
|
95
|
+
* `findQeBridgeSignoffForRound` uses to confirm it falls inside THIS round's own interval. */
|
|
96
|
+
readonly emittedAt?: string;
|
|
54
97
|
}
|
|
55
98
|
|
|
56
99
|
type RoundRefusal = { readonly ok: false; readonly exit: 1 | 2; readonly reason: string };
|
|
@@ -81,10 +124,18 @@ export function openRound(input: {
|
|
|
81
124
|
readonly force: boolean;
|
|
82
125
|
readonly existingOwnerAlive: boolean | null;
|
|
83
126
|
readonly isRunAlive: (runId: string) => boolean | null;
|
|
127
|
+
/** experiment-envelope FR-3(в)/AC-5: when present, validated BEFORE anything else — an invalid
|
|
128
|
+
* envelope refuses the open (exit 2) with the validator's own reason, same as any other malformed
|
|
129
|
+
* input to this command. Absent stays absent (round open without --envelope is unaffected). */
|
|
130
|
+
readonly envelope?: unknown;
|
|
84
131
|
}): { readonly ok: true; readonly state: RoundState; readonly archiveExisting: boolean } | RoundRefusal {
|
|
85
132
|
if (!validSlug(input.slug) || !Number.isInteger(input.round) || input.round < 1 || !nonEmpty(input.topic)) {
|
|
86
133
|
return { ok: false, exit: 2, reason: 'нужны безопасный --slug, положительный --round и непустой --topic' };
|
|
87
134
|
}
|
|
135
|
+
if (input.envelope !== undefined) {
|
|
136
|
+
const v = validateExperimentEnvelope(input.envelope);
|
|
137
|
+
if (!v.ok) return { ok: false, exit: 2, reason: `--envelope invalid — ${v.reason}` };
|
|
138
|
+
}
|
|
88
139
|
const runOwner = input.ownerKind === 'run';
|
|
89
140
|
if (!Number.isFinite(Date.parse(input.startedAt)) || !Number.isInteger(input.ownerPid)
|
|
90
141
|
|| (runOwner ? input.ownerPid !== 0 || !nonEmpty(input.ownerRun) : input.ownerPid < 1)
|
|
@@ -101,6 +152,7 @@ export function openRound(input: {
|
|
|
101
152
|
...(runOwner ? { ownerRun: input.ownerRun!.trim() } : {}),
|
|
102
153
|
...(nonEmpty(input.run) ? { run: input.run.trim() } : {}),
|
|
103
154
|
recalled: [...input.recalled],
|
|
155
|
+
...(input.envelope !== undefined ? { envelope: input.envelope } : {}),
|
|
104
156
|
};
|
|
105
157
|
let existingOwnerAlive = input.existingOwnerAlive;
|
|
106
158
|
if (input.existing?.ownerKind === 'run' && input.force) {
|
|
@@ -138,13 +190,26 @@ export function closeRound(input: {
|
|
|
138
190
|
readonly closedAt: string;
|
|
139
191
|
readonly knownLessonIds: readonly string[];
|
|
140
192
|
readonly stateId?: string | undefined;
|
|
193
|
+
/** measurement-integrity FR-7: `A`, `A-`, `B+`, … — mandatory for `shipped|refuted`, a
|
|
194
|
+
* warned-and-dropped no-op for `blocked|abandoned`. */
|
|
195
|
+
readonly grade?: string | undefined;
|
|
196
|
+
/** measurement-integrity FR-7: the qe-bridge signoff for this slug, read by the CALLER. */
|
|
197
|
+
readonly reviewSidecar?: RoundReviewSidecar | undefined;
|
|
141
198
|
}, io: {
|
|
142
199
|
readonly writeLedger: (row: RoundLedgerRow) => unknown;
|
|
143
200
|
readonly readLedgerTail: () => string;
|
|
144
|
-
}): { readonly ok: true; readonly row: RoundLedgerRow; readonly marker: string } | RoundRefusal {
|
|
201
|
+
}): { readonly ok: true; readonly row: RoundLedgerRow; readonly marker: string; readonly warnings: readonly string[] } | RoundRefusal {
|
|
145
202
|
if (!(ROUND_OUTCOMES as readonly string[]).includes(input.outcome)) {
|
|
146
203
|
return { ok: false, exit: 2, reason: '--outcome: shipped | refuted | blocked | abandoned' };
|
|
147
204
|
}
|
|
205
|
+
// fix-round-1/F6: `openRound` validated the envelope once, at open time, then trusted it verbatim
|
|
206
|
+
// out of the state FILE from then on. A state file is mutable disk state between open and close —
|
|
207
|
+
// corrupted or hand-edited in that window, it would ride an invalid/tampered envelope straight
|
|
208
|
+
// into the ledger row. Re-validating here, right before the row is built, closes that window.
|
|
209
|
+
if (input.state.envelope !== undefined) {
|
|
210
|
+
const v = validateExperimentEnvelope(input.state.envelope);
|
|
211
|
+
if (!v.ok) return { ok: false, exit: 2, reason: `envelope in round state invalid — ${v.reason}` };
|
|
212
|
+
}
|
|
148
213
|
if (!validCount(input.tokens) || !validCount(input.agents)) {
|
|
149
214
|
return { ok: false, exit: 2, reason: '--tokens и --agents должны быть целыми числами не меньше нуля' };
|
|
150
215
|
}
|
|
@@ -160,6 +225,58 @@ export function closeRound(input: {
|
|
|
160
225
|
const missing = lessons.find((id) => !/^teach:[a-z0-9]+$/i.test(id) || !known.has(id));
|
|
161
226
|
if (missing !== undefined) return { ok: false, exit: 1, reason: `урок не найден: ${missing}` };
|
|
162
227
|
|
|
228
|
+
// measurement-integrity FR-7 (ADR-001 D5): grade is mandatory for a FINISHED review
|
|
229
|
+
// (shipped|refuted), a warned-and-dropped no-op for one that never finished (blocked|abandoned —
|
|
230
|
+
// "Отвергнуто: --grade всегда обязателен — заблокированный круг оценки не имеет"). Placed AFTER
|
|
231
|
+
// the outcome/envelope/tokens/lesson checks above (unchanged ordering, unchanged refusal reasons
|
|
232
|
+
// for those) and BEFORE the duration/row-build below.
|
|
233
|
+
const outcome = input.outcome as RoundOutcome;
|
|
234
|
+
const finishedOutcome = outcome === 'shipped' || outcome === 'refuted';
|
|
235
|
+
const gradeFlagRaw = nonEmpty(input.grade) ? input.grade.trim() : null;
|
|
236
|
+
if (gradeFlagRaw !== null && !/^[A-F][+-]?$/.test(gradeFlagRaw)) {
|
|
237
|
+
return { ok: false, exit: 2, reason: `--grade "${gradeFlagRaw}" не распознан — ожидается вид A|A-|B+|C…F` };
|
|
238
|
+
}
|
|
239
|
+
const warnings: string[] = [];
|
|
240
|
+
// measurement-integrity fix-round-1/F10 (Codex r1 MEDIUM #10): for an UNFINISHED outcome, --grade
|
|
241
|
+
// is dropped-with-warning HERE, BEFORE it is ever compared against the sidecar. The OLD ordering
|
|
242
|
+
// ran the conflict check first — so `--outcome blocked --grade B` against a STALE sidecar grade
|
|
243
|
+
// `A` refused outright instead of the promised warning-and-drop, because a value that was about to
|
|
244
|
+
// be discarded still had to survive a conflict check on its way to being discarded. `gradeFlag` is
|
|
245
|
+
// `null` for an unfinished outcome from this point on, exactly as if `--grade` had never been
|
|
246
|
+
// passed — the conflict check below therefore never sees it.
|
|
247
|
+
const gradeFlag = finishedOutcome ? gradeFlagRaw : null;
|
|
248
|
+
if (!finishedOutcome && gradeFlagRaw !== null) {
|
|
249
|
+
warnings.push(`--grade "${gradeFlagRaw}" проигнорирован: outcome=${outcome} не является завершённым ревью, оценка не пишется`);
|
|
250
|
+
}
|
|
251
|
+
const sidecarGrade = input.reviewSidecar !== undefined && nonEmpty(input.reviewSidecar.grade ?? undefined)
|
|
252
|
+
? (input.reviewSidecar.grade as string).trim()
|
|
253
|
+
: null;
|
|
254
|
+
if (gradeFlag !== null && sidecarGrade !== null && gradeFlag !== sidecarGrade) {
|
|
255
|
+
return {
|
|
256
|
+
ok: false,
|
|
257
|
+
exit: 2,
|
|
258
|
+
reason: `--grade "${gradeFlag}" конфликтует с оценкой сайдкара qe-bridge "${sidecarGrade}" для slug ${input.state.slug}`,
|
|
259
|
+
};
|
|
260
|
+
}
|
|
261
|
+
if (finishedOutcome && gradeFlag === null) {
|
|
262
|
+
return { ok: false, exit: 2, reason: `grade required for a finished review: outcome=${outcome} требует --grade <A|A-|B+|…>` };
|
|
263
|
+
}
|
|
264
|
+
const gradeForRow = gradeFlag;
|
|
265
|
+
|
|
266
|
+
// reviewer/reviewMinutes/reviewSource: an explicit --reviewer always wins; the sidecar fills the
|
|
267
|
+
// gap only, and only its OWN two derived fields travel with it (a flag-supplied reviewer never
|
|
268
|
+
// carries a sidecar-sourced `reviewMinutes`/`reviewSource` — that would misattribute where the
|
|
269
|
+
// minutes figure came from).
|
|
270
|
+
let reviewer = nonEmpty(input.reviewer) ? input.reviewer.trim() : null;
|
|
271
|
+
let reviewMinutes: number | null = null;
|
|
272
|
+
let reviewSource: 'qe-bridge' | null = null;
|
|
273
|
+
if (reviewer === null && input.reviewSidecar !== undefined && nonEmpty(input.reviewSidecar.gradedBy)) {
|
|
274
|
+
reviewer = input.reviewSidecar.gradedBy.trim();
|
|
275
|
+
const elapsedMs = input.reviewSidecar.elapsedMs;
|
|
276
|
+
reviewMinutes = Number.isFinite(elapsedMs) && elapsedMs >= 0 ? Math.round((elapsedMs / 60_000) * 10) / 10 : null;
|
|
277
|
+
reviewSource = 'qe-bridge';
|
|
278
|
+
}
|
|
279
|
+
|
|
163
280
|
const startedMs = Date.parse(input.state.startedAt);
|
|
164
281
|
const closedMs = Date.parse(input.closedAt);
|
|
165
282
|
if (!Number.isFinite(startedMs) || !Number.isFinite(closedMs) || closedMs < startedMs) {
|
|
@@ -173,13 +290,13 @@ export function closeRound(input: {
|
|
|
173
290
|
stage: 'round',
|
|
174
291
|
tier: null,
|
|
175
292
|
coder: nonEmpty(input.coder) ? input.coder.trim() : null,
|
|
176
|
-
reviewer
|
|
293
|
+
reviewer,
|
|
177
294
|
lead: null,
|
|
178
295
|
minutes: input.noCost === true ? null : Math.floor((closedMs - startedMs) / 60_000),
|
|
179
296
|
agents: input.noCost === true ? null : input.agents ?? null,
|
|
180
297
|
tokens: input.noCost === true ? null : input.tokens ?? null,
|
|
181
|
-
grade:
|
|
182
|
-
outcome
|
|
298
|
+
grade: gradeForRow,
|
|
299
|
+
outcome,
|
|
183
300
|
reason: nonEmpty(input.reason) ? input.reason.trim() : null,
|
|
184
301
|
round: input.state.round,
|
|
185
302
|
lessons,
|
|
@@ -188,6 +305,9 @@ export function closeRound(input: {
|
|
|
188
305
|
date: null,
|
|
189
306
|
...(input.noCost === true ? { costIn: 'stages' as const } : {}),
|
|
190
307
|
...(nonEmpty(input.stateId) ? { stateId: input.stateId } : {}),
|
|
308
|
+
...(input.state.envelope !== undefined ? { envelope: input.state.envelope } : {}),
|
|
309
|
+
...(reviewSource !== null ? { reviewSource } : {}),
|
|
310
|
+
...(reviewMinutes !== null ? { reviewMinutes } : {}),
|
|
191
311
|
};
|
|
192
312
|
|
|
193
313
|
try {
|
|
@@ -200,7 +320,46 @@ export function closeRound(input: {
|
|
|
200
320
|
if (!tail.includes(marker)) {
|
|
201
321
|
return { ok: false, exit: 1, reason: 'строка не найдена — круг НЕ закрыт' };
|
|
202
322
|
}
|
|
203
|
-
return { ok: true, row, marker };
|
|
323
|
+
return { ok: true, row, marker, warnings };
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
/**
|
|
327
|
+
* measurement-integrity fix-round-1/F8 (Codex r1 CRITICAL #8): validate an ALREADY-WRITTEN ledger
|
|
328
|
+
* row against the CURRENT schema's `shipped|refuted require a grade` rule (ADR-001 D5 / FR-7).
|
|
329
|
+
*
|
|
330
|
+
* This exists for exactly one caller: `dz round close`'s idempotent-retry path. When the predicted
|
|
331
|
+
* marker (or `stateId`) is already found in the ledger tail, the CLI used to skip `closeRound`
|
|
332
|
+
* ENTIRELY and delete the round's state file — so an OLD row written before FR-7 shipped (`shipped`
|
|
333
|
+
* with `grade: null`, the exact defect this feature exists to close) could be "recognised as already
|
|
334
|
+
* closed" and the state removed without ever being checked against the rule that is supposed to be
|
|
335
|
+
* mandatory. This function is that missing check, run on the ALREADY-FOUND row before the CLI is
|
|
336
|
+
* allowed to treat the retry as a success.
|
|
337
|
+
*
|
|
338
|
+
* Deliberately NARROW: it re-checks only the ONE FR-7 invariant (a schema rule with a proving test),
|
|
339
|
+
* not every field `closeRound` validates on the FIRST write (grade format, envelope shape, …) — those
|
|
340
|
+
* were already enforced when the row was ORIGINALLY written; re-validating them here would either
|
|
341
|
+
* duplicate that logic or silently drift from it. Pure, never throws.
|
|
342
|
+
*/
|
|
343
|
+
export function validateClosedRoundLedgerRow(row: unknown): { readonly ok: true } | { readonly ok: false; readonly reason: string } {
|
|
344
|
+
if (typeof row !== 'object' || row === null || Array.isArray(row)) {
|
|
345
|
+
return { ok: false, reason: 'найденная строка леджера не JSON-объект — не удаётся проверить оценку' };
|
|
346
|
+
}
|
|
347
|
+
const r = row as Record<string, unknown>;
|
|
348
|
+
const outcome = r['outcome'];
|
|
349
|
+
if (typeof outcome !== 'string' || !(ROUND_OUTCOMES as readonly string[]).includes(outcome)) {
|
|
350
|
+
return { ok: false, reason: `найденная строка леджера несёт неизвестный outcome ${JSON.stringify(outcome)}` };
|
|
351
|
+
}
|
|
352
|
+
const finishedOutcome = outcome === 'shipped' || outcome === 'refuted';
|
|
353
|
+
if (!finishedOutcome) return { ok: true }; // blocked|abandoned carry no grade requirement (FR-7)
|
|
354
|
+
const grade = r['grade'];
|
|
355
|
+
if (typeof grade !== 'string' || !/^[A-F][+-]?$/.test(grade)) {
|
|
356
|
+
return {
|
|
357
|
+
ok: false,
|
|
358
|
+
reason: `найденная строка леджера — outcome=${outcome} без валидной оценки (grade=${JSON.stringify(grade)}); ` +
|
|
359
|
+
'закрытие отказано (measurement-integrity ADR-001 D5 / FR-7 требует непустую оценку для завершённого ревью)',
|
|
360
|
+
};
|
|
361
|
+
}
|
|
362
|
+
return { ok: true };
|
|
204
363
|
}
|
|
205
364
|
|
|
206
365
|
export function listRounds(states: readonly RoundState[], input: {
|
package/src/run-records.ts
CHANGED
|
@@ -13,7 +13,133 @@
|
|
|
13
13
|
* Pure: payload in, verdict out. The CLI owns paths, the append, the read-back and the exit code.
|
|
14
14
|
*/
|
|
15
15
|
|
|
16
|
+
import { matchCodexRollouts } from './codex-rollouts.js';
|
|
17
|
+
import type { CodexRollout } from './codex-rollouts.js';
|
|
16
18
|
import { redactTrainingPayload } from './feature-adr-checkpoints.js';
|
|
19
|
+
import { validateExperimentEnvelope } from './feature-adr-envelope.js';
|
|
20
|
+
|
|
21
|
+
/** Structural — a caller passes `cost-scoring.ts`'s `ModelPricing`; kept local so `run-records.ts`
|
|
22
|
+
* does not have to import `cost-scoring.ts` just to name a type.
|
|
23
|
+
*
|
|
24
|
+
* measurement-integrity fix-round-1/F6 (Codex r1 HIGH #6): `cacheCreation` used to be dropped here —
|
|
25
|
+
* the snapshot silently lacked the ONE rate a cache-WRITE-heavy row needs to reproduce its own cost
|
|
26
|
+
* later, even though the caller's own `ModelPricing` carries it. Now carried through verbatim. */
|
|
27
|
+
export interface LedgerPriceEntry {
|
|
28
|
+
readonly prompt: number;
|
|
29
|
+
readonly completion: number;
|
|
30
|
+
readonly cachedInput: number;
|
|
31
|
+
readonly cacheCreation: number;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** measurement-integrity fix-round-1/F4 (Codex r1 HIGH #4): a resolvable executor spec, split into
|
|
35
|
+
* its three parts. Never invents a model: {@link parseModelSpec} returns `null` for anything it
|
|
36
|
+
* cannot resolve to exactly one model, rather than guessing. */
|
|
37
|
+
export interface ParsedModelSpec {
|
|
38
|
+
readonly family: 'claude' | 'codex';
|
|
39
|
+
readonly model: string;
|
|
40
|
+
readonly effort: string | null;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/** Bare Claude model names the pipeline actually emits with no `claude:` prefix (`coder: 'sonnet'`,
|
|
44
|
+
* `'opus'`, `'fable'`, real values observed in `.dz/feature-adr/run-cost-ledger.jsonl`). */
|
|
45
|
+
const BARE_CLAUDE_NAMES = new Set(['sonnet', 'opus', 'haiku', 'fable']);
|
|
46
|
+
/** id/effort alphabet a model spec component may use (lead delta after Codex r2, new MEDIUM #3). */
|
|
47
|
+
const SPEC_PART = /^[A-Za-z0-9._-]+$/;
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Parse an executor spec — the shapes actually recorded in `coder`/`reviewer` fields
|
|
51
|
+
* (`codex:gpt-5.6-sol:high`, `claude:sonnet`, bare `sonnet`/`opus`, bare `codex`, bare `claude`) —
|
|
52
|
+
* into `{family, model, effort}`. Pure, never throws.
|
|
53
|
+
*
|
|
54
|
+
* Returns `null` (never a guess) for anything that cannot be resolved to exactly ONE model:
|
|
55
|
+
* - a bare `'codex'` or `'claude'` (family named, no model at all);
|
|
56
|
+
* - an annotated/aggregate field such as `'claude:sonnet x2'` or `'qe-bridge:claude x2 + lead'` (real
|
|
57
|
+
* values this ledger carries for a MULTI-reviewer round) — any embedded whitespace means the field
|
|
58
|
+
* names more than one resolvable spec, and picking one would misattribute to the others;
|
|
59
|
+
* - a bare model id with no family marker that is not one of the known bare Claude names (e.g. a full
|
|
60
|
+
* `'claude-sonnet-5'` — that shape is handled by the OLDER vendor-prefix path in {@link priceLookup}
|
|
61
|
+
* for backward compatibility, not by this parser).
|
|
62
|
+
*/
|
|
63
|
+
export function parseModelSpec(spec: unknown): ParsedModelSpec | null {
|
|
64
|
+
if (typeof spec !== 'string') return null;
|
|
65
|
+
const trimmed = spec.trim();
|
|
66
|
+
if (trimmed === '' || /\s/.test(trimmed)) return null;
|
|
67
|
+
const parts = trimmed.split(':');
|
|
68
|
+
const head = parts[0] ?? '';
|
|
69
|
+
if (head === 'codex') {
|
|
70
|
+
const model = parts[1];
|
|
71
|
+
if (model === undefined || model === '') return null; // bare 'codex' — no reliable model
|
|
72
|
+
// Lead delta after Codex r2 (new MEDIUM #3): exactly 2 or 3 non-empty components, id alphabet
|
|
73
|
+
// only — `codex:gpt-5.6-sol:high:garbage` is a corrupt/aggregate spec, never a reliable model.
|
|
74
|
+
if (parts.length > 3 || !SPEC_PART.test(model) || (parts.length === 3 && (parts[2] === '' || !SPEC_PART.test(parts[2]!)))) return null;
|
|
75
|
+
const effort = parts[2] ?? null;
|
|
76
|
+
return { family: 'codex', model, effort: effort === '' ? null : effort };
|
|
77
|
+
}
|
|
78
|
+
if (head === 'claude') {
|
|
79
|
+
const model = parts[1];
|
|
80
|
+
if (model === undefined || model === '') return null; // bare 'claude' — no reliable model
|
|
81
|
+
return { family: 'claude', model, effort: null };
|
|
82
|
+
}
|
|
83
|
+
if (parts.length === 1 && BARE_CLAUDE_NAMES.has(head)) {
|
|
84
|
+
return { family: 'claude', model: head, effort: null };
|
|
85
|
+
}
|
|
86
|
+
return null;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/** measurement-integrity FR-5/FR-6: enrichment the WRITER supplies at write time — the rollout logs
|
|
90
|
+
* it already read (I/O lives in the CLI; this stays pure) and the price table snapshot. Absent
|
|
91
|
+
* entirely ⇒ zero behavior change from before this feature (NFR-1). */
|
|
92
|
+
export interface LedgerEnrichInput {
|
|
93
|
+
/** Parsed Codex rollout logs for the window the CLI read — usually every rollout from the days the
|
|
94
|
+
* window spans. Pure data; the CLI is the one that walked `~/.codex/sessions`. */
|
|
95
|
+
readonly rollouts?: readonly CodexRollout[];
|
|
96
|
+
/** The stage's own time window — usually [the previous ledger row's `ts`, this write's `ts`], or
|
|
97
|
+
* an explicit `--window-from/--window-to`. Omitted ⇒ no rollout match is even attempted. */
|
|
98
|
+
readonly window?: { readonly from: string; readonly to: string };
|
|
99
|
+
/** Narrows an otherwise-ambiguous match — usually the repo root the stage ran in. */
|
|
100
|
+
readonly cwd?: string;
|
|
101
|
+
/** A model-pricing table SNAPSHOT (FR-6) — the CALLER's table, captured at write time, never the
|
|
102
|
+
* ledger's own idea of "current" pricing (ADR-001 D4 rejects re-pricing after the fact). */
|
|
103
|
+
readonly prices?: Readonly<Record<string, LedgerPriceEntry>>;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
function isCodexFamily(v: unknown): boolean {
|
|
107
|
+
return typeof v === 'string' && /codex/i.test(v);
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
function isRecord(v: unknown): v is Record<string, unknown> {
|
|
111
|
+
return typeof v === 'object' && v !== null && !Array.isArray(v);
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/** Longest-prefix match against the CALLER'S OWN table (never `cost-scoring.ts`'s internal
|
|
115
|
+
* constant) — the whole point of a price SNAPSHOT is that it answers only from what the caller
|
|
116
|
+
* handed in at write time, mirroring `pricingFor`'s matching rule without importing it.
|
|
117
|
+
*
|
|
118
|
+
* measurement-integrity fix-round-1/F7 (Codex r1 HIGH #7): the OLD version stripped only a
|
|
119
|
+
* `vendor/`-shaped prefix, so the REAL recorded shape `codex:gpt-5.6-sol:high` (or `claude:sonnet`)
|
|
120
|
+
* never matched anything and always landed in `prices.unknown[]`. This now runs the id through the
|
|
121
|
+
* SAME {@link parseModelSpec} FR-4's matcher uses, then reconstructs the normalized id the way
|
|
122
|
+
* `cost-scoring.ts`'s `pricingFor`/`hasKnownPricing` key their table (`claude-<model>` for the
|
|
123
|
+
* Claude family; the bare model for Codex — its ids carry no vendor prefix). A spec this parser
|
|
124
|
+
* cannot resolve falls back to the OLD vendor-prefix strip, so an already-working bare id
|
|
125
|
+
* (`claude-sonnet-5`, `gpt-4o`) keeps matching exactly as before (NFR-1). */
|
|
126
|
+
function priceLookup(modelId: string, table: Readonly<Record<string, LedgerPriceEntry>>): LedgerPriceEntry | null {
|
|
127
|
+
if (typeof modelId !== 'string' || modelId.length === 0) return null;
|
|
128
|
+
const parsed = parseModelSpec(modelId);
|
|
129
|
+
const id = parsed !== null
|
|
130
|
+
? (parsed.family === 'claude' ? `claude-${parsed.model}` : parsed.model).toLowerCase()
|
|
131
|
+
: modelId.toLowerCase().replace(/^[a-z0-9-]+\//, '');
|
|
132
|
+
let best: LedgerPriceEntry | null = null;
|
|
133
|
+
let bestLen = 0;
|
|
134
|
+
for (const [key, price] of Object.entries(table)) {
|
|
135
|
+
const k = key.toLowerCase();
|
|
136
|
+
if (id.startsWith(k) && k.length > bestLen) {
|
|
137
|
+
best = price;
|
|
138
|
+
bestLen = k.length;
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
return best;
|
|
142
|
+
}
|
|
17
143
|
|
|
18
144
|
export type RecordKind = 'ledger' | 'training-pair';
|
|
19
145
|
|
|
@@ -65,7 +191,7 @@ const noop = (verdict: 'duplicate' | 'skipped', reason: string): RecordDecision
|
|
|
65
191
|
line: null,
|
|
66
192
|
});
|
|
67
193
|
|
|
68
|
-
function shapeMismatch(kind: RecordKind, payload: Record<string, unknown
|
|
194
|
+
function shapeMismatch(kind: RecordKind, payload: Record<string, unknown>, autoFlag: boolean): string | null {
|
|
69
195
|
// AM-1 FIRST, before the required-field sweep. A ledger row offered as a training pair fails BOTH
|
|
70
196
|
// checks, and the wrong-kind reason is the one that tells the caller what actually happened —
|
|
71
197
|
// "missing field `output`" sends them looking for a field they never meant to send.
|
|
@@ -78,6 +204,39 @@ function shapeMismatch(kind: RecordKind, payload: Record<string, unknown>): stri
|
|
|
78
204
|
if (kind === 'training-pair' && 'tokens' in payload && !('input' in payload)) {
|
|
79
205
|
return 'this payload looks like a ledger row (`tokens` without `input`), not a training pair';
|
|
80
206
|
}
|
|
207
|
+
// experiment-envelope FR-5 / ADR-001 D2: `auto:true` marks a pipeline-written row, and the pipeline
|
|
208
|
+
// is obligated to carry the envelope built once after the Step-0 router — a row missing it is
|
|
209
|
+
// useless for learning and would silently corrupt the sample (the "warn but write" alternative was
|
|
210
|
+
// rejected in the ADR: a warning nobody reads left `tokens=null` unnoticed for years). A MANUAL row
|
|
211
|
+
// (no `auto`) stays compatible: no envelope required, but one that IS present is still validated —
|
|
212
|
+
// never trusted just because a human typed it.
|
|
213
|
+
//
|
|
214
|
+
// fix-round-1/F2 (cross-family review, HIGH #2): the PAYLOAD's own `auto` field used to be the
|
|
215
|
+
// ONLY signal — an automatic producer that forgot it, or sent `"true"`/`false`, was silently
|
|
216
|
+
// accepted as a manual row and skipped the envelope requirement entirely. `autoFlag` is the CLI's
|
|
217
|
+
// own trusted `--auto` argument (never JSON a caller could typo): it is a SECOND, independent
|
|
218
|
+
// trust source, unioned with the payload field rather than replacing it — nothing that used to be
|
|
219
|
+
// gated stops being gated, and a `--auto`-dispatched caller is now gated even if its hand-built
|
|
220
|
+
// payload forgot the field. Independently of either source, a PRESENT `auto` field is checked for
|
|
221
|
+
// shape: only the literal `true` is a legal value — anything else (a string, `false`, a number) is
|
|
222
|
+
// refused outright, because a field whose only sane value is `true` holding something else is a
|
|
223
|
+
// caller bug worth surfacing, not silently downgrading to "manual".
|
|
224
|
+
if (kind === 'ledger') {
|
|
225
|
+
const rawAuto = payload['auto'];
|
|
226
|
+
if (rawAuto !== undefined && rawAuto !== true) {
|
|
227
|
+
return 'a ledger record\'s `auto` field must be `true` or absent';
|
|
228
|
+
}
|
|
229
|
+
const envelope = payload['envelope'];
|
|
230
|
+
const envelopePresent = envelope !== undefined && envelope !== null;
|
|
231
|
+
const isAuto = autoFlag === true || rawAuto === true;
|
|
232
|
+
if (isAuto && !envelopePresent) {
|
|
233
|
+
return 'an auto ledger row must carry `envelope` (experiment-envelope FR-5)';
|
|
234
|
+
}
|
|
235
|
+
if (envelopePresent) {
|
|
236
|
+
const v = validateExperimentEnvelope(envelope);
|
|
237
|
+
if (!v.ok) return `envelope invalid — ${v.reason}`;
|
|
238
|
+
}
|
|
239
|
+
}
|
|
81
240
|
const required = kind === 'ledger' ? LEDGER_REQUIRED : PAIR_REQUIRED;
|
|
82
241
|
for (const field of required) {
|
|
83
242
|
const v = payload[field];
|
|
@@ -125,7 +284,12 @@ export function decideRecordWrite(input: {
|
|
|
125
284
|
/** Who ran it. Supplied by the CALLER, which lives outside the workflow sandbox and can see the
|
|
126
285
|
* host; absent stays absent (see the stamping comment below). */
|
|
127
286
|
runnerId?: string | null;
|
|
287
|
+
/** fix-round-1/F2: the CLI's own trusted `--auto` flag — see the comment inside `shapeMismatch`. */
|
|
288
|
+
auto?: boolean;
|
|
128
289
|
maxChars?: number;
|
|
290
|
+
/** measurement-integrity FR-5/FR-6: rollout-log + price enrichment for a ledger row. Absent ⇒ zero
|
|
291
|
+
* behavior change (NFR-1). */
|
|
292
|
+
enrich?: LedgerEnrichInput;
|
|
129
293
|
}): RecordDecision {
|
|
130
294
|
const { kind, payloadRaw, stage } = input;
|
|
131
295
|
if (kind !== 'ledger' && kind !== 'training-pair') {
|
|
@@ -153,7 +317,7 @@ export function decideRecordWrite(input: {
|
|
|
153
317
|
// the read-back, the refusal texts — ever sees a byte of the profile block.
|
|
154
318
|
const obj = (kind === 'training-pair' ? redactTrainingPayload(payload) : payload) as Record<string, unknown>;
|
|
155
319
|
|
|
156
|
-
const mismatch = shapeMismatch(kind, obj);
|
|
320
|
+
const mismatch = shapeMismatch(kind, obj, input.auto === true);
|
|
157
321
|
if (mismatch !== null) return refuse(mismatch);
|
|
158
322
|
|
|
159
323
|
// The record is filed under `--stage`, and the payload carries its own. A disagreement means the
|
|
@@ -182,6 +346,11 @@ export function decideRecordWrite(input: {
|
|
|
182
346
|
// inside an already-serialised document — text surgery on a structured value, and the exact place
|
|
183
347
|
// a payload containing that literal token could corrupt itself.
|
|
184
348
|
const stamped: Record<string, unknown> = { ...obj };
|
|
349
|
+
// fix-round-1/F2: the CLI's own `--auto` flag is authoritative — when set, the written row MUST
|
|
350
|
+
// carry `auto:true` too (not just gate on it transiently), so every downstream reader of the
|
|
351
|
+
// PERSISTED line keeps seeing the same signal `shapeMismatch` already gated on above. A no-op when
|
|
352
|
+
// the payload already said `auto:true` (shapeMismatch already refused any OTHER value).
|
|
353
|
+
if (kind === 'ledger' && input.auto === true) stamped['auto'] = true;
|
|
185
354
|
const isGap = (v: unknown): boolean => v === null || v === undefined || (typeof v === 'string' && v.trim() === '');
|
|
186
355
|
if (input.timestamp != null && input.timestamp !== '') {
|
|
187
356
|
// An EMPTY STRING is a gap, not a value. Stamping only over null/undefined let
|
|
@@ -273,6 +442,117 @@ export function decideRecordWrite(input: {
|
|
|
273
442
|
}
|
|
274
443
|
}
|
|
275
444
|
|
|
445
|
+
// measurement-integrity FR-5 (rollout enrichment) + FR-6 (price snapshot). Both are ADDITIVE and
|
|
446
|
+
// OPT-IN on `input.enrich` — a caller that never passes it gets byte-identical output to before
|
|
447
|
+
// this feature (NFR-1).
|
|
448
|
+
if (kind === 'ledger' && input.enrich !== undefined) {
|
|
449
|
+
const enrich = input.enrich;
|
|
450
|
+
|
|
451
|
+
// FR-5: only a codex-family coder/reviewer with `tokens: null` is a candidate — a Claude row, or
|
|
452
|
+
// one that already has a token figure, is left untouched. The loose `/codex/i` check below only
|
|
453
|
+
// decides whether this row is WORTH TRYING at all.
|
|
454
|
+
const tokensIsNull = stamped['tokens'] === null;
|
|
455
|
+
const looksCodexFamily = isCodexFamily(stamped['coder']) || isCodexFamily(stamped['reviewer']);
|
|
456
|
+
if (tokensIsNull && looksCodexFamily) {
|
|
457
|
+
// measurement-integrity fix-round-1/F4 (Codex r1 HIGH #4): the matcher REQUIRES a reliable
|
|
458
|
+
// model, parsed the same way FR-7's price lookup parses one — never `/codex/i` alone. If
|
|
459
|
+
// `coder`/`reviewer` do not resolve to exactly ONE codex model between them (a bare `'codex'`
|
|
460
|
+
// with no model at all, or the two fields naming DIFFERENT codex models), the matcher is never
|
|
461
|
+
// even called with an unreliable/omitted model filter — a lone rollout in the window would
|
|
462
|
+
// otherwise be accepted as `'one'` on time+cwd alone and its tokens misattributed to the wrong
|
|
463
|
+
// model's stage.
|
|
464
|
+
const codexModels = new Set(
|
|
465
|
+
[parseModelSpec(stamped['coder']), parseModelSpec(stamped['reviewer'])]
|
|
466
|
+
.filter((s): s is ParsedModelSpec => s !== null && s.family === 'codex')
|
|
467
|
+
.map((s) => s.model),
|
|
468
|
+
);
|
|
469
|
+
if (codexModels.size !== 1) {
|
|
470
|
+
stamped['tokensSource'] = 'codex-rollout:no-model';
|
|
471
|
+
} else if (enrich.window !== undefined) {
|
|
472
|
+
const model = [...codexModels][0]!;
|
|
473
|
+
const match = matchCodexRollouts(enrich.rollouts ?? [], {
|
|
474
|
+
from: enrich.window.from,
|
|
475
|
+
to: enrich.window.to,
|
|
476
|
+
model,
|
|
477
|
+
...(enrich.cwd !== undefined ? { cwd: enrich.cwd } : {}),
|
|
478
|
+
});
|
|
479
|
+
if (match.status === 'one') {
|
|
480
|
+
stamped['tokens'] = match.rollout.totals.total;
|
|
481
|
+
const startMs = match.rollout.startedAt !== null ? Date.parse(match.rollout.startedAt) : NaN;
|
|
482
|
+
const endMs = match.rollout.endedAt !== null ? Date.parse(match.rollout.endedAt) : NaN;
|
|
483
|
+
// measurement-integrity fix-round-1/F6 (Codex r1 HIGH #6): fill-ONLY-null — an existing
|
|
484
|
+
// `minutes` figure (a manually recorded one, say) must never be silently overwritten by a
|
|
485
|
+
// derived rollout duration.
|
|
486
|
+
if (stamped['minutes'] === null && Number.isFinite(startMs) && Number.isFinite(endMs) && endMs >= startMs) {
|
|
487
|
+
stamped['minutes'] = Math.round(((endMs - startMs) / 60000) * 10) / 10;
|
|
488
|
+
}
|
|
489
|
+
stamped['tokensSource'] = 'codex-rollout';
|
|
490
|
+
stamped['rolloutId'] = match.rollout.id;
|
|
491
|
+
} else {
|
|
492
|
+
// `none` or `ambiguous` — NFR-3: an explicit status, never a guessed number.
|
|
493
|
+
stamped['tokensSource'] = `codex-rollout:${match.status}`;
|
|
494
|
+
}
|
|
495
|
+
} else {
|
|
496
|
+
// Eligible in principle (codex family, one reliable model, tokens null) but no window was
|
|
497
|
+
// supplied — AC-4's "old row without a window": no enrichment is even attempted, and that
|
|
498
|
+
// fact is itself recorded rather than left silently absent.
|
|
499
|
+
stamped['tokensSource'] = 'unavailable';
|
|
500
|
+
}
|
|
501
|
+
}
|
|
502
|
+
|
|
503
|
+
// FR-6: the price snapshot, for every model this row names — independent of the FR-5 branch
|
|
504
|
+
// above (a Claude row gets priced too; only tokens enrichment is codex-specific).
|
|
505
|
+
if (enrich.prices !== undefined) {
|
|
506
|
+
const modelIds = new Set<string>();
|
|
507
|
+
for (const v of [stamped['coder'], stamped['reviewer']]) {
|
|
508
|
+
if (typeof v === 'string' && v.trim() !== '') modelIds.add(v.trim());
|
|
509
|
+
}
|
|
510
|
+
const envelope = stamped['envelope'];
|
|
511
|
+
if (isRecord(envelope)) {
|
|
512
|
+
const chosen = envelope['chosen'];
|
|
513
|
+
if (isRecord(chosen)) {
|
|
514
|
+
const stages = chosen['stages'];
|
|
515
|
+
if (isRecord(stages)) {
|
|
516
|
+
for (const v of Object.values(stages)) {
|
|
517
|
+
if (typeof v === 'string' && v.trim() !== '') modelIds.add(v.trim());
|
|
518
|
+
}
|
|
519
|
+
}
|
|
520
|
+
}
|
|
521
|
+
}
|
|
522
|
+
const table: Record<string, LedgerPriceEntry> = {};
|
|
523
|
+
const unknown: string[] = [];
|
|
524
|
+
for (const modelId of modelIds) {
|
|
525
|
+
const price = priceLookup(modelId, enrich.prices);
|
|
526
|
+
if (price === null) unknown.push(modelId);
|
|
527
|
+
else table[modelId] = { prompt: price.prompt, completion: price.completion, cachedInput: price.cachedInput, cacheCreation: price.cacheCreation };
|
|
528
|
+
}
|
|
529
|
+
const computedPrices = {
|
|
530
|
+
snapshotAt: input.timestamp ?? null,
|
|
531
|
+
table,
|
|
532
|
+
...(unknown.length > 0 ? { unknown } : {}),
|
|
533
|
+
};
|
|
534
|
+
// measurement-integrity fix-round-1/F6 (Codex r1 HIGH #6): fill-ONLY-null for `prices` too — an
|
|
535
|
+
// existing snapshot (a previous write already priced this row) is never unconditionally
|
|
536
|
+
// replaced. Equal → left alone (idempotent re-enrichment, common on a retried write). Different
|
|
537
|
+
// → a named `pricesConflict`, never a silent re-price (ADR-001 D4 forbids re-pricing after the
|
|
538
|
+
// fact — a DIFFERING recomputation is exactly that, so it is surfaced, not applied).
|
|
539
|
+
const existingPrices = stamped['prices'];
|
|
540
|
+
if (existingPrices === undefined || existingPrices === null) {
|
|
541
|
+
stamped['prices'] = computedPrices;
|
|
542
|
+
} else if (isRecord(existingPrices) && isRecord(existingPrices['table']) && (existingPrices['unknown'] === undefined || Array.isArray(existingPrices['unknown']))) {
|
|
543
|
+
// Lead delta after Codex r2 (new MEDIUM #4): compare the WHOLE snapshot canonically (table +
|
|
544
|
+
// sorted unknown), not the table alone.
|
|
545
|
+
const canon = (t: unknown, u: unknown): string => JSON.stringify({ table: t, unknown: Array.isArray(u) ? [...u].map(String).sort() : [] });
|
|
546
|
+
if (canon(existingPrices['table'], existingPrices['unknown']) !== canon(table, unknown)) {
|
|
547
|
+
stamped['pricesConflict'] = { existing: existingPrices, recomputed: computedPrices };
|
|
548
|
+
}
|
|
549
|
+
} else {
|
|
550
|
+
// A malformed existing snapshot (no table / bad unknown) is a CONFLICT, never silently trusted.
|
|
551
|
+
stamped['pricesConflict'] = { existing: existingPrices, recomputed: computedPrices, reason: 'existing prices snapshot is malformed' };
|
|
552
|
+
}
|
|
553
|
+
}
|
|
554
|
+
}
|
|
555
|
+
|
|
276
556
|
let line: string;
|
|
277
557
|
try {
|
|
278
558
|
line = JSON.stringify(stamped);
|