@dzhechkov/harness-core 0.8.36 → 0.8.38
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +216 -76
- package/README.md +349 -8
- package/dist/agentdb-index.d.ts +87 -7
- package/dist/agentdb-index.d.ts.map +1 -1
- package/dist/agentdb-index.js +416 -57
- package/dist/agentdb-index.js.map +1 -1
- package/dist/apply-leg.d.ts +19 -1
- package/dist/apply-leg.d.ts.map +1 -1
- package/dist/apply-leg.js +187 -36
- package/dist/apply-leg.js.map +1 -1
- package/dist/codex-rollouts.d.ts +118 -0
- package/dist/codex-rollouts.d.ts.map +1 -0
- package/dist/codex-rollouts.js +297 -0
- package/dist/codex-rollouts.js.map +1 -0
- package/dist/cost-ledger.d.ts +56 -4
- package/dist/cost-ledger.d.ts.map +1 -1
- package/dist/cost-ledger.js +176 -20
- package/dist/cost-ledger.js.map +1 -1
- package/dist/cross-family-control.d.ts +380 -0
- package/dist/cross-family-control.d.ts.map +1 -0
- package/dist/cross-family-control.js +848 -0
- package/dist/cross-family-control.js.map +1 -0
- package/dist/debt-ratchet.d.ts +53 -0
- package/dist/debt-ratchet.d.ts.map +1 -0
- package/dist/debt-ratchet.js +107 -0
- package/dist/debt-ratchet.js.map +1 -0
- package/dist/embedding-config.d.ts +42 -0
- package/dist/embedding-config.d.ts.map +1 -1
- package/dist/embedding-config.js +106 -10
- package/dist/embedding-config.js.map +1 -1
- package/dist/feature-adr-checkpoints.d.ts +6 -0
- package/dist/feature-adr-checkpoints.d.ts.map +1 -1
- package/dist/feature-adr-checkpoints.js +29 -0
- package/dist/feature-adr-checkpoints.js.map +1 -1
- package/dist/feature-adr-decision-recall.d.ts +2 -2
- package/dist/feature-adr-decision-recall.d.ts.map +1 -1
- package/dist/feature-adr-decision-recall.js +5 -3
- package/dist/feature-adr-decision-recall.js.map +1 -1
- package/dist/feature-adr-envelope.d.ts +96 -0
- package/dist/feature-adr-envelope.d.ts.map +1 -0
- package/dist/feature-adr-envelope.js +183 -0
- package/dist/feature-adr-envelope.js.map +1 -0
- package/dist/feature-adr-routing.d.ts +64 -0
- package/dist/feature-adr-routing.d.ts.map +1 -1
- package/dist/feature-adr-routing.js +133 -3
- package/dist/feature-adr-routing.js.map +1 -1
- package/dist/feature-adr-stage-canon.d.ts +79 -0
- package/dist/feature-adr-stage-canon.d.ts.map +1 -0
- package/dist/feature-adr-stage-canon.js +117 -0
- package/dist/feature-adr-stage-canon.js.map +1 -0
- package/dist/index.d.ts +21 -9
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +15 -5
- package/dist/index.js.map +1 -1
- package/dist/loop-blobs.generated.js +4 -4
- package/dist/loop-blobs.generated.js.map +1 -1
- package/dist/mutation-gate.d.ts +51 -0
- package/dist/mutation-gate.d.ts.map +1 -1
- package/dist/mutation-gate.js +295 -0
- package/dist/mutation-gate.js.map +1 -1
- package/dist/qe-bridge.d.ts +8 -0
- package/dist/qe-bridge.d.ts.map +1 -1
- package/dist/qe-bridge.js +4 -2
- package/dist/qe-bridge.js.map +1 -1
- package/dist/qe-findings.d.ts +107 -0
- package/dist/qe-findings.d.ts.map +1 -0
- package/dist/qe-findings.js +417 -0
- package/dist/qe-findings.js.map +1 -0
- package/dist/recap.d.ts +1 -1
- package/dist/recap.d.ts.map +1 -1
- package/dist/recap.js +4 -2
- package/dist/recap.js.map +1 -1
- package/dist/review-cost.d.ts +51 -0
- package/dist/review-cost.d.ts.map +1 -0
- package/dist/review-cost.js +110 -0
- package/dist/review-cost.js.map +1 -0
- package/dist/round.d.ts +207 -1
- package/dist/round.d.ts.map +1 -1
- package/dist/round.js +321 -4
- package/dist/round.js.map +1 -1
- package/dist/run-records.d.ts +97 -0
- package/dist/run-records.d.ts.map +1 -1
- package/dist/run-records.js +336 -2
- package/dist/run-records.js.map +1 -1
- package/dist/score.d.ts +44 -1
- package/dist/score.d.ts.map +1 -1
- package/dist/score.js +78 -5
- package/dist/score.js.map +1 -1
- package/package.json +1 -1
- package/sbom.json +425 -75
- package/src/agentdb-index.ts +423 -60
- package/src/apply-leg.ts +187 -36
- package/src/codex-rollouts.ts +374 -0
- package/src/cost-ledger.ts +232 -24
- package/src/cross-family-control.ts +1038 -0
- package/src/debt-ratchet.ts +143 -0
- package/src/embedding-config.ts +131 -10
- package/src/feature-adr-checkpoints.ts +29 -0
- package/src/feature-adr-decision-recall.ts +6 -4
- package/src/feature-adr-envelope.ts +242 -0
- package/src/feature-adr-routing.ts +150 -3
- package/src/feature-adr-stage-canon.ts +141 -0
- package/src/index.ts +65 -6
- package/src/loop-blobs.generated.ts +4 -4
- package/src/mutation-gate.ts +316 -0
- package/src/qe-bridge.ts +12 -2
- package/src/qe-findings.ts +463 -0
- package/src/recap.ts +10 -3
- package/src/review-cost.ts +139 -0
- package/src/round.ts +481 -6
- package/src/run-records.ts +388 -2
- package/src/score.ts +115 -6
package/src/run-records.ts
CHANGED
|
@@ -13,7 +13,133 @@
|
|
|
13
13
|
* Pure: payload in, verdict out. The CLI owns paths, the append, the read-back and the exit code.
|
|
14
14
|
*/
|
|
15
15
|
|
|
16
|
+
import { matchCodexRollouts } from './codex-rollouts.js';
|
|
17
|
+
import type { CodexRollout } from './codex-rollouts.js';
|
|
16
18
|
import { redactTrainingPayload } from './feature-adr-checkpoints.js';
|
|
19
|
+
import { validateExperimentEnvelope } from './feature-adr-envelope.js';
|
|
20
|
+
|
|
21
|
+
/** Structural — a caller passes `cost-scoring.ts`'s `ModelPricing`; kept local so `run-records.ts`
|
|
22
|
+
* does not have to import `cost-scoring.ts` just to name a type.
|
|
23
|
+
*
|
|
24
|
+
* measurement-integrity fix-round-1/F6 (Codex r1 HIGH #6): `cacheCreation` used to be dropped here —
|
|
25
|
+
* the snapshot silently lacked the ONE rate a cache-WRITE-heavy row needs to reproduce its own cost
|
|
26
|
+
* later, even though the caller's own `ModelPricing` carries it. Now carried through verbatim. */
|
|
27
|
+
export interface LedgerPriceEntry {
|
|
28
|
+
readonly prompt: number;
|
|
29
|
+
readonly completion: number;
|
|
30
|
+
readonly cachedInput: number;
|
|
31
|
+
readonly cacheCreation: number;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** measurement-integrity fix-round-1/F4 (Codex r1 HIGH #4): a resolvable executor spec, split into
|
|
35
|
+
* its three parts. Never invents a model: {@link parseModelSpec} returns `null` for anything it
|
|
36
|
+
* cannot resolve to exactly one model, rather than guessing. */
|
|
37
|
+
export interface ParsedModelSpec {
|
|
38
|
+
readonly family: 'claude' | 'codex';
|
|
39
|
+
readonly model: string;
|
|
40
|
+
readonly effort: string | null;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/** Bare Claude model names the pipeline actually emits with no `claude:` prefix (`coder: 'sonnet'`,
|
|
44
|
+
* `'opus'`, `'fable'`, real values observed in `.dz/feature-adr/run-cost-ledger.jsonl`). */
|
|
45
|
+
const BARE_CLAUDE_NAMES = new Set(['sonnet', 'opus', 'haiku', 'fable']);
|
|
46
|
+
/** id/effort alphabet a model spec component may use (lead delta after Codex r2, new MEDIUM #3). */
|
|
47
|
+
const SPEC_PART = /^[A-Za-z0-9._-]+$/;
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Parse an executor spec — the shapes actually recorded in `coder`/`reviewer` fields
|
|
51
|
+
* (`codex:gpt-5.6-sol:high`, `claude:sonnet`, bare `sonnet`/`opus`, bare `codex`, bare `claude`) —
|
|
52
|
+
* into `{family, model, effort}`. Pure, never throws.
|
|
53
|
+
*
|
|
54
|
+
* Returns `null` (never a guess) for anything that cannot be resolved to exactly ONE model:
|
|
55
|
+
* - a bare `'codex'` or `'claude'` (family named, no model at all);
|
|
56
|
+
* - an annotated/aggregate field such as `'claude:sonnet x2'` or `'qe-bridge:claude x2 + lead'` (real
|
|
57
|
+
* values this ledger carries for a MULTI-reviewer round) — any embedded whitespace means the field
|
|
58
|
+
* names more than one resolvable spec, and picking one would misattribute to the others;
|
|
59
|
+
* - a bare model id with no family marker that is not one of the known bare Claude names (e.g. a full
|
|
60
|
+
* `'claude-sonnet-5'` — that shape is handled by the OLDER vendor-prefix path in {@link priceLookup}
|
|
61
|
+
* for backward compatibility, not by this parser).
|
|
62
|
+
*/
|
|
63
|
+
export function parseModelSpec(spec: unknown): ParsedModelSpec | null {
|
|
64
|
+
if (typeof spec !== 'string') return null;
|
|
65
|
+
const trimmed = spec.trim();
|
|
66
|
+
if (trimmed === '' || /\s/.test(trimmed)) return null;
|
|
67
|
+
const parts = trimmed.split(':');
|
|
68
|
+
const head = parts[0] ?? '';
|
|
69
|
+
if (head === 'codex') {
|
|
70
|
+
const model = parts[1];
|
|
71
|
+
if (model === undefined || model === '') return null; // bare 'codex' — no reliable model
|
|
72
|
+
// Lead delta after Codex r2 (new MEDIUM #3): exactly 2 or 3 non-empty components, id alphabet
|
|
73
|
+
// only — `codex:gpt-5.6-sol:high:garbage` is a corrupt/aggregate spec, never a reliable model.
|
|
74
|
+
if (parts.length > 3 || !SPEC_PART.test(model) || (parts.length === 3 && (parts[2] === '' || !SPEC_PART.test(parts[2]!)))) return null;
|
|
75
|
+
const effort = parts[2] ?? null;
|
|
76
|
+
return { family: 'codex', model, effort: effort === '' ? null : effort };
|
|
77
|
+
}
|
|
78
|
+
if (head === 'claude') {
|
|
79
|
+
const model = parts[1];
|
|
80
|
+
if (model === undefined || model === '') return null; // bare 'claude' — no reliable model
|
|
81
|
+
return { family: 'claude', model, effort: null };
|
|
82
|
+
}
|
|
83
|
+
if (parts.length === 1 && BARE_CLAUDE_NAMES.has(head)) {
|
|
84
|
+
return { family: 'claude', model: head, effort: null };
|
|
85
|
+
}
|
|
86
|
+
return null;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/** measurement-integrity FR-5/FR-6: enrichment the WRITER supplies at write time — the rollout logs
|
|
90
|
+
* it already read (I/O lives in the CLI; this stays pure) and the price table snapshot. Absent
|
|
91
|
+
* entirely ⇒ zero behavior change from before this feature (NFR-1). */
|
|
92
|
+
export interface LedgerEnrichInput {
|
|
93
|
+
/** Parsed Codex rollout logs for the window the CLI read — usually every rollout from the days the
|
|
94
|
+
* window spans. Pure data; the CLI is the one that walked `~/.codex/sessions`. */
|
|
95
|
+
readonly rollouts?: readonly CodexRollout[];
|
|
96
|
+
/** The stage's own time window — usually [the previous ledger row's `ts`, this write's `ts`], or
|
|
97
|
+
* an explicit `--window-from/--window-to`. Omitted ⇒ no rollout match is even attempted. */
|
|
98
|
+
readonly window?: { readonly from: string; readonly to: string };
|
|
99
|
+
/** Narrows an otherwise-ambiguous match — usually the repo root the stage ran in. */
|
|
100
|
+
readonly cwd?: string;
|
|
101
|
+
/** A model-pricing table SNAPSHOT (FR-6) — the CALLER's table, captured at write time, never the
|
|
102
|
+
* ledger's own idea of "current" pricing (ADR-001 D4 rejects re-pricing after the fact). */
|
|
103
|
+
readonly prices?: Readonly<Record<string, LedgerPriceEntry>>;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
function isCodexFamily(v: unknown): boolean {
|
|
107
|
+
return typeof v === 'string' && /codex/i.test(v);
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
function isRecord(v: unknown): v is Record<string, unknown> {
|
|
111
|
+
return typeof v === 'object' && v !== null && !Array.isArray(v);
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/** Longest-prefix match against the CALLER'S OWN table (never `cost-scoring.ts`'s internal
|
|
115
|
+
* constant) — the whole point of a price SNAPSHOT is that it answers only from what the caller
|
|
116
|
+
* handed in at write time, mirroring `pricingFor`'s matching rule without importing it.
|
|
117
|
+
*
|
|
118
|
+
* measurement-integrity fix-round-1/F7 (Codex r1 HIGH #7): the OLD version stripped only a
|
|
119
|
+
* `vendor/`-shaped prefix, so the REAL recorded shape `codex:gpt-5.6-sol:high` (or `claude:sonnet`)
|
|
120
|
+
* never matched anything and always landed in `prices.unknown[]`. This now runs the id through the
|
|
121
|
+
* SAME {@link parseModelSpec} FR-4's matcher uses, then reconstructs the normalized id the way
|
|
122
|
+
* `cost-scoring.ts`'s `pricingFor`/`hasKnownPricing` key their table (`claude-<model>` for the
|
|
123
|
+
* Claude family; the bare model for Codex — its ids carry no vendor prefix). A spec this parser
|
|
124
|
+
* cannot resolve falls back to the OLD vendor-prefix strip, so an already-working bare id
|
|
125
|
+
* (`claude-sonnet-5`, `gpt-4o`) keeps matching exactly as before (NFR-1). */
|
|
126
|
+
function priceLookup(modelId: string, table: Readonly<Record<string, LedgerPriceEntry>>): LedgerPriceEntry | null {
|
|
127
|
+
if (typeof modelId !== 'string' || modelId.length === 0) return null;
|
|
128
|
+
const parsed = parseModelSpec(modelId);
|
|
129
|
+
const id = parsed !== null
|
|
130
|
+
? (parsed.family === 'claude' ? `claude-${parsed.model}` : parsed.model).toLowerCase()
|
|
131
|
+
: modelId.toLowerCase().replace(/^[a-z0-9-]+\//, '');
|
|
132
|
+
let best: LedgerPriceEntry | null = null;
|
|
133
|
+
let bestLen = 0;
|
|
134
|
+
for (const [key, price] of Object.entries(table)) {
|
|
135
|
+
const k = key.toLowerCase();
|
|
136
|
+
if (id.startsWith(k) && k.length > bestLen) {
|
|
137
|
+
best = price;
|
|
138
|
+
bestLen = k.length;
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
return best;
|
|
142
|
+
}
|
|
17
143
|
|
|
18
144
|
export type RecordKind = 'ledger' | 'training-pair';
|
|
19
145
|
|
|
@@ -65,7 +191,7 @@ const noop = (verdict: 'duplicate' | 'skipped', reason: string): RecordDecision
|
|
|
65
191
|
line: null,
|
|
66
192
|
});
|
|
67
193
|
|
|
68
|
-
function shapeMismatch(kind: RecordKind, payload: Record<string, unknown
|
|
194
|
+
function shapeMismatch(kind: RecordKind, payload: Record<string, unknown>, autoFlag: boolean): string | null {
|
|
69
195
|
// AM-1 FIRST, before the required-field sweep. A ledger row offered as a training pair fails BOTH
|
|
70
196
|
// checks, and the wrong-kind reason is the one that tells the caller what actually happened —
|
|
71
197
|
// "missing field `output`" sends them looking for a field they never meant to send.
|
|
@@ -78,6 +204,39 @@ function shapeMismatch(kind: RecordKind, payload: Record<string, unknown>): stri
|
|
|
78
204
|
if (kind === 'training-pair' && 'tokens' in payload && !('input' in payload)) {
|
|
79
205
|
return 'this payload looks like a ledger row (`tokens` without `input`), not a training pair';
|
|
80
206
|
}
|
|
207
|
+
// experiment-envelope FR-5 / ADR-001 D2: `auto:true` marks a pipeline-written row, and the pipeline
|
|
208
|
+
// is obligated to carry the envelope built once after the Step-0 router — a row missing it is
|
|
209
|
+
// useless for learning and would silently corrupt the sample (the "warn but write" alternative was
|
|
210
|
+
// rejected in the ADR: a warning nobody reads left `tokens=null` unnoticed for years). A MANUAL row
|
|
211
|
+
// (no `auto`) stays compatible: no envelope required, but one that IS present is still validated —
|
|
212
|
+
// never trusted just because a human typed it.
|
|
213
|
+
//
|
|
214
|
+
// fix-round-1/F2 (cross-family review, HIGH #2): the PAYLOAD's own `auto` field used to be the
|
|
215
|
+
// ONLY signal — an automatic producer that forgot it, or sent `"true"`/`false`, was silently
|
|
216
|
+
// accepted as a manual row and skipped the envelope requirement entirely. `autoFlag` is the CLI's
|
|
217
|
+
// own trusted `--auto` argument (never JSON a caller could typo): it is a SECOND, independent
|
|
218
|
+
// trust source, unioned with the payload field rather than replacing it — nothing that used to be
|
|
219
|
+
// gated stops being gated, and a `--auto`-dispatched caller is now gated even if its hand-built
|
|
220
|
+
// payload forgot the field. Independently of either source, a PRESENT `auto` field is checked for
|
|
221
|
+
// shape: only the literal `true` is a legal value — anything else (a string, `false`, a number) is
|
|
222
|
+
// refused outright, because a field whose only sane value is `true` holding something else is a
|
|
223
|
+
// caller bug worth surfacing, not silently downgrading to "manual".
|
|
224
|
+
if (kind === 'ledger') {
|
|
225
|
+
const rawAuto = payload['auto'];
|
|
226
|
+
if (rawAuto !== undefined && rawAuto !== true) {
|
|
227
|
+
return 'a ledger record\'s `auto` field must be `true` or absent';
|
|
228
|
+
}
|
|
229
|
+
const envelope = payload['envelope'];
|
|
230
|
+
const envelopePresent = envelope !== undefined && envelope !== null;
|
|
231
|
+
const isAuto = autoFlag === true || rawAuto === true;
|
|
232
|
+
if (isAuto && !envelopePresent) {
|
|
233
|
+
return 'an auto ledger row must carry `envelope` (experiment-envelope FR-5)';
|
|
234
|
+
}
|
|
235
|
+
if (envelopePresent) {
|
|
236
|
+
const v = validateExperimentEnvelope(envelope);
|
|
237
|
+
if (!v.ok) return `envelope invalid — ${v.reason}`;
|
|
238
|
+
}
|
|
239
|
+
}
|
|
81
240
|
const required = kind === 'ledger' ? LEDGER_REQUIRED : PAIR_REQUIRED;
|
|
82
241
|
for (const field of required) {
|
|
83
242
|
const v = payload[field];
|
|
@@ -125,7 +284,16 @@ export function decideRecordWrite(input: {
|
|
|
125
284
|
/** Who ran it. Supplied by the CALLER, which lives outside the workflow sandbox and can see the
|
|
126
285
|
* host; absent stays absent (see the stamping comment below). */
|
|
127
286
|
runnerId?: string | null;
|
|
287
|
+
/** fix-round-1/F2: the CLI's own trusted `--auto` flag — see the comment inside `shapeMismatch`. */
|
|
288
|
+
auto?: boolean;
|
|
128
289
|
maxChars?: number;
|
|
290
|
+
/** measurement-integrity FR-5/FR-6: rollout-log + price enrichment for a ledger row. Absent ⇒ zero
|
|
291
|
+
* behavior change (NFR-1). */
|
|
292
|
+
enrich?: LedgerEnrichInput;
|
|
293
|
+
/** experiment-instrument FR-2/A3 (ADR-001): when `true`, an AUTO ledger row that would be written
|
|
294
|
+
* `complete:false` is refused instead (exit 2, before any write) — a круг-B default candidate, opt-
|
|
295
|
+
* in today so nothing that already writes incomplete auto rows starts failing underfoot (NFR-1). */
|
|
296
|
+
strict?: boolean;
|
|
129
297
|
}): RecordDecision {
|
|
130
298
|
const { kind, payloadRaw, stage } = input;
|
|
131
299
|
if (kind !== 'ledger' && kind !== 'training-pair') {
|
|
@@ -153,7 +321,7 @@ export function decideRecordWrite(input: {
|
|
|
153
321
|
// the read-back, the refusal texts — ever sees a byte of the profile block.
|
|
154
322
|
const obj = (kind === 'training-pair' ? redactTrainingPayload(payload) : payload) as Record<string, unknown>;
|
|
155
323
|
|
|
156
|
-
const mismatch = shapeMismatch(kind, obj);
|
|
324
|
+
const mismatch = shapeMismatch(kind, obj, input.auto === true);
|
|
157
325
|
if (mismatch !== null) return refuse(mismatch);
|
|
158
326
|
|
|
159
327
|
// The record is filed under `--stage`, and the payload carries its own. A disagreement means the
|
|
@@ -182,6 +350,11 @@ export function decideRecordWrite(input: {
|
|
|
182
350
|
// inside an already-serialised document — text surgery on a structured value, and the exact place
|
|
183
351
|
// a payload containing that literal token could corrupt itself.
|
|
184
352
|
const stamped: Record<string, unknown> = { ...obj };
|
|
353
|
+
// fix-round-1/F2: the CLI's own `--auto` flag is authoritative — when set, the written row MUST
|
|
354
|
+
// carry `auto:true` too (not just gate on it transiently), so every downstream reader of the
|
|
355
|
+
// PERSISTED line keeps seeing the same signal `shapeMismatch` already gated on above. A no-op when
|
|
356
|
+
// the payload already said `auto:true` (shapeMismatch already refused any OTHER value).
|
|
357
|
+
if (kind === 'ledger' && input.auto === true) stamped['auto'] = true;
|
|
185
358
|
const isGap = (v: unknown): boolean => v === null || v === undefined || (typeof v === 'string' && v.trim() === '');
|
|
186
359
|
if (input.timestamp != null && input.timestamp !== '') {
|
|
187
360
|
// An EMPTY STRING is a gap, not a value. Stamping only over null/undefined let
|
|
@@ -273,6 +446,168 @@ export function decideRecordWrite(input: {
|
|
|
273
446
|
}
|
|
274
447
|
}
|
|
275
448
|
|
|
449
|
+
// measurement-integrity FR-5 (rollout enrichment) + FR-6 (price snapshot). Both are ADDITIVE and
|
|
450
|
+
// OPT-IN on `input.enrich` — a caller that never passes it gets byte-identical output to before
|
|
451
|
+
// this feature (NFR-1).
|
|
452
|
+
if (kind === 'ledger' && input.enrich !== undefined) {
|
|
453
|
+
const enrich = input.enrich;
|
|
454
|
+
|
|
455
|
+
// FR-5: only a codex-family coder/reviewer with `tokens: null` is a candidate — a Claude row, or
|
|
456
|
+
// one that already has a token figure, is left untouched. The loose `/codex/i` check below only
|
|
457
|
+
// decides whether this row is WORTH TRYING at all.
|
|
458
|
+
const tokensIsNull = stamped['tokens'] === null;
|
|
459
|
+
const looksCodexFamily = isCodexFamily(stamped['coder']) || isCodexFamily(stamped['reviewer']);
|
|
460
|
+
if (tokensIsNull && looksCodexFamily) {
|
|
461
|
+
// measurement-integrity fix-round-1/F4 (Codex r1 HIGH #4): the matcher REQUIRES a reliable
|
|
462
|
+
// model, parsed the same way FR-7's price lookup parses one — never `/codex/i` alone. If
|
|
463
|
+
// `coder`/`reviewer` do not resolve to exactly ONE codex model between them (a bare `'codex'`
|
|
464
|
+
// with no model at all, or the two fields naming DIFFERENT codex models), the matcher is never
|
|
465
|
+
// even called with an unreliable/omitted model filter — a lone rollout in the window would
|
|
466
|
+
// otherwise be accepted as `'one'` on time+cwd alone and its tokens misattributed to the wrong
|
|
467
|
+
// model's stage.
|
|
468
|
+
const codexModels = new Set(
|
|
469
|
+
[parseModelSpec(stamped['coder']), parseModelSpec(stamped['reviewer'])]
|
|
470
|
+
.filter((s): s is ParsedModelSpec => s !== null && s.family === 'codex')
|
|
471
|
+
.map((s) => s.model),
|
|
472
|
+
);
|
|
473
|
+
if (codexModels.size !== 1) {
|
|
474
|
+
stamped['tokensSource'] = 'codex-rollout:no-model';
|
|
475
|
+
} else if (enrich.window !== undefined) {
|
|
476
|
+
const model = [...codexModels][0]!;
|
|
477
|
+
const match = matchCodexRollouts(enrich.rollouts ?? [], {
|
|
478
|
+
from: enrich.window.from,
|
|
479
|
+
to: enrich.window.to,
|
|
480
|
+
model,
|
|
481
|
+
...(enrich.cwd !== undefined ? { cwd: enrich.cwd } : {}),
|
|
482
|
+
});
|
|
483
|
+
if (match.status === 'one') {
|
|
484
|
+
stamped['tokens'] = match.rollout.totals.total;
|
|
485
|
+
const startMs = match.rollout.startedAt !== null ? Date.parse(match.rollout.startedAt) : NaN;
|
|
486
|
+
const endMs = match.rollout.endedAt !== null ? Date.parse(match.rollout.endedAt) : NaN;
|
|
487
|
+
// measurement-integrity fix-round-1/F6 (Codex r1 HIGH #6): fill-ONLY-null — an existing
|
|
488
|
+
// `minutes` figure (a manually recorded one, say) must never be silently overwritten by a
|
|
489
|
+
// derived rollout duration.
|
|
490
|
+
if (stamped['minutes'] === null && Number.isFinite(startMs) && Number.isFinite(endMs) && endMs >= startMs) {
|
|
491
|
+
stamped['minutes'] = Math.round(((endMs - startMs) / 60000) * 10) / 10;
|
|
492
|
+
}
|
|
493
|
+
stamped['tokensSource'] = 'codex-rollout';
|
|
494
|
+
stamped['rolloutId'] = match.rollout.id;
|
|
495
|
+
} else {
|
|
496
|
+
// `none` or `ambiguous` — NFR-3: an explicit status, never a guessed number.
|
|
497
|
+
stamped['tokensSource'] = `codex-rollout:${match.status}`;
|
|
498
|
+
}
|
|
499
|
+
} else {
|
|
500
|
+
// Eligible in principle (codex family, one reliable model, tokens null) but no window was
|
|
501
|
+
// supplied — AC-4's "old row without a window": no enrichment is even attempted, and that
|
|
502
|
+
// fact is itself recorded rather than left silently absent.
|
|
503
|
+
stamped['tokensSource'] = 'unavailable';
|
|
504
|
+
}
|
|
505
|
+
}
|
|
506
|
+
|
|
507
|
+
// FR-6: the price snapshot, for every model this row names — independent of the FR-5 branch
|
|
508
|
+
// above (a Claude row gets priced too; only tokens enrichment is codex-specific).
|
|
509
|
+
if (enrich.prices !== undefined) {
|
|
510
|
+
const modelIds = new Set<string>();
|
|
511
|
+
for (const v of [stamped['coder'], stamped['reviewer']]) {
|
|
512
|
+
if (typeof v === 'string' && v.trim() !== '') modelIds.add(v.trim());
|
|
513
|
+
}
|
|
514
|
+
const envelope = stamped['envelope'];
|
|
515
|
+
if (isRecord(envelope)) {
|
|
516
|
+
const chosen = envelope['chosen'];
|
|
517
|
+
if (isRecord(chosen)) {
|
|
518
|
+
const stages = chosen['stages'];
|
|
519
|
+
if (isRecord(stages)) {
|
|
520
|
+
for (const v of Object.values(stages)) {
|
|
521
|
+
if (typeof v === 'string' && v.trim() !== '') modelIds.add(v.trim());
|
|
522
|
+
}
|
|
523
|
+
}
|
|
524
|
+
}
|
|
525
|
+
}
|
|
526
|
+
const table: Record<string, LedgerPriceEntry> = {};
|
|
527
|
+
const unknown: string[] = [];
|
|
528
|
+
for (const modelId of modelIds) {
|
|
529
|
+
const price = priceLookup(modelId, enrich.prices);
|
|
530
|
+
if (price === null) unknown.push(modelId);
|
|
531
|
+
else table[modelId] = { prompt: price.prompt, completion: price.completion, cachedInput: price.cachedInput, cacheCreation: price.cacheCreation };
|
|
532
|
+
}
|
|
533
|
+
const computedPrices = {
|
|
534
|
+
snapshotAt: input.timestamp ?? null,
|
|
535
|
+
table,
|
|
536
|
+
...(unknown.length > 0 ? { unknown } : {}),
|
|
537
|
+
};
|
|
538
|
+
// measurement-integrity fix-round-1/F6 (Codex r1 HIGH #6): fill-ONLY-null for `prices` too — an
|
|
539
|
+
// existing snapshot (a previous write already priced this row) is never unconditionally
|
|
540
|
+
// replaced. Equal → left alone (idempotent re-enrichment, common on a retried write). Different
|
|
541
|
+
// → a named `pricesConflict`, never a silent re-price (ADR-001 D4 forbids re-pricing after the
|
|
542
|
+
// fact — a DIFFERING recomputation is exactly that, so it is surfaced, not applied).
|
|
543
|
+
const existingPrices = stamped['prices'];
|
|
544
|
+
if (existingPrices === undefined || existingPrices === null) {
|
|
545
|
+
stamped['prices'] = computedPrices;
|
|
546
|
+
} else if (isRecord(existingPrices) && isRecord(existingPrices['table']) && (existingPrices['unknown'] === undefined || Array.isArray(existingPrices['unknown']))) {
|
|
547
|
+
// Lead delta after Codex r2 (new MEDIUM #4): compare the WHOLE snapshot canonically (table +
|
|
548
|
+
// sorted unknown), not the table alone.
|
|
549
|
+
const canon = (t: unknown, u: unknown): string => JSON.stringify({ table: t, unknown: Array.isArray(u) ? [...u].map(String).sort() : [] });
|
|
550
|
+
if (canon(existingPrices['table'], existingPrices['unknown']) !== canon(table, unknown)) {
|
|
551
|
+
stamped['pricesConflict'] = { existing: existingPrices, recomputed: computedPrices };
|
|
552
|
+
}
|
|
553
|
+
} else {
|
|
554
|
+
// A malformed existing snapshot (no table / bad unknown) is a CONFLICT, never silently trusted.
|
|
555
|
+
stamped['pricesConflict'] = { existing: existingPrices, recomputed: computedPrices, reason: 'existing prices snapshot is malformed' };
|
|
556
|
+
}
|
|
557
|
+
}
|
|
558
|
+
}
|
|
559
|
+
|
|
560
|
+
// experiment-instrument FR-2/A3 (ADR-001): completeness of an AUTO ledger row. `minutes` is
|
|
561
|
+
// fill-only-null from a `wallSec` the payload carries (the workflow sandbox has a clock delta even
|
|
562
|
+
// when it has no wall clock of its own — FR-2's `wallSec` field, distinct from `minutesSincePrev`
|
|
563
|
+
// above, which needs a PREVIOUS row and a runId neither of which every auto row has). `tokens` is
|
|
564
|
+
// judged complete when it is a real number OR the row already NAMES why it is not (`tokensSource`,
|
|
565
|
+
// set above by the FR-5 rollout match, or supplied by the caller) — an unexplained non-number is the
|
|
566
|
+
// one shape that is actually incomplete. Gated on `auto` only: a MANUAL row never gains any of these
|
|
567
|
+
// three keys, so it stays byte-identical to before this feature (NFR-1).
|
|
568
|
+
if (kind === 'ledger' && stamped['auto'] === true) {
|
|
569
|
+
const incompleteReasons: string[] = [];
|
|
570
|
+
if (stamped['minutes'] === null || stamped['minutes'] === undefined) {
|
|
571
|
+
const wallSec = stamped['wallSec'];
|
|
572
|
+
if (typeof wallSec === 'number' && Number.isFinite(wallSec) && wallSec >= 0) {
|
|
573
|
+
stamped['minutes'] = Math.round((wallSec / 60) * 10) / 10;
|
|
574
|
+
stamped['minutesSource'] = 'wallSec';
|
|
575
|
+
} else {
|
|
576
|
+
incompleteReasons.push('minutes');
|
|
577
|
+
}
|
|
578
|
+
}
|
|
579
|
+
// r1-10 (Codex r1 HIGH #10): completeness requires a real, finite NUMBER of tokens. Naming why
|
|
580
|
+
// tokens are missing (`tokensSource:'unavailable'`, or any other provenance) is diagnostic, never
|
|
581
|
+
// a substitute for the number itself — ADR-001 says missing tokens makes the row incomplete, full
|
|
582
|
+
// stop. The old check (`tokensSource === undefined`) treated a NAMED absence as if it were data.
|
|
583
|
+
if (typeof stamped['tokens'] !== 'number' || !Number.isFinite(stamped['tokens'])) {
|
|
584
|
+
incompleteReasons.push('tokens');
|
|
585
|
+
}
|
|
586
|
+
// r1-11 (Codex r1 MEDIUM #11): a payload that ALREADY declared itself incomplete (its own
|
|
587
|
+
// `complete:false` + `incompleteReasons`, e.g. a workflow-side `'artifact'` reason this module
|
|
588
|
+
// knows nothing about) must never be overwritten back to `complete:true` just because THIS
|
|
589
|
+
// module's own minutes/tokens checks both passed — that erases a true fact and leaves a
|
|
590
|
+
// contradictory row (`complete:true` alongside a stale `incompleteReasons`). Preserve and MERGE.
|
|
591
|
+
const existingCompleteRaw = stamped['complete'];
|
|
592
|
+
const existingWasIncomplete = existingCompleteRaw === false;
|
|
593
|
+
const existingReasonsRaw = stamped['incompleteReasons'];
|
|
594
|
+
const existingReasons = Array.isArray(existingReasonsRaw)
|
|
595
|
+
? existingReasonsRaw.filter((r): r is string => typeof r === 'string')
|
|
596
|
+
: [];
|
|
597
|
+
const mergedReasons = existingWasIncomplete
|
|
598
|
+
? [...new Set([...existingReasons, ...incompleteReasons])]
|
|
599
|
+
: incompleteReasons;
|
|
600
|
+
const complete = mergedReasons.length === 0 && !existingWasIncomplete;
|
|
601
|
+
// A3: under `--strict`, incompleteness is a REFUSAL — before any write, the target untouched —
|
|
602
|
+
// rather than a loudly-marked write. Without `--strict` (the default today; круг-B may flip it),
|
|
603
|
+
// the row is still written, just honestly marked `complete:false` with its reasons.
|
|
604
|
+
if (input.strict === true && !complete) {
|
|
605
|
+
return refuse(`auto ledger row is incomplete (${mergedReasons.join(', ') || 'previously marked incomplete'}) — refused under --strict before any write`);
|
|
606
|
+
}
|
|
607
|
+
stamped['complete'] = complete;
|
|
608
|
+
if (!complete) stamped['incompleteReasons'] = mergedReasons.length > 0 ? mergedReasons : existingReasons;
|
|
609
|
+
}
|
|
610
|
+
|
|
276
611
|
let line: string;
|
|
277
612
|
try {
|
|
278
613
|
line = JSON.stringify(stamped);
|
|
@@ -324,3 +659,54 @@ export function decideReadBack(appended: string, lastLineOnDisk: string | null):
|
|
|
324
659
|
export function recordVerdictLine(kind: RecordKind, stage: string, d: RecordDecision): string {
|
|
325
660
|
return `feature-adr record (${kind}/${stage}): ${d.verdict.toUpperCase()} — ${d.reason}`;
|
|
326
661
|
}
|
|
662
|
+
|
|
663
|
+
/** experiment-instrument FR-1/FR-3 (ADR-001): what `round.ts`'s `readOpenRoundTaskId` returns — the
|
|
664
|
+
* single source `applyTaskId` fills from. Duplicated here rather than imported so this pure module
|
|
665
|
+
* never depends on `round.ts`'s own shape; the CLI is the one holding both and wiring them together.
|
|
666
|
+
* r1-1/r1-2 (Codex r1 #1/#2): extended with `'derived-legacy'` and `'unavailable'` to stay in
|
|
667
|
+
* lockstep with `round.ts`'s own `readOpenRoundTaskId` return type. */
|
|
668
|
+
export interface TaskIdLookup {
|
|
669
|
+
readonly taskId: string | null;
|
|
670
|
+
readonly source: 'open-round' | 'derived-legacy' | 'no-open-round' | 'ambiguous' | 'unavailable';
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
/**
|
|
674
|
+
* experiment-instrument FR-1/FR-3/A8 (ADR-001): propagate `taskId` onto a ledger/training-pair payload
|
|
675
|
+
* BEFORE it reaches {@link decideRecordWrite} — fill-only-null, never overwritten.
|
|
676
|
+
*
|
|
677
|
+
* - The payload already names a non-empty `taskId` string ⇒ it is authoritative. When it DISAGREES
|
|
678
|
+
* with the round's own current taskId, that disagreement is a real fact worth keeping — recorded as
|
|
679
|
+
* `taskIdConflict: {payload, round}` — never silently resolved either way (A8).
|
|
680
|
+
* - The payload's `taskId` key is absent, or explicitly `null`/`undefined` ⇒ filled from `lookup`,
|
|
681
|
+
* INCLUDING the honest `null` case: no open round (A4) or two of them (A5) still stamps `taskId:
|
|
682
|
+
* null` + `taskIdSource` naming why, rather than leaving the field silently absent — absence with a
|
|
683
|
+
* named reason beats absence with none.
|
|
684
|
+
* - r1-3 (Codex r1 HIGH #3): the payload's `taskId` key is PRESENT with a value that is neither a
|
|
685
|
+
* non-empty string nor null/undefined (a number, a boolean, an object, or a blank/whitespace-only
|
|
686
|
+
* string) ⇒ that is a present-but-INVALID value, a THIRD case distinct from both of the above. It
|
|
687
|
+
* used to be treated exactly like "absent" (`typeof !== 'string'` fell through to the fill branch),
|
|
688
|
+
* silently replacing the caller's own (malformed) value with the round's — violating both
|
|
689
|
+
* fill-only-null and "a present payload value always wins". Now: the row's own value is preserved
|
|
690
|
+
* UNTOUCHED (never replaced with a guess about what the caller meant), and the problem is named in
|
|
691
|
+
* `taskIdInvalid` so a reader can see the row was neither filled nor trusted blindly.
|
|
692
|
+
*
|
|
693
|
+
* Pure: no filesystem, no clock. The CALLER (the cli) is the one that read `.dz/rounds/` to build
|
|
694
|
+
* `lookup` in the first place.
|
|
695
|
+
*/
|
|
696
|
+
export function applyTaskId(row: Record<string, unknown>, lookup: TaskIdLookup): Record<string, unknown> {
|
|
697
|
+
const hasTaskIdKey = Object.prototype.hasOwnProperty.call(row, 'taskId');
|
|
698
|
+
const rawPayloadTaskId = row['taskId'];
|
|
699
|
+
if (hasTaskIdKey && rawPayloadTaskId !== null && rawPayloadTaskId !== undefined) {
|
|
700
|
+
if (typeof rawPayloadTaskId === 'string' && rawPayloadTaskId.trim() !== '') {
|
|
701
|
+
const payloadTaskId = rawPayloadTaskId.trim();
|
|
702
|
+
if (lookup.taskId !== null && lookup.taskId !== payloadTaskId) {
|
|
703
|
+
return { ...row, taskIdConflict: { payload: payloadTaskId, round: lookup.taskId } };
|
|
704
|
+
}
|
|
705
|
+
return { ...row };
|
|
706
|
+
}
|
|
707
|
+
// r1-3: present but not a usable identity (non-string, or blank after trim) — refuse to replace
|
|
708
|
+
// it with a lookup guess; preserve it verbatim and name the problem.
|
|
709
|
+
return { ...row, taskIdInvalid: { value: rawPayloadTaskId, reason: 'taskId present but not a non-empty string' } };
|
|
710
|
+
}
|
|
711
|
+
return { ...row, taskId: lookup.taskId, taskIdSource: lookup.source };
|
|
712
|
+
}
|
package/src/score.ts
CHANGED
|
@@ -21,6 +21,7 @@
|
|
|
21
21
|
*/
|
|
22
22
|
|
|
23
23
|
import { amendmentIdsIn } from './amendment-trace.js';
|
|
24
|
+
import { QE_SEVERITIES, findInvalidQeVerdictLines, parseQeFindings, readQeVerdictLines, type QeFindingsRefusedRow, type QeFindingsSummary } from './qe-findings.js';
|
|
24
25
|
|
|
25
26
|
export type DisciplineVerdict = 'pass' | 'partial' | 'absent';
|
|
26
27
|
|
|
@@ -45,11 +46,29 @@ export interface RunScorecard {
|
|
|
45
46
|
* stay distinguishable.
|
|
46
47
|
*/
|
|
47
48
|
readonly mutationEvidence?: MutationEvidence;
|
|
49
|
+
/**
|
|
50
|
+
* qe-findings-record FR-4: where `qeGrade` came from — a machine-written `QE-VERDICT:` line, the
|
|
51
|
+
* pre-existing prose scan, or neither. ADDITIVE for the same reason as `mutationEvidence`: every
|
|
52
|
+
* scorecard built before this field existed still satisfies `RunScorecard`.
|
|
53
|
+
*/
|
|
54
|
+
readonly gradeSource?: GradeSource;
|
|
55
|
+
/**
|
|
56
|
+
* qe-findings-record FR-4: the report's Findings ledger, when one exists. `absent` for the 406
|
|
57
|
+
* pre-existing reports that carry neither the table nor the heading (NFR-1) — never omitted, for
|
|
58
|
+
* the same "gate never ran vs a field an older scorer never wrote" reason `mutationEvidence` gives.
|
|
59
|
+
*/
|
|
60
|
+
readonly findings?: QeFindingsScoreView;
|
|
48
61
|
readonly passed: number;
|
|
49
62
|
readonly total: number;
|
|
50
63
|
readonly summary: string;
|
|
51
64
|
}
|
|
52
65
|
|
|
66
|
+
/** A lighter projection of `QeFindingsResult` for the scorecard — summary + refused + hollow, never
|
|
67
|
+
* the full row list (that stays in `parseQeFindings`'s own return value for a caller that wants it). */
|
|
68
|
+
export type QeFindingsScoreView =
|
|
69
|
+
| { readonly status: 'absent' }
|
|
70
|
+
| { readonly status: 'present'; readonly hollow: boolean; readonly summary: QeFindingsSummary; readonly refused: readonly QeFindingsRefusedRow[] };
|
|
71
|
+
|
|
53
72
|
/** The artifact texts of one run, keyed by RELATIVE path under `features/<slug>/`. */
|
|
54
73
|
/**
|
|
55
74
|
* The exact heading Step 5 asks for, and the exact heading the check looks for — ONE constant, so
|
|
@@ -306,7 +325,22 @@ function normaliseGradeSign(grade: string): string {
|
|
|
306
325
|
return grade.replace('\u2212', '-');
|
|
307
326
|
}
|
|
308
327
|
|
|
309
|
-
|
|
328
|
+
/**
|
|
329
|
+
* fix-round-1 (codex-r1-verdict finding 2): `'invalid'` is a report that ATTEMPTED a machine verdict
|
|
330
|
+
* (a line starting `QE-VERDICT`, any case/spacing) and got the grammar wrong — `QE-VERDICT: B – final`
|
|
331
|
+
* (en dash + trailing prose), wrong case, no colon, a grade outside A-D. That is a DIFFERENT fact from
|
|
332
|
+
* `'none'` (no attempt at all, legacy prose scan still applies): a malformed attempt must never fall
|
|
333
|
+
* back to guessing a grade from prose — the malformed line is itself evidence the report is unreliable
|
|
334
|
+
* here, and guessing past it would silently launder that unreliability into a confident number.
|
|
335
|
+
*/
|
|
336
|
+
export type GradeReadStatus = 'unique' | 'ambiguous' | 'none' | 'invalid';
|
|
337
|
+
|
|
338
|
+
/**
|
|
339
|
+
* qe-findings-record (ADR-001 D1): where the grade came from. `'verdict-line'` — a machine-written
|
|
340
|
+
* `QE-VERDICT:` line, the source of truth when present. `'prose'` — the pre-existing GRADE_RE scan,
|
|
341
|
+
* unchanged, used only when NO verdict line exists. `'none'` — neither surface names a grade.
|
|
342
|
+
*/
|
|
343
|
+
export type GradeSource = 'verdict-line' | 'prose' | 'none';
|
|
310
344
|
|
|
311
345
|
export interface GradeReading {
|
|
312
346
|
readonly status: GradeReadStatus;
|
|
@@ -314,6 +348,7 @@ export interface GradeReading {
|
|
|
314
348
|
readonly grade: string | null;
|
|
315
349
|
/** Every distinct grade found, normalised — what makes an `ambiguous` verdict inspectable. */
|
|
316
350
|
readonly found: readonly string[];
|
|
351
|
+
readonly source: GradeSource;
|
|
317
352
|
}
|
|
318
353
|
|
|
319
354
|
/**
|
|
@@ -329,6 +364,44 @@ export interface GradeReading {
|
|
|
329
364
|
* report is ambiguous, which is a fact about the report, not a missing number.
|
|
330
365
|
*/
|
|
331
366
|
export function readQeGrade(qeText: string): GradeReading {
|
|
367
|
+
// ADR-001 D1: a machine-written verdict line is the source of truth WHEN it exists — checked
|
|
368
|
+
// first, and the prose scan below never runs when it does. Exactly one -> unique; more than one
|
|
369
|
+
// -> ambiguous ("две строки — ambiguous, никогда «последняя побеждает»", even if both name the
|
|
370
|
+
// SAME grade — two lines is a fact about the report, not a number to reconcile); zero -> the
|
|
371
|
+
// prose scan runs exactly as it always has (NFR-1: 406 pre-existing reports are unaffected) —
|
|
372
|
+
// UNLESS a malformed attempt exists (fix-round-1 finding 2, checked next): the report is `invalid`
|
|
373
|
+
// and the prose fallback is refused, never silently reached.
|
|
374
|
+
const verdictLines = readQeVerdictLines(qeText);
|
|
375
|
+
// Lead delta after Codex r2 (new HIGH #1): a malformed declaration next to a valid one is NOT a
|
|
376
|
+
// unique verdict — `QE-VERDICT: A` + `qe-verdict: D` used to read as unique A. Invalid lines are
|
|
377
|
+
// checked FIRST, whatever the count of valid ones.
|
|
378
|
+
const invalidFirst = findInvalidQeVerdictLines(qeText);
|
|
379
|
+
if (invalidFirst.length > 0) {
|
|
380
|
+
const reasons = invalidFirst.map(
|
|
381
|
+
(l) => `line ${l.line}: ${JSON.stringify(l.text.trim())} does not match "QE-VERDICT: <A|B|C|D><+|-> "`,
|
|
382
|
+
);
|
|
383
|
+
return { status: 'invalid', grade: null, found: [...verdictLines, ...reasons], source: 'verdict-line' };
|
|
384
|
+
}
|
|
385
|
+
if (verdictLines.length === 1) {
|
|
386
|
+
return { status: 'unique', grade: verdictLines[0] as string, found: verdictLines, source: 'verdict-line' };
|
|
387
|
+
}
|
|
388
|
+
if (verdictLines.length > 1) {
|
|
389
|
+
const distinct: string[] = [];
|
|
390
|
+
for (const g of verdictLines) if (!distinct.includes(g)) distinct.push(g);
|
|
391
|
+
return { status: 'ambiguous', grade: null, found: distinct, source: 'verdict-line' };
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
// fix-round-1 finding 2: zero GRAMMATICALLY VALID verdict lines is not yet "no attempt" — a line
|
|
395
|
+
// that clearly opens a verdict declaration but botches the grammar (wrong dash, wrong case, no
|
|
396
|
+
// colon, a grade outside A-D) is `invalid`, and the legacy prose scan below is FORBIDDEN for it.
|
|
397
|
+
// The reason lives inside `found` (GradeReading keeps the same 4-field shape for every status).
|
|
398
|
+
// Lead delta after Codex r2 (#2 PARTIAL): the prose fallback is for LEGACY reports only. A report
|
|
399
|
+
// that already carries the new-format ledger heading but no verdict line is a NEW-format report
|
|
400
|
+
// missing its verdict — `invalid`, never a prose guess.
|
|
401
|
+
if (/^## Findings ledger\s*$/m.test(qeText)) {
|
|
402
|
+
return { status: 'invalid', grade: null, found: ['new-format report (has "## Findings ledger") without a QE-VERDICT line'], source: 'verdict-line' };
|
|
403
|
+
}
|
|
404
|
+
|
|
332
405
|
const found: string[] = [];
|
|
333
406
|
const lines = qeText.split('\n');
|
|
334
407
|
// Line offsets once, so each match maps to ITS line for the negation screen (67d7883d: the
|
|
@@ -352,9 +425,9 @@ export function readQeGrade(qeText: string): GradeReading {
|
|
|
352
425
|
const g = normaliseGradeSign(m[1] as string);
|
|
353
426
|
if (!found.includes(g)) found.push(g);
|
|
354
427
|
}
|
|
355
|
-
if (found.length === 0) return { status: 'none', grade: null, found: [] };
|
|
356
|
-
if (found.length === 1) return { status: 'unique', grade: found[0] as string, found };
|
|
357
|
-
return { status: 'ambiguous', grade: null, found };
|
|
428
|
+
if (found.length === 0) return { status: 'none', grade: null, found: [], source: 'none' };
|
|
429
|
+
if (found.length === 1) return { status: 'unique', grade: found[0] as string, found, source: 'prose' };
|
|
430
|
+
return { status: 'ambiguous', grade: null, found, source: 'prose' };
|
|
358
431
|
}
|
|
359
432
|
|
|
360
433
|
export function extractQeGrade(qeText: string): string | null {
|
|
@@ -451,7 +524,13 @@ export function scoreRun(slug: string, artifacts: RunArtifacts): RunScorecard {
|
|
|
451
524
|
);
|
|
452
525
|
|
|
453
526
|
// 3. Cross-model QE — an independent family reviewed it, and a grade exists.
|
|
454
|
-
const
|
|
527
|
+
const gradeReading = readQeGrade(qeText);
|
|
528
|
+
const grade = gradeReading.grade;
|
|
529
|
+
const findingsParsed = parseQeFindings(qeText);
|
|
530
|
+
const findings: QeFindingsScoreView =
|
|
531
|
+
findingsParsed.status === 'absent'
|
|
532
|
+
? { status: 'absent' }
|
|
533
|
+
: { status: 'present', hollow: findingsParsed.hollow, summary: findingsParsed.summary, refused: findingsParsed.refused };
|
|
455
534
|
if (qeText === '') {
|
|
456
535
|
add('cross-model-qe', 'independent cross-model review with a grade', 'absent', 'no 08_qe_report.md artifact');
|
|
457
536
|
} else {
|
|
@@ -559,11 +638,36 @@ export function scoreRun(slug: string, artifacts: RunArtifacts): RunScorecard {
|
|
|
559
638
|
(grade !== null ? ` · QE grade ${grade}` : ' · no QE grade') +
|
|
560
639
|
(worst.length > 0 ? ` · absent: ${worst.join(', ')}` : '');
|
|
561
640
|
|
|
562
|
-
return { slug, disciplines, qeGrade: grade, mutationEvidence, passed, total, summary };
|
|
641
|
+
return { slug, disciplines, qeGrade: grade, gradeSource: gradeReading.source, findings, mutationEvidence, passed, total, summary };
|
|
563
642
|
}
|
|
564
643
|
|
|
565
644
|
const MARK: Record<DisciplineVerdict, string> = { pass: '✓', partial: '◐', absent: '✗' };
|
|
566
645
|
|
|
646
|
+
/** qe-findings-record FR-4: the one-line summary `dz score` prints for a report's Findings ledger —
|
|
647
|
+
* "findings: 3 HIGH / 2 MEDIUM; 1 refused (line 84: severity "Major" not in dictionary)". Ordered by
|
|
648
|
+
* QE_SEVERITIES (BLOCKER first) so the worst finding always reads first. `null` when there is no
|
|
649
|
+
* table at all — the common case, which earns no noise (same rule as mutationEvidence above). */
|
|
650
|
+
export function renderFindingsLine(findings: QeFindingsScoreView | undefined): string | null {
|
|
651
|
+
if (findings === undefined || findings.status === 'absent') return null;
|
|
652
|
+
// fix-round-1 finding 6: hollow used to short-circuit BEFORE the refused check, so a report whose
|
|
653
|
+
// real ledger table (with CRITICAL rows) was refused as duplicate/outside-section alongside an
|
|
654
|
+
// empty accepted table read as pure "EMPTY" — the refused rows vanished from the printed line, not
|
|
655
|
+
// merely from the tally. Both facts are ALWAYS reported together now; neither hides the other.
|
|
656
|
+
const refusedSuffix = ((): string => {
|
|
657
|
+
if (findings.refused.length === 0) return '';
|
|
658
|
+
const first = findings.refused[0] as QeFindingsRefusedRow;
|
|
659
|
+
const more = findings.refused.length > 1 ? `, +${findings.refused.length - 1} more` : '';
|
|
660
|
+
return `; ${findings.refused.length} refused (line ${first.line}: ${first.reason})${more}`;
|
|
661
|
+
})();
|
|
662
|
+
if (findings.hollow) {
|
|
663
|
+
return `findings: table present but EMPTY (hollow) — worse than no table at all${refusedSuffix}`;
|
|
664
|
+
}
|
|
665
|
+
const bySev = findings.summary.bySeverity;
|
|
666
|
+
const parts = QE_SEVERITIES.filter((s) => (bySev[s] ?? 0) > 0).map((s) => `${bySev[s]} ${s}`);
|
|
667
|
+
const head = parts.length > 0 ? parts.join(' / ') : 'no rows';
|
|
668
|
+
return `findings: ${head}${refusedSuffix}`;
|
|
669
|
+
}
|
|
670
|
+
|
|
567
671
|
export function renderScorecard(card: RunScorecard): string {
|
|
568
672
|
const out: string[] = [];
|
|
569
673
|
out.push(`dz score — ${card.slug} (process scorecard; descriptive-only, never a gate)`);
|
|
@@ -581,6 +685,11 @@ export function renderScorecard(card: RunScorecard): string {
|
|
|
581
685
|
out.push('');
|
|
582
686
|
out.push(` ${card.mutationEvidence.evidence}`);
|
|
583
687
|
}
|
|
688
|
+
const findingsLine = renderFindingsLine(card.findings);
|
|
689
|
+
if (findingsLine !== null) {
|
|
690
|
+
out.push('');
|
|
691
|
+
out.push(` ${findingsLine}${card.gradeSource !== undefined && card.gradeSource !== 'none' ? ` (grade source: ${card.gradeSource})` : ''}`);
|
|
692
|
+
}
|
|
584
693
|
out.push('');
|
|
585
694
|
out.push(` ${card.summary}`);
|
|
586
695
|
return out.join('\n');
|