@dzhechkov/harness-core 0.8.36 → 0.8.38
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +216 -76
- package/README.md +349 -8
- package/dist/agentdb-index.d.ts +87 -7
- package/dist/agentdb-index.d.ts.map +1 -1
- package/dist/agentdb-index.js +416 -57
- package/dist/agentdb-index.js.map +1 -1
- package/dist/apply-leg.d.ts +19 -1
- package/dist/apply-leg.d.ts.map +1 -1
- package/dist/apply-leg.js +187 -36
- package/dist/apply-leg.js.map +1 -1
- package/dist/codex-rollouts.d.ts +118 -0
- package/dist/codex-rollouts.d.ts.map +1 -0
- package/dist/codex-rollouts.js +297 -0
- package/dist/codex-rollouts.js.map +1 -0
- package/dist/cost-ledger.d.ts +56 -4
- package/dist/cost-ledger.d.ts.map +1 -1
- package/dist/cost-ledger.js +176 -20
- package/dist/cost-ledger.js.map +1 -1
- package/dist/cross-family-control.d.ts +380 -0
- package/dist/cross-family-control.d.ts.map +1 -0
- package/dist/cross-family-control.js +848 -0
- package/dist/cross-family-control.js.map +1 -0
- package/dist/debt-ratchet.d.ts +53 -0
- package/dist/debt-ratchet.d.ts.map +1 -0
- package/dist/debt-ratchet.js +107 -0
- package/dist/debt-ratchet.js.map +1 -0
- package/dist/embedding-config.d.ts +42 -0
- package/dist/embedding-config.d.ts.map +1 -1
- package/dist/embedding-config.js +106 -10
- package/dist/embedding-config.js.map +1 -1
- package/dist/feature-adr-checkpoints.d.ts +6 -0
- package/dist/feature-adr-checkpoints.d.ts.map +1 -1
- package/dist/feature-adr-checkpoints.js +29 -0
- package/dist/feature-adr-checkpoints.js.map +1 -1
- package/dist/feature-adr-decision-recall.d.ts +2 -2
- package/dist/feature-adr-decision-recall.d.ts.map +1 -1
- package/dist/feature-adr-decision-recall.js +5 -3
- package/dist/feature-adr-decision-recall.js.map +1 -1
- package/dist/feature-adr-envelope.d.ts +96 -0
- package/dist/feature-adr-envelope.d.ts.map +1 -0
- package/dist/feature-adr-envelope.js +183 -0
- package/dist/feature-adr-envelope.js.map +1 -0
- package/dist/feature-adr-routing.d.ts +64 -0
- package/dist/feature-adr-routing.d.ts.map +1 -1
- package/dist/feature-adr-routing.js +133 -3
- package/dist/feature-adr-routing.js.map +1 -1
- package/dist/feature-adr-stage-canon.d.ts +79 -0
- package/dist/feature-adr-stage-canon.d.ts.map +1 -0
- package/dist/feature-adr-stage-canon.js +117 -0
- package/dist/feature-adr-stage-canon.js.map +1 -0
- package/dist/index.d.ts +21 -9
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +15 -5
- package/dist/index.js.map +1 -1
- package/dist/loop-blobs.generated.js +4 -4
- package/dist/loop-blobs.generated.js.map +1 -1
- package/dist/mutation-gate.d.ts +51 -0
- package/dist/mutation-gate.d.ts.map +1 -1
- package/dist/mutation-gate.js +295 -0
- package/dist/mutation-gate.js.map +1 -1
- package/dist/qe-bridge.d.ts +8 -0
- package/dist/qe-bridge.d.ts.map +1 -1
- package/dist/qe-bridge.js +4 -2
- package/dist/qe-bridge.js.map +1 -1
- package/dist/qe-findings.d.ts +107 -0
- package/dist/qe-findings.d.ts.map +1 -0
- package/dist/qe-findings.js +417 -0
- package/dist/qe-findings.js.map +1 -0
- package/dist/recap.d.ts +1 -1
- package/dist/recap.d.ts.map +1 -1
- package/dist/recap.js +4 -2
- package/dist/recap.js.map +1 -1
- package/dist/review-cost.d.ts +51 -0
- package/dist/review-cost.d.ts.map +1 -0
- package/dist/review-cost.js +110 -0
- package/dist/review-cost.js.map +1 -0
- package/dist/round.d.ts +207 -1
- package/dist/round.d.ts.map +1 -1
- package/dist/round.js +321 -4
- package/dist/round.js.map +1 -1
- package/dist/run-records.d.ts +97 -0
- package/dist/run-records.d.ts.map +1 -1
- package/dist/run-records.js +336 -2
- package/dist/run-records.js.map +1 -1
- package/dist/score.d.ts +44 -1
- package/dist/score.d.ts.map +1 -1
- package/dist/score.js +78 -5
- package/dist/score.js.map +1 -1
- package/package.json +1 -1
- package/sbom.json +425 -75
- package/src/agentdb-index.ts +423 -60
- package/src/apply-leg.ts +187 -36
- package/src/codex-rollouts.ts +374 -0
- package/src/cost-ledger.ts +232 -24
- package/src/cross-family-control.ts +1038 -0
- package/src/debt-ratchet.ts +143 -0
- package/src/embedding-config.ts +131 -10
- package/src/feature-adr-checkpoints.ts +29 -0
- package/src/feature-adr-decision-recall.ts +6 -4
- package/src/feature-adr-envelope.ts +242 -0
- package/src/feature-adr-routing.ts +150 -3
- package/src/feature-adr-stage-canon.ts +141 -0
- package/src/index.ts +65 -6
- package/src/loop-blobs.generated.ts +4 -4
- package/src/mutation-gate.ts +316 -0
- package/src/qe-bridge.ts +12 -2
- package/src/qe-findings.ts +463 -0
- package/src/recap.ts +10 -3
- package/src/review-cost.ts +139 -0
- package/src/round.ts +481 -6
- package/src/run-records.ts +388 -2
- package/src/score.ts +115 -6
|
@@ -0,0 +1,463 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* qe-findings-record (ADR-001, tier M): machine-readable Step-8 verdict + findings ledger.
|
|
3
|
+
*
|
|
4
|
+
* Two prior facts drove this (Step 0, `00_complexity_assessment.md`): scoring reads review PROSE
|
|
5
|
+
* with a regex (`readQeGrade` in score.ts) and 41 of 100 feature reports came back `ambiguous`
|
|
6
|
+
* because a report that was fixed after review states two grades in prose — a real fact about the
|
|
7
|
+
* report, not a parser bug. Findings were not read at all: `dz score` could say "cross-model QE ran"
|
|
8
|
+
* but never how many BLOCKER/HIGH findings it raised or what happened to them.
|
|
9
|
+
*
|
|
10
|
+
* This module adds ONE machine-readable surface on top of the prose, never replacing it:
|
|
11
|
+
* - `QE-VERDICT: <A|A-|A+|B|B+|B-|C|C+|C-|D>` — a single line, the source of truth when present.
|
|
12
|
+
* - a `Findings ledger` table under the EXACT header `QE_FINDINGS_HEADER`, closed dictionaries for
|
|
13
|
+
* Severity/Status/Author. A row outside the dictionary is REFUSED, never coerced to the nearest
|
|
14
|
+
* known value (ADR-001 D2) — that would make the resulting severity/status tally unprovable.
|
|
15
|
+
* - an empty (header-only) table is `hollow: true` — worse than no table at all (ADR-001 D3, the
|
|
16
|
+
* same principle as the mutation table's `present-unproven` in score.ts).
|
|
17
|
+
*
|
|
18
|
+
* Absence of either surface is NOT an error: 406 pre-existing reports carry neither, and this module
|
|
19
|
+
* reports `{status:'absent'}` for them exactly as it always will (ADR-001 D4, NFR-1).
|
|
20
|
+
*
|
|
21
|
+
* fix-round-1 (codex-r1-verdict, Grade C, findings 1/2/5/6/7): a machine artifact recognised inside
|
|
22
|
+
* fenced/indented code, a blockquote or an HTML comment is not a machine artifact — a review that
|
|
23
|
+
* QUOTES the format as an example must not be read as the report's own verdict/table. Both scans
|
|
24
|
+
* below run over TEXT MASKED THE SAME WAY as `amendment-trace.ts` (`maskMarkdown`), plus one block
|
|
25
|
+
* kind that shared masker does not cover — blockquotes — added here as the minimal local pass the
|
|
26
|
+
* ADR calls for. Masking blanks a line to spaces of the SAME LENGTH, so every line number reported
|
|
27
|
+
* in `refused`/`QeVerdictInvalidLine` still points at the real line in the ORIGINAL text.
|
|
28
|
+
*
|
|
29
|
+
* The findings table is additionally accepted ONLY under the single, top-level `## Findings ledger`
|
|
30
|
+
* heading (finding 1); a table found anywhere else is refused `outside ledger section`, never parsed.
|
|
31
|
+
*
|
|
32
|
+
* PURE, deliberately: no node:fs import here (NFR-2, guarded by test/core-boundary.test.ts). File
|
|
33
|
+
* reads belong to the CLI / the calling agent, same as every other core module.
|
|
34
|
+
*/
|
|
35
|
+
|
|
36
|
+
import { maskMarkdown } from './markdown-masker.js';
|
|
37
|
+
|
|
38
|
+
export const QE_SEVERITIES = ['BLOCKER', 'CRITICAL', 'HIGH', 'MEDIUM', 'LOW', 'INFO'] as const;
|
|
39
|
+
export type QeSeverity = (typeof QE_SEVERITIES)[number];
|
|
40
|
+
|
|
41
|
+
export const QE_STATUSES = ['confirmed', 'fixed', 'partial', 'refuted', 'named-limit', 'open'] as const;
|
|
42
|
+
export type QeStatus = (typeof QE_STATUSES)[number];
|
|
43
|
+
|
|
44
|
+
export const QE_AUTHORS = ['codex', 'claude', 'lead'] as const;
|
|
45
|
+
export type QeAuthor = (typeof QE_AUTHORS)[number];
|
|
46
|
+
|
|
47
|
+
/** The ONE exact heading a producer must write and a reader must find — see score.ts's own
|
|
48
|
+
* comment on why this is a single shared constant rather than two strings that can drift apart. */
|
|
49
|
+
export const QE_FINDINGS_HEADER = '| Finding | Severity | Status | Round | Author | Title |';
|
|
50
|
+
|
|
51
|
+
/** The section a Findings ledger table must live directly under — fix-round-1 finding 1: a table
|
|
52
|
+
* found anywhere else (before this heading, after the section ends, or with no/duplicate heading)
|
|
53
|
+
* is `refused` as `outside ledger section`, never parsed as the real ledger. */
|
|
54
|
+
export const QE_LEDGER_HEADING = '## Findings ledger';
|
|
55
|
+
|
|
56
|
+
/** `QE-VERDICT: B`, `QE-VERDICT: A-`, `QE-VERDICT: A+` — a whole line, nothing trailing but
|
|
57
|
+
* whitespace. Only A-D (FR-1); this is deliberately narrower than score.ts's prose GRADE_RE, which
|
|
58
|
+
* also accepts E/F and looser punctuation because prose is written by hand. */
|
|
59
|
+
export const QE_VERDICT_RE = /^QE-VERDICT:\s*([A-D][+−-]?)\s*$/m;
|
|
60
|
+
|
|
61
|
+
/** Any line that OPENS a verdict declaration — any case, any leading whitespace, with or without the
|
|
62
|
+
* colon — checked separately from `QE_VERDICT_RE`'s strict grammar (fix-round-1 finding 2): this is
|
|
63
|
+
* what lets `readQeGrade` tell "no verdict was ever attempted" (legacy prose fallback allowed) apart
|
|
64
|
+
* from "a verdict was attempted and is malformed" (fallback forbidden — the report must say `invalid`
|
|
65
|
+
* rather than silently reading a stale prose grade). */
|
|
66
|
+
const QE_VERDICT_LOOKALIKE_RE = /^\s*QE-VERDICT\b/i;
|
|
67
|
+
|
|
68
|
+
/** U+2212 (minus sign) and the ASCII hyphen spell the same grade sign — the same normalisation
|
|
69
|
+
* score.ts's `normaliseGradeSign` applies to prose grades, duplicated here (not imported) so this
|
|
70
|
+
* module never depends on score.ts and stays the leaf of the dependency graph. */
|
|
71
|
+
function normaliseVerdictSign(grade: string): string {
|
|
72
|
+
return grade.replace('−', '-');
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** A leading Markdown blockquote marker (CommonMark: up to 3 leading spaces, then `>`).
|
|
76
|
+
* `maskMarkdown` (the canonical shared masker, also used by amendment-trace.ts) does not mask
|
|
77
|
+
* blockquotes — only fences, indented code and HTML comments — so this ONE missing block kind is
|
|
78
|
+
* added here as the minimal local pass fix-round-1 calls for, with the same blank-to-spaces
|
|
79
|
+
* contract (line count and every other line's length never change). */
|
|
80
|
+
const BLOCKQUOTE_RE = /^ {0,3}>/;
|
|
81
|
+
/** CommonMark block starters that INTERRUPT a paragraph (so they can never be a lazy continuation
|
|
82
|
+
* of a blockquote): ATX heading, thematic break, fenced code, table row, list item, HTML block. */
|
|
83
|
+
const BLOCK_STARTER_RE = /^ {0,3}(?:#{1,6}(?:\s|$)|(?:[-*_])(?: *[-*_]){2,} *$|```|~~~|\||(?:[-*+]|\d{1,9}[.)])\s|<)/;
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* qe-findings-lazy-blockquote (backlog baba20e3b1c14060): a `>` line only opens the mask — a
|
|
87
|
+
* CommonMark blockquote paragraph can continue onto a FOLLOWING line that carries no `>` marker at
|
|
88
|
+
* all (a "lazy continuation"), and that unmarked line still belongs to the quote. The original
|
|
89
|
+
* single-line `.map()` masked only lines that themselves started with `>`, so a `QE-VERDICT: X` or
|
|
90
|
+
* `## Findings ledger` line written as the *second* line of a quoted example (no blank line, no
|
|
91
|
+
* `>`) was left unmasked and misread as the report's own artifact.
|
|
92
|
+
*
|
|
93
|
+
* Fix: a small two-state line machine (in-quote / not-in-quote). Once a `>` line is seen, every
|
|
94
|
+
* following line is ALSO masked until the first blank line (a line with no non-whitespace
|
|
95
|
+
* character) or end-of-input — the blank line is the ONLY terminator (a deliberate
|
|
96
|
+
* over-approximation of strict CommonMark, which also lets a block-starting line such as an ATX
|
|
97
|
+
* heading end a paragraph early: `01_requirements.md` FR-4 / acid A2 require the opposite, that a
|
|
98
|
+
* `## Findings ledger` heading glued to a quote with no blank separation stays masked, so the
|
|
99
|
+
* ambiguous case is read fail-closed as "still inside the quote" rather than as a real heading).
|
|
100
|
+
* A line already blanked by an earlier pass (fence/indented-code/HTML-comment) has no
|
|
101
|
+
* non-whitespace character either, so it is read as blank and CLOSES an open run rather than
|
|
102
|
+
* extending it — the existing already-masked invariant (line 90-93 above) holds unchanged.
|
|
103
|
+
*/
|
|
104
|
+
function maskBlockquotes(md: string): string {
|
|
105
|
+
let inQuote = false;
|
|
106
|
+
return md
|
|
107
|
+
.split('\n')
|
|
108
|
+
.map((line) => {
|
|
109
|
+
if (BLOCKQUOTE_RE.test(line)) {
|
|
110
|
+
inQuote = true;
|
|
111
|
+
return ' '.repeat(line.length);
|
|
112
|
+
}
|
|
113
|
+
if (!inQuote) return line;
|
|
114
|
+
if (!/\S/.test(line)) {
|
|
115
|
+
inQuote = false; // FR-2: a blank line always ends the lazy-continuation run
|
|
116
|
+
return line;
|
|
117
|
+
}
|
|
118
|
+
if (BLOCK_STARTER_RE.test(line)) {
|
|
119
|
+
// FR-1 clause (b), lead delta after Step 8 (F1/F2): a line that STARTS A NEW BLOCK — ATX
|
|
120
|
+
// heading, thematic break, fence, table row, list item, HTML block — interrupts the quoted
|
|
121
|
+
// paragraph per CommonMark, so it is NOT a lazy continuation: it and everything after it
|
|
122
|
+
// are outside the quote. Without this, a real `## Findings ledger` heading or a real table
|
|
123
|
+
// glued under a `> note` line vanished into a SILENT `absent` — exactly the near-miss-vs-
|
|
124
|
+
// absent failure this module exists to prevent.
|
|
125
|
+
inQuote = false;
|
|
126
|
+
return line;
|
|
127
|
+
}
|
|
128
|
+
return ' '.repeat(line.length); // FR-1: lazy continuation — still inside the quote
|
|
129
|
+
})
|
|
130
|
+
.join('\n');
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/** The one masking pass both scans below share: fenced code, 4-space indented code and HTML
|
|
134
|
+
* comments (the canonical masker) plus blockquotes (the local addition above). Blockquote masking
|
|
135
|
+
* runs SECOND on purpose — a `>` line already blanked by the canonical pass (inside a fence) is all
|
|
136
|
+
* spaces and can never spuriously match `BLOCKQUOTE_RE` again; a live blockquote outside any fence
|
|
137
|
+
* is exactly what still needs masking. */
|
|
138
|
+
/** Lead delta after Codex r2 (#1 PARTIAL): HTML `<blockquote>…</blockquote>` regions are masked too
|
|
139
|
+
* (same-length blanking, newlines kept) — a quoted example inside them must not count. */
|
|
140
|
+
function maskHtmlBlockquotes(md: string): string {
|
|
141
|
+
return md.replace(/<blockquote\b[\s\S]*?<\/blockquote>/gi, (m) => m.replace(/[^\n]/g, ' '));
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
function maskForScan(md: string): string {
|
|
145
|
+
return maskBlockquotes(maskHtmlBlockquotes(maskMarkdown(md, { indentedCode: true, inlineComments: true })));
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/** Every `QE-VERDICT:` line found, normalised, IN DOCUMENT ORDER (duplicates included — the caller
|
|
149
|
+
* decides whether repeats of the SAME grade still count as ambiguous; ADR-001 D1 says they do:
|
|
150
|
+
* "две строки — ambiguous, никогда «последняя побеждает»"). Scans MASKED text (fix-round-1 finding
|
|
151
|
+
* 1): a `QE-VERDICT:` line inside a fenced/indented code block, a blockquote or an HTML comment is
|
|
152
|
+
* an EXAMPLE, not the report's own verdict, and must never be counted. */
|
|
153
|
+
export function readQeVerdictLines(md: string): string[] {
|
|
154
|
+
const found: string[] = [];
|
|
155
|
+
const masked = maskForScan(md);
|
|
156
|
+
const re = new RegExp(QE_VERDICT_RE.source, QE_VERDICT_RE.flags.includes('m') ? 'gm' : 'g');
|
|
157
|
+
for (const m of masked.matchAll(re)) {
|
|
158
|
+
found.push(normaliseVerdictSign(m[1] as string));
|
|
159
|
+
}
|
|
160
|
+
return found;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
export interface QeVerdictInvalidLine {
|
|
164
|
+
/** 1-based line number in the ORIGINAL (unmasked) text. */
|
|
165
|
+
readonly line: number;
|
|
166
|
+
/** The raw line text, exactly as written (from the original, unmasked text). */
|
|
167
|
+
readonly text: string;
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
/**
|
|
171
|
+
* fix-round-1 finding 2: lines that LOOK LIKE a `QE-VERDICT:` declaration (any case, any leading
|
|
172
|
+
* whitespace, with or without the colon) but do not match the strict grammar — `QE-VERDICT: B – final`
|
|
173
|
+
* (en dash + trailing prose), wrong case, a missing colon, a grade outside A-D. Masked the same way as
|
|
174
|
+
* `readQeVerdictLines` — a malformed EXAMPLE inside a fence/quote/comment is still not a real attempt.
|
|
175
|
+
* A genuinely valid line is never reported here (it belongs to `readQeVerdictLines` instead).
|
|
176
|
+
*/
|
|
177
|
+
export function findInvalidQeVerdictLines(md: string): readonly QeVerdictInvalidLine[] {
|
|
178
|
+
const masked = maskForScan(md);
|
|
179
|
+
const maskedLines = masked.split('\n');
|
|
180
|
+
const originalLines = md.split('\n');
|
|
181
|
+
const strict = new RegExp(QE_VERDICT_RE.source);
|
|
182
|
+
const out: QeVerdictInvalidLine[] = [];
|
|
183
|
+
for (let i = 0; i < maskedLines.length; i++) {
|
|
184
|
+
const line = maskedLines[i] as string;
|
|
185
|
+
if (!QE_VERDICT_LOOKALIKE_RE.test(line)) continue;
|
|
186
|
+
if (strict.test(line)) continue; // grammatically valid — readQeVerdictLines already has it
|
|
187
|
+
out.push({ line: i + 1, text: originalLines[i] as string });
|
|
188
|
+
}
|
|
189
|
+
return out;
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
export interface QeFindingRow {
|
|
193
|
+
readonly finding: string;
|
|
194
|
+
readonly severity: QeSeverity;
|
|
195
|
+
readonly status: QeStatus;
|
|
196
|
+
readonly round: number;
|
|
197
|
+
readonly author: QeAuthor;
|
|
198
|
+
readonly title: string;
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
export interface QeFindingsRefusedRow {
|
|
202
|
+
/** 1-based line number in the source text the refused row (or duplicate/misplaced table header)
|
|
203
|
+
* starts at. */
|
|
204
|
+
readonly line: number;
|
|
205
|
+
/** The raw line text, exactly as written. */
|
|
206
|
+
readonly text: string;
|
|
207
|
+
/** Why the row (or table) was refused — human-readable, not a code. */
|
|
208
|
+
readonly reason: string;
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
export interface QeFindingsSummary {
|
|
212
|
+
readonly bySeverity: Readonly<Record<string, number>>;
|
|
213
|
+
readonly byStatus: Readonly<Record<string, number>>;
|
|
214
|
+
readonly total: number;
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
export type QeFindingsResult =
|
|
218
|
+
| { readonly status: 'absent' }
|
|
219
|
+
| {
|
|
220
|
+
readonly status: 'present';
|
|
221
|
+
/** Header-only table (no data rows at all, valid or refused) — worse than no table (D3). */
|
|
222
|
+
readonly hollow: boolean;
|
|
223
|
+
/** accepted | accepted-hollow (header only) | rejected-only (every table was refused) — lead delta after Codex r2 */
|
|
224
|
+
readonly tableStatus: 'accepted' | 'accepted-hollow' | 'rejected-only';
|
|
225
|
+
readonly rows: readonly QeFindingRow[];
|
|
226
|
+
readonly refused: readonly QeFindingsRefusedRow[];
|
|
227
|
+
readonly summary: QeFindingsSummary;
|
|
228
|
+
};
|
|
229
|
+
|
|
230
|
+
const SEVERITY_SET: ReadonlySet<string> = new Set(QE_SEVERITIES);
|
|
231
|
+
const STATUS_SET: ReadonlySet<string> = new Set(QE_STATUSES);
|
|
232
|
+
const AUTHOR_SET: ReadonlySet<string> = new Set(QE_AUTHORS);
|
|
233
|
+
|
|
234
|
+
/** A markdown table separator row, e.g. `|---|---|---|---|---|---|` or `| --- | :--- | ---: |`. */
|
|
235
|
+
const SEPARATOR_RE = /^\|[\s:|-]+\|$/;
|
|
236
|
+
|
|
237
|
+
/** A top-level Markdown heading (`#` through `######`, ATX-style) — used to close a `## Findings
|
|
238
|
+
* ledger` section at the next heading of any level, same as a reader would read the document. */
|
|
239
|
+
const TOP_LEVEL_HEADING_RE = /^#{1,6}\s/;
|
|
240
|
+
|
|
241
|
+
/** Splits the INSIDE of a `| a | b | c |` row on top-level `|` boundaries: a `\|` is a literal pipe
|
|
242
|
+
* (never a boundary — fix-round-1 finding 5), and a `|` inside a single-backtick inline-code span is
|
|
243
|
+
* never a boundary either (`` `a|b` `` stays one cell). Deliberately minimal — not a full CommonMark
|
|
244
|
+
* table parser; nested/nested-backtick edge cases are out of scope. */
|
|
245
|
+
function splitTableCells(inner: string): string[] {
|
|
246
|
+
const cells: string[] = [];
|
|
247
|
+
let current = '';
|
|
248
|
+
let inCode = false;
|
|
249
|
+
let codeFence = 0;
|
|
250
|
+
for (let i = 0; i < inner.length; i++) {
|
|
251
|
+
const ch = inner[i] as string;
|
|
252
|
+
if (ch === '\\' && inner[i + 1] === '|') {
|
|
253
|
+
current += '|';
|
|
254
|
+
i++;
|
|
255
|
+
continue;
|
|
256
|
+
}
|
|
257
|
+
if (ch === '`') {
|
|
258
|
+
// Lead delta after Codex r2 (#5 PARTIAL): a code span opens with N backticks and closes only
|
|
259
|
+
// with a run of the SAME length — ``a|b`` must not flip twice on its double backticks.
|
|
260
|
+
let run = 0;
|
|
261
|
+
while (inner[i + run] === '`') run++;
|
|
262
|
+
if (!inCode) { inCode = true; codeFence = run; }
|
|
263
|
+
else if (run === codeFence) { inCode = false; codeFence = 0; }
|
|
264
|
+
current += '`'.repeat(run);
|
|
265
|
+
i += run - 1;
|
|
266
|
+
continue;
|
|
267
|
+
}
|
|
268
|
+
if (ch === '|' && !inCode) {
|
|
269
|
+
cells.push(current.trim());
|
|
270
|
+
current = '';
|
|
271
|
+
continue;
|
|
272
|
+
}
|
|
273
|
+
current += ch;
|
|
274
|
+
}
|
|
275
|
+
cells.push(current.trim());
|
|
276
|
+
return cells;
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/** Splits `| a | b | c |` into `['a', 'b', 'c']`, trimmed. Returns null when the line does not open
|
|
280
|
+
* and close with a pipe (not a table row at all) or has the wrong column count for this table. */
|
|
281
|
+
function splitRow(line: string, expectedCols: number): string[] | null {
|
|
282
|
+
const t = line.trim();
|
|
283
|
+
if (!t.startsWith('|') || !t.endsWith('|')) return null;
|
|
284
|
+
const cells = splitTableCells(t.slice(1, -1));
|
|
285
|
+
if (cells.length !== expectedCols) return null;
|
|
286
|
+
return cells;
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
function parseRound(raw: string): number | null {
|
|
290
|
+
if (!/^[0-9]+$/.test(raw)) return null;
|
|
291
|
+
const n = Number(raw);
|
|
292
|
+
return n >= 1 ? n : null;
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
/** Walks a table's body from just below its header (same rules as the real parse loop: an optional
|
|
296
|
+
* separator row, then rows starting with `|` until a blank line, a new header, or a non-table line)
|
|
297
|
+
* and counts how many row lines it contains — WITHOUT validating a single cell. Used only to make a
|
|
298
|
+
* refusal (duplicate or misplaced table) NAME how much was ignored (fix-round-1 finding 6), never to
|
|
299
|
+
* parse the rejected table's content. */
|
|
300
|
+
function countIgnoredRows(lines: readonly string[], headerIdx: number): number {
|
|
301
|
+
let cursor = headerIdx + 1;
|
|
302
|
+
if (cursor < lines.length && SEPARATOR_RE.test((lines[cursor] as string).trim())) cursor++;
|
|
303
|
+
let count = 0;
|
|
304
|
+
while (cursor < lines.length) {
|
|
305
|
+
const trimmed = (lines[cursor] as string).trim();
|
|
306
|
+
if (trimmed === '') break;
|
|
307
|
+
if (trimmed === QE_FINDINGS_HEADER) break;
|
|
308
|
+
if (!trimmed.startsWith('|')) break;
|
|
309
|
+
count++;
|
|
310
|
+
cursor++;
|
|
311
|
+
}
|
|
312
|
+
return count;
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
export function parseQeFindings(md: string): QeFindingsResult {
|
|
316
|
+
const lines = md.split('\n');
|
|
317
|
+
const maskedLines = maskForScan(md).split('\n');
|
|
318
|
+
|
|
319
|
+
// fix-round-1 finding 1: locate the single top-level `## Findings ledger` heading (masked-aware —
|
|
320
|
+
// a heading quoted inside a fence/blockquote/comment does not count). Zero or more-than-one such
|
|
321
|
+
// headings means there is NO section a table can be "under", and every table found is refused.
|
|
322
|
+
const headingIdxs: number[] = [];
|
|
323
|
+
for (let i = 0; i < maskedLines.length; i++) {
|
|
324
|
+
if ((maskedLines[i] as string).trim() === QE_LEDGER_HEADING) headingIdxs.push(i);
|
|
325
|
+
}
|
|
326
|
+
let sectionStart = -1;
|
|
327
|
+
let sectionEnd = -1;
|
|
328
|
+
if (headingIdxs.length === 1) {
|
|
329
|
+
sectionStart = (headingIdxs[0] as number) + 1;
|
|
330
|
+
sectionEnd = maskedLines.length;
|
|
331
|
+
for (let i = sectionStart; i < maskedLines.length; i++) {
|
|
332
|
+
if (TOP_LEVEL_HEADING_RE.test(maskedLines[i] as string)) {
|
|
333
|
+
sectionEnd = i;
|
|
334
|
+
break;
|
|
335
|
+
}
|
|
336
|
+
}
|
|
337
|
+
}
|
|
338
|
+
const inSection = (idx: number): boolean => sectionStart >= 0 && idx >= sectionStart && idx < sectionEnd;
|
|
339
|
+
const noSectionReason =
|
|
340
|
+
headingIdxs.length === 0
|
|
341
|
+
? `no '${QE_LEDGER_HEADING}' heading in the report`
|
|
342
|
+
: headingIdxs.length > 1
|
|
343
|
+
? `${headingIdxs.length} '${QE_LEDGER_HEADING}' headings found — the heading must be unique`
|
|
344
|
+
: '';
|
|
345
|
+
|
|
346
|
+
// Header-row detection is ALSO masked-aware (finding 1): a `QE_FINDINGS_HEADER` line inside a
|
|
347
|
+
// fence/indented block/blockquote/HTML comment is example text, never even a candidate.
|
|
348
|
+
const headerIdxs: number[] = [];
|
|
349
|
+
for (let i = 0; i < maskedLines.length; i++) {
|
|
350
|
+
if ((maskedLines[i] as string).trim() === QE_FINDINGS_HEADER) headerIdxs.push(i);
|
|
351
|
+
}
|
|
352
|
+
if (headerIdxs.length === 0) {
|
|
353
|
+
// Lead delta (dogfooding the lead's own 08, 2026-09-17 02:08 UTC): a pipe-table sitting DIRECTLY
|
|
354
|
+
// under the single ledger heading whose header row is NOT the canonical one used to parse as
|
|
355
|
+
// `absent` — indistinguishable from "no ledger at all". It is a table the author MEANT as the
|
|
356
|
+
// ledger, so it is refused LOUDLY with the expected header spelled out (`rejected-only`).
|
|
357
|
+
if (sectionStart >= 0) {
|
|
358
|
+
for (let i = sectionStart; i < sectionEnd; i++) {
|
|
359
|
+
const t = (maskedLines[i] as string).trim();
|
|
360
|
+
if (t === '') continue;
|
|
361
|
+
if (t.startsWith('|')) {
|
|
362
|
+
return {
|
|
363
|
+
status: 'present',
|
|
364
|
+
hollow: false,
|
|
365
|
+
tableStatus: 'rejected-only',
|
|
366
|
+
rows: [],
|
|
367
|
+
refused: [{ line: i + 1, text: lines[i] as string, reason: `non-canonical header row: expected exactly '${QE_FINDINGS_HEADER}'` }],
|
|
368
|
+
summary: { bySeverity: {}, byStatus: {}, total: 0 },
|
|
369
|
+
};
|
|
370
|
+
}
|
|
371
|
+
break;
|
|
372
|
+
}
|
|
373
|
+
}
|
|
374
|
+
return { status: 'absent' };
|
|
375
|
+
}
|
|
376
|
+
|
|
377
|
+
const rows: QeFindingRow[] = [];
|
|
378
|
+
const refused: QeFindingsRefusedRow[] = [];
|
|
379
|
+
let sawAnyRowLine = false;
|
|
380
|
+
let acceptedOne = false;
|
|
381
|
+
|
|
382
|
+
// Lead delta after Codex r2 (#1 PARTIAL): the table must sit DIRECTLY under the heading — only
|
|
383
|
+
// blank lines between `## Findings ledger` and the header row; prose in between disqualifies it.
|
|
384
|
+
const directlyUnderHeading = (idx: number): boolean => {
|
|
385
|
+
if (sectionStart < 0 || idx < sectionStart) return false;
|
|
386
|
+
for (let i = sectionStart; i < idx; i++) if ((maskedLines[i] as string).trim() !== '') return false;
|
|
387
|
+
return true;
|
|
388
|
+
};
|
|
389
|
+
for (const headerIdx of headerIdxs) {
|
|
390
|
+
const accept = !acceptedOne && inSection(headerIdx) && directlyUnderHeading(headerIdx);
|
|
391
|
+
if (!accept) {
|
|
392
|
+
const ignored = countIgnoredRows(lines, headerIdx);
|
|
393
|
+
const ignoredNote = ignored > 0 ? ` (${ignored} row(s) ignored)` : '';
|
|
394
|
+
const reason = inSection(headerIdx) && acceptedOne
|
|
395
|
+
? `duplicate table: a Findings ledger table already appeared earlier in this report${ignoredNote}`
|
|
396
|
+
: inSection(headerIdx)
|
|
397
|
+
? `not directly under the heading: prose sits between '${QE_LEDGER_HEADING}' and the table${ignoredNote}`
|
|
398
|
+
: `outside ledger section: ${noSectionReason || `must appear directly under the single '${QE_LEDGER_HEADING}' heading`}${ignoredNote}`;
|
|
399
|
+
refused.push({ line: headerIdx + 1, text: lines[headerIdx] as string, reason });
|
|
400
|
+
continue;
|
|
401
|
+
}
|
|
402
|
+
acceptedOne = true;
|
|
403
|
+
|
|
404
|
+
let cursor = headerIdx + 1;
|
|
405
|
+
// Optional separator row right after the header.
|
|
406
|
+
if (cursor < lines.length && SEPARATOR_RE.test((lines[cursor] as string).trim())) cursor++;
|
|
407
|
+
|
|
408
|
+
while (cursor < lines.length) {
|
|
409
|
+
const raw = lines[cursor] as string;
|
|
410
|
+
const trimmed = raw.trim();
|
|
411
|
+
if (trimmed === '') break; // a blank line ends the table
|
|
412
|
+
if (trimmed === QE_FINDINGS_HEADER) break; // a second header starts a NEW (duplicate) table
|
|
413
|
+
if (!trimmed.startsWith('|')) break; // a non-table line ends the table
|
|
414
|
+
sawAnyRowLine = true;
|
|
415
|
+
const cells = splitRow(raw, 6);
|
|
416
|
+
if (cells === null) {
|
|
417
|
+
refused.push({ line: cursor + 1, text: raw, reason: 'malformed row: expected 6 columns' });
|
|
418
|
+
cursor++;
|
|
419
|
+
continue;
|
|
420
|
+
}
|
|
421
|
+
const [finding, severity, status, roundRaw, author, title] = cells as [string, string, string, string, string, string];
|
|
422
|
+
const reasons: string[] = [];
|
|
423
|
+
if (finding === '' || /\s/.test(finding)) reasons.push(`Finding id ${JSON.stringify(finding)} must be a non-empty token with no whitespace`);
|
|
424
|
+
if (!SEVERITY_SET.has(severity)) reasons.push(`severity ${JSON.stringify(severity)} not in dictionary`);
|
|
425
|
+
if (!STATUS_SET.has(status)) reasons.push(`status ${JSON.stringify(status)} not in dictionary`);
|
|
426
|
+
const round = parseRound(roundRaw);
|
|
427
|
+
if (round === null) reasons.push(`round ${JSON.stringify(roundRaw)} is not an integer >= 1`);
|
|
428
|
+
if (!AUTHOR_SET.has(author)) reasons.push(`author ${JSON.stringify(author)} not in dictionary`);
|
|
429
|
+
if (reasons.length > 0) {
|
|
430
|
+
refused.push({ line: cursor + 1, text: raw, reason: reasons.join('; ') });
|
|
431
|
+
} else {
|
|
432
|
+
rows.push({
|
|
433
|
+
finding,
|
|
434
|
+
severity: severity as QeSeverity,
|
|
435
|
+
status: status as QeStatus,
|
|
436
|
+
round: round as number,
|
|
437
|
+
author: author as QeAuthor,
|
|
438
|
+
title,
|
|
439
|
+
});
|
|
440
|
+
}
|
|
441
|
+
cursor++;
|
|
442
|
+
}
|
|
443
|
+
}
|
|
444
|
+
|
|
445
|
+
const bySeverity: Record<string, number> = {};
|
|
446
|
+
const byStatus: Record<string, number> = {};
|
|
447
|
+
for (const r of rows) {
|
|
448
|
+
bySeverity[r.severity] = (bySeverity[r.severity] ?? 0) + 1;
|
|
449
|
+
byStatus[r.status] = (byStatus[r.status] ?? 0) + 1;
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
// Lead delta after Codex r2 (new MEDIUM #2): `hollow` means an ACCEPTED table with no rows — a
|
|
453
|
+
// report whose only tables were refused has no accepted table at all (`rejected-only`).
|
|
454
|
+
const tableStatus: 'accepted' | 'accepted-hollow' | 'rejected-only' = !acceptedOne ? 'rejected-only' : (sawAnyRowLine ? 'accepted' : 'accepted-hollow');
|
|
455
|
+
return {
|
|
456
|
+
status: 'present',
|
|
457
|
+
hollow: tableStatus === 'accepted-hollow',
|
|
458
|
+
tableStatus,
|
|
459
|
+
rows,
|
|
460
|
+
refused,
|
|
461
|
+
summary: { bySeverity, byStatus, total: rows.length },
|
|
462
|
+
};
|
|
463
|
+
}
|
package/src/recap.ts
CHANGED
|
@@ -227,7 +227,12 @@ export const FORBIDDEN_METRICS: readonly ForbiddenMetric[] = [
|
|
|
227
227
|
*/
|
|
228
228
|
export type Delivery =
|
|
229
229
|
| { readonly slug: string; readonly createdIso: string; readonly gradeStatus: 'unique'; readonly grade: string }
|
|
230
|
-
|
|
230
|
+
// fix-round-1 (codex-r1-verdict finding 2/3): `readQeGrade` gained an `'invalid'` status — a
|
|
231
|
+
// report that ATTEMPTED a machine `QE-VERDICT:` line and got the grammar wrong. It is counted
|
|
232
|
+
// alongside `'ambiguous'` below (both mean "no single grade can be trusted", for different
|
|
233
|
+
// reasons) rather than getting a fifth bucket in the printed line — the type keeps the real
|
|
234
|
+
// status, `buildRecap` chooses how to group it for display.
|
|
235
|
+
| { readonly slug: string; readonly createdIso: string; readonly gradeStatus: 'ambiguous' | 'invalid' | 'none' | 'no-report'; readonly grade: null };
|
|
231
236
|
|
|
232
237
|
/** A `unique` with no usable grade is not a graded delivery; it is a report we could not read. */
|
|
233
238
|
function normaliseDelivery(d: Delivery): Delivery {
|
|
@@ -346,14 +351,16 @@ export function buildRecap(facts: RecapFacts): RecapReport {
|
|
|
346
351
|
const covered = coveredWindow(v, facts.window);
|
|
347
352
|
const items = (facts.deliveries?.items ?? []).map(normaliseDelivery).filter((d) => withinWindow(covered, d.createdIso));
|
|
348
353
|
const graded = items.filter((d): d is Extract<Delivery, { gradeStatus: 'unique' }> => d.gradeStatus === 'unique');
|
|
349
|
-
|
|
354
|
+
// fix-round-1: 'invalid' (a malformed QE-VERDICT attempt) is folded into the same printed bucket
|
|
355
|
+
// as 'ambiguous' — both are "no single grade can be trusted", just a different reason.
|
|
356
|
+
const ambiguous = items.filter((d) => d.gradeStatus === 'ambiguous' || d.gradeStatus === 'invalid');
|
|
350
357
|
const ungraded = items.filter((d) => d.gradeStatus === 'none' || d.gradeStatus === 'no-report');
|
|
351
358
|
const lines = sectionLines(v, facts.window, (scope) => items.length === 0
|
|
352
359
|
? emptyOrUnavailable(v, `no feature directories were created in ${scope}`)
|
|
353
360
|
: [
|
|
354
361
|
`${items.length} feature director${items.length === 1 ? 'y' : 'ies'} created in ${scope}`,
|
|
355
362
|
`${graded.length} carry a grade an independent review stated unambiguously${graded.length > 0 ? `: ${tally(graded.map((d) => d.grade))}` : ''}`,
|
|
356
|
-
`${ambiguous.length} have a report that states MORE THAN ONE grade — reported as ambiguous, never guessed`,
|
|
363
|
+
`${ambiguous.length} have a report that states MORE THAN ONE grade or a malformed QE-VERDICT line — reported as ambiguous, never guessed`,
|
|
357
364
|
`${ungraded.length} have no letter grade in their report, or no report at all`,
|
|
358
365
|
'cadence is not value: this counts deliveries, not what they were worth',
|
|
359
366
|
]);
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* review-cost-ledger T1/FR-1/A1 (ADR-001 п.1): a pure parser for the qe-bridge reviewer's own cost
|
|
3
|
+
* receipt — the last non-empty line of the Claude CLI's raw stdout, which (for every observed
|
|
4
|
+
* signoff) is a JSON object carrying `total_cost_usd`, `usage.*`, `duration_ms`, `num_turns`.
|
|
5
|
+
*
|
|
6
|
+
* Deliberately narrow and honest: absence of a price is never silently read as a price of zero.
|
|
7
|
+
* `status:'absent'` (no non-empty line at all) and `status:'unparseable'` (a line that IS present
|
|
8
|
+
* but does not parse into a usable cost) are each their own outcome — never collapsed into `ok`, and
|
|
9
|
+
* never guessed. This module owns no filesystem access (NFR-2): the caller (the cli) reads the
|
|
10
|
+
* stdout sidecar file and hands its TEXT in here.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
/** The four cost-bearing token components, plus their sum. `tokensPartial` is present (`true`) only
|
|
14
|
+
* when at least one component was missing/unusable in the source JSON and was substituted with 0 —
|
|
15
|
+
* never silently blended with a genuine zero. */
|
|
16
|
+
export interface QeBridgeCostTokens {
|
|
17
|
+
readonly input: number;
|
|
18
|
+
readonly output: number;
|
|
19
|
+
readonly cacheCreation: number;
|
|
20
|
+
readonly cacheRead: number;
|
|
21
|
+
readonly total: number;
|
|
22
|
+
readonly tokensPartial?: true;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* `ok` — a usable price was found. `absent` — no non-empty line to read at all (an empty/whitespace
|
|
27
|
+
* stdout, or a caller who never had one to offer); `reason` is OPTIONAL here because the parser
|
|
28
|
+
* itself never has anything more specific to say about a literal absence, but a CALLER (the cli's
|
|
29
|
+
* `readQeBridgeCostSidecar`, T3) may attach its own known reason to a synthesized `absent` (e.g. "no
|
|
30
|
+
* qe-bridge signoff for this round") without inventing a THIRD status for that case. `unparseable` —
|
|
31
|
+
* a non-empty last line existed but did not yield a valid cost (not JSON, not an object, no numeric
|
|
32
|
+
* `total_cost_usd`, or a negative/NaN one) — `reason` is mandatory here, naming what failed.
|
|
33
|
+
*/
|
|
34
|
+
export type QeBridgeCost =
|
|
35
|
+
| {
|
|
36
|
+
readonly status: 'ok';
|
|
37
|
+
readonly costUsd: number;
|
|
38
|
+
readonly tokens: QeBridgeCostTokens;
|
|
39
|
+
readonly durationMs: number | null;
|
|
40
|
+
readonly numTurns: number | null;
|
|
41
|
+
}
|
|
42
|
+
| { readonly status: 'absent'; readonly reason?: string }
|
|
43
|
+
| { readonly status: 'unparseable'; readonly reason: string };
|
|
44
|
+
|
|
45
|
+
function finiteNonNegative(value: unknown): value is number {
|
|
46
|
+
return typeof value === 'number' && Number.isFinite(value) && value >= 0;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
function finiteNumberOrNull(value: unknown): number | null {
|
|
50
|
+
return typeof value === 'number' && Number.isFinite(value) ? value : null;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/** fix-round-1 #6 (Codex r1 HIGH #6): a token COMPONENT is a count, never a measurement — it is only
|
|
54
|
+
* ever trustworthy as a nonnegative SAFE integer. `1e308` is `typeof 'number'` and finite, but adding
|
|
55
|
+
* four of those silently produces `Infinity` (which then serializes to `null` on the ledger row while
|
|
56
|
+
* the in-memory type still claims `number`); a fractional or >2^53 value is likewise not a real token
|
|
57
|
+
* count. Distinct from "component ABSENT" (undefined — stays the existing partial-with-zero path). */
|
|
58
|
+
function safeNonNegativeInteger(value: unknown): value is number {
|
|
59
|
+
return typeof value === 'number' && Number.isSafeInteger(value) && value >= 0;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* `text` is the full raw stdout of a qe-bridge reviewer invocation (Claude CLI, `--output-format
|
|
64
|
+
* json`-shaped result line). The COST line is always the LAST non-empty line — Claude CLI streams
|
|
65
|
+
* intermediate events first and finishes with one `type:"result"` object carrying the run's totals.
|
|
66
|
+
*/
|
|
67
|
+
export function parseQeBridgeStdoutCost(text: string): QeBridgeCost {
|
|
68
|
+
const lines = text.split('\n');
|
|
69
|
+
let lastLine = '';
|
|
70
|
+
for (let i = lines.length - 1; i >= 0; i -= 1) {
|
|
71
|
+
const candidate = lines[i]!.trim();
|
|
72
|
+
if (candidate !== '') { lastLine = candidate; break; }
|
|
73
|
+
}
|
|
74
|
+
if (lastLine === '') return { status: 'absent' };
|
|
75
|
+
|
|
76
|
+
let parsed: unknown;
|
|
77
|
+
try {
|
|
78
|
+
parsed = JSON.parse(lastLine);
|
|
79
|
+
} catch (error) {
|
|
80
|
+
return { status: 'unparseable', reason: `last non-empty line is not valid JSON: ${error instanceof Error ? error.message : String(error)}` };
|
|
81
|
+
}
|
|
82
|
+
if (typeof parsed !== 'object' || parsed === null || Array.isArray(parsed)) {
|
|
83
|
+
return { status: 'unparseable', reason: 'last non-empty line did not parse to a JSON object' };
|
|
84
|
+
}
|
|
85
|
+
const rec = parsed as Record<string, unknown>;
|
|
86
|
+
const totalCostUsd = rec['total_cost_usd'];
|
|
87
|
+
if (!finiteNonNegative(totalCostUsd)) {
|
|
88
|
+
return {
|
|
89
|
+
status: 'unparseable',
|
|
90
|
+
reason: typeof totalCostUsd === 'number'
|
|
91
|
+
? `total_cost_usd is not a finite non-negative number: ${totalCostUsd}`
|
|
92
|
+
: `total_cost_usd field is missing or not a number (got ${JSON.stringify(totalCostUsd)})`,
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
const usageRaw = rec['usage'];
|
|
97
|
+
const usage = typeof usageRaw === 'object' && usageRaw !== null && !Array.isArray(usageRaw)
|
|
98
|
+
? (usageRaw as Record<string, unknown>)
|
|
99
|
+
: {};
|
|
100
|
+
let tokensPartial = false;
|
|
101
|
+
// fix-round-1 #6: a MISSING component (undefined) is the pre-existing partial-with-zero case
|
|
102
|
+
// (never touched by this fix); a PRESENT-but-invalid component (adversarial: 1e308, a fraction,
|
|
103
|
+
// 2^53+1, negative, NaN) is a different failure — the whole result turns `unparseable` rather than
|
|
104
|
+
// silently zeroing a value that was actually there. `invalidComponent` names which key failed.
|
|
105
|
+
let invalidComponent: string | null = null;
|
|
106
|
+
const component = (key: string): number => {
|
|
107
|
+
const value = usage[key];
|
|
108
|
+
if (value === undefined) { tokensPartial = true; return 0; }
|
|
109
|
+
if (safeNonNegativeInteger(value)) return value;
|
|
110
|
+
invalidComponent = key;
|
|
111
|
+
return 0;
|
|
112
|
+
};
|
|
113
|
+
const input = component('input_tokens');
|
|
114
|
+
const output = component('output_tokens');
|
|
115
|
+
const cacheCreation = component('cache_creation_input_tokens');
|
|
116
|
+
const cacheRead = component('cache_read_input_tokens');
|
|
117
|
+
if (invalidComponent !== null) {
|
|
118
|
+
return { status: 'unparseable', reason: `usage.${invalidComponent} is not a nonnegative safe integer: ${JSON.stringify(usage[invalidComponent])}` };
|
|
119
|
+
}
|
|
120
|
+
const total = input + output + cacheCreation + cacheRead;
|
|
121
|
+
if (!Number.isSafeInteger(total)) {
|
|
122
|
+
return { status: 'unparseable', reason: `token total ${total} exceeds Number.MAX_SAFE_INTEGER — not a trustworthy count` };
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
return {
|
|
126
|
+
status: 'ok',
|
|
127
|
+
costUsd: totalCostUsd,
|
|
128
|
+
tokens: {
|
|
129
|
+
input,
|
|
130
|
+
output,
|
|
131
|
+
cacheCreation,
|
|
132
|
+
cacheRead,
|
|
133
|
+
total,
|
|
134
|
+
...(tokensPartial ? { tokensPartial: true as const } : {}),
|
|
135
|
+
},
|
|
136
|
+
durationMs: finiteNumberOrNull(rec['duration_ms']),
|
|
137
|
+
numTurns: finiteNumberOrNull(rec['num_turns']),
|
|
138
|
+
};
|
|
139
|
+
}
|