@dzhechkov/harness-core 0.8.36 → 0.8.38
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +216 -76
- package/README.md +349 -8
- package/dist/agentdb-index.d.ts +87 -7
- package/dist/agentdb-index.d.ts.map +1 -1
- package/dist/agentdb-index.js +416 -57
- package/dist/agentdb-index.js.map +1 -1
- package/dist/apply-leg.d.ts +19 -1
- package/dist/apply-leg.d.ts.map +1 -1
- package/dist/apply-leg.js +187 -36
- package/dist/apply-leg.js.map +1 -1
- package/dist/codex-rollouts.d.ts +118 -0
- package/dist/codex-rollouts.d.ts.map +1 -0
- package/dist/codex-rollouts.js +297 -0
- package/dist/codex-rollouts.js.map +1 -0
- package/dist/cost-ledger.d.ts +56 -4
- package/dist/cost-ledger.d.ts.map +1 -1
- package/dist/cost-ledger.js +176 -20
- package/dist/cost-ledger.js.map +1 -1
- package/dist/cross-family-control.d.ts +380 -0
- package/dist/cross-family-control.d.ts.map +1 -0
- package/dist/cross-family-control.js +848 -0
- package/dist/cross-family-control.js.map +1 -0
- package/dist/debt-ratchet.d.ts +53 -0
- package/dist/debt-ratchet.d.ts.map +1 -0
- package/dist/debt-ratchet.js +107 -0
- package/dist/debt-ratchet.js.map +1 -0
- package/dist/embedding-config.d.ts +42 -0
- package/dist/embedding-config.d.ts.map +1 -1
- package/dist/embedding-config.js +106 -10
- package/dist/embedding-config.js.map +1 -1
- package/dist/feature-adr-checkpoints.d.ts +6 -0
- package/dist/feature-adr-checkpoints.d.ts.map +1 -1
- package/dist/feature-adr-checkpoints.js +29 -0
- package/dist/feature-adr-checkpoints.js.map +1 -1
- package/dist/feature-adr-decision-recall.d.ts +2 -2
- package/dist/feature-adr-decision-recall.d.ts.map +1 -1
- package/dist/feature-adr-decision-recall.js +5 -3
- package/dist/feature-adr-decision-recall.js.map +1 -1
- package/dist/feature-adr-envelope.d.ts +96 -0
- package/dist/feature-adr-envelope.d.ts.map +1 -0
- package/dist/feature-adr-envelope.js +183 -0
- package/dist/feature-adr-envelope.js.map +1 -0
- package/dist/feature-adr-routing.d.ts +64 -0
- package/dist/feature-adr-routing.d.ts.map +1 -1
- package/dist/feature-adr-routing.js +133 -3
- package/dist/feature-adr-routing.js.map +1 -1
- package/dist/feature-adr-stage-canon.d.ts +79 -0
- package/dist/feature-adr-stage-canon.d.ts.map +1 -0
- package/dist/feature-adr-stage-canon.js +117 -0
- package/dist/feature-adr-stage-canon.js.map +1 -0
- package/dist/index.d.ts +21 -9
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +15 -5
- package/dist/index.js.map +1 -1
- package/dist/loop-blobs.generated.js +4 -4
- package/dist/loop-blobs.generated.js.map +1 -1
- package/dist/mutation-gate.d.ts +51 -0
- package/dist/mutation-gate.d.ts.map +1 -1
- package/dist/mutation-gate.js +295 -0
- package/dist/mutation-gate.js.map +1 -1
- package/dist/qe-bridge.d.ts +8 -0
- package/dist/qe-bridge.d.ts.map +1 -1
- package/dist/qe-bridge.js +4 -2
- package/dist/qe-bridge.js.map +1 -1
- package/dist/qe-findings.d.ts +107 -0
- package/dist/qe-findings.d.ts.map +1 -0
- package/dist/qe-findings.js +417 -0
- package/dist/qe-findings.js.map +1 -0
- package/dist/recap.d.ts +1 -1
- package/dist/recap.d.ts.map +1 -1
- package/dist/recap.js +4 -2
- package/dist/recap.js.map +1 -1
- package/dist/review-cost.d.ts +51 -0
- package/dist/review-cost.d.ts.map +1 -0
- package/dist/review-cost.js +110 -0
- package/dist/review-cost.js.map +1 -0
- package/dist/round.d.ts +207 -1
- package/dist/round.d.ts.map +1 -1
- package/dist/round.js +321 -4
- package/dist/round.js.map +1 -1
- package/dist/run-records.d.ts +97 -0
- package/dist/run-records.d.ts.map +1 -1
- package/dist/run-records.js +336 -2
- package/dist/run-records.js.map +1 -1
- package/dist/score.d.ts +44 -1
- package/dist/score.d.ts.map +1 -1
- package/dist/score.js +78 -5
- package/dist/score.js.map +1 -1
- package/package.json +1 -1
- package/sbom.json +425 -75
- package/src/agentdb-index.ts +423 -60
- package/src/apply-leg.ts +187 -36
- package/src/codex-rollouts.ts +374 -0
- package/src/cost-ledger.ts +232 -24
- package/src/cross-family-control.ts +1038 -0
- package/src/debt-ratchet.ts +143 -0
- package/src/embedding-config.ts +131 -10
- package/src/feature-adr-checkpoints.ts +29 -0
- package/src/feature-adr-decision-recall.ts +6 -4
- package/src/feature-adr-envelope.ts +242 -0
- package/src/feature-adr-routing.ts +150 -3
- package/src/feature-adr-stage-canon.ts +141 -0
- package/src/index.ts +65 -6
- package/src/loop-blobs.generated.ts +4 -4
- package/src/mutation-gate.ts +316 -0
- package/src/qe-bridge.ts +12 -2
- package/src/qe-findings.ts +463 -0
- package/src/recap.ts +10 -3
- package/src/review-cost.ts +139 -0
- package/src/round.ts +481 -6
- package/src/run-records.ts +388 -2
- package/src/score.ts +115 -6
package/src/cost-ledger.ts
CHANGED
|
@@ -52,6 +52,8 @@ import { existsSync, lstatSync, mkdirSync, readFileSync, readdirSync, realpathSy
|
|
|
52
52
|
import { dirname, join } from 'node:path';
|
|
53
53
|
|
|
54
54
|
import { hasKnownPricing, usageCost } from './cost-scoring.js';
|
|
55
|
+
import { CANONICAL_STAGES, canonicalStage } from './feature-adr-stage-canon.js';
|
|
56
|
+
import type { CanonicalStage } from './feature-adr-stage-canon.js';
|
|
55
57
|
import { claudeProjectsRoot, rawTokenMixOf, weightedTokensOf } from './usage.js';
|
|
56
58
|
|
|
57
59
|
// ── Scope + vocabulary ──────────────────────────────────────
|
|
@@ -91,14 +93,21 @@ export const COST_LEDGER_DEFECT_KINDS: readonly CostLedgerDefectKind[] = [
|
|
|
91
93
|
];
|
|
92
94
|
|
|
93
95
|
/**
|
|
94
|
-
*
|
|
95
|
-
* `verdict !== 'DEFECT'` as "reconciled" (ADR-003).
|
|
96
|
+
* Four values, not two (measurement-integrity ADR-001 D2 grew this from three). `INSUFFICIENT_DATA`
|
|
97
|
+
* is NOT success: a caller must not read `verdict !== 'DEFECT'` as "reconciled" (ADR-003).
|
|
98
|
+
* `INCOMPLETE_INVENTORY` is likewise not success — it names the SPECIFIC case where every unaccounted
|
|
99
|
+
* token traces to a named orphan transcript (a directory listing the record's own inventory does not
|
|
100
|
+
* know about), with no OTHER attribution defect present. It outranks `BALANCED` (an orphan transcript
|
|
101
|
+
* can never read as balanced) and is outranked by `DEFECT` (a genuine attribution defect — double
|
|
102
|
+
* counting, a foreign sample, spend unaccounted for reasons OTHER than a named orphan — is a worse
|
|
103
|
+
* finding than "the inventory is incomplete but everything it does have reconciles").
|
|
96
104
|
*/
|
|
97
|
-
export type CostLedgerVerdict = 'BALANCED' | 'DEFECT' | 'INSUFFICIENT_DATA';
|
|
105
|
+
export type CostLedgerVerdict = 'BALANCED' | 'DEFECT' | 'INCOMPLETE_INVENTORY' | 'INSUFFICIENT_DATA';
|
|
98
106
|
|
|
99
107
|
export const COST_LEDGER_VERDICTS: readonly CostLedgerVerdict[] = [
|
|
100
108
|
'BALANCED',
|
|
101
109
|
'DEFECT',
|
|
110
|
+
'INCOMPLETE_INVENTORY',
|
|
102
111
|
'INSUFFICIENT_DATA',
|
|
103
112
|
];
|
|
104
113
|
|
|
@@ -168,6 +177,17 @@ export interface CostLedgerRow {
|
|
|
168
177
|
readonly slug: string | null;
|
|
169
178
|
/** `stageLabel()` output, verbatim. */
|
|
170
179
|
readonly stage: string;
|
|
180
|
+
/** measurement-integrity FR-2/ADR-001 D1: the canonical bucket `stage` classifies into, NEXT TO
|
|
181
|
+
* the untouched verbatim `stage` — never a replacement for it. `'unknown'` when no rule matches. */
|
|
182
|
+
readonly stageCanonical: CanonicalStage | 'unknown';
|
|
183
|
+
/** measurement-integrity FR-4: 1-based position of THIS occurrence among stage entries that share
|
|
184
|
+
* `stage`'s exact label, in record order — `attempt: 1, attempts: 1` when the label occurs once. */
|
|
185
|
+
readonly attempt: number;
|
|
186
|
+
/** Total number of stage entries in this run that share `stage`'s exact label. A repeated label no
|
|
187
|
+
* longer merges silently into one row (ADR-001 D... measurement-integrity FR-4): each occurrence
|
|
188
|
+
* is its OWN row, and every one of them carries the same `attempts` count so a reader grouping by
|
|
189
|
+
* `stage` can tell there were several without re-deriving it. */
|
|
190
|
+
readonly attempts: number;
|
|
171
191
|
readonly phase: string | null;
|
|
172
192
|
/** The stage's model id, or `'mixed'` when several agents share a label with different models. */
|
|
173
193
|
readonly model: string;
|
|
@@ -214,6 +234,18 @@ export interface CostLedgerReconciliation {
|
|
|
214
234
|
readonly identityHolds: boolean;
|
|
215
235
|
readonly verdict: CostLedgerVerdict;
|
|
216
236
|
readonly defects: readonly CostLedgerDefect[];
|
|
237
|
+
/** measurement-integrity FR-3/ADR-001 D2: transcripts present in the run's directory that have NO
|
|
238
|
+
* `workflowProgress[]` entry in the record — named explicitly rather than dissolved into a generic
|
|
239
|
+
* `unaccountedTokens` remainder. `method: 'per-transcript'` means `tokens` is an exact sum over
|
|
240
|
+
* each orphan's own extracted samples; `'count-fallback'` means only the COUNT (and, where
|
|
241
|
+
* available, the ids) is known and `tokens` falls back to the reconciliation's own
|
|
242
|
+
* `unaccountedTokens` as the best available estimate — always present, always additive. */
|
|
243
|
+
readonly orphanTranscripts: {
|
|
244
|
+
readonly count: number;
|
|
245
|
+
readonly tokens: number;
|
|
246
|
+
readonly ids: readonly string[];
|
|
247
|
+
readonly method: 'per-transcript' | 'count-fallback';
|
|
248
|
+
};
|
|
217
249
|
}
|
|
218
250
|
|
|
219
251
|
export interface CostLedgerReport {
|
|
@@ -223,6 +255,12 @@ export interface CostLedgerReport {
|
|
|
223
255
|
readonly status: string | null;
|
|
224
256
|
readonly startedTs: string | null;
|
|
225
257
|
readonly rows: readonly CostLedgerRow[];
|
|
258
|
+
/** measurement-integrity FR-2/T1: rows re-aggregated by `stageCanonical`, plus two synthetic
|
|
259
|
+
* buckets — `unknown` (rows whose verbatim label matched no canon rule) and `unattributed` (the
|
|
260
|
+
* orphan-transcript tokens from `reconciliation.orphanTranscripts`, which belong to no stage row
|
|
261
|
+
* at all). Every one of the 12 {@link CanonicalStage} values is always present, even at zero, so a
|
|
262
|
+
* reader can iterate a stable key set. */
|
|
263
|
+
readonly byCanonicalStage: Record<CanonicalStage | 'unknown' | 'unattributed', { readonly tokens: number; readonly agents: number; readonly attempts: number }>;
|
|
226
264
|
readonly reconciliation: CostLedgerReconciliation;
|
|
227
265
|
/** The record's cached raw sum — reported, never the invariant's right-hand side (ADR-002). */
|
|
228
266
|
readonly recordTotalTokens: number | null;
|
|
@@ -412,8 +450,17 @@ export interface BuildCostLedgerInput {
|
|
|
412
450
|
readonly stageSamples: readonly { readonly agentId: string; readonly samples: readonly CostLedgerSample[] }[];
|
|
413
451
|
/** RIGHT side — the dedup-union over the run's transcript DIRECTORY (ADR-002). */
|
|
414
452
|
readonly runSamples: readonly CostLedgerSample[];
|
|
415
|
-
/** Agent transcripts present in the run directory with no `workflowProgress[]` entry.
|
|
453
|
+
/** Agent transcripts present in the run directory with no `workflowProgress[]` entry — ids only.
|
|
454
|
+
* Kept for the `Unaccounted` defect's `subjects` listing; superseded by `orphanTranscripts` below
|
|
455
|
+
* for the FR-3 `orphanTranscripts.tokens` figure whenever the caller can supply per-orphan samples. */
|
|
416
456
|
readonly orphanAgentIds?: readonly string[];
|
|
457
|
+
/** measurement-integrity FR-3: the SAME orphan transcripts as `orphanAgentIds`, but carrying each
|
|
458
|
+
* one's own extracted samples so `reconciliation.orphanTranscripts.tokens` is an EXACT sum rather
|
|
459
|
+
* than a derived remainder. When omitted, `orphanAgentIds` alone still produces a named
|
|
460
|
+
* `orphanTranscripts` entry (`method: 'count-fallback'`, `tokens` = the reconciliation's own
|
|
461
|
+
* `unaccountedTokens` as the best available estimate) — the count/ids are never lost even without
|
|
462
|
+
* the precise per-transcript figure. */
|
|
463
|
+
readonly orphanTranscripts?: readonly { readonly agentId: string; readonly samples: readonly CostLedgerSample[] }[];
|
|
417
464
|
/** Fraction of the run total tolerated as unaccounted. Default {@link DEFAULT_COST_LEDGER_EPSILON}. */
|
|
418
465
|
readonly epsilon?: number;
|
|
419
466
|
/** True when the transcript listing hit the file cap — the run total is incomplete (Codex QE HIGH). */
|
|
@@ -453,6 +500,9 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
|
|
|
453
500
|
...input,
|
|
454
501
|
runSamples: input.runSamples.map(clampSample),
|
|
455
502
|
stageSamples: input.stageSamples.map((e) => ({ agentId: e.agentId, samples: e.samples.map(clampSample) })),
|
|
503
|
+
...(input.orphanTranscripts !== undefined
|
|
504
|
+
? { orphanTranscripts: input.orphanTranscripts.map((e) => ({ agentId: e.agentId, samples: e.samples.map(clampSample) })) }
|
|
505
|
+
: {}),
|
|
456
506
|
};
|
|
457
507
|
|
|
458
508
|
// RIGHT — the run's universe, deduped by sample key.
|
|
@@ -467,6 +517,8 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
|
|
|
467
517
|
|
|
468
518
|
interface Bucket {
|
|
469
519
|
stage: string;
|
|
520
|
+
attempt: number;
|
|
521
|
+
attempts: number;
|
|
470
522
|
phase: string | null;
|
|
471
523
|
models: Set<string>;
|
|
472
524
|
agentIds: string[];
|
|
@@ -478,7 +530,15 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
|
|
|
478
530
|
costUsd: number;
|
|
479
531
|
pricingKnown: boolean;
|
|
480
532
|
}
|
|
533
|
+
// measurement-integrity FR-4: a bucket is now keyed by (label, occurrence) rather than by label
|
|
534
|
+
// alone — a label repeated N times in `record.stages` (a retried stage) produces N buckets, each
|
|
535
|
+
// its own row, instead of one row silently summing N attempts together. `labelTotalCount` is a
|
|
536
|
+
// first pass so every occurrence's row can carry the SAME `attempts` total, including the first.
|
|
537
|
+
const labelTotalCount = new Map<string, number>();
|
|
538
|
+
for (const stage of record.stages) labelTotalCount.set(stage.label, (labelTotalCount.get(stage.label) ?? 0) + 1);
|
|
539
|
+
const labelSeen = new Map<string, number>();
|
|
481
540
|
const buckets = new Map<string, Bucket>();
|
|
541
|
+
const bucketOrder: string[] = [];
|
|
482
542
|
const keyOwners = new Map<string, Set<string>>();
|
|
483
543
|
const foreign: string[] = [];
|
|
484
544
|
const conflicting: string[] = [];
|
|
@@ -489,10 +549,23 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
|
|
|
489
549
|
const samples = byAgent.get(stage.agentId) ?? [];
|
|
490
550
|
if (samples.length === 0) missingTranscript.push(`${stage.label} (${stage.agentId})`);
|
|
491
551
|
|
|
492
|
-
|
|
552
|
+
const attempt = (labelSeen.get(stage.label) ?? 0) + 1;
|
|
553
|
+
labelSeen.set(stage.label, attempt);
|
|
554
|
+
const attempts = labelTotalCount.get(stage.label) ?? 1;
|
|
555
|
+
// measurement-integrity fix-round-1/F11 (Codex r1 MEDIUM #11): the key used to be a NUL-
|
|
556
|
+
// delimited template literal (`${stage.label}\0${attempt}`) - readable in an editor as a plain
|
|
557
|
+
// space because NUL renders invisibly, but NUL is a LEGAL JSON-string character (the same
|
|
558
|
+
// delimiter-ambiguity class `stageCostAggregates` below already fixed for its own key). An
|
|
559
|
+
// unambiguous JSON-tuple serialization removes the theoretical collision outright, and matches
|
|
560
|
+
// the pattern already used two functions down in this file.
|
|
561
|
+
const bucketKey = JSON.stringify([stage.label, attempt]);
|
|
562
|
+
|
|
563
|
+
let b = buckets.get(bucketKey);
|
|
493
564
|
if (b === undefined) {
|
|
494
565
|
b = {
|
|
495
566
|
stage: stage.label,
|
|
567
|
+
attempt,
|
|
568
|
+
attempts,
|
|
496
569
|
phase: stage.phase,
|
|
497
570
|
models: new Set<string>(),
|
|
498
571
|
agentIds: [],
|
|
@@ -503,7 +576,8 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
|
|
|
503
576
|
costUsd: 0,
|
|
504
577
|
pricingKnown: true,
|
|
505
578
|
};
|
|
506
|
-
buckets.set(
|
|
579
|
+
buckets.set(bucketKey, b);
|
|
580
|
+
bucketOrder.push(bucketKey);
|
|
507
581
|
}
|
|
508
582
|
b.models.add(stage.model);
|
|
509
583
|
b.agentIds.push(stage.agentId);
|
|
@@ -576,6 +650,9 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
|
|
|
576
650
|
runId: record.runId,
|
|
577
651
|
slug: record.slug,
|
|
578
652
|
stage: b.stage,
|
|
653
|
+
stageCanonical: canonicalStage(b.stage).stage,
|
|
654
|
+
attempt: b.attempt,
|
|
655
|
+
attempts: b.attempts,
|
|
579
656
|
phase: b.phase,
|
|
580
657
|
model: models.length === 1 ? (models[0] ?? 'unknown') : 'mixed',
|
|
581
658
|
agentIds: b.agentIds,
|
|
@@ -591,20 +668,60 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
|
|
|
591
668
|
calls: b.claims.length,
|
|
592
669
|
});
|
|
593
670
|
}
|
|
594
|
-
rows.sort((a, z) => z.weightedTokens - a.weightedTokens || a.stage.localeCompare(z.stage));
|
|
671
|
+
rows.sort((a, z) => z.weightedTokens - a.weightedTokens || a.stage.localeCompare(z.stage) || a.attempt - z.attempt);
|
|
672
|
+
|
|
673
|
+
// ── orphan-transcript inventory (measurement-integrity FR-3 / ADR-001 D2) ──
|
|
674
|
+
// A transcript present in the run's directory with no `workflowProgress[]` entry is named
|
|
675
|
+
// explicitly here, rather than dissolved into the generic `Unaccounted` defect the way it was
|
|
676
|
+
// before this feature. `orphanTranscripts` is ALWAYS present in the report (count 0 when there are
|
|
677
|
+
// none) — additive, never a replacement for `unaccountedTokens`.
|
|
678
|
+
let orphanTokens = 0;
|
|
679
|
+
let orphanIds: string[] = [];
|
|
680
|
+
let orphanMethod: 'per-transcript' | 'count-fallback' = 'count-fallback';
|
|
681
|
+
if (input.orphanTranscripts !== undefined) {
|
|
682
|
+
orphanMethod = 'per-transcript';
|
|
683
|
+
const seenOrphanKeys = new Set<string>();
|
|
684
|
+
for (const o of input.orphanTranscripts) {
|
|
685
|
+
if (typeof o.agentId === 'string' && o.agentId.length > 0) orphanIds.push(o.agentId);
|
|
686
|
+
for (const s of o.samples) {
|
|
687
|
+
if (seenOrphanKeys.has(s.key)) continue;
|
|
688
|
+
seenOrphanKeys.add(s.key);
|
|
689
|
+
orphanTokens += s.weighted;
|
|
690
|
+
}
|
|
691
|
+
}
|
|
692
|
+
} else {
|
|
693
|
+
orphanIds = (input.orphanAgentIds ?? []).filter((x): x is string => typeof x === 'string' && x.length > 0);
|
|
694
|
+
if (orphanIds.length > 0) {
|
|
695
|
+
// No per-transcript samples were supplied — `orphanTokens` is still reported as the
|
|
696
|
+
// reconciliation's own `unaccountedTokens` (an honest BEST ESTIMATE, never invented) so a
|
|
697
|
+
// reader can see roughly how much spend is implicated. measurement-integrity fix-round-1/F1
|
|
698
|
+
// (Codex r1 CRITICAL #1): this figure is NO LONGER used below to shrink the `Unaccounted`
|
|
699
|
+
// defect — count-fallback's `orphanExplained` stays 0. The old behavior treated the WHOLE
|
|
700
|
+
// remainder as "orphan-explained" and could downgrade a genuine `DEFECT` (unattributed spend
|
|
701
|
+
// whose CAUSE is not actually known — a count-fallback orphan is a NAME, not a subtraction
|
|
702
|
+
// proof) into a merely-incomplete `INCOMPLETE_INVENTORY`. Only the EXACT `'per-transcript'`
|
|
703
|
+
// method — which sums real extracted samples — is trusted to reduce the residual.
|
|
704
|
+
orphanTokens = unaccountedTokens;
|
|
705
|
+
}
|
|
706
|
+
}
|
|
595
707
|
|
|
596
708
|
// ── named defects ──
|
|
597
|
-
const
|
|
598
|
-
if (unaccountedTokens >
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
709
|
+
const tolerance = Math.floor(epsilon * runTotalTokens);
|
|
710
|
+
if (unaccountedTokens > tolerance) {
|
|
711
|
+
// ONLY the portion NOT already explained by a named orphan transcript is a genuine `Unaccounted`
|
|
712
|
+
// defect — and ONLY the exact `'per-transcript'` method may explain any of it (F1 above).
|
|
713
|
+
const orphanExplained = orphanMethod === 'per-transcript' ? Math.min(orphanTokens, unaccountedTokens) : 0;
|
|
714
|
+
const residual = unaccountedTokens - orphanExplained;
|
|
715
|
+
if (residual > 0) {
|
|
716
|
+
defects.push({
|
|
717
|
+
kind: 'Unaccounted',
|
|
718
|
+
detail:
|
|
719
|
+
orphanIds.length > 0
|
|
720
|
+
? 'run spend is attributed to no stage, beyond what the named orphan transcripts explain'
|
|
721
|
+
: 'run spend is attributed to no stage',
|
|
722
|
+
tokens: residual,
|
|
723
|
+
});
|
|
724
|
+
}
|
|
608
725
|
}
|
|
609
726
|
const doubleClaimed = [...keyOwners.entries()].filter(([, owners]) => owners.size > 1);
|
|
610
727
|
if (doubleAttributedTokens > 0 || doubleClaimed.length > 0) {
|
|
@@ -651,19 +768,49 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
|
|
|
651
768
|
});
|
|
652
769
|
}
|
|
653
770
|
|
|
771
|
+
// measurement-integrity fix-round-1/F1 (Codex r1 CRITICAL #1): a named orphan transcript makes the
|
|
772
|
+
// inventory incomplete REGARDLESS of tokens or epsilon — an orphan with zero usage samples
|
|
773
|
+
// (`orphanTokens === 0`) or one whose spend happens to fall under a generous epsilon is STILL a
|
|
774
|
+
// transcript the inventory does not know about. Equality of SUMS never proves attribution; the
|
|
775
|
+
// epsilon budget is about tolerating a small unattributable REMAINDER (the generic `Unaccounted`
|
|
776
|
+
// defect above, which still honors `tolerance`), never about excusing a NAMED gap in the inventory
|
|
777
|
+
// itself. `hasOrphan` therefore no longer reads `orphanTokens`/`tolerance` at all.
|
|
778
|
+
const hasOrphan = orphanIds.length > 0;
|
|
779
|
+
|
|
654
780
|
// INSUFFICIENT_DATA is NOT success (ADR-003): no samples means nothing was measured, and a
|
|
655
781
|
// "0 === 0, so it balances" shortcut would let an absent transcript store read as a clean run.
|
|
782
|
+
//
|
|
783
|
+
// measurement-integrity ADR-001 D2: INCOMPLETE_INVENTORY outranks BALANCED (an orphan transcript
|
|
784
|
+
// can never read as balanced) and is outranked by DEFECT (a genuine attribution defect — one NOT
|
|
785
|
+
// fully explained by a named orphan — is worse than "the inventory is incomplete but everything it
|
|
786
|
+
// does have reconciles").
|
|
656
787
|
const verdict: CostLedgerVerdict =
|
|
657
788
|
runTotalTokens === 0 && stageTokensSum === 0
|
|
658
789
|
? 'INSUFFICIENT_DATA'
|
|
659
790
|
: defects.length > 0
|
|
660
791
|
? 'DEFECT'
|
|
661
|
-
:
|
|
792
|
+
: hasOrphan
|
|
793
|
+
? 'INCOMPLETE_INVENTORY'
|
|
794
|
+
: 'BALANCED';
|
|
662
795
|
|
|
663
796
|
let totalCostUsd = 0;
|
|
664
797
|
for (const r of rows) totalCostUsd += r.costUsd;
|
|
665
798
|
const fallbackModels = [...new Set(record.stages.filter((s) => !hasKnownPricing(s.model)).map((s) => s.model))].sort();
|
|
666
799
|
|
|
800
|
+
// ── byCanonicalStage (FR-2/T1 + FR-3/T2 combined: every canon bucket, plus `unknown` for rows
|
|
801
|
+
// whose verbatim label matched no rule, plus `unattributed` for the orphan-transcript tokens
|
|
802
|
+
// that belong to no stage row at all) ──
|
|
803
|
+
interface CanonAgg { tokens: number; agents: number; attempts: number }
|
|
804
|
+
const byCanonicalStage = {} as Record<CanonicalStage | 'unknown' | 'unattributed', CanonAgg>;
|
|
805
|
+
for (const k of [...CANONICAL_STAGES, 'unknown', 'unattributed'] as const) byCanonicalStage[k] = { tokens: 0, agents: 0, attempts: 0 };
|
|
806
|
+
for (const row of rows) {
|
|
807
|
+
const bucket = byCanonicalStage[row.stageCanonical];
|
|
808
|
+
bucket.tokens += row.weightedTokens;
|
|
809
|
+
bucket.agents += row.agentIds.length;
|
|
810
|
+
bucket.attempts += 1;
|
|
811
|
+
}
|
|
812
|
+
byCanonicalStage.unattributed = { tokens: orphanTokens, agents: orphanIds.length, attempts: orphanIds.length };
|
|
813
|
+
|
|
667
814
|
return {
|
|
668
815
|
runId: record.runId,
|
|
669
816
|
slug: record.slug,
|
|
@@ -671,6 +818,7 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
|
|
|
671
818
|
status: record.status,
|
|
672
819
|
startedTs: isoOrNull(record.startedAtMs),
|
|
673
820
|
rows,
|
|
821
|
+
byCanonicalStage,
|
|
674
822
|
reconciliation: {
|
|
675
823
|
runTotalTokens,
|
|
676
824
|
accountedTokens,
|
|
@@ -681,6 +829,7 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
|
|
|
681
829
|
identityHolds,
|
|
682
830
|
verdict,
|
|
683
831
|
defects,
|
|
832
|
+
orphanTranscripts: { count: orphanIds.length, tokens: orphanTokens, ids: orphanIds, method: orphanMethod },
|
|
684
833
|
},
|
|
685
834
|
recordTotalTokens: record.recordTotalTokens,
|
|
686
835
|
totalCostUsd,
|
|
@@ -740,6 +889,10 @@ export function verifyCostLedgerReport(report: CostLedgerReport): readonly CostL
|
|
|
740
889
|
export function stageCostAggregates(reports: readonly CostLedgerReport[]): StageCostAggregate[] {
|
|
741
890
|
const acc = new Map<string, { stage: string; model: string; total: number; cost: number; runs: Set<string> }>();
|
|
742
891
|
for (const report of reports) {
|
|
892
|
+
// measurement-integrity T2: `!== 'BALANCED'` already excludes `INCOMPLETE_INVENTORY` — a run
|
|
893
|
+
// with a named orphan transcript is exactly as unfit for a routing input as one with a DoubleAttributed
|
|
894
|
+
// defect, and this single comparison against the full 4-value vocabulary keeps excluding it
|
|
895
|
+
// without a second branch to forget.
|
|
743
896
|
if (report.reconciliation.verdict !== 'BALANCED') continue;
|
|
744
897
|
for (const row of report.rows) {
|
|
745
898
|
// JSON-tuple key (Codex QE LOW): NUL is a LEGAL JSON-string character, so even a NUL join
|
|
@@ -839,6 +992,26 @@ export function renderCostLedger(report: CostLedgerReport): string {
|
|
|
839
992
|
const subj = d.subjects === undefined || d.subjects.length === 0 ? '' : ` [${d.subjects.slice(0, 6).join(', ')}${d.subjects.length > 6 ? ', …' : ''}]`;
|
|
840
993
|
lines.push(` ${d.kind}: ${d.detail}${tok}${subj}`);
|
|
841
994
|
}
|
|
995
|
+
if (r.orphanTranscripts.count > 0) {
|
|
996
|
+
const idsPreview = r.orphanTranscripts.ids.slice(0, 6).join(', ') + (r.orphanTranscripts.ids.length > 6 ? ', …' : '');
|
|
997
|
+
lines.push(
|
|
998
|
+
` orphanTranscripts: ${r.orphanTranscripts.count} transcript(s) with no inventory row, ${fmt(r.orphanTranscripts.tokens)} weighted tokens ` +
|
|
999
|
+
`(${r.orphanTranscripts.method}) [${idsPreview}]`,
|
|
1000
|
+
);
|
|
1001
|
+
}
|
|
1002
|
+
// measurement-integrity FR-2/T1: the canon block — one line per non-empty bucket, `unattributed`
|
|
1003
|
+
// (orphan tokens) and `unknown` (unrecognised verbatim labels) LAST so the named canon reads first.
|
|
1004
|
+
const canonEntries = Object.entries(report.byCanonicalStage).filter(([, v]) => v.tokens > 0 || v.attempts > 0);
|
|
1005
|
+
if (canonEntries.length > 0) {
|
|
1006
|
+
canonEntries.sort(([a], [z]) => {
|
|
1007
|
+
const rank = (k: string): number => (k === 'unknown' ? 2 : k === 'unattributed' ? 3 : 1);
|
|
1008
|
+
return rank(a) - rank(z) || a.localeCompare(z);
|
|
1009
|
+
});
|
|
1010
|
+
lines.push(' by canonical stage:');
|
|
1011
|
+
for (const [stage, agg] of canonEntries) {
|
|
1012
|
+
lines.push(` ${pad(stage, 14)} ${padLeft(fmt(agg.tokens), 12)} tok ${padLeft(String(agg.agents), 4)} agent(s) ${padLeft(String(agg.attempts), 4)} attempt(s)`);
|
|
1013
|
+
}
|
|
1014
|
+
}
|
|
842
1015
|
if (report.recordTotalTokens !== null) {
|
|
843
1016
|
lines.push(
|
|
844
1017
|
` note: the run record's own totalTokens is ${fmt(report.recordTotalTokens)} — a RAW unweighted cached sum of the same per-agent list, reported for traceability, NOT the invariant's right-hand side`,
|
|
@@ -1037,22 +1210,57 @@ export function deriveCostLedger(opts: DeriveCostLedgerOptions = {}): CostLedger
|
|
|
1037
1210
|
|
|
1038
1211
|
const runSamples: CostLedgerSample[] = [];
|
|
1039
1212
|
const perAgent = new Map<string, CostLedgerSample[]>();
|
|
1040
|
-
|
|
1213
|
+
// measurement-integrity FR-3: a transcript file with NO matching `workflowProgress[]` entry is an
|
|
1214
|
+
// orphan REGARDLESS of whether it happened to log any usage samples — a zero-sample orphan is
|
|
1215
|
+
// still a transcript the inventory does not know about, so it is counted here (0 tokens, still a
|
|
1216
|
+
// named id) rather than silently dropped the way the pre-existing `orphanAgentIds.length > 0`
|
|
1217
|
+
// gate did.
|
|
1218
|
+
//
|
|
1219
|
+
// measurement-integrity fix-round-1/F2 (Codex r1 HIGH #2): the OLD loop compared inventory
|
|
1220
|
+
// MEMBERSHIP by `agentId` alone and treated two anomalies as invisible: a file whose name does not
|
|
1221
|
+
// match the `agent-<id>.jsonl` shape was `continue`d past — dropped from BOTH `perAgent` and
|
|
1222
|
+
// `orphanTranscripts` — and a SECOND file for an agentId already known simply OVERWROTE the first
|
|
1223
|
+
// in `perAgent`, silently discarding one transcript's samples while reading as "the one known
|
|
1224
|
+
// agent". Both are now named inventory anomalies, folded into the SAME `orphanTranscripts` list
|
|
1225
|
+
// (so the existing `hasOrphan`/verdict machinery in `buildCostLedger` already refuses to call
|
|
1226
|
+
// either case BALANCED) with a synthetic, self-describing id — never silently absorbed as "known".
|
|
1227
|
+
const orphanTranscripts: { agentId: string; samples: CostLedgerSample[] }[] = [];
|
|
1228
|
+
const unparseableNames: string[] = [];
|
|
1229
|
+
const duplicateFor: string[] = [];
|
|
1041
1230
|
for (const f of files) {
|
|
1042
1231
|
const samples = extractCostSamples(safeReadText(join(ref.transcriptDir, f)));
|
|
1043
1232
|
runSamples.push(...samples);
|
|
1233
|
+
// `journal.jsonl` is a KNOWN, EXPECTED per-run housekeeping file (present in every real run
|
|
1234
|
+
// directory alongside the `agent-<id>.jsonl` transcripts — verified against live
|
|
1235
|
+
// `roam/claude-state/**/subagents/workflows/wf_*` directories) that carries no usage samples of
|
|
1236
|
+
// its own. Naming it an inventory anomaly would make F2's fix fire on every single run there is
|
|
1237
|
+
// — the false-positive explosion this feature exists to AVOID, not cause.
|
|
1238
|
+
if (f === 'journal.jsonl') continue;
|
|
1044
1239
|
const m = /^agent-(.+)\.jsonl$/.exec(f);
|
|
1045
|
-
if (m === null)
|
|
1240
|
+
if (m === null) {
|
|
1241
|
+
unparseableNames.push(f);
|
|
1242
|
+
orphanTranscripts.push({ agentId: `unparseable:${f}`, samples });
|
|
1243
|
+
continue;
|
|
1244
|
+
}
|
|
1046
1245
|
const agentId = m[1] ?? '';
|
|
1047
|
-
if (stageAgentIds.has(agentId))
|
|
1048
|
-
|
|
1246
|
+
if (!stageAgentIds.has(agentId)) {
|
|
1247
|
+
orphanTranscripts.push({ agentId, samples });
|
|
1248
|
+
} else if (perAgent.has(agentId)) {
|
|
1249
|
+
// A SECOND transcript file for an agentId already claimed — never silently overwrite the
|
|
1250
|
+
// first one's samples nor pretend this file belongs to "the known agent" too.
|
|
1251
|
+
duplicateFor.push(agentId);
|
|
1252
|
+
orphanTranscripts.push({ agentId: `duplicate:${agentId}:${f}`, samples });
|
|
1253
|
+
} else {
|
|
1254
|
+
perAgent.set(agentId, samples);
|
|
1255
|
+
}
|
|
1049
1256
|
}
|
|
1050
1257
|
|
|
1051
1258
|
return buildCostLedger({
|
|
1052
1259
|
record,
|
|
1053
1260
|
stageSamples: [...perAgent.entries()].map(([agentId, samples]) => ({ agentId, samples })),
|
|
1054
1261
|
runSamples,
|
|
1055
|
-
orphanAgentIds,
|
|
1262
|
+
orphanAgentIds: orphanTranscripts.map((o) => o.agentId),
|
|
1263
|
+
orphanTranscripts,
|
|
1056
1264
|
...(listingTruncated ? { transcriptListingTruncated: true } : {}),
|
|
1057
1265
|
...(opts.epsilon !== undefined ? { epsilon: opts.epsilon } : {}),
|
|
1058
1266
|
});
|