@dzhechkov/harness-core 0.8.35 → 0.8.37

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. package/.dz-manifest.json +224 -104
  2. package/README.md +335 -10
  3. package/dist/agentdb-index.d.ts +87 -7
  4. package/dist/agentdb-index.d.ts.map +1 -1
  5. package/dist/agentdb-index.js +416 -57
  6. package/dist/agentdb-index.js.map +1 -1
  7. package/dist/apply-leg.d.ts +57 -1
  8. package/dist/apply-leg.d.ts.map +1 -1
  9. package/dist/apply-leg.js +450 -52
  10. package/dist/apply-leg.js.map +1 -1
  11. package/dist/codex-hooks-assets.d.ts.map +1 -1
  12. package/dist/codex-hooks-assets.js +67 -5
  13. package/dist/codex-hooks-assets.js.map +1 -1
  14. package/dist/codex-hooks.d.ts +13 -1
  15. package/dist/codex-hooks.d.ts.map +1 -1
  16. package/dist/codex-hooks.js +13 -1
  17. package/dist/codex-hooks.js.map +1 -1
  18. package/dist/codex-rollouts.d.ts +118 -0
  19. package/dist/codex-rollouts.d.ts.map +1 -0
  20. package/dist/codex-rollouts.js +297 -0
  21. package/dist/codex-rollouts.js.map +1 -0
  22. package/dist/cost-ledger.d.ts +56 -4
  23. package/dist/cost-ledger.d.ts.map +1 -1
  24. package/dist/cost-ledger.js +176 -20
  25. package/dist/cost-ledger.js.map +1 -1
  26. package/dist/cross-family-control.d.ts +345 -0
  27. package/dist/cross-family-control.d.ts.map +1 -0
  28. package/dist/cross-family-control.js +802 -0
  29. package/dist/cross-family-control.js.map +1 -0
  30. package/dist/debt-ratchet.d.ts +53 -0
  31. package/dist/debt-ratchet.d.ts.map +1 -0
  32. package/dist/debt-ratchet.js +107 -0
  33. package/dist/debt-ratchet.js.map +1 -0
  34. package/dist/embedding-config.d.ts +42 -0
  35. package/dist/embedding-config.d.ts.map +1 -1
  36. package/dist/embedding-config.js +106 -10
  37. package/dist/embedding-config.js.map +1 -1
  38. package/dist/feature-adr-checkpoints.d.ts +6 -0
  39. package/dist/feature-adr-checkpoints.d.ts.map +1 -1
  40. package/dist/feature-adr-checkpoints.js +29 -0
  41. package/dist/feature-adr-checkpoints.js.map +1 -1
  42. package/dist/feature-adr-decision-recall.d.ts +2 -2
  43. package/dist/feature-adr-decision-recall.d.ts.map +1 -1
  44. package/dist/feature-adr-decision-recall.js +5 -3
  45. package/dist/feature-adr-decision-recall.js.map +1 -1
  46. package/dist/feature-adr-envelope.d.ts +96 -0
  47. package/dist/feature-adr-envelope.d.ts.map +1 -0
  48. package/dist/feature-adr-envelope.js +183 -0
  49. package/dist/feature-adr-envelope.js.map +1 -0
  50. package/dist/feature-adr-routing.d.ts +64 -0
  51. package/dist/feature-adr-routing.d.ts.map +1 -1
  52. package/dist/feature-adr-routing.js +122 -2
  53. package/dist/feature-adr-routing.js.map +1 -1
  54. package/dist/feature-adr-stage-canon.d.ts +79 -0
  55. package/dist/feature-adr-stage-canon.d.ts.map +1 -0
  56. package/dist/feature-adr-stage-canon.js +117 -0
  57. package/dist/feature-adr-stage-canon.js.map +1 -0
  58. package/dist/index.d.ts +23 -12
  59. package/dist/index.d.ts.map +1 -1
  60. package/dist/index.js +15 -7
  61. package/dist/index.js.map +1 -1
  62. package/dist/loop-blobs.generated.js +4 -4
  63. package/dist/loop-blobs.generated.js.map +1 -1
  64. package/dist/mutation-gate.d.ts +51 -0
  65. package/dist/mutation-gate.d.ts.map +1 -1
  66. package/dist/mutation-gate.js +295 -0
  67. package/dist/mutation-gate.js.map +1 -1
  68. package/dist/operations.d.ts +1 -0
  69. package/dist/operations.d.ts.map +1 -1
  70. package/dist/operations.js +18 -2
  71. package/dist/operations.js.map +1 -1
  72. package/dist/publish.d.ts +59 -7
  73. package/dist/publish.d.ts.map +1 -1
  74. package/dist/publish.js +205 -32
  75. package/dist/publish.js.map +1 -1
  76. package/dist/qe-bridge.d.ts.map +1 -1
  77. package/dist/qe-bridge.js +4 -2
  78. package/dist/qe-bridge.js.map +1 -1
  79. package/dist/qe-findings.d.ts +107 -0
  80. package/dist/qe-findings.d.ts.map +1 -0
  81. package/dist/qe-findings.js +417 -0
  82. package/dist/qe-findings.js.map +1 -0
  83. package/dist/recap.d.ts +1 -1
  84. package/dist/recap.d.ts.map +1 -1
  85. package/dist/recap.js +4 -2
  86. package/dist/recap.js.map +1 -1
  87. package/dist/release-line.d.ts +16 -0
  88. package/dist/release-line.d.ts.map +1 -1
  89. package/dist/release-line.js +31 -0
  90. package/dist/release-line.js.map +1 -1
  91. package/dist/round.d.ts +74 -1
  92. package/dist/round.d.ts.map +1 -1
  93. package/dist/round.js +112 -4
  94. package/dist/round.js.map +1 -1
  95. package/dist/run-records.d.ts +60 -0
  96. package/dist/run-records.d.ts.map +1 -1
  97. package/dist/run-records.js +244 -2
  98. package/dist/run-records.js.map +1 -1
  99. package/dist/score.d.ts +44 -1
  100. package/dist/score.d.ts.map +1 -1
  101. package/dist/score.js +78 -5
  102. package/dist/score.js.map +1 -1
  103. package/dist/vector-tier.d.ts +34 -3
  104. package/dist/vector-tier.d.ts.map +1 -1
  105. package/dist/vector-tier.js +105 -14
  106. package/dist/vector-tier.js.map +1 -1
  107. package/package.json +2 -2
  108. package/sbom.json +403 -103
  109. package/src/agentdb-index.ts +423 -60
  110. package/src/apply-leg.ts +469 -50
  111. package/src/codex-hooks-assets.ts +67 -5
  112. package/src/codex-hooks.ts +13 -1
  113. package/src/codex-rollouts.ts +374 -0
  114. package/src/cost-ledger.ts +232 -24
  115. package/src/cross-family-control.ts +960 -0
  116. package/src/debt-ratchet.ts +143 -0
  117. package/src/embedding-config.ts +131 -10
  118. package/src/feature-adr-checkpoints.ts +29 -0
  119. package/src/feature-adr-decision-recall.ts +6 -4
  120. package/src/feature-adr-envelope.ts +242 -0
  121. package/src/feature-adr-routing.ts +139 -2
  122. package/src/feature-adr-stage-canon.ts +141 -0
  123. package/src/index.ts +66 -7
  124. package/src/loop-blobs.generated.ts +4 -4
  125. package/src/mutation-gate.ts +316 -0
  126. package/src/operations.ts +18 -3
  127. package/src/publish.ts +247 -30
  128. package/src/qe-bridge.ts +4 -2
  129. package/src/qe-findings.ts +463 -0
  130. package/src/recap.ts +10 -3
  131. package/src/release-line.ts +32 -0
  132. package/src/round.ts +165 -6
  133. package/src/run-records.ts +282 -2
  134. package/src/score.ts +115 -6
  135. package/src/vector-tier.ts +127 -14
@@ -52,6 +52,8 @@ import { existsSync, lstatSync, mkdirSync, readFileSync, readdirSync, realpathSy
52
52
  import { dirname, join } from 'node:path';
53
53
 
54
54
  import { hasKnownPricing, usageCost } from './cost-scoring.js';
55
+ import { CANONICAL_STAGES, canonicalStage } from './feature-adr-stage-canon.js';
56
+ import type { CanonicalStage } from './feature-adr-stage-canon.js';
55
57
  import { claudeProjectsRoot, rawTokenMixOf, weightedTokensOf } from './usage.js';
56
58
 
57
59
  // ── Scope + vocabulary ──────────────────────────────────────
@@ -91,14 +93,21 @@ export const COST_LEDGER_DEFECT_KINDS: readonly CostLedgerDefectKind[] = [
91
93
  ];
92
94
 
93
95
  /**
94
- * Three values, not two. `INSUFFICIENT_DATA` is NOT success: a caller must not read
95
- * `verdict !== 'DEFECT'` as "reconciled" (ADR-003).
96
+ * Four values, not two (measurement-integrity ADR-001 D2 grew this from three). `INSUFFICIENT_DATA`
97
+ * is NOT success: a caller must not read `verdict !== 'DEFECT'` as "reconciled" (ADR-003).
98
+ * `INCOMPLETE_INVENTORY` is likewise not success — it names the SPECIFIC case where every unaccounted
99
+ * token traces to a named orphan transcript (a directory listing the record's own inventory does not
100
+ * know about), with no OTHER attribution defect present. It outranks `BALANCED` (an orphan transcript
101
+ * can never read as balanced) and is outranked by `DEFECT` (a genuine attribution defect — double
102
+ * counting, a foreign sample, spend unaccounted for reasons OTHER than a named orphan — is a worse
103
+ * finding than "the inventory is incomplete but everything it does have reconciles").
96
104
  */
97
- export type CostLedgerVerdict = 'BALANCED' | 'DEFECT' | 'INSUFFICIENT_DATA';
105
+ export type CostLedgerVerdict = 'BALANCED' | 'DEFECT' | 'INCOMPLETE_INVENTORY' | 'INSUFFICIENT_DATA';
98
106
 
99
107
  export const COST_LEDGER_VERDICTS: readonly CostLedgerVerdict[] = [
100
108
  'BALANCED',
101
109
  'DEFECT',
110
+ 'INCOMPLETE_INVENTORY',
102
111
  'INSUFFICIENT_DATA',
103
112
  ];
104
113
 
@@ -168,6 +177,17 @@ export interface CostLedgerRow {
168
177
  readonly slug: string | null;
169
178
  /** `stageLabel()` output, verbatim. */
170
179
  readonly stage: string;
180
+ /** measurement-integrity FR-2/ADR-001 D1: the canonical bucket `stage` classifies into, NEXT TO
181
+ * the untouched verbatim `stage` — never a replacement for it. `'unknown'` when no rule matches. */
182
+ readonly stageCanonical: CanonicalStage | 'unknown';
183
+ /** measurement-integrity FR-4: 1-based position of THIS occurrence among stage entries that share
184
+ * `stage`'s exact label, in record order — `attempt: 1, attempts: 1` when the label occurs once. */
185
+ readonly attempt: number;
186
+ /** Total number of stage entries in this run that share `stage`'s exact label. A repeated label no
187
+ * longer merges silently into one row (ADR-001 D... measurement-integrity FR-4): each occurrence
188
+ * is its OWN row, and every one of them carries the same `attempts` count so a reader grouping by
189
+ * `stage` can tell there were several without re-deriving it. */
190
+ readonly attempts: number;
171
191
  readonly phase: string | null;
172
192
  /** The stage's model id, or `'mixed'` when several agents share a label with different models. */
173
193
  readonly model: string;
@@ -214,6 +234,18 @@ export interface CostLedgerReconciliation {
214
234
  readonly identityHolds: boolean;
215
235
  readonly verdict: CostLedgerVerdict;
216
236
  readonly defects: readonly CostLedgerDefect[];
237
+ /** measurement-integrity FR-3/ADR-001 D2: transcripts present in the run's directory that have NO
238
+ * `workflowProgress[]` entry in the record — named explicitly rather than dissolved into a generic
239
+ * `unaccountedTokens` remainder. `method: 'per-transcript'` means `tokens` is an exact sum over
240
+ * each orphan's own extracted samples; `'count-fallback'` means only the COUNT (and, where
241
+ * available, the ids) is known and `tokens` falls back to the reconciliation's own
242
+ * `unaccountedTokens` as the best available estimate — always present, always additive. */
243
+ readonly orphanTranscripts: {
244
+ readonly count: number;
245
+ readonly tokens: number;
246
+ readonly ids: readonly string[];
247
+ readonly method: 'per-transcript' | 'count-fallback';
248
+ };
217
249
  }
218
250
 
219
251
  export interface CostLedgerReport {
@@ -223,6 +255,12 @@ export interface CostLedgerReport {
223
255
  readonly status: string | null;
224
256
  readonly startedTs: string | null;
225
257
  readonly rows: readonly CostLedgerRow[];
258
+ /** measurement-integrity FR-2/T1: rows re-aggregated by `stageCanonical`, plus two synthetic
259
+ * buckets — `unknown` (rows whose verbatim label matched no canon rule) and `unattributed` (the
260
+ * orphan-transcript tokens from `reconciliation.orphanTranscripts`, which belong to no stage row
261
+ * at all). Every one of the 12 {@link CanonicalStage} values is always present, even at zero, so a
262
+ * reader can iterate a stable key set. */
263
+ readonly byCanonicalStage: Record<CanonicalStage | 'unknown' | 'unattributed', { readonly tokens: number; readonly agents: number; readonly attempts: number }>;
226
264
  readonly reconciliation: CostLedgerReconciliation;
227
265
  /** The record's cached raw sum — reported, never the invariant's right-hand side (ADR-002). */
228
266
  readonly recordTotalTokens: number | null;
@@ -412,8 +450,17 @@ export interface BuildCostLedgerInput {
412
450
  readonly stageSamples: readonly { readonly agentId: string; readonly samples: readonly CostLedgerSample[] }[];
413
451
  /** RIGHT side — the dedup-union over the run's transcript DIRECTORY (ADR-002). */
414
452
  readonly runSamples: readonly CostLedgerSample[];
415
- /** Agent transcripts present in the run directory with no `workflowProgress[]` entry. */
453
+ /** Agent transcripts present in the run directory with no `workflowProgress[]` entry — ids only.
454
+ * Kept for the `Unaccounted` defect's `subjects` listing; superseded by `orphanTranscripts` below
455
+ * for the FR-3 `orphanTranscripts.tokens` figure whenever the caller can supply per-orphan samples. */
416
456
  readonly orphanAgentIds?: readonly string[];
457
+ /** measurement-integrity FR-3: the SAME orphan transcripts as `orphanAgentIds`, but carrying each
458
+ * one's own extracted samples so `reconciliation.orphanTranscripts.tokens` is an EXACT sum rather
459
+ * than a derived remainder. When omitted, `orphanAgentIds` alone still produces a named
460
+ * `orphanTranscripts` entry (`method: 'count-fallback'`, `tokens` = the reconciliation's own
461
+ * `unaccountedTokens` as the best available estimate) — the count/ids are never lost even without
462
+ * the precise per-transcript figure. */
463
+ readonly orphanTranscripts?: readonly { readonly agentId: string; readonly samples: readonly CostLedgerSample[] }[];
417
464
  /** Fraction of the run total tolerated as unaccounted. Default {@link DEFAULT_COST_LEDGER_EPSILON}. */
418
465
  readonly epsilon?: number;
419
466
  /** True when the transcript listing hit the file cap — the run total is incomplete (Codex QE HIGH). */
@@ -453,6 +500,9 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
453
500
  ...input,
454
501
  runSamples: input.runSamples.map(clampSample),
455
502
  stageSamples: input.stageSamples.map((e) => ({ agentId: e.agentId, samples: e.samples.map(clampSample) })),
503
+ ...(input.orphanTranscripts !== undefined
504
+ ? { orphanTranscripts: input.orphanTranscripts.map((e) => ({ agentId: e.agentId, samples: e.samples.map(clampSample) })) }
505
+ : {}),
456
506
  };
457
507
 
458
508
  // RIGHT — the run's universe, deduped by sample key.
@@ -467,6 +517,8 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
467
517
 
468
518
  interface Bucket {
469
519
  stage: string;
520
+ attempt: number;
521
+ attempts: number;
470
522
  phase: string | null;
471
523
  models: Set<string>;
472
524
  agentIds: string[];
@@ -478,7 +530,15 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
478
530
  costUsd: number;
479
531
  pricingKnown: boolean;
480
532
  }
533
+ // measurement-integrity FR-4: a bucket is now keyed by (label, occurrence) rather than by label
534
+ // alone — a label repeated N times in `record.stages` (a retried stage) produces N buckets, each
535
+ // its own row, instead of one row silently summing N attempts together. `labelTotalCount` is a
536
+ // first pass so every occurrence's row can carry the SAME `attempts` total, including the first.
537
+ const labelTotalCount = new Map<string, number>();
538
+ for (const stage of record.stages) labelTotalCount.set(stage.label, (labelTotalCount.get(stage.label) ?? 0) + 1);
539
+ const labelSeen = new Map<string, number>();
481
540
  const buckets = new Map<string, Bucket>();
541
+ const bucketOrder: string[] = [];
482
542
  const keyOwners = new Map<string, Set<string>>();
483
543
  const foreign: string[] = [];
484
544
  const conflicting: string[] = [];
@@ -489,10 +549,23 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
489
549
  const samples = byAgent.get(stage.agentId) ?? [];
490
550
  if (samples.length === 0) missingTranscript.push(`${stage.label} (${stage.agentId})`);
491
551
 
492
- let b = buckets.get(stage.label);
552
+ const attempt = (labelSeen.get(stage.label) ?? 0) + 1;
553
+ labelSeen.set(stage.label, attempt);
554
+ const attempts = labelTotalCount.get(stage.label) ?? 1;
555
+ // measurement-integrity fix-round-1/F11 (Codex r1 MEDIUM #11): the key used to be a NUL-
556
+ // delimited template literal (`${stage.label}\0${attempt}`) - readable in an editor as a plain
557
+ // space because NUL renders invisibly, but NUL is a LEGAL JSON-string character (the same
558
+ // delimiter-ambiguity class `stageCostAggregates` below already fixed for its own key). An
559
+ // unambiguous JSON-tuple serialization removes the theoretical collision outright, and matches
560
+ // the pattern already used two functions down in this file.
561
+ const bucketKey = JSON.stringify([stage.label, attempt]);
562
+
563
+ let b = buckets.get(bucketKey);
493
564
  if (b === undefined) {
494
565
  b = {
495
566
  stage: stage.label,
567
+ attempt,
568
+ attempts,
496
569
  phase: stage.phase,
497
570
  models: new Set<string>(),
498
571
  agentIds: [],
@@ -503,7 +576,8 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
503
576
  costUsd: 0,
504
577
  pricingKnown: true,
505
578
  };
506
- buckets.set(stage.label, b);
579
+ buckets.set(bucketKey, b);
580
+ bucketOrder.push(bucketKey);
507
581
  }
508
582
  b.models.add(stage.model);
509
583
  b.agentIds.push(stage.agentId);
@@ -576,6 +650,9 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
576
650
  runId: record.runId,
577
651
  slug: record.slug,
578
652
  stage: b.stage,
653
+ stageCanonical: canonicalStage(b.stage).stage,
654
+ attempt: b.attempt,
655
+ attempts: b.attempts,
579
656
  phase: b.phase,
580
657
  model: models.length === 1 ? (models[0] ?? 'unknown') : 'mixed',
581
658
  agentIds: b.agentIds,
@@ -591,20 +668,60 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
591
668
  calls: b.claims.length,
592
669
  });
593
670
  }
594
- rows.sort((a, z) => z.weightedTokens - a.weightedTokens || a.stage.localeCompare(z.stage));
671
+ rows.sort((a, z) => z.weightedTokens - a.weightedTokens || a.stage.localeCompare(z.stage) || a.attempt - z.attempt);
672
+
673
+ // ── orphan-transcript inventory (measurement-integrity FR-3 / ADR-001 D2) ──
674
+ // A transcript present in the run's directory with no `workflowProgress[]` entry is named
675
+ // explicitly here, rather than dissolved into the generic `Unaccounted` defect the way it was
676
+ // before this feature. `orphanTranscripts` is ALWAYS present in the report (count 0 when there are
677
+ // none) — additive, never a replacement for `unaccountedTokens`.
678
+ let orphanTokens = 0;
679
+ let orphanIds: string[] = [];
680
+ let orphanMethod: 'per-transcript' | 'count-fallback' = 'count-fallback';
681
+ if (input.orphanTranscripts !== undefined) {
682
+ orphanMethod = 'per-transcript';
683
+ const seenOrphanKeys = new Set<string>();
684
+ for (const o of input.orphanTranscripts) {
685
+ if (typeof o.agentId === 'string' && o.agentId.length > 0) orphanIds.push(o.agentId);
686
+ for (const s of o.samples) {
687
+ if (seenOrphanKeys.has(s.key)) continue;
688
+ seenOrphanKeys.add(s.key);
689
+ orphanTokens += s.weighted;
690
+ }
691
+ }
692
+ } else {
693
+ orphanIds = (input.orphanAgentIds ?? []).filter((x): x is string => typeof x === 'string' && x.length > 0);
694
+ if (orphanIds.length > 0) {
695
+ // No per-transcript samples were supplied — `orphanTokens` is still reported as the
696
+ // reconciliation's own `unaccountedTokens` (an honest BEST ESTIMATE, never invented) so a
697
+ // reader can see roughly how much spend is implicated. measurement-integrity fix-round-1/F1
698
+ // (Codex r1 CRITICAL #1): this figure is NO LONGER used below to shrink the `Unaccounted`
699
+ // defect — count-fallback's `orphanExplained` stays 0. The old behavior treated the WHOLE
700
+ // remainder as "orphan-explained" and could downgrade a genuine `DEFECT` (unattributed spend
701
+ // whose CAUSE is not actually known — a count-fallback orphan is a NAME, not a subtraction
702
+ // proof) into a merely-incomplete `INCOMPLETE_INVENTORY`. Only the EXACT `'per-transcript'`
703
+ // method — which sums real extracted samples — is trusted to reduce the residual.
704
+ orphanTokens = unaccountedTokens;
705
+ }
706
+ }
595
707
 
596
708
  // ── named defects ──
597
- const orphans = (input.orphanAgentIds ?? []).filter((x) => typeof x === 'string' && x.length > 0);
598
- if (unaccountedTokens > Math.floor(epsilon * runTotalTokens)) {
599
- defects.push({
600
- kind: 'Unaccounted',
601
- detail:
602
- orphans.length > 0
603
- ? `${orphans.length} agent transcript(s) in the run directory have no workflowProgress entry`
604
- : 'run spend is attributed to no stage',
605
- tokens: unaccountedTokens,
606
- ...(orphans.length > 0 ? { subjects: orphans } : {}),
607
- });
709
+ const tolerance = Math.floor(epsilon * runTotalTokens);
710
+ if (unaccountedTokens > tolerance) {
711
+ // ONLY the portion NOT already explained by a named orphan transcript is a genuine `Unaccounted`
712
+ // defect — and ONLY the exact `'per-transcript'` method may explain any of it (F1 above).
713
+ const orphanExplained = orphanMethod === 'per-transcript' ? Math.min(orphanTokens, unaccountedTokens) : 0;
714
+ const residual = unaccountedTokens - orphanExplained;
715
+ if (residual > 0) {
716
+ defects.push({
717
+ kind: 'Unaccounted',
718
+ detail:
719
+ orphanIds.length > 0
720
+ ? 'run spend is attributed to no stage, beyond what the named orphan transcripts explain'
721
+ : 'run spend is attributed to no stage',
722
+ tokens: residual,
723
+ });
724
+ }
608
725
  }
609
726
  const doubleClaimed = [...keyOwners.entries()].filter(([, owners]) => owners.size > 1);
610
727
  if (doubleAttributedTokens > 0 || doubleClaimed.length > 0) {
@@ -651,19 +768,49 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
651
768
  });
652
769
  }
653
770
 
771
+ // measurement-integrity fix-round-1/F1 (Codex r1 CRITICAL #1): a named orphan transcript makes the
772
+ // inventory incomplete REGARDLESS of tokens or epsilon — an orphan with zero usage samples
773
+ // (`orphanTokens === 0`) or one whose spend happens to fall under a generous epsilon is STILL a
774
+ // transcript the inventory does not know about. Equality of SUMS never proves attribution; the
775
+ // epsilon budget is about tolerating a small unattributable REMAINDER (the generic `Unaccounted`
776
+ // defect above, which still honors `tolerance`), never about excusing a NAMED gap in the inventory
777
+ // itself. `hasOrphan` therefore no longer reads `orphanTokens`/`tolerance` at all.
778
+ const hasOrphan = orphanIds.length > 0;
779
+
654
780
  // INSUFFICIENT_DATA is NOT success (ADR-003): no samples means nothing was measured, and a
655
781
  // "0 === 0, so it balances" shortcut would let an absent transcript store read as a clean run.
782
+ //
783
+ // measurement-integrity ADR-001 D2: INCOMPLETE_INVENTORY outranks BALANCED (an orphan transcript
784
+ // can never read as balanced) and is outranked by DEFECT (a genuine attribution defect — one NOT
785
+ // fully explained by a named orphan — is worse than "the inventory is incomplete but everything it
786
+ // does have reconciles").
656
787
  const verdict: CostLedgerVerdict =
657
788
  runTotalTokens === 0 && stageTokensSum === 0
658
789
  ? 'INSUFFICIENT_DATA'
659
790
  : defects.length > 0
660
791
  ? 'DEFECT'
661
- : 'BALANCED';
792
+ : hasOrphan
793
+ ? 'INCOMPLETE_INVENTORY'
794
+ : 'BALANCED';
662
795
 
663
796
  let totalCostUsd = 0;
664
797
  for (const r of rows) totalCostUsd += r.costUsd;
665
798
  const fallbackModels = [...new Set(record.stages.filter((s) => !hasKnownPricing(s.model)).map((s) => s.model))].sort();
666
799
 
800
+ // ── byCanonicalStage (FR-2/T1 + FR-3/T2 combined: every canon bucket, plus `unknown` for rows
801
+ // whose verbatim label matched no rule, plus `unattributed` for the orphan-transcript tokens
802
+ // that belong to no stage row at all) ──
803
+ interface CanonAgg { tokens: number; agents: number; attempts: number }
804
+ const byCanonicalStage = {} as Record<CanonicalStage | 'unknown' | 'unattributed', CanonAgg>;
805
+ for (const k of [...CANONICAL_STAGES, 'unknown', 'unattributed'] as const) byCanonicalStage[k] = { tokens: 0, agents: 0, attempts: 0 };
806
+ for (const row of rows) {
807
+ const bucket = byCanonicalStage[row.stageCanonical];
808
+ bucket.tokens += row.weightedTokens;
809
+ bucket.agents += row.agentIds.length;
810
+ bucket.attempts += 1;
811
+ }
812
+ byCanonicalStage.unattributed = { tokens: orphanTokens, agents: orphanIds.length, attempts: orphanIds.length };
813
+
667
814
  return {
668
815
  runId: record.runId,
669
816
  slug: record.slug,
@@ -671,6 +818,7 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
671
818
  status: record.status,
672
819
  startedTs: isoOrNull(record.startedAtMs),
673
820
  rows,
821
+ byCanonicalStage,
674
822
  reconciliation: {
675
823
  runTotalTokens,
676
824
  accountedTokens,
@@ -681,6 +829,7 @@ export function buildCostLedger(input: BuildCostLedgerInput): CostLedgerReport {
681
829
  identityHolds,
682
830
  verdict,
683
831
  defects,
832
+ orphanTranscripts: { count: orphanIds.length, tokens: orphanTokens, ids: orphanIds, method: orphanMethod },
684
833
  },
685
834
  recordTotalTokens: record.recordTotalTokens,
686
835
  totalCostUsd,
@@ -740,6 +889,10 @@ export function verifyCostLedgerReport(report: CostLedgerReport): readonly CostL
740
889
  export function stageCostAggregates(reports: readonly CostLedgerReport[]): StageCostAggregate[] {
741
890
  const acc = new Map<string, { stage: string; model: string; total: number; cost: number; runs: Set<string> }>();
742
891
  for (const report of reports) {
892
+ // measurement-integrity T2: `!== 'BALANCED'` already excludes `INCOMPLETE_INVENTORY` — a run
893
+ // with a named orphan transcript is exactly as unfit for a routing input as one with a DoubleAttributed
894
+ // defect, and this single comparison against the full 4-value vocabulary keeps excluding it
895
+ // without a second branch to forget.
743
896
  if (report.reconciliation.verdict !== 'BALANCED') continue;
744
897
  for (const row of report.rows) {
745
898
  // JSON-tuple key (Codex QE LOW): NUL is a LEGAL JSON-string character, so even a NUL join
@@ -839,6 +992,26 @@ export function renderCostLedger(report: CostLedgerReport): string {
839
992
  const subj = d.subjects === undefined || d.subjects.length === 0 ? '' : ` [${d.subjects.slice(0, 6).join(', ')}${d.subjects.length > 6 ? ', …' : ''}]`;
840
993
  lines.push(` ${d.kind}: ${d.detail}${tok}${subj}`);
841
994
  }
995
+ if (r.orphanTranscripts.count > 0) {
996
+ const idsPreview = r.orphanTranscripts.ids.slice(0, 6).join(', ') + (r.orphanTranscripts.ids.length > 6 ? ', …' : '');
997
+ lines.push(
998
+ ` orphanTranscripts: ${r.orphanTranscripts.count} transcript(s) with no inventory row, ${fmt(r.orphanTranscripts.tokens)} weighted tokens ` +
999
+ `(${r.orphanTranscripts.method}) [${idsPreview}]`,
1000
+ );
1001
+ }
1002
+ // measurement-integrity FR-2/T1: the canon block — one line per non-empty bucket, `unattributed`
1003
+ // (orphan tokens) and `unknown` (unrecognised verbatim labels) LAST so the named canon reads first.
1004
+ const canonEntries = Object.entries(report.byCanonicalStage).filter(([, v]) => v.tokens > 0 || v.attempts > 0);
1005
+ if (canonEntries.length > 0) {
1006
+ canonEntries.sort(([a], [z]) => {
1007
+ const rank = (k: string): number => (k === 'unknown' ? 2 : k === 'unattributed' ? 3 : 1);
1008
+ return rank(a) - rank(z) || a.localeCompare(z);
1009
+ });
1010
+ lines.push(' by canonical stage:');
1011
+ for (const [stage, agg] of canonEntries) {
1012
+ lines.push(` ${pad(stage, 14)} ${padLeft(fmt(agg.tokens), 12)} tok ${padLeft(String(agg.agents), 4)} agent(s) ${padLeft(String(agg.attempts), 4)} attempt(s)`);
1013
+ }
1014
+ }
842
1015
  if (report.recordTotalTokens !== null) {
843
1016
  lines.push(
844
1017
  ` note: the run record's own totalTokens is ${fmt(report.recordTotalTokens)} — a RAW unweighted cached sum of the same per-agent list, reported for traceability, NOT the invariant's right-hand side`,
@@ -1037,22 +1210,57 @@ export function deriveCostLedger(opts: DeriveCostLedgerOptions = {}): CostLedger
1037
1210
 
1038
1211
  const runSamples: CostLedgerSample[] = [];
1039
1212
  const perAgent = new Map<string, CostLedgerSample[]>();
1040
- const orphanAgentIds: string[] = [];
1213
+ // measurement-integrity FR-3: a transcript file with NO matching `workflowProgress[]` entry is an
1214
+ // orphan REGARDLESS of whether it happened to log any usage samples — a zero-sample orphan is
1215
+ // still a transcript the inventory does not know about, so it is counted here (0 tokens, still a
1216
+ // named id) rather than silently dropped the way the pre-existing `orphanAgentIds.length > 0`
1217
+ // gate did.
1218
+ //
1219
+ // measurement-integrity fix-round-1/F2 (Codex r1 HIGH #2): the OLD loop compared inventory
1220
+ // MEMBERSHIP by `agentId` alone and treated two anomalies as invisible: a file whose name does not
1221
+ // match the `agent-<id>.jsonl` shape was `continue`d past — dropped from BOTH `perAgent` and
1222
+ // `orphanTranscripts` — and a SECOND file for an agentId already known simply OVERWROTE the first
1223
+ // in `perAgent`, silently discarding one transcript's samples while reading as "the one known
1224
+ // agent". Both are now named inventory anomalies, folded into the SAME `orphanTranscripts` list
1225
+ // (so the existing `hasOrphan`/verdict machinery in `buildCostLedger` already refuses to call
1226
+ // either case BALANCED) with a synthetic, self-describing id — never silently absorbed as "known".
1227
+ const orphanTranscripts: { agentId: string; samples: CostLedgerSample[] }[] = [];
1228
+ const unparseableNames: string[] = [];
1229
+ const duplicateFor: string[] = [];
1041
1230
  for (const f of files) {
1042
1231
  const samples = extractCostSamples(safeReadText(join(ref.transcriptDir, f)));
1043
1232
  runSamples.push(...samples);
1233
+ // `journal.jsonl` is a KNOWN, EXPECTED per-run housekeeping file (present in every real run
1234
+ // directory alongside the `agent-<id>.jsonl` transcripts — verified against live
1235
+ // `roam/claude-state/**/subagents/workflows/wf_*` directories) that carries no usage samples of
1236
+ // its own. Naming it an inventory anomaly would make F2's fix fire on every single run there is
1237
+ // — the false-positive explosion this feature exists to AVOID, not cause.
1238
+ if (f === 'journal.jsonl') continue;
1044
1239
  const m = /^agent-(.+)\.jsonl$/.exec(f);
1045
- if (m === null) continue;
1240
+ if (m === null) {
1241
+ unparseableNames.push(f);
1242
+ orphanTranscripts.push({ agentId: `unparseable:${f}`, samples });
1243
+ continue;
1244
+ }
1046
1245
  const agentId = m[1] ?? '';
1047
- if (stageAgentIds.has(agentId)) perAgent.set(agentId, samples);
1048
- else if (samples.length > 0) orphanAgentIds.push(agentId);
1246
+ if (!stageAgentIds.has(agentId)) {
1247
+ orphanTranscripts.push({ agentId, samples });
1248
+ } else if (perAgent.has(agentId)) {
1249
+ // A SECOND transcript file for an agentId already claimed — never silently overwrite the
1250
+ // first one's samples nor pretend this file belongs to "the known agent" too.
1251
+ duplicateFor.push(agentId);
1252
+ orphanTranscripts.push({ agentId: `duplicate:${agentId}:${f}`, samples });
1253
+ } else {
1254
+ perAgent.set(agentId, samples);
1255
+ }
1049
1256
  }
1050
1257
 
1051
1258
  return buildCostLedger({
1052
1259
  record,
1053
1260
  stageSamples: [...perAgent.entries()].map(([agentId, samples]) => ({ agentId, samples })),
1054
1261
  runSamples,
1055
- orphanAgentIds,
1262
+ orphanAgentIds: orphanTranscripts.map((o) => o.agentId),
1263
+ orphanTranscripts,
1056
1264
  ...(listingTruncated ? { transcriptListingTruncated: true } : {}),
1057
1265
  ...(opts.epsilon !== undefined ? { epsilon: opts.epsilon } : {}),
1058
1266
  });