@tangle-network/agent-eval 0.128.2 → 0.129.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. package/CHANGELOG.md +265 -0
  2. package/README.md +18 -0
  3. package/dist/analyst/index.d.ts +107 -165
  4. package/dist/analyst/index.js +5 -9
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/belief-state/index.d.ts +2 -19
  7. package/dist/belief-state/index.js +30 -31
  8. package/dist/belief-state/index.js.map +1 -1
  9. package/dist/benchmarks/index.d.ts +5 -8
  10. package/dist/benchmarks/index.js +12 -11
  11. package/dist/builder-eval/index.js +1 -1
  12. package/dist/campaign/index.d.ts +30 -39
  13. package/dist/campaign/index.js +11 -10
  14. package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
  15. package/dist/chunk-2QU3YOPR.js.map +1 -0
  16. package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
  17. package/dist/chunk-3OCR4R5I.js.map +1 -0
  18. package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
  19. package/dist/chunk-56TAVBOK.js.map +1 -0
  20. package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
  21. package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
  22. package/dist/chunk-BSO5JDQH.js.map +1 -0
  23. package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
  24. package/dist/chunk-C6LXANRU.js.map +1 -0
  25. package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
  26. package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
  27. package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
  28. package/dist/chunk-EG66UGL4.js.map +1 -0
  29. package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
  30. package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
  31. package/dist/chunk-G7MGMCZD.js.map +1 -0
  32. package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
  33. package/dist/chunk-H23X7XKK.js.map +1 -0
  34. package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
  35. package/dist/chunk-HPWUNB47.js.map +1 -0
  36. package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
  37. package/dist/chunk-IYCLP2N2.js.map +1 -0
  38. package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
  39. package/dist/chunk-JQSF5DQT.js.map +1 -0
  40. package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
  41. package/dist/chunk-M4YBQKIJ.js.map +1 -0
  42. package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
  43. package/dist/chunk-OIUOT4QD.js +44 -0
  44. package/dist/chunk-OIUOT4QD.js.map +1 -0
  45. package/dist/chunk-OWN5NPMC.js +152 -0
  46. package/dist/chunk-OWN5NPMC.js.map +1 -0
  47. package/dist/chunk-PC5DOSM7.js +579 -0
  48. package/dist/chunk-PC5DOSM7.js.map +1 -0
  49. package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
  50. package/dist/chunk-QB6BDBP2.js.map +1 -0
  51. package/dist/chunk-RXHCETDZ.js +536 -0
  52. package/dist/chunk-RXHCETDZ.js.map +1 -0
  53. package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
  54. package/dist/chunk-SFLLL76A.js.map +1 -0
  55. package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
  56. package/dist/chunk-T6RLYGAD.js.map +1 -0
  57. package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
  58. package/dist/chunk-TJVT4QFF.js.map +1 -0
  59. package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
  60. package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
  61. package/dist/chunk-U4L7JRPZ.js.map +1 -0
  62. package/dist/chunk-U4PHLT2N.js +419 -0
  63. package/dist/chunk-U4PHLT2N.js.map +1 -0
  64. package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
  65. package/dist/chunk-VCZ5FQYW.js.map +1 -0
  66. package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
  67. package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
  68. package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
  69. package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
  70. package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
  71. package/dist/chunk-ZHTZ4EYI.js.map +1 -0
  72. package/dist/cli.js +6 -5
  73. package/dist/cli.js.map +1 -1
  74. package/dist/contract/index.d.ts +47 -87
  75. package/dist/contract/index.js +14 -13
  76. package/dist/contract/index.js.map +1 -1
  77. package/dist/control.js +3 -2
  78. package/dist/fuzz.js +3 -2
  79. package/dist/fuzz.js.map +1 -1
  80. package/dist/index.d.ts +659 -203
  81. package/dist/index.js +145 -117
  82. package/dist/index.js.map +1 -1
  83. package/dist/meta-eval/index.js +2 -2
  84. package/dist/multishot/index.d.ts +3 -4
  85. package/dist/multishot/index.js.map +1 -1
  86. package/dist/openapi.json +1 -1
  87. package/dist/pipelines/index.js +5 -5
  88. package/dist/reporting.d.ts +14 -0
  89. package/dist/reporting.js +7 -6
  90. package/dist/rl.d.ts +652 -82
  91. package/dist/rl.js +415 -171
  92. package/dist/rl.js.map +1 -1
  93. package/dist/rollout/index.d.ts +1071 -32
  94. package/dist/rollout/index.js +68 -10
  95. package/dist/run-campaign-OJJ7CZF4.js +18 -0
  96. package/dist/supervisor-run/index.d.ts +114 -4
  97. package/dist/supervisor-run/index.js +4 -3
  98. package/dist/traces.d.ts +1 -1
  99. package/dist/traces.js +6 -5
  100. package/dist/wire/index.d.ts +10 -11
  101. package/dist/wire/index.js +3 -3
  102. package/docs/feature-guide.md +1 -1
  103. package/docs/rollout.md +116 -2
  104. package/package.json +4 -4
  105. package/dist/chunk-2JX3CFMB.js.map +0 -1
  106. package/dist/chunk-DJKY2TSY.js.map +0 -1
  107. package/dist/chunk-EJGRPCO3.js.map +0 -1
  108. package/dist/chunk-EOSZT7PL.js.map +0 -1
  109. package/dist/chunk-EZJEIH2R.js.map +0 -1
  110. package/dist/chunk-IHQDPH7D.js.map +0 -1
  111. package/dist/chunk-MHELPNRP.js.map +0 -1
  112. package/dist/chunk-NACAGYSY.js.map +0 -1
  113. package/dist/chunk-NKAGIDE2.js.map +0 -1
  114. package/dist/chunk-NYLOYM6N.js.map +0 -1
  115. package/dist/chunk-PBE2LOSS.js.map +0 -1
  116. package/dist/chunk-TT4KNT67.js +0 -124
  117. package/dist/chunk-TT4KNT67.js.map +0 -1
  118. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  119. package/dist/chunk-UWZZKKU7.js +0 -237
  120. package/dist/chunk-UWZZKKU7.js.map +0 -1
  121. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  122. package/dist/chunk-VGRCHJON.js.map +0 -1
  123. package/dist/chunk-VLOATJQ2.js.map +0 -1
  124. package/dist/chunk-VZSRQ272.js.map +0 -1
  125. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  126. package/dist/chunk-XDWDC2MP.js.map +0 -1
  127. package/dist/chunk-XPRT64IE.js.map +0 -1
  128. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  129. /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
  130. /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
  131. /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
  132. /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
  133. /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
  134. /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
  135. /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
  136. /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
  137. /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/CHANGELOG.md CHANGED
@@ -4,6 +4,271 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
4
4
 
5
5
  ---
6
6
 
7
+ ## [0.129.0] - 2026-07-25 - provider-neutral chat and canonical rollout training
8
+
9
+ ### Changed
10
+
11
+ - **Breaking:** benchmark, driver, executor, judge, completion-checker, tracing, and analyst APIs now accept `ChatClient`.
12
+ - **Breaking:** removed the exported provider SDK type, direct provider SDK dependency, provider-specific retry fields, and custom completion-checker error receipt callback.
13
+ - **Breaking:** `LlmClientOptions.maximumAttempts` replaces the misleading `maxRetries` name, which already represented total attempts.
14
+ - **Breaking:** `toGrpoRows`, `toSftRows`, `extractPreferences`, and `buildRlDataset` accept only `MintedRolloutLine[]`.
15
+ Convert run records once with `mintRolloutRows`.
16
+ - **Breaking:** removed `rolloutReward`, record-input training overloads, record-only reward hooks, duplicate line lookup and preference option types, and the duplicate dataset split map.
17
+ - **Breaking:** removed scalar belief-state and off-policy `qHat` fields; contextual estimates require `qHatChosen` and `vHatTarget` together.
18
+ - **Breaking:** removed `CampaignAggregates.totalCostUsd`, `CostLedgerEntry`, `VerifiableReward.breakdown`, and the fixed-prompt `JudgeFn` factories.
19
+ - **Breaking:** removed the unused `OptimizationProposer` alias; use `SurfaceProposer`.
20
+ - **Breaking:** `CampaignStorage.append` is required; read/write-only storage adapters are no longer accepted.
21
+ - **Breaking:** `RawAnalystFinding` now has one plural `evidence` field.
22
+ Removed the duplicate `CanonicalRawAnalystFinding` names, singular-evidence adapters, the second
23
+ recovery callback, `AnalystRunSummary.cost_usd`, and finding-metadata cost accounting.
24
+ - Paid calls read canonical `ChatResponse.content`, usage, model, duration, and cost.
25
+ - Cost reservations derive provider retries from `ChatClient.maximumAttempts`; capped calls reject clients that do not declare a finite attempt count.
26
+ - `createChatClient({ transport: 'custom' })` adapts external SDKs and transports without importing them into Agent Eval.
27
+ - Updated Agent Core to `0.4.22` and Agent Interface to `0.34.0`.
28
+ - No compatibility aliases, overloads, environment fallbacks, or alternate readers preserve these
29
+ removed fields and functions.
30
+ - Settled cost events accept one current receipt shape, and execution summaries read only
31
+ `outcome.raw.execution_error_count`.
32
+ - Single-run locks accept only structured owner records; plain-PID lock files are rejected.
33
+
34
+ ### Fixed
35
+
36
+ - **A run flagged as gamed exported at full positive reward through every RL path.** The realness gate
37
+ (`outcome.realness.gated`) existed in exactly one function, `rolloutReward`, called from exactly one
38
+ place — `mintRolloutRows`. The same derivation, `outcome.holdoutScore ?? outcome.searchScore`, was
39
+ hand-rolled at 20 other sites with no gate. Six of those sites feed exported training data: the GRPO
40
+ default reward and the SFT row metadata (`rl/exporters.ts`), the DPO preference ordering
41
+ (`rl/preferences.ts`), the probabilistic verifiable-reward fallback (`rl/verifiable-reward.ts`), the
42
+ corpus `minScore` filter (`rl/corpus.ts`), and the published datasheet's reward statistics
43
+ (`rl/dataset.ts`). A gamed run therefore trained at its claimed score, and in DPO it became the
44
+ *chosen* side of a pair against its honest sibling. Every one of those six now derives its reward
45
+ through the gate.
46
+
47
+ ### Added
48
+
49
+ - `trainingScore`, `trainingReward`, `observedScore`, `isRealnessGated`, and the `ScorePreference`
50
+ type — the score derivation now lives once, in `src/rollout/reward.ts`, behind two names that force
51
+ the caller to state intent. `trainingScore` / `trainingReward` are gated and required for anything a
52
+ trainer or an exported dataset consumes; `observedScore` is raw and documented as unsafe for
53
+ training data. Raw is a legitimate choice — reward-hack detection, scorecards, and curriculum
54
+ allocation need the ungated number, and gating a detector's proxy would make it report "clean" on
55
+ precisely the population being gamed — so the fix names the choice rather than removing it.
56
+ - A regression test (`src/rollout/reward-invariant.test.ts`) with two halves: a source-level check that
57
+ the bare derivation appears nowhere outside `rollout/reward.ts`, and a behavioural check that pushes
58
+ one gated record with a 0.95 score through mint, GRPO, SFT, DPO, verifiable reward, the dataset
59
+ bundle, and the corpus filter, asserting 0 in each. Against the pre-fix tree the source check reports
60
+ 21 offending lines and 7 of the 10 tests fail.
61
+
62
+ ### Changed
63
+
64
+ - `rolloutReward` was removed.
65
+ Use `trainingReward` for score derivation or `rolloutRewardFields` when producing a rollout outcome.
66
+ - The 14 analysis, reporting, and detection sites that legitimately want the raw number now call
67
+ `observedScore` explicitly. Behaviour is unchanged at all 14, including the two sites that
68
+ deliberately prefer the search split (`rl/active-curriculum.ts`, via the new `ScorePreference`
69
+ argument). (`description-length-gate.ts` reads through `runTaskScore`, which 0.127.0 stripped of
70
+ its obsolete `raw.score` fallback.)
71
+
72
+ ### Fixed — second pass (the waist now enforces its own invariant)
73
+
74
+ An adversarial review of the pass above found the hole still open on 13 paths. Its core finding:
75
+ `validateRolloutLine({outcome: {reward: 0.95, realness_gated: true}})` returned **zero errors**. It
76
+ type-checked `reward` and it type-checked `realness_gated`, and never once checked the RELATIONSHIP
77
+ between them. `RolloutLine` was a plain structural interface, so any object literal of that shape WAS
78
+ one — "the input is a rollout line" guaranteed nothing, and the ledger round-tripped a poisoned line
79
+ unchanged.
80
+
81
+ - **The invariant now lives in the validator.** `validateRolloutLine` / `assertRolloutLine` reject
82
+ `reward > 0` together with `realness_gated: true`, with an error that explains the rule rather than
83
+ naming the fields. Because `writeRolloutLedger` and `readRolloutLedger` both assert, a poisoned line
84
+ can neither enter a ledger nor leave one — which closes the published-CLI path (`agent-eval
85
+ rollout-release`) at its entrance: `buildHfDataset` reads through `readRolloutLedger`, so the gated
86
+ line is refused before `verifiers/train.jsonl` and `rft/train.jsonl` are written. The dataset card's
87
+ claim about the flag was a **false claim on a published artifact** until now; it is now enforced, and
88
+ the card states the count of gated lines it shipped.
89
+ - **And in the type system.** `MintedRolloutLine` brands the line with a phantom `unique symbol`
90
+ (nothing at runtime, identical JSON). It is produced only by `mintRolloutRows`, `readRolloutLedger`,
91
+ or an explicit `assertMinted` / `assertMintedLines`. The training exporters now require it:
92
+ `rollout/exporters` (`toSftRows`, `toRewardRows`, `toVerifiersRolloutOutput(s)`, `toRftItem(s)`),
93
+ `rl/exporters` (`toGrpoRows`, `toSftRows`, `PrmLineContext.lines`), `rl/preferences.extractPreferences`,
94
+ and `rl/dataset.buildRlDataset`. Belt and braces on purpose: the brand closes first-party call sites
95
+ at compile time, the validator closes data arriving at runtime.
96
+ - **The regex guard is replaced, not extended.** `src/rollout/score-derivation-guard.ts` walks the
97
+ TypeScript AST of `src/**` and flags every READ of `outcome.holdoutScore` / `outcome.searchScore`
98
+ outside a *counted* allowlist (writes and declarations are untouched). The old line regex caught 2 of
99
+ the 7 re-derivations the review planted; the AST rule catches all 7, and they are kept as a permanent
100
+ fixture in `reward-invariant.test.ts` rather than a one-time demonstration.
101
+ - **`supervisorRunRolloutLines` was a second minting door**, writing `outcome.reward` from the judge
102
+ score and omitting `realness_gated` entirely. It now states the flag explicitly on every supervisor
103
+ and worker row, and its rows are plain `RolloutLine`s — a caller putting them into a training export
104
+ has to run them through `assertMinted` first.
105
+ - **`EvalTraceStore.getBest` ranked few-shot exemplars on the ungated score** while its doc comment
106
+ claimed otherwise. `runScore` is now gated, and `getBest` drops realness-gated runs outright instead
107
+ of ranking them: whatever it returns is pasted into the next agent's prompt as an example to imitate,
108
+ so the SFT rule applies. When every run for a scenario is gated the answer is `null`, not the
109
+ least-bad fake.
110
+ - **`release-confidence.passRate` counted a gamed run as a pass.** Gated runs are now excluded from
111
+ both numerator and denominator, and the count ships beside the rate as `metrics.realnessGatedRuns`.
112
+ `HeldOutGate` gets the same treatment: gated runs are dropped from both sides before pairing, with
113
+ `evidence.realnessGatedRuns` surfacing how many. Never a silent 0 — a shrunken denominator has to say
114
+ by how much.
115
+ - **Ten remaining hand-rolled derivations routed by classification**: `rl/sim-fidelity.ts` (×2, RAW —
116
+ gating a sim-vs-production divergence measure would report the simulator as more faithful precisely
117
+ where it is gamed), `belief-state/code-agent-corpus.ts` (GATED — its output becomes corpus labels),
118
+ `eval-trace-store.ts` (GATED, above), `contract/analyze-runs.ts` (×3, RAW), `summary-report.ts` (×4,
119
+ RAW), `release-confidence.ts` (RAW), `held-out-gate.ts` (RAW, over an already-degated set).
120
+ - **`trainingReward` no longer collapses an unscored record to 0.** It returns `reward: null`, matching
121
+ the schema's own "a labeled gap, never 0" rule; a gated run still returns 0, because that IS a
122
+ verdict. Previously a run nobody graded was indistinguishable from one graded a total failure.
123
+
124
+ ### Fixed — the published dataset (`agent-eval rollout-release`)
125
+
126
+ - **The card's gate claim is no longer a sentence; it is a rendering of measured counts.** A README that
127
+ STATES what the build does is a claim about bytes it never reads, and it drifts the moment an exporter
128
+ changes — to whoever downloads the dataset. `buildHfDataset` now exports the rows for every selected
129
+ format, measures the realness-gated rows among them (`measureFormatGate`, matched on `rollout_id`),
130
+ checks the measurement against the declared per-format policy, and only then writes. `buildDatasetCard`
131
+ requires that report, renders it, and **throws** if it disagrees with the lines it describes or with the
132
+ policy. A card that contradicts its own data files cannot be produced without failing the build first.
133
+ The measurement also ships on `BuildSummary.gate` and in the CLI's stdout JSON.
134
+ - **Nothing is written when any config would ship a gated row above reward 0.** Formats used to be
135
+ exported and written one at a time; a build that failed halfway left a poisoned config on disk for
136
+ someone to `--push`. Rows are now computed and gate-checked for every format before the first byte.
137
+ - **The per-format decision is stated once as data**, in `src/rollout/release/gate-report.ts`:
138
+ `sft: 'exclude'` (an SFT row is imitated verbatim — a gamed trajectory must not appear at any weight),
139
+ `verifiers` / `rft` / `raw`: `'zero-and-flag'`. Keeping gated rows in the last three is deliberate: in
140
+ `verifiers` the reward is a signed learning signal, so a gamed trajectory at reward 0 is a correct
141
+ negative, and dropping it would bias the negative population toward honest failures and leave a trainer
142
+ no example of gaming being penalized; `rft` re-samples the completion, so only the prompt and the grader
143
+ reference ship; `raw` is an audit dump, where the gated row is the one an auditor most wants.
144
+ - **Reward 0 is never the only label.** Zeroing without the flag makes a faked success indistinguishable
145
+ from an honest failure — it hides the gamed population instead of disclosing it. `VerifiersRolloutOutput.
146
+ info.realness_gated`, `RftItem.reference.realness_gated`, and `RewardRow.metadata.realness_gated` are new
147
+ and always present, so a consumer can filter the population out or select it for a gaming detector.
148
+
149
+ ### Fixed — Harbor ATIF interchange (`src/rollout/interchange/harbor.ts`)
150
+
151
+ - **Export emitted documents that violate ATIF MUST rule 2.** Tool results were folded into the
152
+ `observation` of whichever step happened to PRECEDE them, so an assistant turn that declared no tool
153
+ calls could carry a `source_call_id`, a result could be attached to a step that declared a different
154
+ call, and an unanswered result rode a synthetic `system` step carrying a `source_call_id` a system
155
+ step can never declare. Results now attach only to the step that declared their `tool_call_id`;
156
+ everything else becomes a carrier step whose observation states no call id and escrows it instead.
157
+ Message order is preserved exactly in every case, and `ruleTwoViolations` checks the whole tree in
158
+ the tests.
159
+ - **`logprobs`, `prompt_token_ids`, `completion_token_ids` and per-step `llm_call_count` were adopted
160
+ onto our types but never wired through the interchange.** They were escrowed under
161
+ `extra.tangle.spans`, where no foreign consumer looks, and ATIF's own `step.metrics` /
162
+ `step.llm_call_count` were left empty in both directions — so a Harbor-native file's logprobs were
163
+ read, validated, and dropped. They now travel on the native channel both ways (escrow still wins on
164
+ import for exactness); a step carrying none of them still produces no span.
165
+ - **`session_id` was invocation-scoped.** ATIF's `session_id` is RUN-scoped; export set it to the root
166
+ LINE's `rollout_id`, so two roots of one run got different session ids and foreign tooling grouping
167
+ by session split the run. It is now `run_id`, on every node.
168
+ - **`is_copied_context` (RFC rule 7) was silently dropped on import.** It is now a field on
169
+ `ChatMessage`, validated, carried both ways on ATIF's native step field, and — the part the RFC
170
+ actually mandates — `toSftRows` excludes those turns, dropping the row entirely if nothing else is
171
+ left.
172
+ - **An escrowed split was trusted from any document.** `extra.tangle.task.split: 'search'` in a
173
+ hand-written or third-party file imported as a TRAINABLE split; the escrow key is namespaced, not
174
+ authenticated. Import now forces `holdout` unconditionally, and promotion is an explicit, greppable
175
+ step (`relabelImportedSplit`) so `grep` enumerates every place foreign data was declared trainable.
176
+ - **Round-tripping was not idempotent.** `provenance.gap` accreted one copy of the import note per
177
+ pass, and imported messages were assembled in an order that depended on which optional fields were
178
+ present, so a ledger hashed on serialized bytes saw a diff. The gap is now composed as a
179
+ de-duplicated ordered set and every imported message is built in the canonical schema key order.
180
+ - **The interchange was not root-exported.** `import { toHarborTrajectory } from
181
+ '@tangle-network/agent-eval'` failed — the symbols existed only on the `/rollout` subpath while every
182
+ other rollout symbol was on both.
183
+ - **The reward-absence test was a substring scan.** `expect(serialized).not.toContain('reward')` passed
184
+ only because the fixture happened to have no reward-shaped key in `outcome.metrics`; it says nothing
185
+ about WHERE a match is and false-positives on any metric named e.g. `reward_hack_rate`. It is now a
186
+ structural walk that reports the PATH of every label-shaped key, exempting the escrowed metrics bag
187
+ by path.
188
+
189
+ ### Known gap (not fixed here)
190
+
191
+ - `mintRolloutRows` hardcodes `tool_defs: []`, so `agent.tool_definitions` is absent on every minted
192
+ line's ATIF export. This is a capture-side gap, not an interchange one: neither `RunRecord` nor the
193
+ trace-span projection carries a tool schema, so there is nothing for mint to read. Fixing it means
194
+ recording the harness's tool definitions at capture time.
195
+
196
+ ### Fixed — unrelated flake encountered on the way
197
+
198
+ - `node:sqlite` is loaded through `createRequire` in `rollout/readers/opencode-sqlite.ts` and its test.
199
+ esbuild and Vite both rewrite an `import()` of a builtin and strip the `node:` prefix, producing a bogus
200
+ `sqlite` package lookup; composing the specifier at runtime did not reliably defeat it, so the failure
201
+ moved between workers whenever a test file was added. A require obtained from `createRequire` is not an
202
+ analyzable module reference in either tool.
203
+
204
+ ### Added — second pass
205
+
206
+ - `MintedRolloutLine`, `MintedRolloutOutcome`, `assertMinted`, `assertMintedLines` (rollout barrel +
207
+ root barrel).
208
+ - `observedSplitScore` (the raw score on ONE split, no cross-split fallback — what every split-scoped
209
+ report and promotion gate actually wants) and `scoreOrigin` / `ScoreOrigin` (which split carried the
210
+ score, or that none did — the provenance `reward_source` is built from).
211
+ - `malformedRolloutLine` in `rollout/fixtures` for tests whose subject is the validator itself;
212
+ `fixtureRolloutLine` now validates on every construction and returns a `MintedRolloutLine`.
213
+ - `rollout/release/gate-report`: `FORMAT_GATE_DISPOSITION`, `GateDisposition`, `GateReport`,
214
+ `FormatGateCounts`, `ReleaseRowRef`, `gatedRolloutIds`, `releaseRowRefs`, `measureFormatGate`,
215
+ `assertGateReport` (rollout barrel). `BuildSummary.gate` and `DatasetCardInputs.gate` are new;
216
+ `DatasetCardInputs.gate` is required, so a card cannot be rendered without the measurement.
217
+
218
+ ### Fixed — third pass (one canonical training input)
219
+
220
+ Trainer-facing APIs now accept canonical minted lines only.
221
+ Artifacts that carry run ids without embedded reward state still require explicit line context.
222
+
223
+ - **Record-input preference and trainer overloads were removed.**
224
+ Their custom reward hooks and independent filtering rules created alternate paths around the canonical rollout checks.
225
+ Callers now mint once and every downstream transform reads the same reward and split fields.
226
+ - **`extractVerifiableRewardsFromRecords` gated only its judge-fallback branch.** The deterministic
227
+ branch — the highest-credibility channel the module emits, `determinism: 'deterministic'`,
228
+ `confidence: 1`, and what the module header calls "the RL training signal" — returned the layer
229
+ score untouched, so a gated run carrying `outcome.raw['layer.test'] = 1.0` exported at value 1 and
230
+ `filterDeterministicallyRewarded` kept it. That is exactly the shape of a reward-hacked coding run:
231
+ `realness.gated` means the success signal was faked, and a test suite reporting green on a stubbed
232
+ integration IS the deterministic layer being the thing that got faked. The gate applies to that
233
+ channel most, not least. `value` and every `components` entry are now 0 on a gated run — zeroing
234
+ `value` alone would let a consumer re-weighting per source reconstruct the refused reward — and the
235
+ new `VerifiableReward.realnessGated` distinguishes "measured a genuine failure" from "claimed a
236
+ success we refuse to believe", which a bare 0 cannot.
237
+ - **`toPrmRows(triples, lookups)` — the deprecated 2-arg form — applied no gate and now fails
238
+ closed.** A `PrmTrainingTriple` carries a bare `chosenReward` number, so without the minted lines
239
+ the exporter has no way to learn that its chosen step belongs to a run that faked its success; the
240
+ rows it produced trained a process-reward model to prefer the gaming move at the exact step the
241
+ gaming happened. The overload is removed (TypeScript callers fail to compile) and the runtime
242
+ throws for everyone else.
243
+ - **`supervisorRunRolloutLines` no longer writes the reward pair by hand.** `reward` and
244
+ `realness_gated` come out of one call in the module that owns the gate — `rolloutRewardFields` for
245
+ `mintRolloutRows`, `unscreenedRewardFields` for a producer with a score but no `RunRecord` behind
246
+ it. Two minting doors is the same class of defect as two reward derivations; there is now one
247
+ writer of the pair, so a future third door cannot state one field and forget the other.
248
+ - **`EvalTraceStore.compareRuns` counted a gamed run as a silent zero.** Gating `runScore` (second
249
+ pass, above) fixed few-shot seeding and quietly changed this: a gamed run entered the paired
250
+ comparison at 0, which reads as "this candidate failed the scenario" when what happened is "this
251
+ candidate's result is not evidence". Gated runs are now excluded and counted in
252
+ `CandidateComparison.realnessGatedRuns` — the same never-a-silent-0 rule `passRate` follows.
253
+
254
+ #### Deliberately NOT gated
255
+
256
+ `rl/reward-hacking.ts` reads the deterministic reward through the new
257
+ `VerifiableRewardExtractionOptions.applyRealnessGate: false`, which preserves its previous behaviour
258
+ exactly. Its `judge_drift` and `reward_disagreement` signals measure the GAP between the judge reward
259
+ and the deterministic one; a deterministic reward another gate already forced to 0 opens that gap by
260
+ construction on the gamed population, so the detector would fire on its own input rather than on
261
+ evidence it found. Same reasoning as its ungated `DEFAULT_PROXY`. The option defaults to `true` and an
262
+ empty options object gates — the opt-out is explicit and greppable.
263
+
264
+ ### Known, not fixed here
265
+
266
+ - `description-length-gate.ts` gives a gated run claiming `score: 1.0` the largest possible improvement
267
+ to its objective, and `product-benchmark/export.ts` publishes a gated run with `pass: true`. Each
268
+ site carries a comment naming the hole.
269
+ - `extractVerifiableReward(report)` — the `VerificationReport` signature — cannot gate and does not
270
+ claim to: `realness` lives on the `RunRecord`, not on the report. Documented on the function.
271
+
7
272
  ## [0.128.2] - 2026-07-25 - current core contract
8
273
 
9
274
  ### Changed
package/README.md CHANGED
@@ -24,6 +24,24 @@ Model calls occur only through the clients and agents you configure.
24
24
  pnpm add @tangle-network/agent-eval
25
25
  ```
26
26
 
27
+ ## Configure Model Calls
28
+
29
+ Benchmarks, user drivers, executors, built-in judges, completion checkers, and judge adapters accept the same `ChatClient`.
30
+
31
+ ```ts
32
+ import { createChatClient } from '@tangle-network/agent-eval'
33
+
34
+ const chat = createChatClient({
35
+ transport: 'router',
36
+ apiKey: process.env.TANGLE_API_KEY!,
37
+ defaultModel: 'openai/gpt-4.1',
38
+ maximumAttempts: 3,
39
+ })
40
+ ```
41
+
42
+ Use `direct-provider` for an OpenAI-compatible endpoint, `cli-bridge` for a local subscription, `sandbox-sdk` for Sandbox, or `custom` to adapt another SDK.
43
+ A custom adapter must return `ChatResponse` and declare `maximumAttempts` before a capped cost ledger can dispatch it.
44
+
27
45
  The official optimizers use the Python bridge.
28
46
  Install only the optimizer you plan to run:
29
47