@tangle-network/agent-eval 0.128.2 → 0.129.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +265 -0
- package/README.md +18 -0
- package/dist/analyst/index.d.ts +107 -165
- package/dist/analyst/index.js +5 -9
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +2 -19
- package/dist/belief-state/index.js +30 -31
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +5 -8
- package/dist/benchmarks/index.js +12 -11
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +30 -39
- package/dist/campaign/index.js +11 -10
- package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
- package/dist/chunk-2QU3YOPR.js.map +1 -0
- package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
- package/dist/chunk-3OCR4R5I.js.map +1 -0
- package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
- package/dist/chunk-56TAVBOK.js.map +1 -0
- package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
- package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
- package/dist/chunk-BSO5JDQH.js.map +1 -0
- package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
- package/dist/chunk-C6LXANRU.js.map +1 -0
- package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
- package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
- package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
- package/dist/chunk-EG66UGL4.js.map +1 -0
- package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
- package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
- package/dist/chunk-G7MGMCZD.js.map +1 -0
- package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
- package/dist/chunk-H23X7XKK.js.map +1 -0
- package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
- package/dist/chunk-HPWUNB47.js.map +1 -0
- package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
- package/dist/chunk-IYCLP2N2.js.map +1 -0
- package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
- package/dist/chunk-JQSF5DQT.js.map +1 -0
- package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
- package/dist/chunk-M4YBQKIJ.js.map +1 -0
- package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
- package/dist/chunk-OIUOT4QD.js +44 -0
- package/dist/chunk-OIUOT4QD.js.map +1 -0
- package/dist/chunk-OWN5NPMC.js +152 -0
- package/dist/chunk-OWN5NPMC.js.map +1 -0
- package/dist/chunk-PC5DOSM7.js +579 -0
- package/dist/chunk-PC5DOSM7.js.map +1 -0
- package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
- package/dist/chunk-QB6BDBP2.js.map +1 -0
- package/dist/chunk-RXHCETDZ.js +536 -0
- package/dist/chunk-RXHCETDZ.js.map +1 -0
- package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
- package/dist/chunk-SFLLL76A.js.map +1 -0
- package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
- package/dist/chunk-T6RLYGAD.js.map +1 -0
- package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
- package/dist/chunk-TJVT4QFF.js.map +1 -0
- package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
- package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
- package/dist/chunk-U4L7JRPZ.js.map +1 -0
- package/dist/chunk-U4PHLT2N.js +419 -0
- package/dist/chunk-U4PHLT2N.js.map +1 -0
- package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
- package/dist/chunk-VCZ5FQYW.js.map +1 -0
- package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
- package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
- package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
- package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
- package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
- package/dist/chunk-ZHTZ4EYI.js.map +1 -0
- package/dist/cli.js +6 -5
- package/dist/cli.js.map +1 -1
- package/dist/contract/index.d.ts +47 -87
- package/dist/contract/index.js +14 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +3 -2
- package/dist/fuzz.js +3 -2
- package/dist/fuzz.js.map +1 -1
- package/dist/index.d.ts +659 -203
- package/dist/index.js +145 -117
- package/dist/index.js.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +3 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +5 -5
- package/dist/reporting.d.ts +14 -0
- package/dist/reporting.js +7 -6
- package/dist/rl.d.ts +652 -82
- package/dist/rl.js +415 -171
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1071 -32
- package/dist/rollout/index.js +68 -10
- package/dist/run-campaign-OJJ7CZF4.js +18 -0
- package/dist/supervisor-run/index.d.ts +114 -4
- package/dist/supervisor-run/index.js +4 -3
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +6 -5
- package/dist/wire/index.d.ts +10 -11
- package/dist/wire/index.js +3 -3
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +4 -4
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
- /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
- /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
- /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
- /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
- /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
- /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
- /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
- /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,271 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.129.0] - 2026-07-25 - provider-neutral chat and canonical rollout training
|
|
8
|
+
|
|
9
|
+
### Changed
|
|
10
|
+
|
|
11
|
+
- **Breaking:** benchmark, driver, executor, judge, completion-checker, tracing, and analyst APIs now accept `ChatClient`.
|
|
12
|
+
- **Breaking:** removed the exported provider SDK type, direct provider SDK dependency, provider-specific retry fields, and custom completion-checker error receipt callback.
|
|
13
|
+
- **Breaking:** `LlmClientOptions.maximumAttempts` replaces the misleading `maxRetries` name, which already represented total attempts.
|
|
14
|
+
- **Breaking:** `toGrpoRows`, `toSftRows`, `extractPreferences`, and `buildRlDataset` accept only `MintedRolloutLine[]`.
|
|
15
|
+
Convert run records once with `mintRolloutRows`.
|
|
16
|
+
- **Breaking:** removed `rolloutReward`, record-input training overloads, record-only reward hooks, duplicate line lookup and preference option types, and the duplicate dataset split map.
|
|
17
|
+
- **Breaking:** removed scalar belief-state and off-policy `qHat` fields; contextual estimates require `qHatChosen` and `vHatTarget` together.
|
|
18
|
+
- **Breaking:** removed `CampaignAggregates.totalCostUsd`, `CostLedgerEntry`, `VerifiableReward.breakdown`, and the fixed-prompt `JudgeFn` factories.
|
|
19
|
+
- **Breaking:** removed the unused `OptimizationProposer` alias; use `SurfaceProposer`.
|
|
20
|
+
- **Breaking:** `CampaignStorage.append` is required; read/write-only storage adapters are no longer accepted.
|
|
21
|
+
- **Breaking:** `RawAnalystFinding` now has one plural `evidence` field.
|
|
22
|
+
Removed the duplicate `CanonicalRawAnalystFinding` names, singular-evidence adapters, the second
|
|
23
|
+
recovery callback, `AnalystRunSummary.cost_usd`, and finding-metadata cost accounting.
|
|
24
|
+
- Paid calls read canonical `ChatResponse.content`, usage, model, duration, and cost.
|
|
25
|
+
- Cost reservations derive provider retries from `ChatClient.maximumAttempts`; capped calls reject clients that do not declare a finite attempt count.
|
|
26
|
+
- `createChatClient({ transport: 'custom' })` adapts external SDKs and transports without importing them into Agent Eval.
|
|
27
|
+
- Updated Agent Core to `0.4.22` and Agent Interface to `0.34.0`.
|
|
28
|
+
- No compatibility aliases, overloads, environment fallbacks, or alternate readers preserve these
|
|
29
|
+
removed fields and functions.
|
|
30
|
+
- Settled cost events accept one current receipt shape, and execution summaries read only
|
|
31
|
+
`outcome.raw.execution_error_count`.
|
|
32
|
+
- Single-run locks accept only structured owner records; plain-PID lock files are rejected.
|
|
33
|
+
|
|
34
|
+
### Fixed
|
|
35
|
+
|
|
36
|
+
- **A run flagged as gamed exported at full positive reward through every RL path.** The realness gate
|
|
37
|
+
(`outcome.realness.gated`) existed in exactly one function, `rolloutReward`, called from exactly one
|
|
38
|
+
place — `mintRolloutRows`. The same derivation, `outcome.holdoutScore ?? outcome.searchScore`, was
|
|
39
|
+
hand-rolled at 20 other sites with no gate. Six of those sites feed exported training data: the GRPO
|
|
40
|
+
default reward and the SFT row metadata (`rl/exporters.ts`), the DPO preference ordering
|
|
41
|
+
(`rl/preferences.ts`), the probabilistic verifiable-reward fallback (`rl/verifiable-reward.ts`), the
|
|
42
|
+
corpus `minScore` filter (`rl/corpus.ts`), and the published datasheet's reward statistics
|
|
43
|
+
(`rl/dataset.ts`). A gamed run therefore trained at its claimed score, and in DPO it became the
|
|
44
|
+
*chosen* side of a pair against its honest sibling. Every one of those six now derives its reward
|
|
45
|
+
through the gate.
|
|
46
|
+
|
|
47
|
+
### Added
|
|
48
|
+
|
|
49
|
+
- `trainingScore`, `trainingReward`, `observedScore`, `isRealnessGated`, and the `ScorePreference`
|
|
50
|
+
type — the score derivation now lives once, in `src/rollout/reward.ts`, behind two names that force
|
|
51
|
+
the caller to state intent. `trainingScore` / `trainingReward` are gated and required for anything a
|
|
52
|
+
trainer or an exported dataset consumes; `observedScore` is raw and documented as unsafe for
|
|
53
|
+
training data. Raw is a legitimate choice — reward-hack detection, scorecards, and curriculum
|
|
54
|
+
allocation need the ungated number, and gating a detector's proxy would make it report "clean" on
|
|
55
|
+
precisely the population being gamed — so the fix names the choice rather than removing it.
|
|
56
|
+
- A regression test (`src/rollout/reward-invariant.test.ts`) with two halves: a source-level check that
|
|
57
|
+
the bare derivation appears nowhere outside `rollout/reward.ts`, and a behavioural check that pushes
|
|
58
|
+
one gated record with a 0.95 score through mint, GRPO, SFT, DPO, verifiable reward, the dataset
|
|
59
|
+
bundle, and the corpus filter, asserting 0 in each. Against the pre-fix tree the source check reports
|
|
60
|
+
21 offending lines and 7 of the 10 tests fail.
|
|
61
|
+
|
|
62
|
+
### Changed
|
|
63
|
+
|
|
64
|
+
- `rolloutReward` was removed.
|
|
65
|
+
Use `trainingReward` for score derivation or `rolloutRewardFields` when producing a rollout outcome.
|
|
66
|
+
- The 14 analysis, reporting, and detection sites that legitimately want the raw number now call
|
|
67
|
+
`observedScore` explicitly. Behaviour is unchanged at all 14, including the two sites that
|
|
68
|
+
deliberately prefer the search split (`rl/active-curriculum.ts`, via the new `ScorePreference`
|
|
69
|
+
argument). (`description-length-gate.ts` reads through `runTaskScore`, which 0.127.0 stripped of
|
|
70
|
+
its obsolete `raw.score` fallback.)
|
|
71
|
+
|
|
72
|
+
### Fixed — second pass (the waist now enforces its own invariant)
|
|
73
|
+
|
|
74
|
+
An adversarial review of the pass above found the hole still open on 13 paths. Its core finding:
|
|
75
|
+
`validateRolloutLine({outcome: {reward: 0.95, realness_gated: true}})` returned **zero errors**. It
|
|
76
|
+
type-checked `reward` and it type-checked `realness_gated`, and never once checked the RELATIONSHIP
|
|
77
|
+
between them. `RolloutLine` was a plain structural interface, so any object literal of that shape WAS
|
|
78
|
+
one — "the input is a rollout line" guaranteed nothing, and the ledger round-tripped a poisoned line
|
|
79
|
+
unchanged.
|
|
80
|
+
|
|
81
|
+
- **The invariant now lives in the validator.** `validateRolloutLine` / `assertRolloutLine` reject
|
|
82
|
+
`reward > 0` together with `realness_gated: true`, with an error that explains the rule rather than
|
|
83
|
+
naming the fields. Because `writeRolloutLedger` and `readRolloutLedger` both assert, a poisoned line
|
|
84
|
+
can neither enter a ledger nor leave one — which closes the published-CLI path (`agent-eval
|
|
85
|
+
rollout-release`) at its entrance: `buildHfDataset` reads through `readRolloutLedger`, so the gated
|
|
86
|
+
line is refused before `verifiers/train.jsonl` and `rft/train.jsonl` are written. The dataset card's
|
|
87
|
+
claim about the flag was a **false claim on a published artifact** until now; it is now enforced, and
|
|
88
|
+
the card states the count of gated lines it shipped.
|
|
89
|
+
- **And in the type system.** `MintedRolloutLine` brands the line with a phantom `unique symbol`
|
|
90
|
+
(nothing at runtime, identical JSON). It is produced only by `mintRolloutRows`, `readRolloutLedger`,
|
|
91
|
+
or an explicit `assertMinted` / `assertMintedLines`. The training exporters now require it:
|
|
92
|
+
`rollout/exporters` (`toSftRows`, `toRewardRows`, `toVerifiersRolloutOutput(s)`, `toRftItem(s)`),
|
|
93
|
+
`rl/exporters` (`toGrpoRows`, `toSftRows`, `PrmLineContext.lines`), `rl/preferences.extractPreferences`,
|
|
94
|
+
and `rl/dataset.buildRlDataset`. Belt and braces on purpose: the brand closes first-party call sites
|
|
95
|
+
at compile time, the validator closes data arriving at runtime.
|
|
96
|
+
- **The regex guard is replaced, not extended.** `src/rollout/score-derivation-guard.ts` walks the
|
|
97
|
+
TypeScript AST of `src/**` and flags every READ of `outcome.holdoutScore` / `outcome.searchScore`
|
|
98
|
+
outside a *counted* allowlist (writes and declarations are untouched). The old line regex caught 2 of
|
|
99
|
+
the 7 re-derivations the review planted; the AST rule catches all 7, and they are kept as a permanent
|
|
100
|
+
fixture in `reward-invariant.test.ts` rather than a one-time demonstration.
|
|
101
|
+
- **`supervisorRunRolloutLines` was a second minting door**, writing `outcome.reward` from the judge
|
|
102
|
+
score and omitting `realness_gated` entirely. It now states the flag explicitly on every supervisor
|
|
103
|
+
and worker row, and its rows are plain `RolloutLine`s — a caller putting them into a training export
|
|
104
|
+
has to run them through `assertMinted` first.
|
|
105
|
+
- **`EvalTraceStore.getBest` ranked few-shot exemplars on the ungated score** while its doc comment
|
|
106
|
+
claimed otherwise. `runScore` is now gated, and `getBest` drops realness-gated runs outright instead
|
|
107
|
+
of ranking them: whatever it returns is pasted into the next agent's prompt as an example to imitate,
|
|
108
|
+
so the SFT rule applies. When every run for a scenario is gated the answer is `null`, not the
|
|
109
|
+
least-bad fake.
|
|
110
|
+
- **`release-confidence.passRate` counted a gamed run as a pass.** Gated runs are now excluded from
|
|
111
|
+
both numerator and denominator, and the count ships beside the rate as `metrics.realnessGatedRuns`.
|
|
112
|
+
`HeldOutGate` gets the same treatment: gated runs are dropped from both sides before pairing, with
|
|
113
|
+
`evidence.realnessGatedRuns` surfacing how many. Never a silent 0 — a shrunken denominator has to say
|
|
114
|
+
by how much.
|
|
115
|
+
- **Ten remaining hand-rolled derivations routed by classification**: `rl/sim-fidelity.ts` (×2, RAW —
|
|
116
|
+
gating a sim-vs-production divergence measure would report the simulator as more faithful precisely
|
|
117
|
+
where it is gamed), `belief-state/code-agent-corpus.ts` (GATED — its output becomes corpus labels),
|
|
118
|
+
`eval-trace-store.ts` (GATED, above), `contract/analyze-runs.ts` (×3, RAW), `summary-report.ts` (×4,
|
|
119
|
+
RAW), `release-confidence.ts` (RAW), `held-out-gate.ts` (RAW, over an already-degated set).
|
|
120
|
+
- **`trainingReward` no longer collapses an unscored record to 0.** It returns `reward: null`, matching
|
|
121
|
+
the schema's own "a labeled gap, never 0" rule; a gated run still returns 0, because that IS a
|
|
122
|
+
verdict. Previously a run nobody graded was indistinguishable from one graded a total failure.
|
|
123
|
+
|
|
124
|
+
### Fixed — the published dataset (`agent-eval rollout-release`)
|
|
125
|
+
|
|
126
|
+
- **The card's gate claim is no longer a sentence; it is a rendering of measured counts.** A README that
|
|
127
|
+
STATES what the build does is a claim about bytes it never reads, and it drifts the moment an exporter
|
|
128
|
+
changes — to whoever downloads the dataset. `buildHfDataset` now exports the rows for every selected
|
|
129
|
+
format, measures the realness-gated rows among them (`measureFormatGate`, matched on `rollout_id`),
|
|
130
|
+
checks the measurement against the declared per-format policy, and only then writes. `buildDatasetCard`
|
|
131
|
+
requires that report, renders it, and **throws** if it disagrees with the lines it describes or with the
|
|
132
|
+
policy. A card that contradicts its own data files cannot be produced without failing the build first.
|
|
133
|
+
The measurement also ships on `BuildSummary.gate` and in the CLI's stdout JSON.
|
|
134
|
+
- **Nothing is written when any config would ship a gated row above reward 0.** Formats used to be
|
|
135
|
+
exported and written one at a time; a build that failed halfway left a poisoned config on disk for
|
|
136
|
+
someone to `--push`. Rows are now computed and gate-checked for every format before the first byte.
|
|
137
|
+
- **The per-format decision is stated once as data**, in `src/rollout/release/gate-report.ts`:
|
|
138
|
+
`sft: 'exclude'` (an SFT row is imitated verbatim — a gamed trajectory must not appear at any weight),
|
|
139
|
+
`verifiers` / `rft` / `raw`: `'zero-and-flag'`. Keeping gated rows in the last three is deliberate: in
|
|
140
|
+
`verifiers` the reward is a signed learning signal, so a gamed trajectory at reward 0 is a correct
|
|
141
|
+
negative, and dropping it would bias the negative population toward honest failures and leave a trainer
|
|
142
|
+
no example of gaming being penalized; `rft` re-samples the completion, so only the prompt and the grader
|
|
143
|
+
reference ship; `raw` is an audit dump, where the gated row is the one an auditor most wants.
|
|
144
|
+
- **Reward 0 is never the only label.** Zeroing without the flag makes a faked success indistinguishable
|
|
145
|
+
from an honest failure — it hides the gamed population instead of disclosing it. `VerifiersRolloutOutput.
|
|
146
|
+
info.realness_gated`, `RftItem.reference.realness_gated`, and `RewardRow.metadata.realness_gated` are new
|
|
147
|
+
and always present, so a consumer can filter the population out or select it for a gaming detector.
|
|
148
|
+
|
|
149
|
+
### Fixed — Harbor ATIF interchange (`src/rollout/interchange/harbor.ts`)
|
|
150
|
+
|
|
151
|
+
- **Export emitted documents that violate ATIF MUST rule 2.** Tool results were folded into the
|
|
152
|
+
`observation` of whichever step happened to PRECEDE them, so an assistant turn that declared no tool
|
|
153
|
+
calls could carry a `source_call_id`, a result could be attached to a step that declared a different
|
|
154
|
+
call, and an unanswered result rode a synthetic `system` step carrying a `source_call_id` a system
|
|
155
|
+
step can never declare. Results now attach only to the step that declared their `tool_call_id`;
|
|
156
|
+
everything else becomes a carrier step whose observation states no call id and escrows it instead.
|
|
157
|
+
Message order is preserved exactly in every case, and `ruleTwoViolations` checks the whole tree in
|
|
158
|
+
the tests.
|
|
159
|
+
- **`logprobs`, `prompt_token_ids`, `completion_token_ids` and per-step `llm_call_count` were adopted
|
|
160
|
+
onto our types but never wired through the interchange.** They were escrowed under
|
|
161
|
+
`extra.tangle.spans`, where no foreign consumer looks, and ATIF's own `step.metrics` /
|
|
162
|
+
`step.llm_call_count` were left empty in both directions — so a Harbor-native file's logprobs were
|
|
163
|
+
read, validated, and dropped. They now travel on the native channel both ways (escrow still wins on
|
|
164
|
+
import for exactness); a step carrying none of them still produces no span.
|
|
165
|
+
- **`session_id` was invocation-scoped.** ATIF's `session_id` is RUN-scoped; export set it to the root
|
|
166
|
+
LINE's `rollout_id`, so two roots of one run got different session ids and foreign tooling grouping
|
|
167
|
+
by session split the run. It is now `run_id`, on every node.
|
|
168
|
+
- **`is_copied_context` (RFC rule 7) was silently dropped on import.** It is now a field on
|
|
169
|
+
`ChatMessage`, validated, carried both ways on ATIF's native step field, and — the part the RFC
|
|
170
|
+
actually mandates — `toSftRows` excludes those turns, dropping the row entirely if nothing else is
|
|
171
|
+
left.
|
|
172
|
+
- **An escrowed split was trusted from any document.** `extra.tangle.task.split: 'search'` in a
|
|
173
|
+
hand-written or third-party file imported as a TRAINABLE split; the escrow key is namespaced, not
|
|
174
|
+
authenticated. Import now forces `holdout` unconditionally, and promotion is an explicit, greppable
|
|
175
|
+
step (`relabelImportedSplit`) so `grep` enumerates every place foreign data was declared trainable.
|
|
176
|
+
- **Round-tripping was not idempotent.** `provenance.gap` accreted one copy of the import note per
|
|
177
|
+
pass, and imported messages were assembled in an order that depended on which optional fields were
|
|
178
|
+
present, so a ledger hashed on serialized bytes saw a diff. The gap is now composed as a
|
|
179
|
+
de-duplicated ordered set and every imported message is built in the canonical schema key order.
|
|
180
|
+
- **The interchange was not root-exported.** `import { toHarborTrajectory } from
|
|
181
|
+
'@tangle-network/agent-eval'` failed — the symbols existed only on the `/rollout` subpath while every
|
|
182
|
+
other rollout symbol was on both.
|
|
183
|
+
- **The reward-absence test was a substring scan.** `expect(serialized).not.toContain('reward')` passed
|
|
184
|
+
only because the fixture happened to have no reward-shaped key in `outcome.metrics`; it says nothing
|
|
185
|
+
about WHERE a match is and false-positives on any metric named e.g. `reward_hack_rate`. It is now a
|
|
186
|
+
structural walk that reports the PATH of every label-shaped key, exempting the escrowed metrics bag
|
|
187
|
+
by path.
|
|
188
|
+
|
|
189
|
+
### Known gap (not fixed here)
|
|
190
|
+
|
|
191
|
+
- `mintRolloutRows` hardcodes `tool_defs: []`, so `agent.tool_definitions` is absent on every minted
|
|
192
|
+
line's ATIF export. This is a capture-side gap, not an interchange one: neither `RunRecord` nor the
|
|
193
|
+
trace-span projection carries a tool schema, so there is nothing for mint to read. Fixing it means
|
|
194
|
+
recording the harness's tool definitions at capture time.
|
|
195
|
+
|
|
196
|
+
### Fixed — unrelated flake encountered on the way
|
|
197
|
+
|
|
198
|
+
- `node:sqlite` is loaded through `createRequire` in `rollout/readers/opencode-sqlite.ts` and its test.
|
|
199
|
+
esbuild and Vite both rewrite an `import()` of a builtin and strip the `node:` prefix, producing a bogus
|
|
200
|
+
`sqlite` package lookup; composing the specifier at runtime did not reliably defeat it, so the failure
|
|
201
|
+
moved between workers whenever a test file was added. A require obtained from `createRequire` is not an
|
|
202
|
+
analyzable module reference in either tool.
|
|
203
|
+
|
|
204
|
+
### Added — second pass
|
|
205
|
+
|
|
206
|
+
- `MintedRolloutLine`, `MintedRolloutOutcome`, `assertMinted`, `assertMintedLines` (rollout barrel +
|
|
207
|
+
root barrel).
|
|
208
|
+
- `observedSplitScore` (the raw score on ONE split, no cross-split fallback — what every split-scoped
|
|
209
|
+
report and promotion gate actually wants) and `scoreOrigin` / `ScoreOrigin` (which split carried the
|
|
210
|
+
score, or that none did — the provenance `reward_source` is built from).
|
|
211
|
+
- `malformedRolloutLine` in `rollout/fixtures` for tests whose subject is the validator itself;
|
|
212
|
+
`fixtureRolloutLine` now validates on every construction and returns a `MintedRolloutLine`.
|
|
213
|
+
- `rollout/release/gate-report`: `FORMAT_GATE_DISPOSITION`, `GateDisposition`, `GateReport`,
|
|
214
|
+
`FormatGateCounts`, `ReleaseRowRef`, `gatedRolloutIds`, `releaseRowRefs`, `measureFormatGate`,
|
|
215
|
+
`assertGateReport` (rollout barrel). `BuildSummary.gate` and `DatasetCardInputs.gate` are new;
|
|
216
|
+
`DatasetCardInputs.gate` is required, so a card cannot be rendered without the measurement.
|
|
217
|
+
|
|
218
|
+
### Fixed — third pass (one canonical training input)
|
|
219
|
+
|
|
220
|
+
Trainer-facing APIs now accept canonical minted lines only.
|
|
221
|
+
Artifacts that carry run ids without embedded reward state still require explicit line context.
|
|
222
|
+
|
|
223
|
+
- **Record-input preference and trainer overloads were removed.**
|
|
224
|
+
Their custom reward hooks and independent filtering rules created alternate paths around the canonical rollout checks.
|
|
225
|
+
Callers now mint once and every downstream transform reads the same reward and split fields.
|
|
226
|
+
- **`extractVerifiableRewardsFromRecords` gated only its judge-fallback branch.** The deterministic
|
|
227
|
+
branch — the highest-credibility channel the module emits, `determinism: 'deterministic'`,
|
|
228
|
+
`confidence: 1`, and what the module header calls "the RL training signal" — returned the layer
|
|
229
|
+
score untouched, so a gated run carrying `outcome.raw['layer.test'] = 1.0` exported at value 1 and
|
|
230
|
+
`filterDeterministicallyRewarded` kept it. That is exactly the shape of a reward-hacked coding run:
|
|
231
|
+
`realness.gated` means the success signal was faked, and a test suite reporting green on a stubbed
|
|
232
|
+
integration IS the deterministic layer being the thing that got faked. The gate applies to that
|
|
233
|
+
channel most, not least. `value` and every `components` entry are now 0 on a gated run — zeroing
|
|
234
|
+
`value` alone would let a consumer re-weighting per source reconstruct the refused reward — and the
|
|
235
|
+
new `VerifiableReward.realnessGated` distinguishes "measured a genuine failure" from "claimed a
|
|
236
|
+
success we refuse to believe", which a bare 0 cannot.
|
|
237
|
+
- **`toPrmRows(triples, lookups)` — the deprecated 2-arg form — applied no gate and now fails
|
|
238
|
+
closed.** A `PrmTrainingTriple` carries a bare `chosenReward` number, so without the minted lines
|
|
239
|
+
the exporter has no way to learn that its chosen step belongs to a run that faked its success; the
|
|
240
|
+
rows it produced trained a process-reward model to prefer the gaming move at the exact step the
|
|
241
|
+
gaming happened. The overload is removed (TypeScript callers fail to compile) and the runtime
|
|
242
|
+
throws for everyone else.
|
|
243
|
+
- **`supervisorRunRolloutLines` no longer writes the reward pair by hand.** `reward` and
|
|
244
|
+
`realness_gated` come out of one call in the module that owns the gate — `rolloutRewardFields` for
|
|
245
|
+
`mintRolloutRows`, `unscreenedRewardFields` for a producer with a score but no `RunRecord` behind
|
|
246
|
+
it. Two minting doors is the same class of defect as two reward derivations; there is now one
|
|
247
|
+
writer of the pair, so a future third door cannot state one field and forget the other.
|
|
248
|
+
- **`EvalTraceStore.compareRuns` counted a gamed run as a silent zero.** Gating `runScore` (second
|
|
249
|
+
pass, above) fixed few-shot seeding and quietly changed this: a gamed run entered the paired
|
|
250
|
+
comparison at 0, which reads as "this candidate failed the scenario" when what happened is "this
|
|
251
|
+
candidate's result is not evidence". Gated runs are now excluded and counted in
|
|
252
|
+
`CandidateComparison.realnessGatedRuns` — the same never-a-silent-0 rule `passRate` follows.
|
|
253
|
+
|
|
254
|
+
#### Deliberately NOT gated
|
|
255
|
+
|
|
256
|
+
`rl/reward-hacking.ts` reads the deterministic reward through the new
|
|
257
|
+
`VerifiableRewardExtractionOptions.applyRealnessGate: false`, which preserves its previous behaviour
|
|
258
|
+
exactly. Its `judge_drift` and `reward_disagreement` signals measure the GAP between the judge reward
|
|
259
|
+
and the deterministic one; a deterministic reward another gate already forced to 0 opens that gap by
|
|
260
|
+
construction on the gamed population, so the detector would fire on its own input rather than on
|
|
261
|
+
evidence it found. Same reasoning as its ungated `DEFAULT_PROXY`. The option defaults to `true` and an
|
|
262
|
+
empty options object gates — the opt-out is explicit and greppable.
|
|
263
|
+
|
|
264
|
+
### Known, not fixed here
|
|
265
|
+
|
|
266
|
+
- `description-length-gate.ts` gives a gated run claiming `score: 1.0` the largest possible improvement
|
|
267
|
+
to its objective, and `product-benchmark/export.ts` publishes a gated run with `pass: true`. Each
|
|
268
|
+
site carries a comment naming the hole.
|
|
269
|
+
- `extractVerifiableReward(report)` — the `VerificationReport` signature — cannot gate and does not
|
|
270
|
+
claim to: `realness` lives on the `RunRecord`, not on the report. Documented on the function.
|
|
271
|
+
|
|
7
272
|
## [0.128.2] - 2026-07-25 - current core contract
|
|
8
273
|
|
|
9
274
|
### Changed
|
package/README.md
CHANGED
|
@@ -24,6 +24,24 @@ Model calls occur only through the clients and agents you configure.
|
|
|
24
24
|
pnpm add @tangle-network/agent-eval
|
|
25
25
|
```
|
|
26
26
|
|
|
27
|
+
## Configure Model Calls
|
|
28
|
+
|
|
29
|
+
Benchmarks, user drivers, executors, built-in judges, completion checkers, and judge adapters accept the same `ChatClient`.
|
|
30
|
+
|
|
31
|
+
```ts
|
|
32
|
+
import { createChatClient } from '@tangle-network/agent-eval'
|
|
33
|
+
|
|
34
|
+
const chat = createChatClient({
|
|
35
|
+
transport: 'router',
|
|
36
|
+
apiKey: process.env.TANGLE_API_KEY!,
|
|
37
|
+
defaultModel: 'openai/gpt-4.1',
|
|
38
|
+
maximumAttempts: 3,
|
|
39
|
+
})
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Use `direct-provider` for an OpenAI-compatible endpoint, `cli-bridge` for a local subscription, `sandbox-sdk` for Sandbox, or `custom` to adapt another SDK.
|
|
43
|
+
A custom adapter must return `ChatResponse` and declare `maximumAttempts` before a capped cost ledger can dispatch it.
|
|
44
|
+
|
|
27
45
|
The official optimizers use the Python bridge.
|
|
28
46
|
Install only the optimizer you plan to run:
|
|
29
47
|
|