@tangle-network/agent-eval 0.128.2 → 0.129.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +265 -0
- package/README.md +18 -0
- package/dist/analyst/index.d.ts +107 -165
- package/dist/analyst/index.js +5 -9
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +2 -19
- package/dist/belief-state/index.js +30 -31
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +5 -8
- package/dist/benchmarks/index.js +12 -11
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +30 -39
- package/dist/campaign/index.js +11 -10
- package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
- package/dist/chunk-2QU3YOPR.js.map +1 -0
- package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
- package/dist/chunk-3OCR4R5I.js.map +1 -0
- package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
- package/dist/chunk-56TAVBOK.js.map +1 -0
- package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
- package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
- package/dist/chunk-BSO5JDQH.js.map +1 -0
- package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
- package/dist/chunk-C6LXANRU.js.map +1 -0
- package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
- package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
- package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
- package/dist/chunk-EG66UGL4.js.map +1 -0
- package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
- package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
- package/dist/chunk-G7MGMCZD.js.map +1 -0
- package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
- package/dist/chunk-H23X7XKK.js.map +1 -0
- package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
- package/dist/chunk-HPWUNB47.js.map +1 -0
- package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
- package/dist/chunk-IYCLP2N2.js.map +1 -0
- package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
- package/dist/chunk-JQSF5DQT.js.map +1 -0
- package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
- package/dist/chunk-M4YBQKIJ.js.map +1 -0
- package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
- package/dist/chunk-OIUOT4QD.js +44 -0
- package/dist/chunk-OIUOT4QD.js.map +1 -0
- package/dist/chunk-OWN5NPMC.js +152 -0
- package/dist/chunk-OWN5NPMC.js.map +1 -0
- package/dist/chunk-PC5DOSM7.js +579 -0
- package/dist/chunk-PC5DOSM7.js.map +1 -0
- package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
- package/dist/chunk-QB6BDBP2.js.map +1 -0
- package/dist/chunk-RXHCETDZ.js +536 -0
- package/dist/chunk-RXHCETDZ.js.map +1 -0
- package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
- package/dist/chunk-SFLLL76A.js.map +1 -0
- package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
- package/dist/chunk-T6RLYGAD.js.map +1 -0
- package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
- package/dist/chunk-TJVT4QFF.js.map +1 -0
- package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
- package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
- package/dist/chunk-U4L7JRPZ.js.map +1 -0
- package/dist/chunk-U4PHLT2N.js +419 -0
- package/dist/chunk-U4PHLT2N.js.map +1 -0
- package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
- package/dist/chunk-VCZ5FQYW.js.map +1 -0
- package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
- package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
- package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
- package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
- package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
- package/dist/chunk-ZHTZ4EYI.js.map +1 -0
- package/dist/cli.js +6 -5
- package/dist/cli.js.map +1 -1
- package/dist/contract/index.d.ts +47 -87
- package/dist/contract/index.js +14 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +3 -2
- package/dist/fuzz.js +3 -2
- package/dist/fuzz.js.map +1 -1
- package/dist/index.d.ts +659 -203
- package/dist/index.js +145 -117
- package/dist/index.js.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +3 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +5 -5
- package/dist/reporting.d.ts +14 -0
- package/dist/reporting.js +7 -6
- package/dist/rl.d.ts +652 -82
- package/dist/rl.js +415 -171
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1071 -32
- package/dist/rollout/index.js +68 -10
- package/dist/run-campaign-OJJ7CZF4.js +18 -0
- package/dist/supervisor-run/index.d.ts +114 -4
- package/dist/supervisor-run/index.js +4 -3
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +6 -5
- package/dist/wire/index.d.ts +10 -11
- package/dist/wire/index.js +3 -3
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +4 -4
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
- /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
- /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
- /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
- /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
- /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
- /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
- /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
- /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/dist/rollout/index.d.ts
CHANGED
|
@@ -25,6 +25,28 @@ import { DatabaseSync } from 'node:sqlite';
|
|
|
25
25
|
* `outcome.reward` is THE single scalar (null = no verdict exists — a
|
|
26
26
|
* labeled gap, never 0). `outcome.realness_gated` is the anti-Goodhart
|
|
27
27
|
* flag: a gated line must never export as a positive training example.
|
|
28
|
+
*
|
|
29
|
+
* That last sentence is enforced here, by `validateRolloutLine`, not merely
|
|
30
|
+
* documented. Validating `reward` and `realness_gated` independently — each a
|
|
31
|
+
* well-typed field, their COMBINATION unchecked — is what let a line claiming
|
|
32
|
+
* `{reward: 0.95, realness_gated: true}` validate clean and walk into every
|
|
33
|
+
* training export. The relationship between the two IS the invariant, so it is
|
|
34
|
+
* checked where every other structural claim about a line is checked.
|
|
35
|
+
*
|
|
36
|
+
* The invariant is about the OUTCOME, not about one field of it. Zeroing
|
|
37
|
+
* `reward` while `outcome.metrics` still carried the per-layer scores that
|
|
38
|
+
* reward was computed from exported the gamed signal anyway, in the dict the
|
|
39
|
+
* verifiers format reads as its per-rubric scores. So `gateGamedOutcome`
|
|
40
|
+
* transforms the whole outcome once, at `assertMinted` — the funnel every
|
|
41
|
+
* minted line passes — and the reward-bearing components are relocated to
|
|
42
|
+
* `provenance.gated_evidence`, which no exporter projects.
|
|
43
|
+
*
|
|
44
|
+
* WHICH checks each door applies is not decided in this file. `./gate-checks`
|
|
45
|
+
* owns the canonical list and the total per-entry-point policy; the three doors
|
|
46
|
+
* below (`validateRolloutLine`, `assertRewardGate`, `assertMinted`) each call
|
|
47
|
+
* `gateErrors` with their declared policy, so a check added to that list applies
|
|
48
|
+
* here without anyone editing this file, and a check deliberately skipped has to
|
|
49
|
+
* name itself there.
|
|
28
50
|
*/
|
|
29
51
|
declare const ROLLOUT_SCHEMA = "tangle.rollout.v1";
|
|
30
52
|
/** `agent` = a solo evaluation run (no multi-agent topology). */
|
|
@@ -59,6 +81,14 @@ interface ChatMessage {
|
|
|
59
81
|
/** Required on role:"tool" — the ChatToolCall this result answers. */
|
|
60
82
|
tool_call_id?: string;
|
|
61
83
|
name?: string;
|
|
84
|
+
/**
|
|
85
|
+
* Harbor ATIF `is_copied_context` (RFC 0001 rule 7): this turn was COPIED IN
|
|
86
|
+
* from another trajectory's context, not produced by the agent on this line.
|
|
87
|
+
* The RFC makes excluding it from SFT a MUST, and `toSftRows` does — training
|
|
88
|
+
* on it teaches the model to author text it never authored, and credits this
|
|
89
|
+
* run for another one's work. Absent = false (authored here).
|
|
90
|
+
*/
|
|
91
|
+
is_copied_context?: boolean;
|
|
62
92
|
}
|
|
63
93
|
interface ToolDef {
|
|
64
94
|
type: 'function';
|
|
@@ -82,6 +112,21 @@ interface RolloutStep {
|
|
|
82
112
|
output?: string;
|
|
83
113
|
status?: 'ok' | 'error';
|
|
84
114
|
durationMs?: number;
|
|
115
|
+
/**
|
|
116
|
+
* LLM inferences this span represents. 0 = deterministic dispatch with no
|
|
117
|
+
* model call — distinct from absent, which means the producer did not track it.
|
|
118
|
+
*/
|
|
119
|
+
llm_call_count?: number;
|
|
120
|
+
/** Exact prompt tokenization. Removes the ambiguity of re-tokenizing text at train time. */
|
|
121
|
+
prompt_token_ids?: number[];
|
|
122
|
+
/** Exact completion tokenization; aligns index-wise with `logprobs`. */
|
|
123
|
+
completion_token_ids?: number[];
|
|
124
|
+
/**
|
|
125
|
+
* Per-completion-token log probabilities under the sampling policy. Required
|
|
126
|
+
* for off-policy correction (importance weighting) when the rollout was
|
|
127
|
+
* generated by a policy other than the one being trained.
|
|
128
|
+
*/
|
|
129
|
+
logprobs?: number[];
|
|
85
130
|
}
|
|
86
131
|
interface RolloutTask {
|
|
87
132
|
/** Benchmark/suite id (e.g. "swe-bench-verified") or the experiment id. */
|
|
@@ -126,11 +171,39 @@ interface RolloutOutcome {
|
|
|
126
171
|
is_truncated: boolean;
|
|
127
172
|
error: string | null;
|
|
128
173
|
/**
|
|
129
|
-
* Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run
|
|
130
|
-
*
|
|
131
|
-
* line never qualifies for
|
|
174
|
+
* Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run faked
|
|
175
|
+
* its success signal. `true` requires `reward` to be 0 or null — the
|
|
176
|
+
* validator rejects the line otherwise — and the line never qualifies for
|
|
177
|
+
* SFT. Required on the wire: a line that does not state the flag does not
|
|
178
|
+
* validate, so no producer can dodge the gate by omitting it.
|
|
179
|
+
*
|
|
180
|
+
* `true` ALSO requires `metrics` to be empty and `verdict` to be null: the
|
|
181
|
+
* numbers the reward was computed from are relocated to
|
|
182
|
+
* `provenance.gated_evidence` by `gateGamedOutcome`. See that function for
|
|
183
|
+
* why zeroing the scalar alone was not enough.
|
|
132
184
|
*/
|
|
133
185
|
realness_gated: boolean;
|
|
186
|
+
/**
|
|
187
|
+
* Whether an authenticity SCREEN ever RAN on this reward — a different claim
|
|
188
|
+
* from `realness_gated`, which is the screen's VERDICT.
|
|
189
|
+
*
|
|
190
|
+
* `realness_gated: false` reads as "we looked and nothing fired". A producer
|
|
191
|
+
* with no screen at all was emitting exactly that, so a never-screened reward
|
|
192
|
+
* was indistinguishable on the wire from a screened-clean one, and the whole
|
|
193
|
+
* anti-Goodhart apparatus silently treated the first as the second. The two
|
|
194
|
+
* claims are now separable:
|
|
195
|
+
*
|
|
196
|
+
* - `true` — a screen ran; `realness_gated` is its verdict.
|
|
197
|
+
* - `false` — the producer declares it HAS no screen (`unscreenedRewardFields`).
|
|
198
|
+
* `assertMinted` REFUSES such a line when its reward is above
|
|
199
|
+
* zero: an unscreened positive reward is precisely the signal
|
|
200
|
+
* the gate exists to qualify, and nothing has qualified it.
|
|
201
|
+
* - absent — not stated. Pre-unification ledgers land here, as does a
|
|
202
|
+
* `RunRecord` carrying no `outcome.realness` at all. Absent is
|
|
203
|
+
* read as "unknown", never as `false` (which would refuse most
|
|
204
|
+
* of the existing corpus) and never as `true`.
|
|
205
|
+
*/
|
|
206
|
+
realness_screened?: boolean;
|
|
134
207
|
}
|
|
135
208
|
interface RolloutCostBlock {
|
|
136
209
|
usd: number | null;
|
|
@@ -140,6 +213,11 @@ interface RolloutCostBlock {
|
|
|
140
213
|
cache_read: number | null;
|
|
141
214
|
cache_write: number | null;
|
|
142
215
|
wall_s: number | null;
|
|
216
|
+
/**
|
|
217
|
+
* Total LLM inferences across the invocation (ATIF `llm_call_count`,
|
|
218
|
+
* aggregated). Optional and additive: absent = not tracked, never 0.
|
|
219
|
+
*/
|
|
220
|
+
llm_call_count?: number | null;
|
|
143
221
|
}
|
|
144
222
|
interface RolloutArtifacts {
|
|
145
223
|
patch_path: string | null;
|
|
@@ -147,11 +225,43 @@ interface RolloutArtifacts {
|
|
|
147
225
|
/** Source-of-truth transcript pointer (session id / jsonl path) for audit. */
|
|
148
226
|
transcript_ref: string | null;
|
|
149
227
|
}
|
|
228
|
+
/**
|
|
229
|
+
* The reward-bearing half of a GATED line's outcome, moved off `outcome` and
|
|
230
|
+
* parked here verbatim. Diagnostics, never training input — see
|
|
231
|
+
* `gateGamedOutcome`.
|
|
232
|
+
*/
|
|
233
|
+
interface GatedEvidence {
|
|
234
|
+
/** `outcome.metrics` exactly as the producer measured it. */
|
|
235
|
+
metrics?: Record<string, unknown>;
|
|
236
|
+
/** `outcome.verdict` verbatim — the judge record that claimed the success. */
|
|
237
|
+
verdict?: unknown;
|
|
238
|
+
/**
|
|
239
|
+
* The per-step fields `tangle.rollout.v1` does not declare, parked here when
|
|
240
|
+
* the gate projected `steps[]` down to the schema's own key set.
|
|
241
|
+
*
|
|
242
|
+
* A per-step reward is training signal exactly like the scalar, and `steps`
|
|
243
|
+
* rides through `toRewardRows` verbatim — so a gated line was shipping its
|
|
244
|
+
* step-level credit assignment at full value beside a `reward` of 0.
|
|
245
|
+
*/
|
|
246
|
+
steps?: unknown;
|
|
247
|
+
}
|
|
150
248
|
interface RolloutProvenance {
|
|
151
249
|
captured_at: string;
|
|
152
250
|
capture: RolloutCapture;
|
|
153
|
-
/**
|
|
251
|
+
/**
|
|
252
|
+
* Why this line is incomplete. Required when `messages` is empty (the
|
|
253
|
+
* transcript could not be recovered); also set by interchange importers to
|
|
254
|
+
* name a MISSING LABEL — an imported trajectory carries no verdict, so
|
|
255
|
+
* `outcome.reward` is null and this says why.
|
|
256
|
+
*/
|
|
154
257
|
gap?: string;
|
|
258
|
+
/**
|
|
259
|
+
* Present only on a realness-gated line: the outcome fields the gate
|
|
260
|
+
* relocated, kept so an auditor can still see WHY the run was gated and what
|
|
261
|
+
* it claimed. Deliberately OUTSIDE `outcome`, because every training exporter
|
|
262
|
+
* reads `outcome` and none reads `provenance`.
|
|
263
|
+
*/
|
|
264
|
+
gated_evidence?: GatedEvidence;
|
|
155
265
|
}
|
|
156
266
|
interface RolloutLine {
|
|
157
267
|
schema: typeof ROLLOUT_SCHEMA;
|
|
@@ -180,9 +290,120 @@ interface RolloutLine {
|
|
|
180
290
|
artifacts: RolloutArtifacts;
|
|
181
291
|
provenance: RolloutProvenance;
|
|
182
292
|
}
|
|
293
|
+
/**
|
|
294
|
+
* THE anti-Goodhart gate, applied to the WHOLE outcome as a TRANSFORMATION.
|
|
295
|
+
*
|
|
296
|
+
* Two prior rounds enforced the gate as a CHECK ON ONE FIELD at N call sites,
|
|
297
|
+
* and each round the next reward-bearing field leaked. The one that shipped:
|
|
298
|
+
* `mintRolloutRows` bulk-copied `RunRecord.outcome.raw` into `outcome.metrics`
|
|
299
|
+
* with no gate, so a gated run exported `reward: 0` (correct) while the
|
|
300
|
+
* deterministic per-layer scores that reward was COMPUTED FROM — the
|
|
301
|
+
* `layer.*` keys `rl/verifiable-reward.ts` calls the RL training signal —
|
|
302
|
+
* shipped at 1.0, in the top-level `metrics` dict of the Prime Intellect
|
|
303
|
+
* verifiers format, which IS that format's per-rubric score dict. `verdict`
|
|
304
|
+
* leaks the same way into `toRftItem`'s `reference.verdict`, where a grader
|
|
305
|
+
* author reads `resolved: true` off a run that faked it.
|
|
306
|
+
*
|
|
307
|
+
* So the rule is no longer "zero the field we remembered". It is: if the gate
|
|
308
|
+
* fired, the outcome that leaves here carries NOTHING positive that was derived
|
|
309
|
+
* from the reward, whichever field a present or future exporter decides to
|
|
310
|
+
* read. `reward` is already forced to 0 upstream (`trainingReward`) and
|
|
311
|
+
* REJECTED here if it is not; `metrics` and `verdict` are relocated.
|
|
312
|
+
*
|
|
313
|
+
* WHERE they go, and why relocation rather than deletion: zeroing destroys the
|
|
314
|
+
* audit trail that shows why the run was gated and what it claimed, which is
|
|
315
|
+
* the row an auditor most wants and the labeled example a gaming DETECTOR
|
|
316
|
+
* trains on. `provenance.gated_evidence` keeps every byte, at a path no
|
|
317
|
+
* training exporter reads — all four release configs and every `rl/exporters`
|
|
318
|
+
* shape project from `outcome`, `messages`, `cost` and `task`; none projects
|
|
319
|
+
* `provenance`. Auditability preserved, training signal removed, and a future
|
|
320
|
+
* exporter that reads a field nobody thought of is safe by construction because
|
|
321
|
+
* the field is empty rather than because the exporter remembered to check.
|
|
322
|
+
*
|
|
323
|
+
* Idempotent: a second application finds nothing left to move and returns the
|
|
324
|
+
* line unchanged, so re-minting a line read back off a ledger cannot clobber
|
|
325
|
+
* the evidence it already carries.
|
|
326
|
+
*/
|
|
327
|
+
declare function gateGamedOutcome(line: RolloutLine): RolloutLine;
|
|
183
328
|
declare function validateRolloutLine(value: unknown): string[];
|
|
184
329
|
declare function assertRolloutLine(value: unknown, context?: string): asserts value is RolloutLine;
|
|
185
330
|
declare function isRolloutLine(value: unknown): value is RolloutLine;
|
|
331
|
+
/**
|
|
332
|
+
* Phantom property. `declare const` means it exists only in the type system:
|
|
333
|
+
* nothing is written at runtime, so a branded line still serializes to exactly
|
|
334
|
+
* the same JSON as a plain one.
|
|
335
|
+
*/
|
|
336
|
+
declare const MINTED_ROLLOUT: unique symbol;
|
|
337
|
+
/**
|
|
338
|
+
* A minted outcome states the gate verdict — it is not allowed to stay silent —
|
|
339
|
+
* and, when that verdict is `true`, carries nothing else the reward was derived
|
|
340
|
+
* from (`gateGamedOutcome` has run).
|
|
341
|
+
*/
|
|
342
|
+
interface MintedRolloutOutcome extends RolloutOutcome {
|
|
343
|
+
realness_gated: boolean;
|
|
344
|
+
}
|
|
345
|
+
/**
|
|
346
|
+
* A `RolloutLine` whose reward has been checked against the anti-Goodhart
|
|
347
|
+
* invariant. The type every training-data exporter takes.
|
|
348
|
+
*
|
|
349
|
+
* Why a brand and not just the interface: `RolloutLine` is structural, so any
|
|
350
|
+
* hand-built object literal of the right shape IS one — which is how a line
|
|
351
|
+
* declaring `{reward: 0.95, realness_gated: true}` reached the exporters
|
|
352
|
+
* despite them "only accepting a minted line". The phantom symbol makes the
|
|
353
|
+
* type nominal: it cannot be produced by writing an object literal, only by
|
|
354
|
+
* `mintRolloutRows` (which applies the gate), `readRolloutLedger` (which
|
|
355
|
+
* validates every line off disk), or an explicit, greppable `assertMinted`.
|
|
356
|
+
*
|
|
357
|
+
* Belt and braces on purpose. The brand closes first-party call sites at
|
|
358
|
+
* COMPILE time; `validateRolloutLine` closes data arriving at RUNTIME (ledger
|
|
359
|
+
* files, foreign imports, JSON from another process) where types are absent.
|
|
360
|
+
* Neither alone is enough.
|
|
361
|
+
*
|
|
362
|
+
* Assignable to `RolloutLine` in one direction only: readers, analysis, and
|
|
363
|
+
* the ledger writer keep taking the plain type.
|
|
364
|
+
*/
|
|
365
|
+
type MintedRolloutLine = Omit<RolloutLine, 'outcome'> & {
|
|
366
|
+
readonly [MINTED_ROLLOUT]: true;
|
|
367
|
+
outcome: MintedRolloutOutcome;
|
|
368
|
+
};
|
|
369
|
+
/**
|
|
370
|
+
* Promote a line to the type the training exporters accept, applying the
|
|
371
|
+
* anti-Goodhart gate to the WHOLE outcome on the way through. THE escape hatch
|
|
372
|
+
* — grep `assertMinted` to enumerate every place a line enters the training
|
|
373
|
+
* path without coming from mint or a ledger.
|
|
374
|
+
*
|
|
375
|
+
* The gate runs HERE, once, rather than at each producer, because this is the
|
|
376
|
+
* single funnel every minted line passes: `mintRolloutRows` calls it,
|
|
377
|
+
* `readRolloutLedger` calls it per line off disk, `scrubLines` calls it on the
|
|
378
|
+
* way out of a release, and a hand-built line has no other door. One
|
|
379
|
+
* transformation at the funnel means an already-published ledger holding a
|
|
380
|
+
* gated line with populated `metrics` is RE-GATED when it is read, instead of
|
|
381
|
+
* being rejected (which would make every such artifact unreadable) or trusted
|
|
382
|
+
* (which is the leak). Three steps, in this order:
|
|
383
|
+
*
|
|
384
|
+
* 1. VALIDATE the schema.
|
|
385
|
+
* 2. REFUSE every check `GATE_POLICIES.assertMinted` marks `enforce` — today
|
|
386
|
+
* the reward relationship (which stays a REJECTION: a caller claiming
|
|
387
|
+
* `{reward: 0.95, realness_gated: true}` is a producer defect and must fail
|
|
388
|
+
* loudly, since laundering it into `reward: 0` here would hide the
|
|
389
|
+
* producer) and a positive reward the producer declared it never screened.
|
|
390
|
+
* 3. TRANSFORM the one check that policy marks `repair` — relocate the
|
|
391
|
+
* reward's components off `outcome` (`gateGamedOutcome`), so no exporter
|
|
392
|
+
* can leak them whichever field it reads.
|
|
393
|
+
*
|
|
394
|
+
* Step 2 enumerates nothing by hand: a check added to `GATE_CHECKS` is enforced
|
|
395
|
+
* here the moment its disposition in that policy says so.
|
|
396
|
+
*
|
|
397
|
+
* Also normalizes the optional wire flag to an explicit boolean.
|
|
398
|
+
* `realness_gated` is absent on pre-unification ledgers and absent means "not
|
|
399
|
+
* flagged" per the schema, so filling it in states a claim the line was already
|
|
400
|
+
* making, and makes the flag readable on every published row instead of most of
|
|
401
|
+
* them. `realness_screened` is NOT filled in: absent means "unknown", and
|
|
402
|
+
* inventing either value there would be the same overclaim this round removed.
|
|
403
|
+
*/
|
|
404
|
+
declare function assertMinted(value: unknown, context?: string): MintedRolloutLine;
|
|
405
|
+
/** `assertMinted` over a batch, naming the offending index in the error. */
|
|
406
|
+
declare function assertMintedLines(values: readonly unknown[], context?: string): MintedRolloutLine[];
|
|
186
407
|
|
|
187
408
|
/**
|
|
188
409
|
* Pure exporters over `tangle.rollout.v1` lines → the training-data shapes
|
|
@@ -195,14 +416,72 @@ declare function isRolloutLine(value: unknown): value is RolloutLine;
|
|
|
195
416
|
* All exporters are pure functions of the lines — filtering (never train on
|
|
196
417
|
* holdout, reward thresholds, the realness gate) happens HERE, on inline
|
|
197
418
|
* labels, no joins.
|
|
419
|
+
*
|
|
420
|
+
* Every exporter takes `MintedRolloutLine[]`, not `RolloutLine[]`: the reward
|
|
421
|
+
* on a minted line has been checked against the anti-Goodhart invariant, and
|
|
422
|
+
* the brand is what stops a hand-built object literal claiming a positive
|
|
423
|
+
* reward on a gamed run from being handed to an exporter that copies it
|
|
424
|
+
* verbatim into training data.
|
|
198
425
|
*/
|
|
199
426
|
|
|
427
|
+
/**
|
|
428
|
+
* The gate's two claims, which travel TOGETHER on every emitted row.
|
|
429
|
+
*
|
|
430
|
+
* `realness_gated` alone is ambiguous, and the ambiguity is exploitable:
|
|
431
|
+
* `false` reads as "we screened it and nothing fired", so a producer that has no
|
|
432
|
+
* screen at all emitted rows indistinguishable from screened-clean ones, and
|
|
433
|
+
* every consumer of the published dataset read them as clean. The second field
|
|
434
|
+
* is what separates the two claims, and it only removes the ambiguity if it
|
|
435
|
+
* reaches the WIRE — for a round it existed on `RolloutOutcome` and on no
|
|
436
|
+
* exported row shape at all, which left the published rows exactly as ambiguous
|
|
437
|
+
* as before.
|
|
438
|
+
*
|
|
439
|
+
* So there is one helper and every row shape spreads it. A row that states one
|
|
440
|
+
* claim without the other is not constructible by copying the pattern, and
|
|
441
|
+
* `exporters.test.ts` walks every emitted shape to prove none does.
|
|
442
|
+
*/
|
|
443
|
+
interface RealnessLabels {
|
|
444
|
+
/** The screen's VERDICT: the run faked its success signal. */
|
|
445
|
+
realness_gated: boolean;
|
|
446
|
+
/**
|
|
447
|
+
* Whether a screen RAN at all. `true` = it ran, so `realness_gated` is its
|
|
448
|
+
* verdict. `false` = the producer declares it has none. `null` = not stated
|
|
449
|
+
* (pre-unification producers), which is "unknown" and never "clean".
|
|
450
|
+
*/
|
|
451
|
+
realness_screened: boolean | null;
|
|
452
|
+
}
|
|
453
|
+
declare function realnessLabels(line: MintedRolloutLine): RealnessLabels;
|
|
200
454
|
interface TrainingExportOptions {
|
|
201
455
|
/** Include held-out evaluation data in training output. Default false. */
|
|
202
456
|
allowHeldOutTrainingData?: boolean;
|
|
203
457
|
/** Require reward to be strictly greater than this value. Default 0. */
|
|
204
458
|
minimumQualityExclusive?: number;
|
|
205
459
|
}
|
|
460
|
+
/**
|
|
461
|
+
* What a signed-signal exporter (verifiers, RFT) does with lines that are not
|
|
462
|
+
* clean trainable successes — realness-gated lines above all.
|
|
463
|
+
*
|
|
464
|
+
* - 'exclude' — the default, the same fail-closed policy as every
|
|
465
|
+
* other training export: positive, completed,
|
|
466
|
+
* non-gated rows on a trainable split.
|
|
467
|
+
* - 'zero-and-flag' — keep them, at their non-positive (or null) reward,
|
|
468
|
+
* with `RealnessLabels` on the row. The dataset release
|
|
469
|
+
* sets this per `FORMAT_GATE_DISPOSITION`: in these
|
|
470
|
+
* formats the reward is a signed learning signal, so a
|
|
471
|
+
* gamed trajectory at reward 0 is a correct negative,
|
|
472
|
+
* and dropping it would bias the negative population
|
|
473
|
+
* toward honest failures and leave a trainer no example
|
|
474
|
+
* of gaming being penalized. The split policy is NOT
|
|
475
|
+
* relaxed: held-out lines still need the named opt-in.
|
|
476
|
+
*
|
|
477
|
+
* SFT deliberately has no such option — an SFT row is an imitation target and
|
|
478
|
+
* a gamed trajectory must never appear in one at any weight.
|
|
479
|
+
*/
|
|
480
|
+
type GatedLineDisposition = 'exclude' | 'zero-and-flag';
|
|
481
|
+
interface SignedSignalExportOptions extends TrainingExportOptions {
|
|
482
|
+
/** Disposition for non-trainable lines. Default 'exclude'. */
|
|
483
|
+
gatedLines?: GatedLineDisposition;
|
|
484
|
+
}
|
|
206
485
|
type SftExportOptions = TrainingExportOptions;
|
|
207
486
|
interface SftRow {
|
|
208
487
|
messages: ChatMessage[];
|
|
@@ -212,15 +491,24 @@ interface SftRow {
|
|
|
212
491
|
candidate_id: string | null;
|
|
213
492
|
instance_id: string;
|
|
214
493
|
reward: number;
|
|
215
|
-
};
|
|
494
|
+
} & RealnessLabels;
|
|
216
495
|
}
|
|
217
496
|
/**
|
|
218
497
|
* Supervised fine-tune rows: the completed conversation of each qualifying
|
|
219
498
|
* line. Fail-closed filters: trainable split only (never holdout/canary),
|
|
220
|
-
*
|
|
221
|
-
* no trainable content
|
|
499
|
+
* reward strictly above `minimumQualityExclusive` (default 0), realness-gated
|
|
500
|
+
* lines never qualify, gap lines carry no trainable content, and
|
|
501
|
+
* copied-context turns are dropped from the transcript (Harbor ATIF RFC 0001
|
|
502
|
+
* rule 7 — see `ChatMessage.is_copied_context`).
|
|
503
|
+
*
|
|
504
|
+
* `realness_gated` is therefore always `false` on an emitted row. It is carried
|
|
505
|
+
* anyway: an SFT row is a pure imitation target, so the row states its realness
|
|
506
|
+
* claims instead of making the reader know the format's policy, and carrying
|
|
507
|
+
* both flags on all four shapes is what lets the release accounting measure
|
|
508
|
+
* every config with one rule rather than skipping the one whose row shape
|
|
509
|
+
* happened to omit the field.
|
|
222
510
|
*/
|
|
223
|
-
declare function toSftRows(lines:
|
|
511
|
+
declare function toSftRows(lines: MintedRolloutLine[], options?: SftExportOptions): SftRow[];
|
|
224
512
|
interface RewardRow {
|
|
225
513
|
/** First user turn — the task prompt. */
|
|
226
514
|
prompt: string;
|
|
@@ -232,12 +520,12 @@ interface RewardRow {
|
|
|
232
520
|
candidate_id: string | null;
|
|
233
521
|
instance_id: string;
|
|
234
522
|
split: RolloutSplit;
|
|
235
|
-
};
|
|
523
|
+
} & RealnessLabels;
|
|
236
524
|
}
|
|
237
525
|
/**
|
|
238
526
|
* Reward-labeled rows for completed, positive-quality training runs.
|
|
239
527
|
*/
|
|
240
|
-
declare function toRewardRows(lines:
|
|
528
|
+
declare function toRewardRows(lines: MintedRolloutLine[], options?: TrainingExportOptions): RewardRow[];
|
|
241
529
|
interface VerifiersTokenUsage {
|
|
242
530
|
input_tokens: number | null;
|
|
243
531
|
output_tokens: number | null;
|
|
@@ -264,10 +552,10 @@ interface VerifiersRolloutOutput {
|
|
|
264
552
|
generation: number | null;
|
|
265
553
|
candidate_index: number | null;
|
|
266
554
|
role: RolloutLine['role'];
|
|
267
|
-
};
|
|
555
|
+
} & RealnessLabels;
|
|
268
556
|
}
|
|
269
|
-
declare function toVerifiersRolloutOutput(line:
|
|
270
|
-
declare function toVerifiersRolloutOutputs(lines:
|
|
557
|
+
declare function toVerifiersRolloutOutput(line: MintedRolloutLine): VerifiersRolloutOutput;
|
|
558
|
+
declare function toVerifiersRolloutOutputs(lines: MintedRolloutLine[], options?: SignedSignalExportOptions): VerifiersRolloutOutput[];
|
|
271
559
|
interface RftItem {
|
|
272
560
|
/** Prompt turns only — the graded completion is re-sampled during RFT. */
|
|
273
561
|
messages: ChatMessage[];
|
|
@@ -280,18 +568,450 @@ interface RftItem {
|
|
|
280
568
|
suite: string;
|
|
281
569
|
split: RolloutSplit;
|
|
282
570
|
rollout_id: string;
|
|
283
|
-
};
|
|
571
|
+
} & RealnessLabels;
|
|
284
572
|
}
|
|
285
|
-
declare function toRftItem(line:
|
|
573
|
+
declare function toRftItem(line: MintedRolloutLine): RftItem;
|
|
286
574
|
/** RFT needs a real prompt: lines whose transcript starts with prompt turns. */
|
|
287
|
-
declare function toRftItems(lines:
|
|
575
|
+
declare function toRftItems(lines: MintedRolloutLine[], options?: SignedSignalExportOptions): RftItem[];
|
|
288
576
|
declare function toJsonl(rows: ReadonlyArray<unknown>): string;
|
|
289
577
|
|
|
578
|
+
/**
|
|
579
|
+
* THE canonical list of anti-Goodhart gate checks, plus the TOTAL policy every
|
|
580
|
+
* entry point has to declare over it.
|
|
581
|
+
*
|
|
582
|
+
* Four rounds of adversarial review found the same defect four times, and it
|
|
583
|
+
* was never the check itself: it was the COMPOSITION. `validateRolloutLine`
|
|
584
|
+
* composed one check, `assertMinted` composed two, `assertRewardGate` composed
|
|
585
|
+
* two of the three, `assertGateReport` composed its own pair — each by hand, in
|
|
586
|
+
* its own file. So a check added to the package applied wherever its author
|
|
587
|
+
* happened to remember, and the guard that forgot it looked exactly like the
|
|
588
|
+
* guard that didn't. The last leak was literally that: `assertRewardGate`
|
|
589
|
+
* composed `reward-relationship` + `gated-evidence` and not `unscreened-reward`,
|
|
590
|
+
* so a never-screened positive reward that `assertMinted` correctly REFUSED
|
|
591
|
+
* walked through all four waist exporters at full value.
|
|
592
|
+
*
|
|
593
|
+
* The fix is to make hand-composition impossible rather than to add a third
|
|
594
|
+
* call to the two places that had two:
|
|
595
|
+
*
|
|
596
|
+
* - `GATE_CHECKS` is a TOTAL map over `GateCheckId`. A new id with no
|
|
597
|
+
* implementation does not compile.
|
|
598
|
+
* - `GatePolicy` is a TOTAL map over `GateCheckId`. Every entry point
|
|
599
|
+
* declares one, so a new id makes EVERY entry point's policy a type error
|
|
600
|
+
* until it is wired. Wiring it means writing `enforced`, `repairedBy(...)`
|
|
601
|
+
* or `omittedBecause(...)` — and the last two force a written reason, so a
|
|
602
|
+
* silent gap is not expressible.
|
|
603
|
+
* - every check carries a `tripwire`: the minimal outcome it must refuse.
|
|
604
|
+
* `gate-checks.test.ts` feeds each tripwire to every entry point that
|
|
605
|
+
* declares `enforced` and requires a rejection, so wiring a check to the
|
|
606
|
+
* wrong disposition is a TEST failure even when it type-checks.
|
|
607
|
+
*
|
|
608
|
+
* Adding a check is therefore: append the id, write the check, and the compiler
|
|
609
|
+
* enumerates every place that has to decide about it.
|
|
610
|
+
*/
|
|
611
|
+
|
|
612
|
+
/**
|
|
613
|
+
* Every gate check in the package, in the order they are applied.
|
|
614
|
+
*
|
|
615
|
+
* Order is load-bearing only for which message a caller sees first: a line that
|
|
616
|
+
* trips two checks reports the earlier one, and `reward-relationship` is first
|
|
617
|
+
* because it is the invariant the other three protect.
|
|
618
|
+
*/
|
|
619
|
+
declare const GATE_CHECK_IDS: readonly ["reward-relationship", "gated-evidence", "undeclared-step-payload", "unscreened-reward"];
|
|
620
|
+
type GateCheckId = (typeof GATE_CHECK_IDS)[number];
|
|
621
|
+
/**
|
|
622
|
+
* An outcome as it reaches a check.
|
|
623
|
+
*
|
|
624
|
+
* Deliberately accepts a raw record as well as the typed shape: the checks are
|
|
625
|
+
* the RUNTIME half of the gate, and the callers they exist for — JSON off a
|
|
626
|
+
* ledger, a plain-JavaScript consumer of the published package — arrive with no
|
|
627
|
+
* types at all. `Partial` because a tripwire states only the fields it trips on.
|
|
628
|
+
*/
|
|
629
|
+
type GateCheckedOutcome = Partial<RolloutOutcome> | Readonly<Record<string, unknown>>;
|
|
630
|
+
/**
|
|
631
|
+
* What a gate check reads: the reward-bearing surface of ONE LINE.
|
|
632
|
+
*
|
|
633
|
+
* For three rounds the subject was the OUTCOME alone, and that assumption is
|
|
634
|
+
* what produced the next leak rather than any missing check: `steps[]` sits on
|
|
635
|
+
* the LINE, outside `outcome`, so a per-step reward on a gated line was read by
|
|
636
|
+
* no check at all while `toRewardRows` copied it out verbatim — through the
|
|
637
|
+
* MINTED door, not merely the raw one. Widening the subject is what makes
|
|
638
|
+
* "somewhere else on the line" a place the checks can see.
|
|
639
|
+
*
|
|
640
|
+
* `outcome` is REQUIRED, and that is the point: a bare `RolloutOutcome` is then
|
|
641
|
+
* not assignable to a subject, so every call site that used to pass one is a
|
|
642
|
+
* COMPILE error until it passes the line instead. A subject with an optional
|
|
643
|
+
* `outcome` would have let the old call sites keep compiling while silently
|
|
644
|
+
* checking nothing — the exact failure this module exists to make impossible.
|
|
645
|
+
*/
|
|
646
|
+
interface GateSubject {
|
|
647
|
+
outcome: GateCheckedOutcome;
|
|
648
|
+
/** The line's trajectory steps, when it carries any. */
|
|
649
|
+
steps?: unknown;
|
|
650
|
+
}
|
|
651
|
+
interface GateCheck {
|
|
652
|
+
id: GateCheckId;
|
|
653
|
+
/** One sentence: what this check refuses. */
|
|
654
|
+
refuses: string;
|
|
655
|
+
/** One dotted-path message per defect; `[]` when the line is clean. */
|
|
656
|
+
errors: (subject: GateSubject) => string[];
|
|
657
|
+
/**
|
|
658
|
+
* Every minimal subject that MUST trip `errors` — the executable form of
|
|
659
|
+
* `refuses`, and the reason a check cannot be added without being provable.
|
|
660
|
+
* The calibration test feeds each one to every entry point declaring
|
|
661
|
+
* `enforced`.
|
|
662
|
+
*
|
|
663
|
+
* A LIST rather than one case: a check that refuses two distinct populations
|
|
664
|
+
* (a positive reward AND a reward it cannot read as a number) proved able to
|
|
665
|
+
* hold for the first while silently passing the second, so each population
|
|
666
|
+
* states its own tripwire and each is exercised separately.
|
|
667
|
+
*/
|
|
668
|
+
tripwires: GateSubject[];
|
|
669
|
+
}
|
|
670
|
+
/**
|
|
671
|
+
* The reward-bearing outcome fields that are NOT the scalar: the numbers the
|
|
672
|
+
* reward was computed from, and the verdict record that claimed it.
|
|
673
|
+
*
|
|
674
|
+
* Returned as one block rather than filtered key-by-key. A key-name heuristic
|
|
675
|
+
* ("zero anything matching `layer.*` or `/score/`") is the same defect shape as
|
|
676
|
+
* the line-oriented regex the AST score guard replaced: it holds until someone
|
|
677
|
+
* names a metric `pass_fraction`, and the next reward-shaped key ships at full
|
|
678
|
+
* value. The producer's OWN classification — "this is the scalar, that is
|
|
679
|
+
* everything else" — is the only partition that cannot be out-guessed.
|
|
680
|
+
*/
|
|
681
|
+
declare function gatedEvidenceOf(subject: GateSubject): GatedEvidence | undefined;
|
|
682
|
+
/**
|
|
683
|
+
* The registry. Total over `GateCheckId`, so an id with no check does not
|
|
684
|
+
* compile, and `GATE_CHECK_IDS` stays the single enumeration everything
|
|
685
|
+
* iterates.
|
|
686
|
+
*/
|
|
687
|
+
declare const GATE_CHECKS: {
|
|
688
|
+
readonly [K in GateCheckId]: GateCheck;
|
|
689
|
+
};
|
|
690
|
+
/**
|
|
691
|
+
* What ONE entry point does about ONE check.
|
|
692
|
+
*
|
|
693
|
+
* `repair` and `omit` both carry a mandatory sentence, which is the mechanism
|
|
694
|
+
* that keeps a legitimate omission distinguishable from a forgotten one: you
|
|
695
|
+
* cannot skip a check without writing down why, and the reasons are readable
|
|
696
|
+
* side by side in `GATE_POLICIES`.
|
|
697
|
+
*/
|
|
698
|
+
type GateCheckDisposition = {
|
|
699
|
+
readonly kind: 'enforce';
|
|
700
|
+
}
|
|
701
|
+
/** Resolved by TRANSFORMING the line instead of rejecting it; `by` names the function. */
|
|
702
|
+
| {
|
|
703
|
+
readonly kind: 'repair';
|
|
704
|
+
readonly by: string;
|
|
705
|
+
}
|
|
706
|
+
/** Deliberately not applied here; `because` states the reason. */
|
|
707
|
+
| {
|
|
708
|
+
readonly kind: 'omit';
|
|
709
|
+
readonly because: string;
|
|
710
|
+
};
|
|
711
|
+
/** Total over `GateCheckId`: a new check makes every policy literal a type error. */
|
|
712
|
+
type GatePolicy = {
|
|
713
|
+
readonly [K in GateCheckId]: GateCheckDisposition;
|
|
714
|
+
};
|
|
715
|
+
/**
|
|
716
|
+
* Every entry point that decides about the gate, and what it decides.
|
|
717
|
+
*
|
|
718
|
+
* Read this as the package's gate policy in one screen. The four entry points
|
|
719
|
+
* are not interchangeable — a validator that rejects, a mint funnel that
|
|
720
|
+
* repairs, a runtime backstop for untyped callers, and a release certifier over
|
|
721
|
+
* emitted rows — and the dispositions say which is which.
|
|
722
|
+
*/
|
|
723
|
+
declare const GATE_POLICIES: {
|
|
724
|
+
/**
|
|
725
|
+
* The schema validator. Rejects the reward relationship and NOTHING ELSE, on
|
|
726
|
+
* purpose: it runs on every line read off disk, and the other two conditions
|
|
727
|
+
* describe artifacts that already exist.
|
|
728
|
+
*/
|
|
729
|
+
readonly validateRolloutLine: {
|
|
730
|
+
readonly 'reward-relationship': {
|
|
731
|
+
readonly kind: "enforce";
|
|
732
|
+
};
|
|
733
|
+
readonly 'gated-evidence': GateCheckDisposition;
|
|
734
|
+
readonly 'undeclared-step-payload': GateCheckDisposition;
|
|
735
|
+
readonly 'unscreened-reward': GateCheckDisposition;
|
|
736
|
+
};
|
|
737
|
+
/**
|
|
738
|
+
* The mint funnel — the single door every `MintedRolloutLine` passes. Validates
|
|
739
|
+
* first (so the reward relationship has already been rejected), then refuses
|
|
740
|
+
* what cannot be repaired, then repairs what can.
|
|
741
|
+
*/
|
|
742
|
+
readonly assertMinted: {
|
|
743
|
+
readonly 'reward-relationship': {
|
|
744
|
+
readonly kind: "enforce";
|
|
745
|
+
};
|
|
746
|
+
readonly 'gated-evidence': GateCheckDisposition;
|
|
747
|
+
readonly 'undeclared-step-payload': GateCheckDisposition;
|
|
748
|
+
readonly 'unscreened-reward': {
|
|
749
|
+
readonly kind: "enforce";
|
|
750
|
+
};
|
|
751
|
+
};
|
|
752
|
+
/**
|
|
753
|
+
* The runtime backstop, and the one entry point with no license to omit
|
|
754
|
+
* anything: it exists for callers the type system never saw (plain JavaScript
|
|
755
|
+
* handing an object literal to a published exporter), so a check it skips is a
|
|
756
|
+
* check that does not run at all for them. This is where the fourth leak was.
|
|
757
|
+
*/
|
|
758
|
+
readonly assertRewardGate: {
|
|
759
|
+
readonly 'reward-relationship': {
|
|
760
|
+
readonly kind: "enforce";
|
|
761
|
+
};
|
|
762
|
+
readonly 'gated-evidence': {
|
|
763
|
+
readonly kind: "enforce";
|
|
764
|
+
};
|
|
765
|
+
readonly 'undeclared-step-payload': {
|
|
766
|
+
readonly kind: "enforce";
|
|
767
|
+
};
|
|
768
|
+
readonly 'unscreened-reward': {
|
|
769
|
+
readonly kind: "enforce";
|
|
770
|
+
};
|
|
771
|
+
};
|
|
772
|
+
/**
|
|
773
|
+
* The release certifier. Same checks, measured over the rows a release is
|
|
774
|
+
* ABOUT TO WRITE rather than over one line's outcome — see `REPORT_MEASURES`
|
|
775
|
+
* in `release/gate-report.ts`, which is the second total map this policy
|
|
776
|
+
* drives.
|
|
777
|
+
*/
|
|
778
|
+
readonly assertGateReport: {
|
|
779
|
+
readonly 'reward-relationship': {
|
|
780
|
+
readonly kind: "enforce";
|
|
781
|
+
};
|
|
782
|
+
readonly 'gated-evidence': {
|
|
783
|
+
readonly kind: "enforce";
|
|
784
|
+
};
|
|
785
|
+
readonly 'undeclared-step-payload': {
|
|
786
|
+
readonly kind: "enforce";
|
|
787
|
+
};
|
|
788
|
+
readonly 'unscreened-reward': {
|
|
789
|
+
readonly kind: "enforce";
|
|
790
|
+
};
|
|
791
|
+
};
|
|
792
|
+
};
|
|
793
|
+
/** Every entry point that declares a gate policy. */
|
|
794
|
+
type GateEntryPoint = keyof typeof GATE_POLICIES;
|
|
795
|
+
/**
|
|
796
|
+
* Run the checks one entry point enforces. The ONLY way an entry point should
|
|
797
|
+
* obtain gate errors — hand-composing two of the three is the bug this module
|
|
798
|
+
* exists to remove.
|
|
799
|
+
*/
|
|
800
|
+
declare function gateErrors(subject: GateSubject, policy: GatePolicy): string[];
|
|
801
|
+
|
|
802
|
+
/**
|
|
803
|
+
* Harbor ATIF-v1.7 interchange — `tangle.rollout.v1` ⇄ Agent Trajectory
|
|
804
|
+
* Interchange Format.
|
|
805
|
+
*
|
|
806
|
+
* ATIF is the portability format (spec:
|
|
807
|
+
* https://www.harborframework.com/docs/agents/trajectory-format, normative
|
|
808
|
+
* RFC: harbor-framework/harbor `rfcs/0001-trajectory-format.md`). It sits
|
|
809
|
+
* BELOW the waist of the rollout hourglass in both directions — export reads
|
|
810
|
+
* `RolloutLine[]`, import writes `RolloutLine[]` — and it is never a source
|
|
811
|
+
* of training labels:
|
|
812
|
+
*
|
|
813
|
+
* ATIF models NO reward, NO judge verdict, NO task/split coordinates.
|
|
814
|
+
*
|
|
815
|
+
* Consequences, both deliberate:
|
|
816
|
+
* - EXPORT drops `outcome.reward`, `outcome.reward_source` and
|
|
817
|
+
* `outcome.verdict` entirely. They are not smuggled into `extra`: a
|
|
818
|
+
* third-party reading our ATIF file must not be able to mistake an
|
|
819
|
+
* agent-eval judge score for something ATIF sanctioned.
|
|
820
|
+
* - IMPORT therefore mints UNLABELED lines: `reward: null` (the existing
|
|
821
|
+
* "null reward is a labeled gap, never 0" semantics), `verdict: null`,
|
|
822
|
+
* and a `provenance.gap` naming the missing label. An imported
|
|
823
|
+
* trajectory is not a training example until a judge scores it.
|
|
824
|
+
*
|
|
825
|
+
* Everything else we own that ATIF has no field for travels in a namespaced
|
|
826
|
+
* escrow at `extra.tangle.*`, so our own round-trip is exact while a foreign
|
|
827
|
+
* reader can ignore it. Fields that neither ATIF nor the escrow can carry
|
|
828
|
+
* come back explicitly null / fail-closed, never invented.
|
|
829
|
+
*
|
|
830
|
+
* THE ESCROW IS NAMESPACED, NOT AUTHENTICATED. Anyone can write
|
|
831
|
+
* `extra.tangle.*` into a file. So the escrow may restore what a value IS, but
|
|
832
|
+
* never what a line is ALLOWED to do: `task.split` is forced to `holdout` on
|
|
833
|
+
* every import regardless of what the document claims, and promoting an
|
|
834
|
+
* imported trajectory to a trainable split is an explicit, greppable act
|
|
835
|
+
* (`relabelImportedSplit`) rather than a property of the file. The document
|
|
836
|
+
* keeps its claim — the claim just is not authority.
|
|
837
|
+
*
|
|
838
|
+
* Multi-agent shape differs on purpose. ATIF EMBEDS children in
|
|
839
|
+
* `subagent_trajectories`; we keep a flat ledger with a normalized
|
|
840
|
+
* `parent_rollout_id` edge. Export assembles the tree, import flattens it.
|
|
841
|
+
* `session_id` is RUN-scoped in ATIF, so it carries `run_id` — the coordinate
|
|
842
|
+
* that is shared by every invocation of one run — not `rollout_id`, which
|
|
843
|
+
* identifies a single invocation and would split one run across session ids.
|
|
844
|
+
*
|
|
845
|
+
* ROUND-TRIPPING IS IDEMPOTENT: `import(export(import(export(x))))` is
|
|
846
|
+
* byte-identical to `import(export(x))`. Import composes `provenance.gap` as a
|
|
847
|
+
* de-duplicated ordered set rather than appending, and it emits every
|
|
848
|
+
* `ChatMessage` with keys in the canonical schema order (role, content,
|
|
849
|
+
* reasoning_content, tool_calls, tool_call_id, name, is_copied_context), so a
|
|
850
|
+
* ledger hashed on serialized bytes sees no diff across further passes. The
|
|
851
|
+
* FIRST import may re-order a producer's keys — that is the canonicalization.
|
|
852
|
+
*
|
|
853
|
+
* NOT building a Letta converter. Letta's trajectory-v1 is a strict subset of
|
|
854
|
+
* what we need from ATIF here — no per-step or aggregate cost, no
|
|
855
|
+
* multi-agent/subagent structure, no token-id or logprob channel — so a Letta
|
|
856
|
+
* sink would carry less than this one and add a second format to keep
|
|
857
|
+
* correct. Decision recorded in docs/rollout.md; do not re-litigate without a
|
|
858
|
+
* concrete consumer that reads Letta and cannot read ATIF.
|
|
859
|
+
*/
|
|
860
|
+
|
|
861
|
+
declare const ATIF_SCHEMA_VERSION = "ATIF-v1.7";
|
|
862
|
+
/** Gap note on every imported line — ATIF carries no verdict, so nothing is scored. */
|
|
863
|
+
declare const HARBOR_IMPORT_GAP = "imported from Harbor ATIF; no verdict";
|
|
864
|
+
type HarborStepSource = 'system' | 'user' | 'agent';
|
|
865
|
+
interface HarborImageSource {
|
|
866
|
+
media_type: string;
|
|
867
|
+
path: string;
|
|
868
|
+
}
|
|
869
|
+
interface HarborContentPart {
|
|
870
|
+
type: 'text' | 'image';
|
|
871
|
+
text?: string;
|
|
872
|
+
source?: HarborImageSource;
|
|
873
|
+
}
|
|
874
|
+
interface HarborToolCall {
|
|
875
|
+
tool_call_id: string;
|
|
876
|
+
function_name: string;
|
|
877
|
+
/** ATIF requires a decoded JSON object here, unlike our raw argument string. */
|
|
878
|
+
arguments: Record<string, unknown>;
|
|
879
|
+
extra?: Record<string, unknown>;
|
|
880
|
+
}
|
|
881
|
+
interface HarborSubagentTrajectoryRef {
|
|
882
|
+
trajectory_id?: string;
|
|
883
|
+
trajectory_path?: string;
|
|
884
|
+
/** Informational only since v1.7 — never a resolution key. */
|
|
885
|
+
session_id?: string;
|
|
886
|
+
extra?: Record<string, unknown>;
|
|
887
|
+
}
|
|
888
|
+
interface HarborObservationResult {
|
|
889
|
+
source_call_id?: string;
|
|
890
|
+
content?: string | HarborContentPart[];
|
|
891
|
+
subagent_trajectory_ref?: HarborSubagentTrajectoryRef[];
|
|
892
|
+
extra?: Record<string, unknown>;
|
|
893
|
+
}
|
|
894
|
+
interface HarborObservation {
|
|
895
|
+
results: HarborObservationResult[];
|
|
896
|
+
}
|
|
897
|
+
interface HarborMetrics {
|
|
898
|
+
prompt_tokens?: number;
|
|
899
|
+
completion_tokens?: number;
|
|
900
|
+
cached_tokens?: number;
|
|
901
|
+
cost_usd?: number;
|
|
902
|
+
prompt_token_ids?: number[];
|
|
903
|
+
completion_token_ids?: number[];
|
|
904
|
+
logprobs?: number[];
|
|
905
|
+
extra?: Record<string, unknown>;
|
|
906
|
+
}
|
|
907
|
+
interface HarborStep {
|
|
908
|
+
/** Ordinal, sequential from 1. */
|
|
909
|
+
step_id: number;
|
|
910
|
+
timestamp?: string;
|
|
911
|
+
source: HarborStepSource;
|
|
912
|
+
model_name?: string;
|
|
913
|
+
reasoning_effort?: string | number;
|
|
914
|
+
message: string | HarborContentPart[];
|
|
915
|
+
reasoning_content?: string;
|
|
916
|
+
tool_calls?: HarborToolCall[];
|
|
917
|
+
observation?: HarborObservation;
|
|
918
|
+
metrics?: HarborMetrics;
|
|
919
|
+
llm_call_count?: number;
|
|
920
|
+
is_copied_context?: boolean;
|
|
921
|
+
extra?: Record<string, unknown>;
|
|
922
|
+
}
|
|
923
|
+
interface HarborAgent {
|
|
924
|
+
name: string;
|
|
925
|
+
version: string;
|
|
926
|
+
model_name?: string;
|
|
927
|
+
/** OpenAI function-calling schema — byte-identical to our `ToolDef`. */
|
|
928
|
+
tool_definitions?: ToolDef[];
|
|
929
|
+
extra?: Record<string, unknown>;
|
|
930
|
+
}
|
|
931
|
+
interface HarborFinalMetrics {
|
|
932
|
+
total_prompt_tokens?: number;
|
|
933
|
+
total_completion_tokens?: number;
|
|
934
|
+
total_cached_tokens?: number;
|
|
935
|
+
total_cost_usd?: number;
|
|
936
|
+
total_steps?: number;
|
|
937
|
+
extra?: Record<string, unknown>;
|
|
938
|
+
}
|
|
939
|
+
interface HarborTrajectory {
|
|
940
|
+
schema_version: string;
|
|
941
|
+
session_id?: string;
|
|
942
|
+
/** Required on embedded subagents; we always set it so lines stay joinable. */
|
|
943
|
+
trajectory_id?: string;
|
|
944
|
+
agent: HarborAgent;
|
|
945
|
+
steps: HarborStep[];
|
|
946
|
+
notes?: string;
|
|
947
|
+
final_metrics?: HarborFinalMetrics;
|
|
948
|
+
continued_trajectory_ref?: string;
|
|
949
|
+
subagent_trajectories?: HarborTrajectory[];
|
|
950
|
+
extra?: Record<string, unknown>;
|
|
951
|
+
}
|
|
952
|
+
/**
|
|
953
|
+
* Assemble one episode's flat lines into a single ATIF trajectory tree,
|
|
954
|
+
* linked by `parent_rollout_id`.
|
|
955
|
+
*
|
|
956
|
+
* Reward, verdict and split are NOT emitted (ATIF models none of them); the
|
|
957
|
+
* split and the rest of the task coordinates survive only in `extra.tangle`.
|
|
958
|
+
*
|
|
959
|
+
* We deliberately do NOT synthesize an `observation.subagent_trajectory_ref`
|
|
960
|
+
* pointing at each child: our ledger records WHICH invocation spawned a
|
|
961
|
+
* worker, not which STEP did, and attaching the ref to a guessed step would
|
|
962
|
+
* fabricate a causal claim. Children are embedded in `subagent_trajectories`
|
|
963
|
+
* (each with the `trajectory_id` the spec requires) and the edge is stated in
|
|
964
|
+
* the child's escrowed `parent_rollout_id`.
|
|
965
|
+
*
|
|
966
|
+
* Throws when the lines are not one tree — use `toHarborTrajectories` for a forest.
|
|
967
|
+
*/
|
|
968
|
+
declare function toHarborTrajectory(lines: RolloutLine[]): HarborTrajectory;
|
|
969
|
+
/** Every independent tree in the input, one ATIF document each. */
|
|
970
|
+
declare function toHarborTrajectories(lines: RolloutLine[]): HarborTrajectory[];
|
|
971
|
+
interface FromHarborOptions {
|
|
972
|
+
/** Injected clock for deterministic output when the source carries no capture time. */
|
|
973
|
+
now?: () => Date;
|
|
974
|
+
}
|
|
975
|
+
/**
|
|
976
|
+
* Flatten an ATIF trajectory tree back into `tangle.rollout.v1` lines, parent
|
|
977
|
+
* first, each child carrying `parent_rollout_id`.
|
|
978
|
+
*
|
|
979
|
+
* Every line comes back UNLABELED: `reward`, `reward_source` and `verdict` are
|
|
980
|
+
* null and `provenance.gap` says why. ATIF models no verdict, so scoring an
|
|
981
|
+
* imported trajectory is a judge's job, not this function's. Every line lands
|
|
982
|
+
* on `holdout` whatever the document claims — see `relabelImportedSplit`.
|
|
983
|
+
*/
|
|
984
|
+
declare function fromHarborTrajectory(trajectory: HarborTrajectory, options?: FromHarborOptions): RolloutLine[];
|
|
985
|
+
/**
|
|
986
|
+
* THE explicit door out of `holdout` for imported lines.
|
|
987
|
+
*
|
|
988
|
+
* Import forces `holdout` because a document's own claim about its split is not
|
|
989
|
+
* evidence — anyone can write `extra.tangle.task.split`. Promoting a file to a
|
|
990
|
+
* trainable split is an operator's decision about provenance they verified, so
|
|
991
|
+
* it is a separate, greppable call: `grep relabelImportedSplit` enumerates
|
|
992
|
+
* every place foreign data was declared trainable, which is exactly the audit
|
|
993
|
+
* the trusted-escrow version made impossible.
|
|
994
|
+
*
|
|
995
|
+
* Returns plain `RolloutLine`s. They still have to pass `assertMinted` (and its
|
|
996
|
+
* anti-Goodhart check) to reach an exporter — re-labeling a split is not
|
|
997
|
+
* minting a reward.
|
|
998
|
+
*/
|
|
999
|
+
declare function relabelImportedSplit(lines: readonly RolloutLine[], split: RolloutSplit): RolloutLine[];
|
|
1000
|
+
|
|
290
1001
|
/**
|
|
291
1002
|
* Rollout-ledger file API — append-only JSONL of validated `tangle.rollout.v1`
|
|
292
1003
|
* lines. Writes validate BEFORE touching disk (a bad line never lands);
|
|
293
1004
|
* reads validate line-by-line and fail loud with the line number, because a
|
|
294
1005
|
* silently-skipped rollout is a corrupted dataset.
|
|
1006
|
+
*
|
|
1007
|
+
* "Validate" includes the anti-Goodhart invariant (a realness-gated line may
|
|
1008
|
+
* not carry a positive reward), so a poisoned line can neither enter a ledger
|
|
1009
|
+
* nor leave one.
|
|
1010
|
+
*
|
|
1011
|
+
* Two read modes, matching the two write-side row classes: `readRolloutLedger`
|
|
1012
|
+
* re-validates under the mint policy (training data), `readRolloutJournal`
|
|
1013
|
+
* under the write policy (supervision journals, whose unscreened positive
|
|
1014
|
+
* rewards are writable and must stay readable).
|
|
295
1015
|
*/
|
|
296
1016
|
|
|
297
1017
|
/** Replace the ledger file with exactly `lines`. */
|
|
@@ -301,8 +1021,29 @@ declare function appendRolloutLines(path: string, lines: RolloutLine[]): Promise
|
|
|
301
1021
|
/**
|
|
302
1022
|
* Read and validate every line. Throws on the first malformed/invalid line
|
|
303
1023
|
* (with its 1-based line number) — fail-closed, never a silent drop.
|
|
1024
|
+
*
|
|
1025
|
+
* Validation includes the anti-Goodhart invariant, which is why the result is
|
|
1026
|
+
* `MintedRolloutLine[]`: a ledger file is the main way a rollout reaches this
|
|
1027
|
+
* process from outside the type system (another run, another machine, a
|
|
1028
|
+
* hand-edited JSONL), so this read is the runtime boundary where a poisoned
|
|
1029
|
+
* line is refused rather than exported.
|
|
304
1030
|
*/
|
|
305
|
-
declare function readRolloutLedger(path: string): Promise<
|
|
1031
|
+
declare function readRolloutLedger(path: string): Promise<MintedRolloutLine[]>;
|
|
1032
|
+
/**
|
|
1033
|
+
* Read a ledger under the WRITE-side policy (`validateRolloutLine`), which
|
|
1034
|
+
* omits the unscreened-reward check. `writeRolloutLedger` accepts a
|
|
1035
|
+
* supervision-journal row (`realness_screened: false` with a positive reward
|
|
1036
|
+
* — the documented `unscreenedRewardFields` shape), and `GATE_POLICIES` says
|
|
1037
|
+
* such rows "must stay writable, readable and reportable"; a read API that
|
|
1038
|
+
* only re-validated under `assertMinted` made every such file unreadable —
|
|
1039
|
+
* write-accepted but read-refused is a data-loss trap.
|
|
1040
|
+
*
|
|
1041
|
+
* The result is `RolloutLine[]`, NOT `MintedRolloutLine[]`: nothing read here
|
|
1042
|
+
* can reach a training exporter without passing `assertMinted`, so the
|
|
1043
|
+
* promotion gate (which DOES enforce unscreened-reward) is exactly as closed
|
|
1044
|
+
* as before. Use `readRolloutLedger` when the file is training data.
|
|
1045
|
+
*/
|
|
1046
|
+
declare function readRolloutJournal(path: string): Promise<RolloutLine[]>;
|
|
306
1047
|
|
|
307
1048
|
type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
|
|
308
1049
|
type AgentProfileDimensionValue = string | number | boolean | null;
|
|
@@ -775,6 +1516,100 @@ interface TraceStore {
|
|
|
775
1516
|
artifacts(runId: string): Promise<Artifact[]>;
|
|
776
1517
|
}
|
|
777
1518
|
|
|
1519
|
+
/**
|
|
1520
|
+
* The two named score derivations every consumer must choose between.
|
|
1521
|
+
*
|
|
1522
|
+
* The anti-Goodhart gate (`outcome.realness.gated`) only holds if it is
|
|
1523
|
+
* impossible to read a run's score WITHOUT deciding whether the gate applies.
|
|
1524
|
+
* A bare `outcome.holdoutScore ?? outcome.searchScore` makes that decision
|
|
1525
|
+
* invisible — and silently answers "no gate", which is the wrong default on
|
|
1526
|
+
* every path that produces training data. So the expression lives here, once,
|
|
1527
|
+
* behind two names that force the caller to state the intent:
|
|
1528
|
+
*
|
|
1529
|
+
* - `trainingScore` / `trainingReward` — GATED. Anything that becomes
|
|
1530
|
+
* training data, or a reward a trainer consumes, uses these.
|
|
1531
|
+
* - `observedScore` — RAW. Analysis, reporting, and reward-hack DETECTION
|
|
1532
|
+
* need the ungated number; that is how a gamed run is visible at all.
|
|
1533
|
+
*
|
|
1534
|
+
* A leaf module on purpose: it imports only the `RunRecord` type, so gate and
|
|
1535
|
+
* reporting code can depend on it without pulling in the trace store that
|
|
1536
|
+
* `mint.ts` needs.
|
|
1537
|
+
*/
|
|
1538
|
+
|
|
1539
|
+
/**
|
|
1540
|
+
* Which split's score wins when a record carries both. `'holdout'` is the
|
|
1541
|
+
* canonical "real signal" default; `'search'` exists because some callers
|
|
1542
|
+
* deliberately score on the search split when both are present.
|
|
1543
|
+
*/
|
|
1544
|
+
type ScorePreference = 'holdout' | 'search';
|
|
1545
|
+
/** Only the outcome is read, so every accessor here accepts anything carrying one. */
|
|
1546
|
+
type Scored = Pick<RunRecord, 'outcome'>;
|
|
1547
|
+
/** True when the authenticity gate flagged the run as gamed (`realness.gated`). */
|
|
1548
|
+
declare function isRealnessGated(record: Scored): boolean;
|
|
1549
|
+
/**
|
|
1550
|
+
* The RAW score recorded on ONE split, with no cross-split fallback and no
|
|
1551
|
+
* anti-Goodhart gate.
|
|
1552
|
+
*
|
|
1553
|
+
* The narrowest of the three raw readers, and the one every split-scoped
|
|
1554
|
+
* consumer wants: a per-split report, a promotion gate, or a paired comparison
|
|
1555
|
+
* asks "what did this run score on the split I am summarising", and answering
|
|
1556
|
+
* it with the other split's number silently mixes populations. `undefined` =
|
|
1557
|
+
* that split was never scored.
|
|
1558
|
+
*
|
|
1559
|
+
* Same warning as `observedScore`: this INCLUDES runs flagged as gamed. Never
|
|
1560
|
+
* feed it into training data.
|
|
1561
|
+
*/
|
|
1562
|
+
declare function observedSplitScore(record: Scored, split: ScorePreference): number | undefined;
|
|
1563
|
+
/**
|
|
1564
|
+
* The RAW split score the run carries, with NO anti-Goodhart gate applied.
|
|
1565
|
+
*
|
|
1566
|
+
* INCLUDES RUNS FLAGGED AS GAMED (`outcome.realness.gated === true`); NEVER
|
|
1567
|
+
* feed this into training data — a fine-tune that sees it learns from gamed
|
|
1568
|
+
* successes. It is exported anyway because analysis, reporting, and
|
|
1569
|
+
* reward-hacking detection legitimately need the ungated number: forcing a
|
|
1570
|
+
* gamed run to 0 collapses the proxy signal toward ground truth and makes a
|
|
1571
|
+
* detector report "clean" on exactly the population that is being gamed.
|
|
1572
|
+
*
|
|
1573
|
+
* Returns `undefined` when the record carries neither score — an unscored run
|
|
1574
|
+
* is a labeled gap, not a measured zero, and each caller picks its own
|
|
1575
|
+
* sentinel (`?? 0`, `?? null`, skip, throw). Non-finite values are returned
|
|
1576
|
+
* as-is; callers that care keep their own `Number.isFinite` guard.
|
|
1577
|
+
*/
|
|
1578
|
+
declare function observedScore(record: Scored, prefer?: ScorePreference): number | undefined;
|
|
1579
|
+
/** Which split actually carried the score, or that none did. */
|
|
1580
|
+
type ScoreOrigin = 'holdout' | 'search' | 'unscored';
|
|
1581
|
+
/**
|
|
1582
|
+
* Where `observedScore` / `trainingScore` read their number from — the
|
|
1583
|
+
* provenance label a rollout line's `reward_source` is built from, and the
|
|
1584
|
+
* only supported way to ask "was this run scored at all" without respelling
|
|
1585
|
+
* the field access.
|
|
1586
|
+
*/
|
|
1587
|
+
declare function scoreOrigin(record: Scored, prefer?: ScorePreference): ScoreOrigin;
|
|
1588
|
+
/**
|
|
1589
|
+
* The GATED score — the only derivation allowed to reach training data.
|
|
1590
|
+
*
|
|
1591
|
+
* A realness-gated run scores 0 no matter what it claims, so a fine-tune
|
|
1592
|
+
* cannot learn from a gamed success. An unscored run stays `undefined` (a
|
|
1593
|
+
* labeled gap), keeping "we never measured this" distinct from "we measured
|
|
1594
|
+
* zero"; callers that need a number apply their own sentinel.
|
|
1595
|
+
*/
|
|
1596
|
+
declare function trainingScore(record: Scored, prefer?: ScorePreference): number | undefined;
|
|
1597
|
+
/**
|
|
1598
|
+
* `{reward, gated}` as written onto a minted `RolloutLine` — `trainingScore`
|
|
1599
|
+
* plus the flag itself, so the gate travels into the exported row and a
|
|
1600
|
+
* downstream filter can drop or down-weight the line.
|
|
1601
|
+
*
|
|
1602
|
+
* An unscored record yields `reward: null`, matching the schema's "no verdict
|
|
1603
|
+
* exists — a labeled gap, never 0" rule. It previously collapsed to 0, which
|
|
1604
|
+
* made a run nobody graded indistinguishable from one graded as a total
|
|
1605
|
+
* failure, and taught any trainer reading the row that the trajectory was bad.
|
|
1606
|
+
* A gated run still yields 0, because that IS a verdict: the gate decided.
|
|
1607
|
+
*/
|
|
1608
|
+
declare function trainingReward(record: Scored): {
|
|
1609
|
+
reward: number | null;
|
|
1610
|
+
gated: boolean;
|
|
1611
|
+
};
|
|
1612
|
+
|
|
778
1613
|
/**
|
|
779
1614
|
* Rollout minting — `tangle.rollout.v1` lines joined from the records the
|
|
780
1615
|
* substrate ALREADY keeps. There is no separate rollout store: a rollout
|
|
@@ -787,10 +1622,20 @@ interface TraceStore {
|
|
|
787
1622
|
* - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)
|
|
788
1623
|
* - PRM / reward-model → `reward-model-export.ts`
|
|
789
1624
|
*
|
|
790
|
-
* Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true
|
|
791
|
-
*
|
|
792
|
-
* training data (`reward` forced
|
|
793
|
-
*
|
|
1625
|
+
* Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is
|
|
1626
|
+
* never exported with a positive reward OR with any of the numbers that reward
|
|
1627
|
+
* was computed from. The gate travels into the training data (`reward` forced
|
|
1628
|
+
* to 0, `realness_gated: true`) and the whole outcome is transformed by
|
|
1629
|
+
* `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and
|
|
1630
|
+
* `verdict` to `provenance.gated_evidence`. Mint returns
|
|
1631
|
+
* `MintedRolloutLine[]`: the brand the training exporters require, which only
|
|
1632
|
+
* this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.
|
|
1633
|
+
*
|
|
1634
|
+
* A record carrying NEITHER split score is REJECTED (`ValidationError`), never
|
|
1635
|
+
* minted at 0 — "nobody graded this" is not the same claim as "graded a total
|
|
1636
|
+
* failure", and a trainer reading 0 learns the second. Lines that already
|
|
1637
|
+
* carry `reward: null` (interchange imports, existing ledgers) remain valid on
|
|
1638
|
+
* the wire; only the RunRecord→line door refuses.
|
|
794
1639
|
*
|
|
795
1640
|
* Records without spans become labeled GAP LINES (messages: [],
|
|
796
1641
|
* provenance.gap) — present in the output AND surfaced in
|
|
@@ -811,14 +1656,11 @@ interface MintRolloutOptions {
|
|
|
811
1656
|
now?: () => Date;
|
|
812
1657
|
}
|
|
813
1658
|
interface MintRolloutResult {
|
|
814
|
-
rows:
|
|
1659
|
+
rows: MintedRolloutLine[];
|
|
815
1660
|
/** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */
|
|
816
1661
|
missingTraces: string[];
|
|
817
1662
|
}
|
|
818
|
-
|
|
819
|
-
reward: number;
|
|
820
|
-
gated: boolean;
|
|
821
|
-
};
|
|
1663
|
+
|
|
822
1664
|
/**
|
|
823
1665
|
* Join RunRecords with their traces into canonical rollout lines. Records
|
|
824
1666
|
* without spans are emitted as labeled gap lines and reported in
|
|
@@ -924,6 +1766,187 @@ declare function findOpencodeSessionById(db: DatabaseSync, sessionId: string): O
|
|
|
924
1766
|
*/
|
|
925
1767
|
declare function readOpencodeSessionMessages(db: DatabaseSync, sessionId: string): ChatMessage[];
|
|
926
1768
|
|
|
1769
|
+
/**
|
|
1770
|
+
* Per-format accounting of the anti-Goodhart gate for one dataset release.
|
|
1771
|
+
*
|
|
1772
|
+
* The defect this exists to make impossible: a dataset card that STATES what
|
|
1773
|
+
* the gate does while the build does something else. A sentence in a README is
|
|
1774
|
+
* a claim about bytes it never reads, so it drifts the moment an exporter
|
|
1775
|
+
* changes — and the drift ships to whoever downloads the dataset.
|
|
1776
|
+
*
|
|
1777
|
+
* So the card is not allowed to assert anything about the gate. The build
|
|
1778
|
+
* measures the rows it is ABOUT TO WRITE (`measureFormatGate`), the measurement
|
|
1779
|
+
* is checked against the declared per-format disposition (`assertGateReport`,
|
|
1780
|
+
* which throws rather than warns), and the card renders only numbers handed to
|
|
1781
|
+
* it. A card that disagrees with its own data files cannot be produced without
|
|
1782
|
+
* failing the build first.
|
|
1783
|
+
*
|
|
1784
|
+
* The dispositions themselves are the release policy, stated once as data:
|
|
1785
|
+
*
|
|
1786
|
+
* sft EXCLUDE — an SFT row is an imitation target. A gamed
|
|
1787
|
+
* trajectory must never be imitated, at any weight.
|
|
1788
|
+
* verifiers ZERO_AND_FLAG — reward is a signed learning signal here, so a
|
|
1789
|
+
* gamed trajectory at reward 0 is a correct
|
|
1790
|
+
* negative. Dropping it would bias the negative
|
|
1791
|
+
* population toward honest failures and leave a
|
|
1792
|
+
* trainer no example of what gaming looks like
|
|
1793
|
+
* when it is penalized.
|
|
1794
|
+
* rft ZERO_AND_FLAG — RFT re-samples the completion; only the prompt
|
|
1795
|
+
* and the grader's `reference.*` verdict ship, so
|
|
1796
|
+
* nothing gamed is imitated. The flag is what lets
|
|
1797
|
+
* a grader author skip the instance.
|
|
1798
|
+
* raw ZERO_AND_FLAG — a faithful audit dump. Removing rows from it
|
|
1799
|
+
* would defeat its only purpose, and the gated
|
|
1800
|
+
* row is the one an auditor most wants.
|
|
1801
|
+
*
|
|
1802
|
+
* `ZERO_AND_FLAG` is never `reward: 0` alone. Zeroing without the label makes a
|
|
1803
|
+
* faked success indistinguishable from an honest failure — it hides the gamed
|
|
1804
|
+
* population from the buyer instead of disclosing it. Every included format
|
|
1805
|
+
* carries `realness_gated` on the row itself.
|
|
1806
|
+
*
|
|
1807
|
+
* And `ZERO_AND_FLAG` means the whole outcome, not the scalar. A gated row that
|
|
1808
|
+
* ships `reward: 0` beside the per-layer verifier scores the reward was
|
|
1809
|
+
* computed from has not been zeroed in any sense a trainer respects; the
|
|
1810
|
+
* accounting therefore measures every reward-derived number each format writes,
|
|
1811
|
+
* not just the one field.
|
|
1812
|
+
*/
|
|
1813
|
+
|
|
1814
|
+
/** What a format does with a line the realness gate flagged. */
|
|
1815
|
+
type GateDisposition = 'exclude' | 'zero-and-flag';
|
|
1816
|
+
declare const FORMAT_GATE_DISPOSITION: Record<ReleaseFormat, GateDisposition>;
|
|
1817
|
+
/** What the gate accounting reads off an emitted row, per format. */
|
|
1818
|
+
interface ReleaseRowRef {
|
|
1819
|
+
rollout_id: string;
|
|
1820
|
+
reward: number | null;
|
|
1821
|
+
/**
|
|
1822
|
+
* The rest of the row that was DERIVED from the reward — the per-layer score
|
|
1823
|
+
* dict, the judge verdict record, whatever this format ships beside the
|
|
1824
|
+
* scalar. Walked for positive numbers, so the certification is about the
|
|
1825
|
+
* whole outcome rather than one field.
|
|
1826
|
+
*
|
|
1827
|
+
* Absent when the format's row carries nothing but the scalar. NOT the whole
|
|
1828
|
+
* row: `cost.tokens_in`, `wall_s` and `total_steps` are positive numbers that
|
|
1829
|
+
* have nothing to do with the reward, and a certification that flags them is
|
|
1830
|
+
* a certification nobody can act on.
|
|
1831
|
+
*/
|
|
1832
|
+
evidence?: unknown;
|
|
1833
|
+
/**
|
|
1834
|
+
* The screen claim AS EMITTED — read off the row, not off the line it came
|
|
1835
|
+
* from, because what ships is what matters. Required, not optional: an
|
|
1836
|
+
* optional field is how a format quietly opts out of the check that reads it,
|
|
1837
|
+
* and every emitted row shape carries `RealnessLabels` precisely so no adapter
|
|
1838
|
+
* has to.
|
|
1839
|
+
*/
|
|
1840
|
+
realness_screened: boolean | null;
|
|
1841
|
+
/**
|
|
1842
|
+
* The part of an emitted `steps[]` the wire format does not declare.
|
|
1843
|
+
*
|
|
1844
|
+
* Separate from `evidence` because the declared step fields are FULL of
|
|
1845
|
+
* legitimate positive numbers — `durationMs`, `llm_call_count`,
|
|
1846
|
+
* `prompt_token_ids` — and a certification that flags those is one nobody can
|
|
1847
|
+
* act on. Only the undeclared remainder is unclassified reward-bearing
|
|
1848
|
+
* payload, which is the same partition the check applies.
|
|
1849
|
+
*
|
|
1850
|
+
* Set only by formats whose row carries steps: today `raw` alone.
|
|
1851
|
+
*/
|
|
1852
|
+
stepEvidence?: unknown;
|
|
1853
|
+
}
|
|
1854
|
+
/** A positive number found inside an emitted gated row, with where it was. */
|
|
1855
|
+
interface EmittedEvidence {
|
|
1856
|
+
/** JSON-ish path from the row's evidence root, e.g. `metrics['layer.tests']`. */
|
|
1857
|
+
path: string;
|
|
1858
|
+
value: number;
|
|
1859
|
+
}
|
|
1860
|
+
interface FormatGateCounts {
|
|
1861
|
+
/** Gated lines that reached this format's exporter. */
|
|
1862
|
+
input: number;
|
|
1863
|
+
/** Gated rows the format actually wrote. */
|
|
1864
|
+
emitted: number;
|
|
1865
|
+
/**
|
|
1866
|
+
* Gated lines this format did not write. Not all of these are the gate:
|
|
1867
|
+
* `verifiers` also drops gap lines (empty transcript) and `rft` drops lines
|
|
1868
|
+
* with no prompt turn, so an excluded count can mix both causes.
|
|
1869
|
+
*/
|
|
1870
|
+
excluded: number;
|
|
1871
|
+
/** Highest reward on an emitted gated row; `null` when none was emitted. */
|
|
1872
|
+
maxEmittedReward: number | null;
|
|
1873
|
+
/**
|
|
1874
|
+
* The largest positive number found in the reward-DERIVED payload of an
|
|
1875
|
+
* emitted gated row, and its path; `null` when there is none.
|
|
1876
|
+
*
|
|
1877
|
+
* This column exists because the release once certified CLEAN while leaking.
|
|
1878
|
+
* `assertGateReport` inspected `outcome.reward` alone, so a gated row shipping
|
|
1879
|
+
* `reward: 0` next to `metrics['layer.tests']: 1` — the deterministic verifier
|
|
1880
|
+
* score the reward was computed from, and the per-rubric score dict of the
|
|
1881
|
+
* Prime Intellect verifiers format — passed, and the card rendered "max reward
|
|
1882
|
+
* | 0" over a file that carried the gamed signal at full value. A wrong
|
|
1883
|
+
* certification is worse than the leak: it is the leak plus a document saying
|
|
1884
|
+
* there isn't one.
|
|
1885
|
+
*/
|
|
1886
|
+
maxEmittedEvidence: EmittedEvidence | null;
|
|
1887
|
+
/**
|
|
1888
|
+
* Rows this format wrote carrying a positive reward whose producer DECLARED
|
|
1889
|
+
* that no authenticity screen ever ran on it (`realness_screened: false`).
|
|
1890
|
+
*
|
|
1891
|
+
* Measured over EVERY emitted row, not just the gated ones: an unscreened
|
|
1892
|
+
* reward is by definition one the gate never had a verdict on, so it is not in
|
|
1893
|
+
* the gated set and a measurement scoped to that set would report 0 forever.
|
|
1894
|
+
* `assertMinted` already refuses these, which is exactly why the release still
|
|
1895
|
+
* measures them — the last door before a public dataset does not get to assume
|
|
1896
|
+
* the earlier doors held.
|
|
1897
|
+
*/
|
|
1898
|
+
unscreenedPositiveRows: number;
|
|
1899
|
+
/** Highest reward on such a row; `null` when there is none. */
|
|
1900
|
+
maxUnscreenedReward: number | null;
|
|
1901
|
+
/**
|
|
1902
|
+
* The largest positive number found in an emitted gated row's UNDECLARED
|
|
1903
|
+
* per-step payload, and its path; `null` when there is none.
|
|
1904
|
+
*
|
|
1905
|
+
* The column exists because the gate read `outcome` and nothing else for
|
|
1906
|
+
* three rounds, so a gated line shipping `steps: [{kind, name, reward: 0.86}]`
|
|
1907
|
+
* certified clean — the release accounting agreed with the exporter that a
|
|
1908
|
+
* per-step reward was not a reward.
|
|
1909
|
+
*/
|
|
1910
|
+
maxEmittedStepEvidence: EmittedEvidence | null;
|
|
1911
|
+
}
|
|
1912
|
+
interface GateReport {
|
|
1913
|
+
/** Gated lines in the release input, after the split/proposer filters. */
|
|
1914
|
+
gatedLines: number;
|
|
1915
|
+
byFormat: Partial<Record<ReleaseFormat, FormatGateCounts>>;
|
|
1916
|
+
}
|
|
1917
|
+
/** Rollout ids of every gated line, the key the emitted rows are matched on. */
|
|
1918
|
+
declare function gatedRolloutIds(lines: readonly MintedRolloutLine[]): Set<string>;
|
|
1919
|
+
/**
|
|
1920
|
+
* Row refs per format. Written as one adapter per format so that the knowledge
|
|
1921
|
+
* of WHERE the id and reward live in each published shape sits next to the
|
|
1922
|
+
* assertion that uses it — an exporter that moves either field breaks here
|
|
1923
|
+
* rather than silently reporting zero gated rows.
|
|
1924
|
+
*/
|
|
1925
|
+
declare const releaseRowRefs: {
|
|
1926
|
+
sft: (rows: readonly SftRow[]) => ReleaseRowRef[];
|
|
1927
|
+
verifiers: (rows: readonly VerifiersRolloutOutput[]) => ReleaseRowRef[];
|
|
1928
|
+
rft: (rows: readonly RftItem[]) => ReleaseRowRef[];
|
|
1929
|
+
raw: (lines: readonly MintedRolloutLine[]) => ReleaseRowRef[];
|
|
1930
|
+
};
|
|
1931
|
+
/** Measure one format's gated rows from the refs of the rows about to be written. */
|
|
1932
|
+
declare function measureFormatGate(gated: ReadonlySet<string>, refs: readonly ReleaseRowRef[]): FormatGateCounts;
|
|
1933
|
+
/**
|
|
1934
|
+
* Fail the build when the measurement disagrees with the declared policy.
|
|
1935
|
+
*
|
|
1936
|
+
* Throws, never filters: an emitted positive reward on a gated row means an
|
|
1937
|
+
* exporter upstream stopped applying the gate, and silently dropping the row
|
|
1938
|
+
* would hide the producer that made it — the producer is the actual defect.
|
|
1939
|
+
*
|
|
1940
|
+
* Certifies the whole emitted outcome, not `reward` alone. The earlier version
|
|
1941
|
+
* checked one field and therefore certified a release CLEAN while its
|
|
1942
|
+
* `verifiers/train.jsonl` shipped the gamed run's per-layer scores at 1.0 in
|
|
1943
|
+
* the top-level `metrics` dict — the card then rendered "max reward | 0" over
|
|
1944
|
+
* exactly that file. A certification that is wrong is worse than an
|
|
1945
|
+
* uncertified leak, so the checks it runs are no longer written down here at
|
|
1946
|
+
* all: it iterates `GATE_CHECK_IDS` under its own declared policy.
|
|
1947
|
+
*/
|
|
1948
|
+
declare function assertGateReport(report: GateReport): void;
|
|
1949
|
+
|
|
927
1950
|
/**
|
|
928
1951
|
* Deterministic scrubbing pass over rollout-ledger lines before public release.
|
|
929
1952
|
*
|
|
@@ -948,10 +1971,17 @@ type ScrubCounts = Record<string, number>;
|
|
|
948
1971
|
declare function emptyScrubCounts(): ScrubCounts;
|
|
949
1972
|
declare function addScrubCounts(into: ScrubCounts, from: ScrubCounts): ScrubCounts;
|
|
950
1973
|
declare function scrubText(text: string, counts: ScrubCounts): string;
|
|
951
|
-
/**
|
|
952
|
-
|
|
953
|
-
|
|
954
|
-
|
|
1974
|
+
/**
|
|
1975
|
+
* Scrub every string value in a line; structure and key order are preserved.
|
|
1976
|
+
*
|
|
1977
|
+
* `assertMinted` on the way out rather than a cast: scrubbing rebuilds the
|
|
1978
|
+
* object, so the brand has to be re-earned, and re-validating proves the rules
|
|
1979
|
+
* did not rewrite a field the schema constrains (`reward` is a number, not a
|
|
1980
|
+
* string, so no rule should ever touch it — this is what checks that).
|
|
1981
|
+
*/
|
|
1982
|
+
declare function scrubRolloutLine(line: MintedRolloutLine, counts: ScrubCounts): MintedRolloutLine;
|
|
1983
|
+
declare function scrubLines(lines: MintedRolloutLine[]): {
|
|
1984
|
+
lines: MintedRolloutLine[];
|
|
955
1985
|
counts: ScrubCounts;
|
|
956
1986
|
};
|
|
957
1987
|
/**
|
|
@@ -979,7 +2009,7 @@ type ReleaseFormat = (typeof RELEASE_FORMATS)[number];
|
|
|
979
2009
|
declare const FORMAT_FILES: Record<ReleaseFormat, string>;
|
|
980
2010
|
interface DatasetCardInputs {
|
|
981
2011
|
/** Scrubbed, release-filtered lines (what actually ships). */
|
|
982
|
-
lines:
|
|
2012
|
+
lines: MintedRolloutLine[];
|
|
983
2013
|
formats: ReleaseFormat[];
|
|
984
2014
|
includeProposers: boolean;
|
|
985
2015
|
/** Source ledger basenames, for provenance. */
|
|
@@ -990,6 +2020,13 @@ interface DatasetCardInputs {
|
|
|
990
2020
|
nonTrain: number;
|
|
991
2021
|
};
|
|
992
2022
|
formatCounts: Partial<Record<ReleaseFormat, number>>;
|
|
2023
|
+
/**
|
|
2024
|
+
* Per-format anti-Goodhart accounting MEASURED on the rows the build wrote.
|
|
2025
|
+
* Required, not optional: the card's only statement about the gate is a
|
|
2026
|
+
* render of these numbers, so a card cannot be produced without them and
|
|
2027
|
+
* cannot drift from the data files it ships beside.
|
|
2028
|
+
*/
|
|
2029
|
+
gate: GateReport;
|
|
993
2030
|
}
|
|
994
2031
|
declare function buildDatasetCard(inputs: DatasetCardInputs): string;
|
|
995
2032
|
|
|
@@ -1031,6 +2068,8 @@ interface BuildSummary {
|
|
|
1031
2068
|
kept: number;
|
|
1032
2069
|
scrub: ScrubReport;
|
|
1033
2070
|
formatCounts: Partial<Record<ReleaseFormat, number>>;
|
|
2071
|
+
/** Per-format anti-Goodhart accounting, measured on the rows written. */
|
|
2072
|
+
gate: GateReport;
|
|
1034
2073
|
files: string[];
|
|
1035
2074
|
}
|
|
1036
2075
|
declare function buildHfDataset(inputs: string[], options: BuildOptions): Promise<BuildSummary>;
|
|
@@ -1045,4 +2084,4 @@ declare function parseRolloutReleaseArgs(argv: string[]): RolloutReleaseCliArgs;
|
|
|
1045
2084
|
/** CLI driver for `agent-eval rollout-release`. Returns the process exit code. */
|
|
1046
2085
|
declare function runRolloutReleaseCli(argv: string[]): Promise<number>;
|
|
1047
2086
|
|
|
1048
|
-
export { type BuildOptions, type BuildSummary, CHAT_ROLES, type ChatMessage, type ChatRole, type ChatToolCall, type ClaudeTranscript, type ClaudeTranscriptRef, type ClaudeUsageTotals, DEFAULT_CLAUDE_PROJECTS_DIR, DEFAULT_OPENCODE_DB, type DatasetCardInputs, FORMAT_FILES, type MintRolloutOptions, type MintRolloutResult, type OpencodeSessionRow, RELEASE_FORMATS, ROLLOUT_CAPTURES, ROLLOUT_RELEASE_USAGE, ROLLOUT_ROLES, ROLLOUT_SCHEMA, ROLLOUT_SPLITS, type ReleaseFormat, type RewardRow, type RftItem, type RolloutArtifacts, type RolloutCapture, type RolloutCostBlock, type RolloutLine, type RolloutOutcome, type RolloutPolicy, type RolloutProvenance, type RolloutReleaseCliArgs, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RolloutTask, SCRUB_RULES, type ScrubCounts, type ScrubReport, type ScrubRule, type SftExportOptions, type SftRow, TRAINABLE_SPLITS, type ToolDef, type VerifiersRolloutOutput, type VerifiersTokenUsage, addScrubCounts, appendRolloutLines, assertRolloutLine, buildDatasetCard, buildHfDataset, claudeProjectSlug, defaultRolloutScrubber, emptyScrubCounts, findClaudeTranscripts, findOpencodeSessionById, findOpencodeSessionsByDirectory, isRolloutLine, isTrainableSplit, mintRolloutRows, openOpencodeDb, parseRolloutReleaseArgs, planPushCommand, pushDataset, readClaudeTranscript, readOpencodeSessionMessages, readRolloutLedger,
|
|
2087
|
+
export { ATIF_SCHEMA_VERSION, type BuildOptions, type BuildSummary, CHAT_ROLES, type ChatMessage, type ChatRole, type ChatToolCall, type ClaudeTranscript, type ClaudeTranscriptRef, type ClaudeUsageTotals, DEFAULT_CLAUDE_PROJECTS_DIR, DEFAULT_OPENCODE_DB, type DatasetCardInputs, type EmittedEvidence, FORMAT_FILES, FORMAT_GATE_DISPOSITION, type FormatGateCounts, type FromHarborOptions, GATE_CHECKS, GATE_CHECK_IDS, GATE_POLICIES, type GateCheck, type GateCheckDisposition, type GateCheckId, type GateCheckedOutcome, type GateDisposition, type GateEntryPoint, type GatePolicy, type GateReport, type GatedEvidence, HARBOR_IMPORT_GAP, type HarborAgent, type HarborContentPart, type HarborFinalMetrics, type HarborImageSource, type HarborMetrics, type HarborObservation, type HarborObservationResult, type HarborStep, type HarborStepSource, type HarborSubagentTrajectoryRef, type HarborToolCall, type HarborTrajectory, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type MintedRolloutOutcome, type OpencodeSessionRow, RELEASE_FORMATS, ROLLOUT_CAPTURES, ROLLOUT_RELEASE_USAGE, ROLLOUT_ROLES, ROLLOUT_SCHEMA, ROLLOUT_SPLITS, type RealnessLabels, type ReleaseFormat, type ReleaseRowRef, type RewardRow, type RftItem, type RolloutArtifacts, type RolloutCapture, type RolloutCostBlock, type RolloutLine, type RolloutOutcome, type RolloutPolicy, type RolloutProvenance, type RolloutReleaseCliArgs, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RolloutTask, SCRUB_RULES, type ScoreOrigin, type ScorePreference, type ScrubCounts, type ScrubReport, type ScrubRule, type SftExportOptions, type SftRow, TRAINABLE_SPLITS, type ToolDef, type VerifiersRolloutOutput, type VerifiersTokenUsage, addScrubCounts, appendRolloutLines, assertGateReport, assertMinted, assertMintedLines, assertRolloutLine, buildDatasetCard, buildHfDataset, claudeProjectSlug, defaultRolloutScrubber, emptyScrubCounts, findClaudeTranscripts, findOpencodeSessionById, findOpencodeSessionsByDirectory, fromHarborTrajectory, gateErrors, gateGamedOutcome, gatedEvidenceOf, gatedRolloutIds, isRealnessGated, isRolloutLine, isTrainableSplit, measureFormatGate, mintRolloutRows, observedScore, observedSplitScore, openOpencodeDb, parseRolloutReleaseArgs, planPushCommand, pushDataset, readClaudeTranscript, readOpencodeSessionMessages, readRolloutJournal, readRolloutLedger, realnessLabels, relabelImportedSplit, releaseRowRefs, runRolloutReleaseCli, scoreOrigin, scrubLines, scrubRolloutLine, scrubText, toHarborTrajectories, toHarborTrajectory, toJsonl, toRewardRows, toRftItem, toRftItems, toSftRows, toVerifiersRolloutOutput, toVerifiersRolloutOutputs, trainingReward, trainingScore, validateRolloutLine, writeRolloutLedger };
|