@tangle-network/agent-eval 0.128.2 → 0.129.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. package/CHANGELOG.md +265 -0
  2. package/README.md +18 -0
  3. package/dist/analyst/index.d.ts +107 -165
  4. package/dist/analyst/index.js +5 -9
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/belief-state/index.d.ts +2 -19
  7. package/dist/belief-state/index.js +30 -31
  8. package/dist/belief-state/index.js.map +1 -1
  9. package/dist/benchmarks/index.d.ts +5 -8
  10. package/dist/benchmarks/index.js +12 -11
  11. package/dist/builder-eval/index.js +1 -1
  12. package/dist/campaign/index.d.ts +30 -39
  13. package/dist/campaign/index.js +11 -10
  14. package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
  15. package/dist/chunk-2QU3YOPR.js.map +1 -0
  16. package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
  17. package/dist/chunk-3OCR4R5I.js.map +1 -0
  18. package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
  19. package/dist/chunk-56TAVBOK.js.map +1 -0
  20. package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
  21. package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
  22. package/dist/chunk-BSO5JDQH.js.map +1 -0
  23. package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
  24. package/dist/chunk-C6LXANRU.js.map +1 -0
  25. package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
  26. package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
  27. package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
  28. package/dist/chunk-EG66UGL4.js.map +1 -0
  29. package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
  30. package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
  31. package/dist/chunk-G7MGMCZD.js.map +1 -0
  32. package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
  33. package/dist/chunk-H23X7XKK.js.map +1 -0
  34. package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
  35. package/dist/chunk-HPWUNB47.js.map +1 -0
  36. package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
  37. package/dist/chunk-IYCLP2N2.js.map +1 -0
  38. package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
  39. package/dist/chunk-JQSF5DQT.js.map +1 -0
  40. package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
  41. package/dist/chunk-M4YBQKIJ.js.map +1 -0
  42. package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
  43. package/dist/chunk-OIUOT4QD.js +44 -0
  44. package/dist/chunk-OIUOT4QD.js.map +1 -0
  45. package/dist/chunk-OWN5NPMC.js +152 -0
  46. package/dist/chunk-OWN5NPMC.js.map +1 -0
  47. package/dist/chunk-PC5DOSM7.js +579 -0
  48. package/dist/chunk-PC5DOSM7.js.map +1 -0
  49. package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
  50. package/dist/chunk-QB6BDBP2.js.map +1 -0
  51. package/dist/chunk-RXHCETDZ.js +536 -0
  52. package/dist/chunk-RXHCETDZ.js.map +1 -0
  53. package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
  54. package/dist/chunk-SFLLL76A.js.map +1 -0
  55. package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
  56. package/dist/chunk-T6RLYGAD.js.map +1 -0
  57. package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
  58. package/dist/chunk-TJVT4QFF.js.map +1 -0
  59. package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
  60. package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
  61. package/dist/chunk-U4L7JRPZ.js.map +1 -0
  62. package/dist/chunk-U4PHLT2N.js +419 -0
  63. package/dist/chunk-U4PHLT2N.js.map +1 -0
  64. package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
  65. package/dist/chunk-VCZ5FQYW.js.map +1 -0
  66. package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
  67. package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
  68. package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
  69. package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
  70. package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
  71. package/dist/chunk-ZHTZ4EYI.js.map +1 -0
  72. package/dist/cli.js +6 -5
  73. package/dist/cli.js.map +1 -1
  74. package/dist/contract/index.d.ts +47 -87
  75. package/dist/contract/index.js +14 -13
  76. package/dist/contract/index.js.map +1 -1
  77. package/dist/control.js +3 -2
  78. package/dist/fuzz.js +3 -2
  79. package/dist/fuzz.js.map +1 -1
  80. package/dist/index.d.ts +659 -203
  81. package/dist/index.js +145 -117
  82. package/dist/index.js.map +1 -1
  83. package/dist/meta-eval/index.js +2 -2
  84. package/dist/multishot/index.d.ts +3 -4
  85. package/dist/multishot/index.js.map +1 -1
  86. package/dist/openapi.json +1 -1
  87. package/dist/pipelines/index.js +5 -5
  88. package/dist/reporting.d.ts +14 -0
  89. package/dist/reporting.js +7 -6
  90. package/dist/rl.d.ts +652 -82
  91. package/dist/rl.js +415 -171
  92. package/dist/rl.js.map +1 -1
  93. package/dist/rollout/index.d.ts +1071 -32
  94. package/dist/rollout/index.js +68 -10
  95. package/dist/run-campaign-OJJ7CZF4.js +18 -0
  96. package/dist/supervisor-run/index.d.ts +114 -4
  97. package/dist/supervisor-run/index.js +4 -3
  98. package/dist/traces.d.ts +1 -1
  99. package/dist/traces.js +6 -5
  100. package/dist/wire/index.d.ts +10 -11
  101. package/dist/wire/index.js +3 -3
  102. package/docs/feature-guide.md +1 -1
  103. package/docs/rollout.md +116 -2
  104. package/package.json +4 -4
  105. package/dist/chunk-2JX3CFMB.js.map +0 -1
  106. package/dist/chunk-DJKY2TSY.js.map +0 -1
  107. package/dist/chunk-EJGRPCO3.js.map +0 -1
  108. package/dist/chunk-EOSZT7PL.js.map +0 -1
  109. package/dist/chunk-EZJEIH2R.js.map +0 -1
  110. package/dist/chunk-IHQDPH7D.js.map +0 -1
  111. package/dist/chunk-MHELPNRP.js.map +0 -1
  112. package/dist/chunk-NACAGYSY.js.map +0 -1
  113. package/dist/chunk-NKAGIDE2.js.map +0 -1
  114. package/dist/chunk-NYLOYM6N.js.map +0 -1
  115. package/dist/chunk-PBE2LOSS.js.map +0 -1
  116. package/dist/chunk-TT4KNT67.js +0 -124
  117. package/dist/chunk-TT4KNT67.js.map +0 -1
  118. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  119. package/dist/chunk-UWZZKKU7.js +0 -237
  120. package/dist/chunk-UWZZKKU7.js.map +0 -1
  121. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  122. package/dist/chunk-VGRCHJON.js.map +0 -1
  123. package/dist/chunk-VLOATJQ2.js.map +0 -1
  124. package/dist/chunk-VZSRQ272.js.map +0 -1
  125. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  126. package/dist/chunk-XDWDC2MP.js.map +0 -1
  127. package/dist/chunk-XPRT64IE.js.map +0 -1
  128. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  129. /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
  130. /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
  131. /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
  132. /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
  133. /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
  134. /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
  135. /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
  136. /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
  137. /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/dist/rl.d.ts CHANGED
@@ -939,7 +939,369 @@ declare function injectIrrelevantClause<S extends {
939
939
  }>(clause: string, position?: 'prefix' | 'suffix'): ScenarioPerturbation<S>;
940
940
 
941
941
  /**
942
- * Preference dataset extraction bridge from `RunRecord[]` to RL training.
942
+ * `tangle.rollout.v1`THE canonical rollout serialization, owned by
943
+ * agent-eval. One JSONL line per agent invocation (a solo eval run, a
944
+ * supervisor episode, a worker session, a proposer shot, a judge call, an
945
+ * analyst pass), labeled with its task/split coordinates and a single
946
+ * scalar reward, carrying the FULL message transcript inline.
947
+ *
948
+ * This schema is the reconciliation of two prior producers:
949
+ * - agent-eval's RunRecord-joined rollout rows (PR #410): identity,
950
+ * provenance hashes, the realness gate travelling into the reward,
951
+ * trace-derived steps.
952
+ * - the bench rollout-ledger (agent-runtime PR #591): the wire shape —
953
+ * role, task.split/rep, parent_rollout_id, policy provenance, capture
954
+ * provenance, inline canonical chat-with-tools messages.
955
+ * Where the two conflicted, RunRecord-derived semantics won; the wire
956
+ * field names follow the ledger (snake_case). See `docs/rollout.md` for
957
+ * the field-by-field decision table.
958
+ *
959
+ * Messages are inlined — never referenced — because every harness store a
960
+ * rollout can be recovered from is mutable or garbage-collected. A line
961
+ * must stay a complete training/eval example on its own.
962
+ *
963
+ * `outcome.reward` is THE single scalar (null = no verdict exists — a
964
+ * labeled gap, never 0). `outcome.realness_gated` is the anti-Goodhart
965
+ * flag: a gated line must never export as a positive training example.
966
+ *
967
+ * That last sentence is enforced here, by `validateRolloutLine`, not merely
968
+ * documented. Validating `reward` and `realness_gated` independently — each a
969
+ * well-typed field, their COMBINATION unchecked — is what let a line claiming
970
+ * `{reward: 0.95, realness_gated: true}` validate clean and walk into every
971
+ * training export. The relationship between the two IS the invariant, so it is
972
+ * checked where every other structural claim about a line is checked.
973
+ *
974
+ * The invariant is about the OUTCOME, not about one field of it. Zeroing
975
+ * `reward` while `outcome.metrics` still carried the per-layer scores that
976
+ * reward was computed from exported the gamed signal anyway, in the dict the
977
+ * verifiers format reads as its per-rubric scores. So `gateGamedOutcome`
978
+ * transforms the whole outcome once, at `assertMinted` — the funnel every
979
+ * minted line passes — and the reward-bearing components are relocated to
980
+ * `provenance.gated_evidence`, which no exporter projects.
981
+ *
982
+ * WHICH checks each door applies is not decided in this file. `./gate-checks`
983
+ * owns the canonical list and the total per-entry-point policy; the three doors
984
+ * below (`validateRolloutLine`, `assertRewardGate`, `assertMinted`) each call
985
+ * `gateErrors` with their declared policy, so a check added to that list applies
986
+ * here without anyone editing this file, and a check deliberately skipped has to
987
+ * name itself there.
988
+ */
989
+ declare const ROLLOUT_SCHEMA = "tangle.rollout.v1";
990
+ /** `agent` = a solo evaluation run (no multi-agent topology). */
991
+ type RolloutRole = 'agent' | 'supervisor' | 'worker' | 'proposer' | 'judge' | 'analyst';
992
+ /** Split vocabulary follows `RunRecord.splitTag`, extended with `canary`. */
993
+ type RolloutSplit = 'search' | 'dev' | 'holdout' | 'canary';
994
+ /** 'mint' = joined live from RunRecord + trace by `mintRolloutRows`. */
995
+ type RolloutCapture = 'mint' | 'settle-time' | 'backfill';
996
+ type ChatRole = 'system' | 'user' | 'assistant' | 'tool';
997
+ interface ChatToolCall {
998
+ id: string;
999
+ type: 'function';
1000
+ function: {
1001
+ name: string;
1002
+ /** JSON-encoded argument object, exactly as the model emitted it. */
1003
+ arguments: string;
1004
+ };
1005
+ }
1006
+ interface ChatMessage {
1007
+ role: ChatRole;
1008
+ content: string | null;
1009
+ /** Reasoning/thinking channel where the harness captured it (full fidelity). */
1010
+ reasoning_content?: string;
1011
+ tool_calls?: ChatToolCall[];
1012
+ /** Required on role:"tool" — the ChatToolCall this result answers. */
1013
+ tool_call_id?: string;
1014
+ name?: string;
1015
+ /**
1016
+ * Harbor ATIF `is_copied_context` (RFC 0001 rule 7): this turn was COPIED IN
1017
+ * from another trajectory's context, not produced by the agent on this line.
1018
+ * The RFC makes excluding it from SFT a MUST, and `toSftRows` does — training
1019
+ * on it teaches the model to author text it never authored, and credits this
1020
+ * run for another one's work. Absent = false (authored here).
1021
+ */
1022
+ is_copied_context?: boolean;
1023
+ }
1024
+ interface ToolDef {
1025
+ type: 'function';
1026
+ function: {
1027
+ name: string;
1028
+ description?: string;
1029
+ parameters?: Record<string, unknown>;
1030
+ };
1031
+ }
1032
+ /**
1033
+ * Compact trace-span projection (llm/tool step) carried alongside the
1034
+ * conversation when the line was minted from a trace. Optional: lines
1035
+ * recovered from harness stores have no span structure.
1036
+ */
1037
+ interface RolloutStep {
1038
+ kind: string;
1039
+ name: string;
1040
+ /** llm: last-message summary · tool: stringified args. Scrubbed. */
1041
+ input?: string;
1042
+ /** llm: output text · tool: stringified result. Scrubbed. */
1043
+ output?: string;
1044
+ status?: 'ok' | 'error';
1045
+ durationMs?: number;
1046
+ /**
1047
+ * LLM inferences this span represents. 0 = deterministic dispatch with no
1048
+ * model call — distinct from absent, which means the producer did not track it.
1049
+ */
1050
+ llm_call_count?: number;
1051
+ /** Exact prompt tokenization. Removes the ambiguity of re-tokenizing text at train time. */
1052
+ prompt_token_ids?: number[];
1053
+ /** Exact completion tokenization; aligns index-wise with `logprobs`. */
1054
+ completion_token_ids?: number[];
1055
+ /**
1056
+ * Per-completion-token log probabilities under the sampling policy. Required
1057
+ * for off-policy correction (importance weighting) when the rollout was
1058
+ * generated by a policy other than the one being trained.
1059
+ */
1060
+ logprobs?: number[];
1061
+ }
1062
+ interface RolloutTask {
1063
+ /** Benchmark/suite id (e.g. "swe-bench-verified") or the experiment id. */
1064
+ suite: string;
1065
+ instance_id: string;
1066
+ split: RolloutSplit;
1067
+ /** Sampling seed the campaign pinned; null = not recorded. */
1068
+ seed: number | null;
1069
+ /** Replicate index (0-based). */
1070
+ rep: number;
1071
+ }
1072
+ interface RolloutPolicy {
1073
+ /** Harness that drove the invocation (e.g. "opencode", "claude", "pi-loops"). */
1074
+ harness: string | null;
1075
+ harness_version: string | null;
1076
+ model: string | null;
1077
+ provider: string | null;
1078
+ /** Commit of the agent profile / candidate under evaluation. */
1079
+ profile_commit: string | null;
1080
+ /** sha256 of the effective prompt (post-steering), when recorded. */
1081
+ prompt_hash?: string | null;
1082
+ /** sha256 of the effective run config, when recorded. */
1083
+ config_hash?: string | null;
1084
+ /** Canonical agent-profile cell identity, when the run carries one. */
1085
+ agent_profile_cell_id?: string | null;
1086
+ /** Sampling params (temperature, top_p, max_tokens…); null = not recorded. */
1087
+ sampling: Record<string, unknown> | null;
1088
+ }
1089
+ interface RolloutOutcome {
1090
+ /**
1091
+ * THE single scalar training signal — the official verdict.
1092
+ * null = no verdict exists for this invocation (a labeled gap, never 0).
1093
+ */
1094
+ reward: number | null;
1095
+ /** Where the reward came from (judge id; "/inherited" = parent episode's). */
1096
+ reward_source: string | null;
1097
+ /** Raw judge verdict record, verbatim. */
1098
+ verdict: unknown;
1099
+ /** Everything that is NOT the scalar reward. */
1100
+ metrics: Record<string, unknown>;
1101
+ is_completed: boolean;
1102
+ is_truncated: boolean;
1103
+ error: string | null;
1104
+ /**
1105
+ * Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run faked
1106
+ * its success signal. `true` requires `reward` to be 0 or null — the
1107
+ * validator rejects the line otherwise — and the line never qualifies for
1108
+ * SFT. Required on the wire: a line that does not state the flag does not
1109
+ * validate, so no producer can dodge the gate by omitting it.
1110
+ *
1111
+ * `true` ALSO requires `metrics` to be empty and `verdict` to be null: the
1112
+ * numbers the reward was computed from are relocated to
1113
+ * `provenance.gated_evidence` by `gateGamedOutcome`. See that function for
1114
+ * why zeroing the scalar alone was not enough.
1115
+ */
1116
+ realness_gated: boolean;
1117
+ /**
1118
+ * Whether an authenticity SCREEN ever RAN on this reward — a different claim
1119
+ * from `realness_gated`, which is the screen's VERDICT.
1120
+ *
1121
+ * `realness_gated: false` reads as "we looked and nothing fired". A producer
1122
+ * with no screen at all was emitting exactly that, so a never-screened reward
1123
+ * was indistinguishable on the wire from a screened-clean one, and the whole
1124
+ * anti-Goodhart apparatus silently treated the first as the second. The two
1125
+ * claims are now separable:
1126
+ *
1127
+ * - `true` — a screen ran; `realness_gated` is its verdict.
1128
+ * - `false` — the producer declares it HAS no screen (`unscreenedRewardFields`).
1129
+ * `assertMinted` REFUSES such a line when its reward is above
1130
+ * zero: an unscreened positive reward is precisely the signal
1131
+ * the gate exists to qualify, and nothing has qualified it.
1132
+ * - absent — not stated. Pre-unification ledgers land here, as does a
1133
+ * `RunRecord` carrying no `outcome.realness` at all. Absent is
1134
+ * read as "unknown", never as `false` (which would refuse most
1135
+ * of the existing corpus) and never as `true`.
1136
+ */
1137
+ realness_screened?: boolean;
1138
+ }
1139
+ interface RolloutCostBlock {
1140
+ usd: number | null;
1141
+ tokens_in: number | null;
1142
+ tokens_out: number | null;
1143
+ tokens_reasoning: number | null;
1144
+ cache_read: number | null;
1145
+ cache_write: number | null;
1146
+ wall_s: number | null;
1147
+ /**
1148
+ * Total LLM inferences across the invocation (ATIF `llm_call_count`,
1149
+ * aggregated). Optional and additive: absent = not tracked, never 0.
1150
+ */
1151
+ llm_call_count?: number | null;
1152
+ }
1153
+ interface RolloutArtifacts {
1154
+ patch_path: string | null;
1155
+ run_dir: string | null;
1156
+ /** Source-of-truth transcript pointer (session id / jsonl path) for audit. */
1157
+ transcript_ref: string | null;
1158
+ }
1159
+ /**
1160
+ * The reward-bearing half of a GATED line's outcome, moved off `outcome` and
1161
+ * parked here verbatim. Diagnostics, never training input — see
1162
+ * `gateGamedOutcome`.
1163
+ */
1164
+ interface GatedEvidence {
1165
+ /** `outcome.metrics` exactly as the producer measured it. */
1166
+ metrics?: Record<string, unknown>;
1167
+ /** `outcome.verdict` verbatim — the judge record that claimed the success. */
1168
+ verdict?: unknown;
1169
+ /**
1170
+ * The per-step fields `tangle.rollout.v1` does not declare, parked here when
1171
+ * the gate projected `steps[]` down to the schema's own key set.
1172
+ *
1173
+ * A per-step reward is training signal exactly like the scalar, and `steps`
1174
+ * rides through `toRewardRows` verbatim — so a gated line was shipping its
1175
+ * step-level credit assignment at full value beside a `reward` of 0.
1176
+ */
1177
+ steps?: unknown;
1178
+ }
1179
+ interface RolloutProvenance {
1180
+ captured_at: string;
1181
+ capture: RolloutCapture;
1182
+ /**
1183
+ * Why this line is incomplete. Required when `messages` is empty (the
1184
+ * transcript could not be recovered); also set by interchange importers to
1185
+ * name a MISSING LABEL — an imported trajectory carries no verdict, so
1186
+ * `outcome.reward` is null and this says why.
1187
+ */
1188
+ gap?: string;
1189
+ /**
1190
+ * Present only on a realness-gated line: the outcome fields the gate
1191
+ * relocated, kept so an auditor can still see WHY the run was gated and what
1192
+ * it claimed. Deliberately OUTSIDE `outcome`, because every training exporter
1193
+ * reads `outcome` and none reads `provenance`.
1194
+ */
1195
+ gated_evidence?: GatedEvidence;
1196
+ }
1197
+ interface RolloutLine {
1198
+ schema: typeof ROLLOUT_SCHEMA;
1199
+ rollout_id: string;
1200
+ /** Spawning invocation within the same episode (worker → supervisor). */
1201
+ parent_rollout_id: string | null;
1202
+ run_id: string;
1203
+ /** Logical experiment grouping from `RunRecord.experimentId`; null = not recorded. */
1204
+ experiment_id: string | null;
1205
+ /** Stable candidate identity from `RunRecord.candidateId`; null = not recorded. */
1206
+ candidate_id: string | null;
1207
+ /** Improvement-loop generation (-1 = baseline); null = not an improvement loop. */
1208
+ generation: number | null;
1209
+ /** Improvement-loop candidate index (-1 = baseline); null = not an improvement loop. */
1210
+ candidate_index: number | null;
1211
+ role: RolloutRole;
1212
+ task: RolloutTask;
1213
+ policy: RolloutPolicy;
1214
+ /** Full transcript, inline. [] = gap line (see provenance.gap). */
1215
+ messages: ChatMessage[];
1216
+ tool_defs: ToolDef[];
1217
+ /** Trace-span projections, when minted from a trace. */
1218
+ steps?: RolloutStep[];
1219
+ outcome: RolloutOutcome;
1220
+ cost: RolloutCostBlock;
1221
+ artifacts: RolloutArtifacts;
1222
+ provenance: RolloutProvenance;
1223
+ }
1224
+ /**
1225
+ * Phantom property. `declare const` means it exists only in the type system:
1226
+ * nothing is written at runtime, so a branded line still serializes to exactly
1227
+ * the same JSON as a plain one.
1228
+ */
1229
+ declare const MINTED_ROLLOUT: unique symbol;
1230
+ /**
1231
+ * A minted outcome states the gate verdict — it is not allowed to stay silent —
1232
+ * and, when that verdict is `true`, carries nothing else the reward was derived
1233
+ * from (`gateGamedOutcome` has run).
1234
+ */
1235
+ interface MintedRolloutOutcome extends RolloutOutcome {
1236
+ realness_gated: boolean;
1237
+ }
1238
+ /**
1239
+ * A `RolloutLine` whose reward has been checked against the anti-Goodhart
1240
+ * invariant. The type every training-data exporter takes.
1241
+ *
1242
+ * Why a brand and not just the interface: `RolloutLine` is structural, so any
1243
+ * hand-built object literal of the right shape IS one — which is how a line
1244
+ * declaring `{reward: 0.95, realness_gated: true}` reached the exporters
1245
+ * despite them "only accepting a minted line". The phantom symbol makes the
1246
+ * type nominal: it cannot be produced by writing an object literal, only by
1247
+ * `mintRolloutRows` (which applies the gate), `readRolloutLedger` (which
1248
+ * validates every line off disk), or an explicit, greppable `assertMinted`.
1249
+ *
1250
+ * Belt and braces on purpose. The brand closes first-party call sites at
1251
+ * COMPILE time; `validateRolloutLine` closes data arriving at RUNTIME (ledger
1252
+ * files, foreign imports, JSON from another process) where types are absent.
1253
+ * Neither alone is enough.
1254
+ *
1255
+ * Assignable to `RolloutLine` in one direction only: readers, analysis, and
1256
+ * the ledger writer keep taking the plain type.
1257
+ */
1258
+ type MintedRolloutLine = Omit<RolloutLine, 'outcome'> & {
1259
+ readonly [MINTED_ROLLOUT]: true;
1260
+ outcome: MintedRolloutOutcome;
1261
+ };
1262
+
1263
+ /**
1264
+ * Shared checks for trainer exports over canonical minted rollout lines.
1265
+ *
1266
+ * Exporters accept only `MintedRolloutLine[]`. Callers convert run records with
1267
+ * `mintRolloutRows` before deriving preferences or trainer files.
1268
+ */
1269
+
1270
+ /**
1271
+ * The minted lines behind a LINE-LESS training artifact.
1272
+ *
1273
+ * `PreferenceTriple`, `PrmTrainingTriple` and `StepReward` all carry a bare
1274
+ * reward number plus run ids, and nothing that says whether those runs faked
1275
+ * their success. An exporter over them therefore has no way, from its input
1276
+ * alone, to learn that its chosen side is a run the gate flagged — it will
1277
+ * happily emit the gaming trajectory as the preferred one. Supplying the lines
1278
+ * is what gives it eyes.
1279
+ */
1280
+ interface RolloutLineContext {
1281
+ /**
1282
+ * Minted lines for every INVOCATION the artifacts reference.
1283
+ *
1284
+ * Not "one line per run": `tangle.rollout.v1` models many invocations per
1285
+ * `run_id` — that is what `rollout_id` and `parent_rollout_id` are for, and
1286
+ * `supervisorRunRolloutLines` emits a supervisor node plus one per worker, all
1287
+ * sharing a single `run_id`. A reference is resolved against `rollout_id`
1288
+ * first and falls back to `run_id` only when that run has exactly one
1289
+ * invocation; see `resolveInvocation`.
1290
+ */
1291
+ lines: MintedRolloutLine[];
1292
+ }
1293
+ /** How one exporter names itself and its context type in the failure messages. */
1294
+ interface LineContextRequirement {
1295
+ /** Exporter label, e.g. `'DPO export'`. */
1296
+ exporter: string;
1297
+ /** The context type the caller must pass, e.g. `'DpoLineContext'`. */
1298
+ contextType: string;
1299
+ /** Why this exporter cannot see the gate without lines. One sentence. */
1300
+ because: string;
1301
+ }
1302
+
1303
+ /**
1304
+ * Preference dataset extraction from canonical minted rollout lines.
943
1305
  *
944
1306
  * Production RLHF / DPO / KTO / SimPO pipelines need preference triples:
945
1307
  * `(prompt, chosen, rejected)`. The campaign artifact already contains the
@@ -973,7 +1335,16 @@ declare function injectIrrelevantClause<S extends {
973
1335
  * per scenario but biggest score gap per pair. Useful for early
974
1336
  * bootstrapping when you have few variants.
975
1337
  *
976
- * Resolve `PreferenceTriple` text with `toDpoRows` from `./exporters`.
1338
+ * The output `PreferenceTriple` is *agent-eval-canonical* but trivially
1339
+ * mappable to TRL's `DPODataset` shape (`prompt`, `chosen`, `rejected`)
1340
+ * via the `toTRLFormat` helper, which resolves real prompt/completion text
1341
+ * through the same lookups `toDpoRows` takes (`./exporters` carries the
1342
+ * richer row with margin + metadata).
1343
+ *
1344
+ * Input discipline: the function accepts only `MintedRolloutLine[]`, whose
1345
+ * reward and authenticity fields have already been validated. `search` is the
1346
+ * default split; held-out pairing requires an explicit opt-in, while `dev` and
1347
+ * `canary` remain evaluation-only.
977
1348
  */
978
1349
 
979
1350
  type PreferenceStrategy = 'paired-by-scenario-and-seed' | 'paired-by-scenario' | 'top-vs-bottom';
@@ -1000,7 +1371,7 @@ interface PreferenceTriple {
1000
1371
  /** Tie-breaker — when multiple seeds match this scenario, the one used. */
1001
1372
  seed?: number;
1002
1373
  /**
1003
- * Free-form metadata propagated from the run records e.g. original
1374
+ * Free-form metadata propagated from the rollout lines, such as original
1004
1375
  * prompt-hash, model, etc. Lets the RL trainer reconstruct the prompt.
1005
1376
  */
1006
1377
  meta: {
@@ -1012,7 +1383,7 @@ interface PreferenceTriple {
1012
1383
  rejectedModel: string;
1013
1384
  };
1014
1385
  }
1015
- interface ExtractPreferencesOptions extends TrainingRunSelectionOptions {
1386
+ interface ExtractPreferencesOptions {
1016
1387
  strategy?: PreferenceStrategy;
1017
1388
  /**
1018
1389
  * Minimum score gap required to admit a pair. Pairs below this are
@@ -1020,16 +1391,13 @@ interface ExtractPreferencesOptions extends TrainingRunSelectionOptions {
1020
1391
  */
1021
1392
  minMargin?: number;
1022
1393
  /**
1023
- * Optional split tag filter. Without one, only search is included.
1024
- * Holdout requires `allowHeldOutTrainingData: true`; dev is evaluation-only.
1394
+ * Optional split filter. Without one, only search is included.
1395
+ * Holdout requires `allowHeldOutTrainingData: true`; dev and canary are
1396
+ * evaluation-only.
1025
1397
  */
1026
- splitTag?: RunRecord['splitTag'];
1027
- /**
1028
- * Optional reward extractor that overrides `outcome.holdoutScore` /
1029
- * `outcome.searchScore`. Use to drive preferences off a verifiable
1030
- * reward instead of the headline score.
1031
- */
1032
- rewardOf?: (run: RunRecord) => number | null;
1398
+ split?: RolloutSplit;
1399
+ /** Named opt-in required before held-out lines may be paired. */
1400
+ allowHeldOutTrainingData?: boolean;
1033
1401
  }
1034
1402
  interface PreferenceExtractionReport {
1035
1403
  pairs: PreferenceTriple[];
@@ -1041,26 +1409,64 @@ interface PreferenceExtractionReport {
1041
1409
  cellsSingleton: number;
1042
1410
  /** Strategy used. */
1043
1411
  strategy: PreferenceStrategy;
1412
+ /**
1413
+ * Lines dropped before pairing because they carry no `candidate_id`. A
1414
+ * preference is a statement about two candidates, so a line that names none
1415
+ * cannot be paired.
1416
+ */
1417
+ linesWithoutCandidateId: number;
1044
1418
  }
1045
1419
  /**
1046
- * Convert `RunRecord[]` to preference triples for RL training.
1420
+ * Convert rollout lines to preference triples for RL training.
1047
1421
  *
1048
1422
  * Returns a structured report so callers can see how much data was
1049
1423
  * dropped and why (low-margin pairs, singleton cells). For production
1050
1424
  * pipelines, you usually want to:
1051
1425
  *
1052
1426
  * 1. Run a campaign producing 5–10 variants × 50–200 scenarios × 3 seeds
1053
- * 2. Call this with `strategy: 'paired-by-scenario-and-seed'` and a
1054
- * verifiable-reward extractor as `rewardOf`
1055
- * 3. Pass `report.pairs` to `toDpoRows` with prompt/completion resolvers
1427
+ * 2. Mint the runs with `mintRolloutRows` and call this with
1428
+ * `strategy: 'paired-by-scenario-and-seed'`
1429
+ * 3. Pass `report.pairs` to `toDpoRows` (or `toTRLFormat`) with
1430
+ * prompt/completion resolvers and pipe to your DPO trainer
1431
+ *
1432
+ * The gate is what makes a preference dataset safe: ordered on an ungated
1433
+ * score, a gamed run with an inflated number becomes the `chosen` side and DPO
1434
+ * is trained to prefer the gaming trajectory over its honest sibling. A gated
1435
+ * line arrives here already scored 0, so it sinks to `rejected`.
1436
+ */
1437
+ declare function extractPreferences(lines: MintedRolloutLine[], opts?: ExtractPreferencesOptions): PreferenceExtractionReport;
1438
+ /**
1439
+ * TRL-compatible export. TRL's `DPODataset` is `{ prompt, chosen, rejected }`
1440
+ * where `chosen`/`rejected` are completion TEXT — a trainer fed prompt hashes
1441
+ * would optimize the policy toward emitting hex digests. Neither the prompt
1442
+ * nor the completions live on the triple (it carries only run ids and hashes),
1443
+ * so the caller supplies the same `promptOf`/`completionOf` lookups `toDpoRows`
1444
+ * takes, keyed by run id, and this function resolves real text.
1445
+ *
1446
+ * The chosen and rejected sides of a valid pair share one prompt; resolving
1447
+ * both and comparing catches lookup bugs (a stale map keyed by the wrong id)
1448
+ * before they ship a row whose prompt does not match its rejected completion.
1449
+ *
1450
+ * `context` is REQUIRED: this is the third exporter over the identical
1451
+ * line-less input class, and the round that hardened `toPrmRows` while leaving
1452
+ * `toDpoRows` open is why every one of them now takes the same argument and
1453
+ * runs the same admission rule.
1056
1454
  */
1057
- declare function extractPreferences(runs: RunRecord[], opts?: ExtractPreferencesOptions): PreferenceExtractionReport;
1455
+ declare function toTRLFormat(triples: PreferenceTriple[], lookups: DpoLookups, context: RolloutLineContext): Promise<Array<{
1456
+ prompt: string;
1457
+ chosen: string;
1458
+ rejected: string;
1459
+ }>>;
1058
1460
  /**
1059
1461
  * Anthropic finetuning JSONL export — `{ system, user, assistant_chosen, assistant_rejected }`
1060
1462
  * shape. Same caveat as TRL: prompt + outputs are content the caller has
1061
1463
  * to map back from the run record / raw event log.
1464
+ *
1465
+ * `context` is REQUIRED — see `toTRLFormat`. The emitted `margin` is a number
1466
+ * derived from the two runs' rewards, so this row is training signal even
1467
+ * though it ships no completion text.
1062
1468
  */
1063
- declare function toAnthropicFormat(triples: PreferenceTriple[]): Array<{
1469
+ declare function toAnthropicFormat(triples: PreferenceTriple[], context: RolloutLineContext): Array<{
1064
1470
  scenarioId: string;
1065
1471
  chosenRunId: string;
1066
1472
  rejectedRunId: string;
@@ -1239,7 +1645,7 @@ declare function prmTrainingPairs(stepRewardsByRun: Map<string, StepReward[]>, o
1239
1645
  /**
1240
1646
  * Trainer-format exporters.
1241
1647
  *
1242
- * agent-eval produces canonical artifacts (`RunRecord[]`, `PreferenceTriple[]`,
1648
+ * agent-eval produces canonical artifacts (`MintedRolloutLine[]`, `PreferenceTriple[]`,
1243
1649
  * `StepReward[]`, `PrmTrainingTriple[]`). RL training pipelines consume
1244
1650
  * different shapes — Hugging Face TRL, Prime Intellect's prime-rl, OpenAI
1245
1651
  * fine-tuning, Anthropic finetuning, OpenRLHF, verl. Each has its own
@@ -1259,7 +1665,7 @@ declare function prmTrainingPairs(stepRewardsByRun: Map<string, StepReward[]>, o
1259
1665
  * Consumed by Lightman-style PRM trainers and prime-rl's PRM mode.
1260
1666
  *
1261
1667
  * Why ship this in agent-eval rather than a separate adapter package: the
1262
- * canonical artifacts (`RunRecord[]`, `PreferenceTriple[]`, etc.) are
1668
+ * canonical artifacts (`MintedRolloutLine[]`, `PreferenceTriple[]`, etc.) are
1263
1669
  * agent-eval's contract; without first-party exporters consumers reverse-
1264
1670
  * engineer the mapping every release. The exporters codify it.
1265
1671
  *
@@ -1267,6 +1673,10 @@ declare function prmTrainingPairs(stepRewardsByRun: Map<string, StepReward[]>, o
1267
1673
  * artifact (specifically: prompt + completion text, since the package
1268
1674
  * stores only their hashes by design — full text is the consumer's
1269
1675
  * trace store / raw event log).
1676
+ *
1677
+ * Every exporter that produces a training row accepts canonical minted rollout
1678
+ * lines. Convert run records once with `mintRolloutRows`; downstream transforms
1679
+ * then share one reward, split, and authenticity contract.
1270
1680
  */
1271
1681
 
1272
1682
  interface DpoLookups {
@@ -1284,25 +1694,43 @@ interface DpoExportRow {
1284
1694
  /** Free-form metadata for downstream filtering / sharding. */
1285
1695
  meta?: Record<string, unknown>;
1286
1696
  }
1697
+ /** The minted lines for the runs a `PreferenceTriple` names on each side. */
1698
+ type DpoLineContext = RolloutLineContext;
1699
+ declare const DPO_CONTEXT_REQUIREMENT: LineContextRequirement;
1287
1700
  /**
1288
1701
  * Convert preference triples to TRL-compatible DPO rows. The shape
1289
1702
  * `{prompt, chosen, rejected}` is the canonical HuggingFace DPODataset
1290
1703
  * entry; every major DPO trainer accepts it.
1704
+ *
1705
+ * `context` is REQUIRED, and for the same reason it is required on the sibling
1706
+ * `toPrmRows`: a triple is a line-less artifact. It names two run ids and a
1707
+ * margin, and nothing on it says whether either run was flagged as gamed —
1708
+ * so a two-argument call applied NO gate at all and emitted the row verbatim,
1709
+ * reachable straight through the published bundle builder
1710
+ * (`buildRlDataset(lines, lookups, {formats:['dpo']}, {triples, lookups})`).
1711
+ * Triples whose chosen or rejected side is realness-gated are dropped; a triple
1712
+ * naming a run with no supplied line is refused. See `admitUngatedByInvocation` for
1713
+ * why dropping, not zeroing, is the right disposition for a preference pair.
1291
1714
  */
1292
- declare function toDpoRows(triples: PreferenceTriple[], lookups: DpoLookups): Promise<DpoExportRow[]>;
1715
+ declare function toDpoRows(triples: PreferenceTriple[], lookups: DpoLookups, context: DpoLineContext): Promise<DpoExportRow[]>;
1293
1716
  /** Serialize DPO rows as JSONL. One line per row. */
1294
1717
  declare function toDpoJsonl(rows: DpoExportRow[]): string;
1295
- interface TrainingRunSelectionOptions {
1718
+ interface TrainingLineSelectionOptions {
1296
1719
  /** Include held-out evaluation data in training output. Default false. */
1297
1720
  allowHeldOutTrainingData?: boolean;
1298
1721
  /** Require quality to be strictly greater than this value. Default 0. */
1299
1722
  minimumQualityExclusive?: number;
1723
+ /**
1724
+ * Explicit split selection, replacing the default trainable-split rule.
1725
+ * Use this only when producing a deliberately named non-training slice.
1726
+ */
1727
+ splitFilter?: RolloutSplit[];
1300
1728
  }
1301
- interface GrpoLookups extends TrainingRunSelectionOptions {
1729
+ interface GrpoLookups extends Pick<TrainingLineSelectionOptions, 'allowHeldOutTrainingData' | 'splitFilter'> {
1730
+ /** Resolve the prompt text for a rollout, keyed by `line.run_id`. */
1302
1731
  promptOf: (runId: string) => string | Promise<string>;
1732
+ /** Resolve the assistant completion text for a rollout. */
1303
1733
  completionOf: (runId: string) => string | Promise<string>;
1304
- /** Optional: derive a custom reward from the run. Defaults to score. */
1305
- rewardOf?: (run: RunRecord) => number | null;
1306
1734
  }
1307
1735
  interface GrpoExportRow {
1308
1736
  prompt: string;
@@ -1313,23 +1741,35 @@ interface GrpoExportRow {
1313
1741
  meta?: Record<string, unknown>;
1314
1742
  }
1315
1743
  /**
1316
- * Convert RunRecord[] grouped by canonical `(scenarioId, promptHash)` identity
1317
- * into GRPO offline rows.
1744
+ * Convert rollout lines grouped by `task.instance_id` into GRPO offline rows —
1745
+ * one row per scenario, with one completion per rollout on that scenario.
1746
+ * A scenario with fewer than two rewarded completions emits no row because a
1747
+ * group of one has no relative baseline.
1318
1748
  *
1319
1749
  * GRPO (Shao et al. 2024 / DeepSeek-R1) trains on relative advantages
1320
1750
  * within a group of completions for the same prompt; this is the
1321
- * canonical input format. A scenario containing multiple prompt hashes, or a
1322
- * prompt hash that resolves to different text, is rejected rather than mixed.
1751
+ * canonical input format. That relative baseline is exactly why the gate has
1752
+ * to hold here: one gamed sibling exporting at full reward shifts the advantage
1753
+ * of every honest run beside it.
1754
+ *
1755
+ * On the line path a realness-gated line stays in its group at reward 0 rather
1756
+ * than being dropped. 0 is the honest label for a faked success and is usable
1757
+ * signal; removing the line would also move the group's baseline, just in the
1758
+ * other direction. (SFT differs — see `toSftRows`.)
1323
1759
  */
1324
- declare function toGrpoRows(runs: RunRecord[], lookups: GrpoLookups): Promise<GrpoExportRow[]>;
1760
+ declare function toGrpoRows(lines: MintedRolloutLine[], lookups: GrpoLookups): Promise<GrpoExportRow[]>;
1325
1761
  declare function toGrpoJsonl(rows: GrpoExportRow[]): string;
1326
- interface SftLookups extends TrainingRunSelectionOptions {
1762
+ interface SftLookups extends TrainingLineSelectionOptions {
1763
+ /** Resolve the prompt text for a rollout, keyed by `line.run_id`. */
1327
1764
  promptOf: (runId: string) => string | Promise<string>;
1765
+ /** Resolve the assistant completion text for a rollout. */
1328
1766
  completionOf: (runId: string) => string | Promise<string>;
1329
1767
  /** Optional system message. Default omits. */
1330
- systemOf?: (run: RunRecord) => string | null | undefined;
1331
- /** Filter return false to skip the run (e.g., low score, failed cases). */
1332
- include?: (run: RunRecord) => boolean;
1768
+ systemOf?: (line: MintedRolloutLine) => string | null | undefined;
1769
+ /** Extra filter on top of the realness gate (e.g., low score, failed cases). */
1770
+ include?: (line: MintedRolloutLine) => boolean;
1771
+ /** Include held-out lines under the default split rule. Default false. */
1772
+ allowHeldOutTrainingData?: boolean;
1333
1773
  }
1334
1774
  interface SftExportRow {
1335
1775
  messages: Array<{
@@ -1339,11 +1779,24 @@ interface SftExportRow {
1339
1779
  meta?: Record<string, unknown>;
1340
1780
  }
1341
1781
  /**
1342
- * Convert RunRecord[] into Hugging Face / OpenAI / Anthropic-style
1343
- * conversational SFT rows. By default, only completed, positive-quality
1344
- * search runs are eligible. Pass `include` for additional filtering.
1782
+ * Convert rollout lines into Hugging Face / OpenAI / Anthropic-style
1783
+ * conversational SFT rows. By default every qualifying line becomes one row;
1784
+ * pass `include` to filter further (e.g., keep only `reward >= 0.8` for
1785
+ * rejection-sampling SFT).
1786
+ *
1787
+ * Realness-gated lines are dropped outright, not zeroed. SFT is imitation
1788
+ * learning: unlike GRPO, where a 0 reward teaches "this trajectory was bad",
1789
+ * every row here is a target to copy, so a gamed trajectory must not be in the
1790
+ * file at all. Mirrors the waist filter in `rollout/exporters.toSftRows`.
1791
+ *
1792
+ * The exporter is fail-closed on the split, same rule as
1793
+ * `rollout/exporters.toSftRows` (`isSplitEligible`): `search` ships by
1794
+ * default, held-out lines need `allowHeldOutTrainingData: true`, `dev` and
1795
+ * `canary` never pass the default rule. A non-training bundle that wants an
1796
+ * explicit slice (e.g. a holdout-only eval bundle) names it with
1797
+ * `splitFilter: ['holdout']` — explicit selection replaces the default rule.
1345
1798
  */
1346
- declare function toSftRows(runs: RunRecord[], lookups: SftLookups): Promise<SftExportRow[]>;
1799
+ declare function toSftRows(lines: MintedRolloutLine[], lookups: SftLookups): Promise<SftExportRow[]>;
1347
1800
  declare function toSftJsonl(rows: SftExportRow[]): string;
1348
1801
  interface PrmLookups {
1349
1802
  /** Resolve the prompt text for a run. */
@@ -1365,11 +1818,48 @@ interface PrmExportRow {
1365
1818
  marginScore: number;
1366
1819
  meta?: Record<string, unknown>;
1367
1820
  }
1821
+ interface PrmLineContext extends RolloutLineContext {
1822
+ /**
1823
+ * The `maxSteps` cap the lines were minted with, if any.
1824
+ *
1825
+ * `mintRolloutRows` drops the MIDDLE of an over-long trajectory and leaves no
1826
+ * marker on the line, so a capped trajectory is indistinguishable from a
1827
+ * short one. Declaring the cap lets this exporter refuse any line sitting at
1828
+ * it — a process-reward model trained on a trajectory with a hole in it
1829
+ * learns credit assignment that never happened.
1830
+ */
1831
+ mintedWithMaxSteps?: number;
1832
+ }
1368
1833
  /**
1369
1834
  * Convert PRM training triples to JSONL rows. Caller's `stepTextOf`
1370
1835
  * callback resolves span text from the consumer's trace store.
1836
+ *
1837
+ * Every referenced run is checked against its minted line before any row is
1838
+ * emitted, and the export FAILS LOUD on a trajectory that was never fully
1839
+ * captured (see `assertPrmTrainableLine`). Triples whose chosen or rejected
1840
+ * side is realness-gated are dropped instead: a capture defect is the caller's
1841
+ * mint configuration and must be fixed, whereas a gamed run is exactly the
1842
+ * condition the gate exists to filter.
1843
+ *
1844
+ * `context` is REQUIRED. A two-argument call used to be accepted and produced
1845
+ * rows with no gate applied at all — a `PrmTrainingTriple` carries a bare
1846
+ * `chosenReward` number and nothing that says which run it came from is honest,
1847
+ * so with no lines this exporter has no way to learn that its chosen step is a
1848
+ * step from a run that faked its success. It now throws: fail closed, because
1849
+ * the alternative is a process-reward model taught to prefer the gaming move at
1850
+ * the exact step the gaming happened.
1851
+ */
1852
+ declare function toPrmRows(triples: PrmTrainingTriple[], lookups: PrmLookups, context: PrmLineContext): Promise<PrmExportRow[]>;
1853
+ /**
1854
+ * Refuse to build a process-reward row from a trajectory we do not fully have.
1855
+ *
1856
+ * PRM training assigns credit step by step, so a missing or silently shortened
1857
+ * step list is not degraded data — it is data about a trajectory that never
1858
+ * existed. Every condition below throws rather than filters, because each one
1859
+ * means the CALLER's capture or mint configuration is wrong.
1371
1860
  */
1372
- declare function toPrmRows(triples: PrmTrainingTriple[], lookups: PrmLookups): Promise<PrmExportRow[]>;
1861
+ declare function assertPrmTrainableLine(line: MintedRolloutLine, mintedWithMaxSteps?: number): void;
1862
+ declare const PRM_CONTEXT_REQUIREMENT: LineContextRequirement;
1373
1863
  declare function toPrmJsonl(rows: PrmExportRow[]): string;
1374
1864
  interface StepRewardJsonlRow {
1375
1865
  runId: string;
@@ -1379,8 +1869,18 @@ interface StepRewardJsonlRow {
1379
1869
  determinism: 'deterministic' | 'probabilistic';
1380
1870
  weight: number;
1381
1871
  }
1382
- declare function stepRewardsToJsonl(stepRewards: StepReward[]): string;
1383
- declare function isTrainingRunEligible(run: RunRecord, quality: number | null | undefined, options?: TrainingRunSelectionOptions): quality is number;
1872
+ declare const STEP_REWARD_CONTEXT_REQUIREMENT: LineContextRequirement;
1873
+ /**
1874
+ * Step-level reward rows as JSONL.
1875
+ *
1876
+ * `context` is REQUIRED for the same reason it is on `toDpoRows` and
1877
+ * `toPrmRows`: this is a line-less input carrying a reward number. Steps
1878
+ * belonging to a realness-gated run are dropped rather than zeroed — a
1879
+ * per-step reward of 0 across a whole trajectory is a claim that every step was
1880
+ * bad, which is a different (and false) statement from "this run's success was
1881
+ * fabricated, so its step-level credit assignment is meaningless".
1882
+ */
1883
+ declare function stepRewardsToJsonl(stepRewards: StepReward[], context: RolloutLineContext): string;
1384
1884
 
1385
1885
  /**
1386
1886
  * RL dataset packaging + datasheet — the publishable, sellable bundle.
@@ -1392,13 +1892,19 @@ declare function isTrainingRunEligible(run: RunRecord, quality: number | null |
1392
1892
  * reward was derived (deterministic verifiable vs probabilistic judge — the
1393
1893
  * credibility axis a buyer checks first), the split discipline, the reward
1394
1894
  * distribution, the quality gates, the license, and the intended/out-of-scope
1395
- * uses. This module computes those facts from the `RunRecord[]` and renders a
1895
+ * uses. This module computes those facts from `MintedRolloutLine[]` and renders a
1396
1896
  * "Datasheet for Datasets" (Gebru et al. 2018) card alongside the format files.
1397
1897
  *
1398
1898
  * It composes the existing `rl/exporters` — it does not reimplement any trainer
1399
1899
  * format. The renderers token-identity step (DeepSeek/Kimi/Qwen tokenization
1400
1900
  * with per-token loss masks) is a downstream Python stage that consumes the
1401
1901
  * `messages`/`completions` this bundle emits.
1902
+ *
1903
+ * Input discipline: the bundle is built from `MintedRolloutLine[]`. The datasheet's
1904
+ * reward distribution ships INSIDE the published artifact, so it has to be
1905
+ * derived from exactly the same gated number as the rows it describes —
1906
+ * otherwise the provenance a buyer checks first is a lie. Taking the same
1907
+ * gated input as the exporters is what guarantees that.
1402
1908
  */
1403
1909
 
1404
1910
  type RewardKind = 'deterministic' | 'probabilistic' | 'mixed';
@@ -1446,10 +1952,10 @@ interface RewardStats {
1446
1952
  }
1447
1953
  interface RlDatasetStats {
1448
1954
  records: number;
1449
- /** Records carrying an explicit task-quality score. */
1955
+ /** Rollouts carrying an explicit task-quality score. */
1450
1956
  scoredRecords: number;
1451
- /** Record count per split — a publishable dataset must declare its holdout. */
1452
- splits: Record<RunSplitTag, number>;
1957
+ /** Rollout count per split. */
1958
+ splits: Record<RolloutSplit, number>;
1453
1959
  reward: RewardStats;
1454
1960
  /** Distinct snapshot-pinned models that produced the trajectories. */
1455
1961
  models: string[];
@@ -1461,6 +1967,12 @@ interface RlDatasetStats {
1461
1967
  output: number;
1462
1968
  };
1463
1969
  totalCostUsd: number;
1970
+ /**
1971
+ * Rollouts whose USD cost was never captured (`cost.usd === null`). When
1972
+ * non-zero, `totalCostUsd` is a floor, not the bill — a published dataset
1973
+ * must not present an unbilled run as a $0 one.
1974
+ */
1975
+ rolloutsWithoutCost: number;
1464
1976
  }
1465
1977
  interface RlDatasetManifest extends RlDatasetConfig {
1466
1978
  formats: DatasetFormat[];
@@ -1473,13 +1985,13 @@ interface RlDatasetBundle {
1473
1985
  files: Record<string, string>;
1474
1986
  }
1475
1987
  /**
1476
- * Package graded `RunRecord[]` into a publishable RL dataset bundle: the
1988
+ * Package graded rollout lines into a publishable RL dataset bundle: the
1477
1989
  * trainer-format JSONL files + a manifest + a datasheet. DPO requires
1478
1990
  * pre-extracted preference triples (pass `preferences`); GRPO/SFT derive from
1479
- * the records directly via the supplied lookups. Throws on an empty corpus —
1991
+ * the lines directly via the supplied lookups. Throws on an empty corpus —
1480
1992
  * an empty dataset must never be published.
1481
1993
  */
1482
- declare function buildRlDataset(records: RunRecord[], lookups: GrpoLookups & SftLookups, config: RlDatasetConfig, preferences?: {
1994
+ declare function buildRlDataset(lines: MintedRolloutLine[], lookups: GrpoLookups & SftLookups, config: RlDatasetConfig, preferences?: {
1483
1995
  triples: PreferenceTriple[];
1484
1996
  lookups: DpoLookups;
1485
1997
  }): Promise<RlDatasetBundle>;
@@ -1539,6 +2051,12 @@ interface HarvestOptions {
1539
2051
  * missing either are excluded (a graded score with no trajectory can't train).
1540
2052
  * Optionally filters by score / split. Throws (via buildRlDataset) if nothing
1541
2053
  * survives — an empty dataset must never be published.
2054
+ *
2055
+ * `minScore` is applied to the GATED reward (`trainingScore`), so a gamed run
2056
+ * cannot buy its way into the published bundle with its claimed score —
2057
+ * `minScore` is exactly the door a reward-hacked run would otherwise clear for
2058
+ * SFT. Unscored records are dropped before packaging: a missing label is not a
2059
+ * zero, and it is not publishable either.
1542
2060
  */
1543
2061
  declare function buildDatasetFromCorpus(corpusPath: string, config: RlDatasetConfig, opts?: HarvestOptions): Promise<RlDatasetBundle>;
1544
2062
 
@@ -1611,20 +2129,12 @@ interface OffPolicyTrajectory {
1611
2129
  * values must come from a model cross-fitted or trained outside this row.
1612
2130
  */
1613
2131
  vHatTarget?: number | null;
1614
- /**
1615
- * @deprecated Use `qHatChosen` and `vHatTarget` together. When the new pair
1616
- * is absent, this scalar is used as both terms to preserve existing results.
1617
- * When the new pair is present, this field is ignored.
1618
- */
1619
- qHat?: number | null;
1620
2132
  }
1621
2133
  interface OffPolicyContributionCounts {
1622
2134
  /** Contributions using the contextual-bandit doubly-robust formula. */
1623
2135
  dr: number;
1624
2136
  /** Contributions using exact IPS because no reward-model estimate was supplied. */
1625
2137
  ipsFallback: number;
1626
- /** Contributions using the deprecated single-scalar formula. */
1627
- legacyScalar: number;
1628
2138
  }
1629
2139
  interface OffPolicyEstimate {
1630
2140
  /** Estimated value of the target policy. */
@@ -1683,9 +2193,8 @@ declare function selfNormalizedImportanceWeighting(trajectories: OffPolicyTrajec
1683
2193
  * default in production OPE pipelines.
1684
2194
  *
1685
2195
  * `qHatChosen` and `vHatTarget` must be supplied together. Rows with neither
1686
- * use the exact IPS contribution. Deprecated `qHat` rows preserve the scalar
1687
- * formula, and a complete new pair takes precedence when both forms exist.
1688
- * `contributionCounts` makes the mix explicit in the result.
2196
+ * use the exact IPS contribution. `contributionCounts` makes the mix explicit
2197
+ * in the result.
1689
2198
  * Callers must cross-fit the Q-function or train it on independent rows;
1690
2199
  * fitting and evaluating Q on the same outcomes leaks the answer.
1691
2200
  */
@@ -1899,6 +2408,13 @@ interface GateEvidence {
1899
2408
  /** Median per-task USD cost across the baseline runs, for
1900
2409
  * symmetric reporting. */
1901
2410
  medianBaselineCost: number | null;
2411
+ /**
2412
+ * Runs (candidate + baseline) dropped before pairing because the
2413
+ * authenticity gate flagged them as gamed. Surfaced rather than silent: a
2414
+ * promotion decision computed over a shrunken pool has to say by how much,
2415
+ * and a nonzero count here is itself the finding.
2416
+ */
2417
+ realnessGatedRuns: number;
1902
2418
  }
1903
2419
  interface GateDecision$1 {
1904
2420
  /** Final promote/no-promote verdict. */
@@ -2268,12 +2784,34 @@ interface VerifiableReward {
2268
2784
  */
2269
2785
  components: Record<string, number>;
2270
2786
  /**
2271
- * @deprecated Read `components` for per-source reward values. Kept for
2272
- * published-API compatibility: single-source rewards carry the layer's
2273
- * diagnostics here (e.g. `{ tests_passed: 7 }`); composite rewards carry
2274
- * the same per-layer scores `components` now holds.
2787
+ * The run carries `outcome.realness.gated` the authenticity gate flagged
2788
+ * its success signal as faked.
2789
+ *
2790
+ * With the gate applied (the default) `value` and every `components` entry
2791
+ * are 0 on such a run; with `applyRealnessGate: false` the observed numbers
2792
+ * come back untouched and this flag is the only marker that they are not to
2793
+ * be trusted. Either way it distinguishes "measured a genuine failure" from
2794
+ * "claimed a success we refuse to believe", which a bare 0 cannot.
2275
2795
  */
2276
- breakdown?: Record<string, number>;
2796
+ realnessGated?: boolean;
2797
+ /**
2798
+ * Whether an authenticity screen COULD run on this reward at all — the same
2799
+ * distinction `RolloutOutcome.realness_screened` draws, for the same reason.
2800
+ *
2801
+ * `false` on every reward from `extractVerifiableReward`, because a
2802
+ * `VerificationReport` carries layer scores and nothing else: there is no
2803
+ * `outcome.realness` to consult, so no gate has run, and `realnessGated`
2804
+ * being absent there means "unknown", NOT "clean". Absent on the
2805
+ * `RunRecord` path when the record itself carries no realness verdict.
2806
+ *
2807
+ * This matters most exactly where it is easiest to miss: a report whose
2808
+ * deterministic layers all passed yields `determinism: 'deterministic'`,
2809
+ * `confidence: 1` — the highest-credibility reward this module can emit —
2810
+ * and a stubbed integration reporting green is precisely what a gamed run
2811
+ * looks like. Consumers driving training off this shape must screen the run
2812
+ * themselves; the flag is what tells them nobody has.
2813
+ */
2814
+ realnessScreened?: boolean;
2277
2815
  }
2278
2816
  interface VerifiableRewardExtractionOptions {
2279
2817
  /**
@@ -2300,6 +2838,18 @@ interface VerifiableRewardExtractionOptions {
2300
2838
  * doesn't report one. Default `0.7`.
2301
2839
  */
2302
2840
  judgeConfidenceFloor?: number;
2841
+ /**
2842
+ * Whether the anti-Goodhart realness gate applies. Default `true`, and the
2843
+ * default is the one every training path must keep.
2844
+ *
2845
+ * Set `false` ONLY for detection and analysis. `rl/reward-hacking.ts` does,
2846
+ * for the same reason it reads `observedScore` for its proxy: it measures the
2847
+ * DIVERGENCE between the judge signal and the deterministic one, and a
2848
+ * deterministic reward that another gate already forced to 0 manufactures
2849
+ * exactly that divergence on exactly the gamed population. The detector would
2850
+ * then be re-reporting a verdict it was supposed to reach independently.
2851
+ */
2852
+ applyRealnessGate?: boolean;
2303
2853
  }
2304
2854
  /**
2305
2855
  * Extract a `VerifiableReward` from a `VerificationReport`.
@@ -2308,6 +2858,11 @@ interface VerifiableRewardExtractionOptions {
2308
2858
  * schema → sandbox), fall back to the judge layer if `fallbackToJudge` is
2309
2859
  * true, return `null` if no signal qualifies. When multiple deterministic
2310
2860
  * layers contribute, return a `'composite'` source with a weighted blend.
2861
+ *
2862
+ * NO realness gate is applied and none can be: a `VerificationReport` carries
2863
+ * layer scores and nothing about whether the run faked them — `realness` lives
2864
+ * on the `RunRecord`. Use `extractVerifiableRewardsFromRecords` for anything
2865
+ * that becomes training data; this signature is for scoring a report in hand.
2311
2866
  */
2312
2867
  declare function extractVerifiableReward(report: VerificationReport, opts?: VerifiableRewardExtractionOptions): VerifiableReward | null;
2313
2868
  /**
@@ -2321,12 +2876,30 @@ declare function extractVerifiableReward(report: VerificationReport, opts?: Veri
2321
2876
  * verifiable reward becomes a training datum, every record that doesn't
2322
2877
  * gets filtered out (or kept with `'probabilistic'` determinism for
2323
2878
  * separate downstream handling).
2879
+ *
2880
+ * The realness gate applies to EVERY channel here, and to the deterministic one
2881
+ * MOST. It is tempting to reason that a decidable signal cannot be gamed, so
2882
+ * the gate is redundant on it — that reasoning is backwards. `realness.gated`
2883
+ * means the run's success signal was FAKED, and a test suite reporting green on
2884
+ * a stubbed integration is precisely what that looks like: the deterministic
2885
+ * layer is the thing that got faked. Exporting it ungated hands a trainer the
2886
+ * highest-credibility reward the module can emit (`determinism: 'deterministic'`,
2887
+ * `confidence: 1`) for the one population the gate exists to catch. Pass
2888
+ * `applyRealnessGate: false` only to look at the ungated numbers for detection.
2324
2889
  */
2325
2890
  declare function extractVerifiableRewardsFromRecords(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
2326
2891
  runId: string;
2327
2892
  reward: VerifiableReward | null;
2328
2893
  }>;
2329
- /** Filter `RunRecord[]` to those with deterministic verifiable rewards. */
2894
+ /**
2895
+ * Filter `RunRecord[]` to those with deterministic verifiable rewards.
2896
+ *
2897
+ * A realness-gated run is KEPT, at reward 0 with `realnessGated: true` — the
2898
+ * same rule GRPO uses on a gated line. 0 is the honest label for a faked
2899
+ * success and is usable signal, whereas dropping the run would move a group
2900
+ * baseline without saying so. (SFT differs: there every row is a target to
2901
+ * imitate, so a gated row is removed outright.)
2902
+ */
2330
2903
  declare function filterDeterministicallyRewarded(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
2331
2904
  run: RunRecord;
2332
2905
  reward: VerifiableReward;
@@ -2574,9 +3147,8 @@ type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
2574
3147
  * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
2575
3148
  * )
2576
3149
  *
2577
- * This is THE llm-calling seam for agent-eval primitives that need structured
2578
- * output (semantic concept judge, reviewer directives, critic scores). Primitives
2579
- * that need free-form text use `callLlm` and parse output themselves.
3150
+ * `createChatClient` wraps this implementation for provider-neutral package
3151
+ * entry points. Direct callers can use `callLlm` or `callLlmJson`.
2580
3152
  */
2581
3153
 
2582
3154
  type LlmThinkingMode = 'enabled' | 'disabled';
@@ -2654,8 +3226,8 @@ interface LlmClientOptions {
2654
3226
  * total attempts × `timeoutMs`.
2655
3227
  */
2656
3228
  deadlineMs?: number;
2657
- /** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
2658
- maxRetries?: number;
3229
+ /** Total provider attempts. Default 3. */
3230
+ maximumAttempts?: number;
2659
3231
  /** Token rates used when the provider omits cost or package pricing does not cover the model. */
2660
3232
  customTokenPricing?: CustomTokenPricing;
2661
3233
  /**
@@ -3474,7 +4046,7 @@ interface InterimReleaseConfidence {
3474
4046
  *
3475
4047
  * Wires:
3476
4048
  * 1. `runEvalCampaign` for the matrix run (capture, integrity, hooks)
3477
- * 2. `extractVerifiableReward` over each run, separating deterministic
4049
+ * 2. `extractVerifiableRewardsFromRecords` over the runs, separating deterministic
3478
4050
  * from probabilistic reward sources for the trainer
3479
4051
  * 3. `extractPreferences` to produce DPO/PPO/KTO triples
3480
4052
  * 4. `evaluateInterimReleaseConfidence` over paired deltas (anytime-valid)
@@ -3484,8 +4056,9 @@ interface InterimReleaseConfidence {
3484
4056
  *
3485
4057
  * The output `RLCampaignResult` is a single, audit-ready artifact: every
3486
4058
  * stage's output is in there. The consumer's downstream fits in a single
3487
- * line: pass `result.preferences` to their DPO trainer, `result.grpoRows`
3488
- * to GRPO, `result.runs` plus `result.rewardSignals` to a custom RL loop.
4059
+ * line: pass `result.preferences.pairs` to a DPO trainer,
4060
+ * `result.trainerRows.grpo` to GRPO, or `result.campaign.runs` plus
4061
+ * `result.rewardSignals` to a custom RL loop.
3489
4062
  */
3490
4063
 
3491
4064
  interface RunRLCampaignOptions<V> extends EvalCampaignOptions<V> {
@@ -3512,7 +4085,7 @@ interface RunRLCampaignOptions<V> extends EvalCampaignOptions<V> {
3512
4085
  sft?: SftLookups;
3513
4086
  };
3514
4087
  }
3515
- interface RLCampaignResult<V> {
4088
+ interface RLCampaignResult {
3516
4089
  campaign: EvalCampaignResult;
3517
4090
  /** Per-run verifiable reward (deterministic when available, probabilistic fallback otherwise). */
3518
4091
  rewardSignals: Array<{
@@ -3541,9 +4114,8 @@ interface RLCampaignResult<V> {
3541
4114
  * Convenience type-tag — consumers can branch on `result.kind`.
3542
4115
  */
3543
4116
  kind: 'agent-eval-rl-campaign';
3544
- unusedVariant?: V;
3545
4117
  }
3546
- declare function runRLCampaign<V>(opts: RunRLCampaignOptions<V>): Promise<RLCampaignResult<V>>;
4118
+ declare function runRLCampaign<V>(opts: RunRLCampaignOptions<V>): Promise<RLCampaignResult>;
3547
4119
 
3548
4120
  /**
3549
4121
  * Pass A substrate types — `runCampaign` is the one primitive every
@@ -3582,7 +4154,7 @@ interface CampaignScenarioIdentity extends Pick<Scenario, 'id' | 'kind'> {
3582
4154
  /** The canonical judge verdict shape — one declaration, shared by campaign
3583
4155
  * judges and the multishot judge runner (which re-exports this type).
3584
4156
  *
3585
- * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the legacy
4157
+ * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
3586
4158
  * multishot runner emits 0-10. Cross-scale comparison must go through
3587
4159
  * `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
3588
4160
  * promotion-policy) — never renormalize a producer's values in place, as
@@ -3736,8 +4308,6 @@ interface CampaignAggregates {
3736
4308
  byScenario: Record<string, ScenarioAggregate>;
3737
4309
  /** Canonical campaign accounting, including worker and judge calls. */
3738
4310
  cost: CostLedgerSummary;
3739
- /** Compatibility alias of `cost.totalCostUsd`. */
3740
- totalCostUsd: number;
3741
4311
  /** Cells whose dispatch completed, including cells whose later judge failed. */
3742
4312
  cellsExecuted: number;
3743
4313
  cellsSkipped: number;
@@ -3748,7 +4318,7 @@ interface CampaignAggregates {
3748
4318
  cellsDispatchFailed?: number;
3749
4319
  /** Present on results that record failure stages. */
3750
4320
  cellsJudgeFailed?: number;
3751
- /** Legacy failures whose stage was not recorded. */
4321
+ /** Failures whose stage could not be classified. */
3752
4322
  cellsUnclassifiedFailed?: number;
3753
4323
  }
3754
4324
  interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scenario> {
@@ -4113,4 +4683,4 @@ interface BuildPairwiseFromCampaignInput {
4113
4683
  }
4114
4684
  declare function buildPairwiseFromCampaign(input: BuildPairwiseFromCampaignInput): PairwiseOutcome[];
4115
4685
 
4116
- export { ABSENT_CATEGORY, type AdaptationCurve, type AdaptationPoint, type AdaptationRunner, type AdapterContext, type AdversarialMutation, type BehaviorFeatures, type BradleyTerryFit, type BradleyTerryRating, type BuildPairwiseFromCampaignInput, type CellObservation, type CompareCurvesResult, type ComputeBestOfNOptions, type ComputeBestOfNResult, type ComputeCurve, type ComputeCurveBudget, type ComputeCurvePoint, type ContaminationProbeInput, type ContaminationProbeOptions, type ContaminationProbeReport, type CorpusAppendResult, type CorpusRecord, type CurriculumAllocation, DEFAULT_MIN_N_PER_FEATURE, DEFAULT_QUANTILE_BUCKETS, type DatasetFormat, type DeploymentOutcome, type DetectRewardHackingInput, type DpoExportRow, type DpoLookups, type EasyModeOptions, type EasyModeReport, type EloOptions, type ExtractPreferencesOptions, type ExtractStepRewardsOptions, type FeatureDivergence, type FeatureShift, type FidelityReport, type FidelityVerdict, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type GrpoExportRow, type GrpoLookups, type HarvestOptions, InMemoryOutcomeStore, type OffPolicyContributionCounts, type OffPolicyEstimate, type OffPolicyOptions, type OffPolicyTrajectory, type OutcomeStore, type PairwiseOutcome, type ParetoPointInput, PredictiveValidityResearcher, type PredictiveValidityResearcherOptions, type PreferenceExtractionReport, type PreferenceStrategy, type PreferenceTriple, type PrmExportRow, type PrmLookups, type PrmTrainingTriple, REPRESENTATIVE_MIN_FIDELITY, type RLCampaignResult, type RewardHackingFinding, type RewardHackingReport, type RewardHackingSignal, type RewardKind, type RewardStats, type RlDatasetBundle, type RlDatasetConfig, type RlDatasetManifest, type RlDatasetStats, type RunAdaptationCurveOptions, type RunComputeCurveOptions, type RunRLCampaignOptions, type RunwiseStepSummary, type ScenarioPerturbation, type ScenarioPerturbationKind, type SelfConsistencyOptions, type SelfConsistencyResult, type SftExportRow, type SftLookups, type SimFidelityOptions, type StepReward, type StepRewardJsonlRow, type StepScorer, type ThompsonCurriculumOptions, type TrainingRunSelectionOptions, type VarianceCurriculumOptions, type VerifiableReward, type VerifiableRewardExtractionOptions, type VerifiableRewardSource, appendToCorpus, applyEloUpdate, bestOfN, bucketLabel, buildDatasetFromCorpus, buildPairwiseFromCampaign, buildRlDataset, campaignToRunRecords, compareAdaptationCurves, datasheetToMarkdown, defaultBehaviorFeatures, detectRewardHacking, doublyRobust, easyModeCheck, extractPreferences, extractStepRewards, extractVerifiableReward, extractVerifiableRewardsFromRecords, filterDeterministicallyRewarded, firstPassK, fitBradleyTerry, injectIrrelevantClause, inverseProbabilityWeighting, isTrainingRunEligible, jsDivergence, observationsFromRunRecords, offPolicyEstimateAll, paretoFrontier, prmTrainingPairs, quantileEdges, readCorpus, renameVariables, runAdaptationCurve, runComputeCurve, runContaminationProbe, runEvalCampaign, runRLCampaign, runwiseStepRewardSummary, selfConsistency, selfNormalizedImportanceWeighting, shuffleOrder, simFidelityReport, stepRewardsToJsonl, thompsonCurriculum, toAnthropicFormat, toDpoJsonl, toDpoRows, toGrpoJsonl, toGrpoRows, toPrmJsonl, toPrmRows, toSftJsonl, toSftRows, validateDatasetFormats, varianceBasedCurriculum, verificationReportToRunRecord };
4686
+ export { ABSENT_CATEGORY, type AdaptationCurve, type AdaptationPoint, type AdaptationRunner, type AdapterContext, type AdversarialMutation, type BehaviorFeatures, type BradleyTerryFit, type BradleyTerryRating, type BuildPairwiseFromCampaignInput, type CellObservation, type CompareCurvesResult, type ComputeBestOfNOptions, type ComputeBestOfNResult, type ComputeCurve, type ComputeCurveBudget, type ComputeCurvePoint, type ContaminationProbeInput, type ContaminationProbeOptions, type ContaminationProbeReport, type CorpusAppendResult, type CorpusRecord, type CurriculumAllocation, DEFAULT_MIN_N_PER_FEATURE, DEFAULT_QUANTILE_BUCKETS, DPO_CONTEXT_REQUIREMENT, type DatasetFormat, type DeploymentOutcome, type DetectRewardHackingInput, type DpoExportRow, type DpoLineContext, type DpoLookups, type EasyModeOptions, type EasyModeReport, type EloOptions, type ExtractPreferencesOptions, type ExtractStepRewardsOptions, type FeatureDivergence, type FeatureShift, type FidelityReport, type FidelityVerdict, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type GrpoExportRow, type GrpoLookups, type HarvestOptions, InMemoryOutcomeStore, type OffPolicyContributionCounts, type OffPolicyEstimate, type OffPolicyOptions, type OffPolicyTrajectory, type OutcomeStore, PRM_CONTEXT_REQUIREMENT, type PairwiseOutcome, type ParetoPointInput, PredictiveValidityResearcher, type PredictiveValidityResearcherOptions, type PreferenceExtractionReport, type PreferenceStrategy, type PreferenceTriple, type PrmExportRow, type PrmLineContext, type PrmLookups, type PrmTrainingTriple, REPRESENTATIVE_MIN_FIDELITY, type RLCampaignResult, type RewardHackingFinding, type RewardHackingReport, type RewardHackingSignal, type RewardKind, type RewardStats, type RlDatasetBundle, type RlDatasetConfig, type RlDatasetManifest, type RlDatasetStats, type RolloutLineContext, type RunAdaptationCurveOptions, type RunComputeCurveOptions, type RunRLCampaignOptions, type RunwiseStepSummary, STEP_REWARD_CONTEXT_REQUIREMENT, type ScenarioPerturbation, type ScenarioPerturbationKind, type SelfConsistencyOptions, type SelfConsistencyResult, type SftExportRow, type SftLookups, type SimFidelityOptions, type StepReward, type StepRewardJsonlRow, type StepScorer, type ThompsonCurriculumOptions, type TrainingLineSelectionOptions, type VarianceCurriculumOptions, type VerifiableReward, type VerifiableRewardExtractionOptions, type VerifiableRewardSource, appendToCorpus, applyEloUpdate, assertPrmTrainableLine, bestOfN, bucketLabel, buildDatasetFromCorpus, buildPairwiseFromCampaign, buildRlDataset, campaignToRunRecords, compareAdaptationCurves, datasheetToMarkdown, defaultBehaviorFeatures, detectRewardHacking, doublyRobust, easyModeCheck, extractPreferences, extractStepRewards, extractVerifiableReward, extractVerifiableRewardsFromRecords, filterDeterministicallyRewarded, firstPassK, fitBradleyTerry, injectIrrelevantClause, inverseProbabilityWeighting, jsDivergence, observationsFromRunRecords, offPolicyEstimateAll, paretoFrontier, prmTrainingPairs, quantileEdges, readCorpus, renameVariables, runAdaptationCurve, runComputeCurve, runContaminationProbe, runEvalCampaign, runRLCampaign, runwiseStepRewardSummary, selfConsistency, selfNormalizedImportanceWeighting, shuffleOrder, simFidelityReport, stepRewardsToJsonl, thompsonCurriculum, toAnthropicFormat, toDpoJsonl, toDpoRows, toGrpoJsonl, toGrpoRows, toPrmJsonl, toPrmRows, toSftJsonl, toSftRows, toTRLFormat, validateDatasetFormats, varianceBasedCurriculum, verificationReportToRunRecord };