@tangle-network/agent-eval 0.128.2 → 0.129.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +265 -0
- package/README.md +18 -0
- package/dist/analyst/index.d.ts +107 -165
- package/dist/analyst/index.js +5 -9
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +2 -19
- package/dist/belief-state/index.js +30 -31
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +5 -8
- package/dist/benchmarks/index.js +12 -11
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +30 -39
- package/dist/campaign/index.js +11 -10
- package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
- package/dist/chunk-2QU3YOPR.js.map +1 -0
- package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
- package/dist/chunk-3OCR4R5I.js.map +1 -0
- package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
- package/dist/chunk-56TAVBOK.js.map +1 -0
- package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
- package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
- package/dist/chunk-BSO5JDQH.js.map +1 -0
- package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
- package/dist/chunk-C6LXANRU.js.map +1 -0
- package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
- package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
- package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
- package/dist/chunk-EG66UGL4.js.map +1 -0
- package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
- package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
- package/dist/chunk-G7MGMCZD.js.map +1 -0
- package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
- package/dist/chunk-H23X7XKK.js.map +1 -0
- package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
- package/dist/chunk-HPWUNB47.js.map +1 -0
- package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
- package/dist/chunk-IYCLP2N2.js.map +1 -0
- package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
- package/dist/chunk-JQSF5DQT.js.map +1 -0
- package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
- package/dist/chunk-M4YBQKIJ.js.map +1 -0
- package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
- package/dist/chunk-OIUOT4QD.js +44 -0
- package/dist/chunk-OIUOT4QD.js.map +1 -0
- package/dist/chunk-OWN5NPMC.js +152 -0
- package/dist/chunk-OWN5NPMC.js.map +1 -0
- package/dist/chunk-PC5DOSM7.js +579 -0
- package/dist/chunk-PC5DOSM7.js.map +1 -0
- package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
- package/dist/chunk-QB6BDBP2.js.map +1 -0
- package/dist/chunk-RXHCETDZ.js +536 -0
- package/dist/chunk-RXHCETDZ.js.map +1 -0
- package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
- package/dist/chunk-SFLLL76A.js.map +1 -0
- package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
- package/dist/chunk-T6RLYGAD.js.map +1 -0
- package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
- package/dist/chunk-TJVT4QFF.js.map +1 -0
- package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
- package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
- package/dist/chunk-U4L7JRPZ.js.map +1 -0
- package/dist/chunk-U4PHLT2N.js +419 -0
- package/dist/chunk-U4PHLT2N.js.map +1 -0
- package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
- package/dist/chunk-VCZ5FQYW.js.map +1 -0
- package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
- package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
- package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
- package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
- package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
- package/dist/chunk-ZHTZ4EYI.js.map +1 -0
- package/dist/cli.js +6 -5
- package/dist/cli.js.map +1 -1
- package/dist/contract/index.d.ts +47 -87
- package/dist/contract/index.js +14 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +3 -2
- package/dist/fuzz.js +3 -2
- package/dist/fuzz.js.map +1 -1
- package/dist/index.d.ts +659 -203
- package/dist/index.js +145 -117
- package/dist/index.js.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +3 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +5 -5
- package/dist/reporting.d.ts +14 -0
- package/dist/reporting.js +7 -6
- package/dist/rl.d.ts +652 -82
- package/dist/rl.js +415 -171
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1071 -32
- package/dist/rollout/index.js +68 -10
- package/dist/run-campaign-OJJ7CZF4.js +18 -0
- package/dist/supervisor-run/index.d.ts +114 -4
- package/dist/supervisor-run/index.js +4 -3
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +6 -5
- package/dist/wire/index.d.ts +10 -11
- package/dist/wire/index.js +3 -3
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +4 -4
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
- /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
- /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
- /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
- /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
- /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
- /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
- /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
- /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/dist/rl.d.ts
CHANGED
|
@@ -939,7 +939,369 @@ declare function injectIrrelevantClause<S extends {
|
|
|
939
939
|
}>(clause: string, position?: 'prefix' | 'suffix'): ScenarioPerturbation<S>;
|
|
940
940
|
|
|
941
941
|
/**
|
|
942
|
-
*
|
|
942
|
+
* `tangle.rollout.v1` — THE canonical rollout serialization, owned by
|
|
943
|
+
* agent-eval. One JSONL line per agent invocation (a solo eval run, a
|
|
944
|
+
* supervisor episode, a worker session, a proposer shot, a judge call, an
|
|
945
|
+
* analyst pass), labeled with its task/split coordinates and a single
|
|
946
|
+
* scalar reward, carrying the FULL message transcript inline.
|
|
947
|
+
*
|
|
948
|
+
* This schema is the reconciliation of two prior producers:
|
|
949
|
+
* - agent-eval's RunRecord-joined rollout rows (PR #410): identity,
|
|
950
|
+
* provenance hashes, the realness gate travelling into the reward,
|
|
951
|
+
* trace-derived steps.
|
|
952
|
+
* - the bench rollout-ledger (agent-runtime PR #591): the wire shape —
|
|
953
|
+
* role, task.split/rep, parent_rollout_id, policy provenance, capture
|
|
954
|
+
* provenance, inline canonical chat-with-tools messages.
|
|
955
|
+
* Where the two conflicted, RunRecord-derived semantics won; the wire
|
|
956
|
+
* field names follow the ledger (snake_case). See `docs/rollout.md` for
|
|
957
|
+
* the field-by-field decision table.
|
|
958
|
+
*
|
|
959
|
+
* Messages are inlined — never referenced — because every harness store a
|
|
960
|
+
* rollout can be recovered from is mutable or garbage-collected. A line
|
|
961
|
+
* must stay a complete training/eval example on its own.
|
|
962
|
+
*
|
|
963
|
+
* `outcome.reward` is THE single scalar (null = no verdict exists — a
|
|
964
|
+
* labeled gap, never 0). `outcome.realness_gated` is the anti-Goodhart
|
|
965
|
+
* flag: a gated line must never export as a positive training example.
|
|
966
|
+
*
|
|
967
|
+
* That last sentence is enforced here, by `validateRolloutLine`, not merely
|
|
968
|
+
* documented. Validating `reward` and `realness_gated` independently — each a
|
|
969
|
+
* well-typed field, their COMBINATION unchecked — is what let a line claiming
|
|
970
|
+
* `{reward: 0.95, realness_gated: true}` validate clean and walk into every
|
|
971
|
+
* training export. The relationship between the two IS the invariant, so it is
|
|
972
|
+
* checked where every other structural claim about a line is checked.
|
|
973
|
+
*
|
|
974
|
+
* The invariant is about the OUTCOME, not about one field of it. Zeroing
|
|
975
|
+
* `reward` while `outcome.metrics` still carried the per-layer scores that
|
|
976
|
+
* reward was computed from exported the gamed signal anyway, in the dict the
|
|
977
|
+
* verifiers format reads as its per-rubric scores. So `gateGamedOutcome`
|
|
978
|
+
* transforms the whole outcome once, at `assertMinted` — the funnel every
|
|
979
|
+
* minted line passes — and the reward-bearing components are relocated to
|
|
980
|
+
* `provenance.gated_evidence`, which no exporter projects.
|
|
981
|
+
*
|
|
982
|
+
* WHICH checks each door applies is not decided in this file. `./gate-checks`
|
|
983
|
+
* owns the canonical list and the total per-entry-point policy; the three doors
|
|
984
|
+
* below (`validateRolloutLine`, `assertRewardGate`, `assertMinted`) each call
|
|
985
|
+
* `gateErrors` with their declared policy, so a check added to that list applies
|
|
986
|
+
* here without anyone editing this file, and a check deliberately skipped has to
|
|
987
|
+
* name itself there.
|
|
988
|
+
*/
|
|
989
|
+
declare const ROLLOUT_SCHEMA = "tangle.rollout.v1";
|
|
990
|
+
/** `agent` = a solo evaluation run (no multi-agent topology). */
|
|
991
|
+
type RolloutRole = 'agent' | 'supervisor' | 'worker' | 'proposer' | 'judge' | 'analyst';
|
|
992
|
+
/** Split vocabulary follows `RunRecord.splitTag`, extended with `canary`. */
|
|
993
|
+
type RolloutSplit = 'search' | 'dev' | 'holdout' | 'canary';
|
|
994
|
+
/** 'mint' = joined live from RunRecord + trace by `mintRolloutRows`. */
|
|
995
|
+
type RolloutCapture = 'mint' | 'settle-time' | 'backfill';
|
|
996
|
+
type ChatRole = 'system' | 'user' | 'assistant' | 'tool';
|
|
997
|
+
interface ChatToolCall {
|
|
998
|
+
id: string;
|
|
999
|
+
type: 'function';
|
|
1000
|
+
function: {
|
|
1001
|
+
name: string;
|
|
1002
|
+
/** JSON-encoded argument object, exactly as the model emitted it. */
|
|
1003
|
+
arguments: string;
|
|
1004
|
+
};
|
|
1005
|
+
}
|
|
1006
|
+
interface ChatMessage {
|
|
1007
|
+
role: ChatRole;
|
|
1008
|
+
content: string | null;
|
|
1009
|
+
/** Reasoning/thinking channel where the harness captured it (full fidelity). */
|
|
1010
|
+
reasoning_content?: string;
|
|
1011
|
+
tool_calls?: ChatToolCall[];
|
|
1012
|
+
/** Required on role:"tool" — the ChatToolCall this result answers. */
|
|
1013
|
+
tool_call_id?: string;
|
|
1014
|
+
name?: string;
|
|
1015
|
+
/**
|
|
1016
|
+
* Harbor ATIF `is_copied_context` (RFC 0001 rule 7): this turn was COPIED IN
|
|
1017
|
+
* from another trajectory's context, not produced by the agent on this line.
|
|
1018
|
+
* The RFC makes excluding it from SFT a MUST, and `toSftRows` does — training
|
|
1019
|
+
* on it teaches the model to author text it never authored, and credits this
|
|
1020
|
+
* run for another one's work. Absent = false (authored here).
|
|
1021
|
+
*/
|
|
1022
|
+
is_copied_context?: boolean;
|
|
1023
|
+
}
|
|
1024
|
+
interface ToolDef {
|
|
1025
|
+
type: 'function';
|
|
1026
|
+
function: {
|
|
1027
|
+
name: string;
|
|
1028
|
+
description?: string;
|
|
1029
|
+
parameters?: Record<string, unknown>;
|
|
1030
|
+
};
|
|
1031
|
+
}
|
|
1032
|
+
/**
|
|
1033
|
+
* Compact trace-span projection (llm/tool step) carried alongside the
|
|
1034
|
+
* conversation when the line was minted from a trace. Optional: lines
|
|
1035
|
+
* recovered from harness stores have no span structure.
|
|
1036
|
+
*/
|
|
1037
|
+
interface RolloutStep {
|
|
1038
|
+
kind: string;
|
|
1039
|
+
name: string;
|
|
1040
|
+
/** llm: last-message summary · tool: stringified args. Scrubbed. */
|
|
1041
|
+
input?: string;
|
|
1042
|
+
/** llm: output text · tool: stringified result. Scrubbed. */
|
|
1043
|
+
output?: string;
|
|
1044
|
+
status?: 'ok' | 'error';
|
|
1045
|
+
durationMs?: number;
|
|
1046
|
+
/**
|
|
1047
|
+
* LLM inferences this span represents. 0 = deterministic dispatch with no
|
|
1048
|
+
* model call — distinct from absent, which means the producer did not track it.
|
|
1049
|
+
*/
|
|
1050
|
+
llm_call_count?: number;
|
|
1051
|
+
/** Exact prompt tokenization. Removes the ambiguity of re-tokenizing text at train time. */
|
|
1052
|
+
prompt_token_ids?: number[];
|
|
1053
|
+
/** Exact completion tokenization; aligns index-wise with `logprobs`. */
|
|
1054
|
+
completion_token_ids?: number[];
|
|
1055
|
+
/**
|
|
1056
|
+
* Per-completion-token log probabilities under the sampling policy. Required
|
|
1057
|
+
* for off-policy correction (importance weighting) when the rollout was
|
|
1058
|
+
* generated by a policy other than the one being trained.
|
|
1059
|
+
*/
|
|
1060
|
+
logprobs?: number[];
|
|
1061
|
+
}
|
|
1062
|
+
interface RolloutTask {
|
|
1063
|
+
/** Benchmark/suite id (e.g. "swe-bench-verified") or the experiment id. */
|
|
1064
|
+
suite: string;
|
|
1065
|
+
instance_id: string;
|
|
1066
|
+
split: RolloutSplit;
|
|
1067
|
+
/** Sampling seed the campaign pinned; null = not recorded. */
|
|
1068
|
+
seed: number | null;
|
|
1069
|
+
/** Replicate index (0-based). */
|
|
1070
|
+
rep: number;
|
|
1071
|
+
}
|
|
1072
|
+
interface RolloutPolicy {
|
|
1073
|
+
/** Harness that drove the invocation (e.g. "opencode", "claude", "pi-loops"). */
|
|
1074
|
+
harness: string | null;
|
|
1075
|
+
harness_version: string | null;
|
|
1076
|
+
model: string | null;
|
|
1077
|
+
provider: string | null;
|
|
1078
|
+
/** Commit of the agent profile / candidate under evaluation. */
|
|
1079
|
+
profile_commit: string | null;
|
|
1080
|
+
/** sha256 of the effective prompt (post-steering), when recorded. */
|
|
1081
|
+
prompt_hash?: string | null;
|
|
1082
|
+
/** sha256 of the effective run config, when recorded. */
|
|
1083
|
+
config_hash?: string | null;
|
|
1084
|
+
/** Canonical agent-profile cell identity, when the run carries one. */
|
|
1085
|
+
agent_profile_cell_id?: string | null;
|
|
1086
|
+
/** Sampling params (temperature, top_p, max_tokens…); null = not recorded. */
|
|
1087
|
+
sampling: Record<string, unknown> | null;
|
|
1088
|
+
}
|
|
1089
|
+
interface RolloutOutcome {
|
|
1090
|
+
/**
|
|
1091
|
+
* THE single scalar training signal — the official verdict.
|
|
1092
|
+
* null = no verdict exists for this invocation (a labeled gap, never 0).
|
|
1093
|
+
*/
|
|
1094
|
+
reward: number | null;
|
|
1095
|
+
/** Where the reward came from (judge id; "/inherited" = parent episode's). */
|
|
1096
|
+
reward_source: string | null;
|
|
1097
|
+
/** Raw judge verdict record, verbatim. */
|
|
1098
|
+
verdict: unknown;
|
|
1099
|
+
/** Everything that is NOT the scalar reward. */
|
|
1100
|
+
metrics: Record<string, unknown>;
|
|
1101
|
+
is_completed: boolean;
|
|
1102
|
+
is_truncated: boolean;
|
|
1103
|
+
error: string | null;
|
|
1104
|
+
/**
|
|
1105
|
+
* Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run faked
|
|
1106
|
+
* its success signal. `true` requires `reward` to be 0 or null — the
|
|
1107
|
+
* validator rejects the line otherwise — and the line never qualifies for
|
|
1108
|
+
* SFT. Required on the wire: a line that does not state the flag does not
|
|
1109
|
+
* validate, so no producer can dodge the gate by omitting it.
|
|
1110
|
+
*
|
|
1111
|
+
* `true` ALSO requires `metrics` to be empty and `verdict` to be null: the
|
|
1112
|
+
* numbers the reward was computed from are relocated to
|
|
1113
|
+
* `provenance.gated_evidence` by `gateGamedOutcome`. See that function for
|
|
1114
|
+
* why zeroing the scalar alone was not enough.
|
|
1115
|
+
*/
|
|
1116
|
+
realness_gated: boolean;
|
|
1117
|
+
/**
|
|
1118
|
+
* Whether an authenticity SCREEN ever RAN on this reward — a different claim
|
|
1119
|
+
* from `realness_gated`, which is the screen's VERDICT.
|
|
1120
|
+
*
|
|
1121
|
+
* `realness_gated: false` reads as "we looked and nothing fired". A producer
|
|
1122
|
+
* with no screen at all was emitting exactly that, so a never-screened reward
|
|
1123
|
+
* was indistinguishable on the wire from a screened-clean one, and the whole
|
|
1124
|
+
* anti-Goodhart apparatus silently treated the first as the second. The two
|
|
1125
|
+
* claims are now separable:
|
|
1126
|
+
*
|
|
1127
|
+
* - `true` — a screen ran; `realness_gated` is its verdict.
|
|
1128
|
+
* - `false` — the producer declares it HAS no screen (`unscreenedRewardFields`).
|
|
1129
|
+
* `assertMinted` REFUSES such a line when its reward is above
|
|
1130
|
+
* zero: an unscreened positive reward is precisely the signal
|
|
1131
|
+
* the gate exists to qualify, and nothing has qualified it.
|
|
1132
|
+
* - absent — not stated. Pre-unification ledgers land here, as does a
|
|
1133
|
+
* `RunRecord` carrying no `outcome.realness` at all. Absent is
|
|
1134
|
+
* read as "unknown", never as `false` (which would refuse most
|
|
1135
|
+
* of the existing corpus) and never as `true`.
|
|
1136
|
+
*/
|
|
1137
|
+
realness_screened?: boolean;
|
|
1138
|
+
}
|
|
1139
|
+
interface RolloutCostBlock {
|
|
1140
|
+
usd: number | null;
|
|
1141
|
+
tokens_in: number | null;
|
|
1142
|
+
tokens_out: number | null;
|
|
1143
|
+
tokens_reasoning: number | null;
|
|
1144
|
+
cache_read: number | null;
|
|
1145
|
+
cache_write: number | null;
|
|
1146
|
+
wall_s: number | null;
|
|
1147
|
+
/**
|
|
1148
|
+
* Total LLM inferences across the invocation (ATIF `llm_call_count`,
|
|
1149
|
+
* aggregated). Optional and additive: absent = not tracked, never 0.
|
|
1150
|
+
*/
|
|
1151
|
+
llm_call_count?: number | null;
|
|
1152
|
+
}
|
|
1153
|
+
interface RolloutArtifacts {
|
|
1154
|
+
patch_path: string | null;
|
|
1155
|
+
run_dir: string | null;
|
|
1156
|
+
/** Source-of-truth transcript pointer (session id / jsonl path) for audit. */
|
|
1157
|
+
transcript_ref: string | null;
|
|
1158
|
+
}
|
|
1159
|
+
/**
|
|
1160
|
+
* The reward-bearing half of a GATED line's outcome, moved off `outcome` and
|
|
1161
|
+
* parked here verbatim. Diagnostics, never training input — see
|
|
1162
|
+
* `gateGamedOutcome`.
|
|
1163
|
+
*/
|
|
1164
|
+
interface GatedEvidence {
|
|
1165
|
+
/** `outcome.metrics` exactly as the producer measured it. */
|
|
1166
|
+
metrics?: Record<string, unknown>;
|
|
1167
|
+
/** `outcome.verdict` verbatim — the judge record that claimed the success. */
|
|
1168
|
+
verdict?: unknown;
|
|
1169
|
+
/**
|
|
1170
|
+
* The per-step fields `tangle.rollout.v1` does not declare, parked here when
|
|
1171
|
+
* the gate projected `steps[]` down to the schema's own key set.
|
|
1172
|
+
*
|
|
1173
|
+
* A per-step reward is training signal exactly like the scalar, and `steps`
|
|
1174
|
+
* rides through `toRewardRows` verbatim — so a gated line was shipping its
|
|
1175
|
+
* step-level credit assignment at full value beside a `reward` of 0.
|
|
1176
|
+
*/
|
|
1177
|
+
steps?: unknown;
|
|
1178
|
+
}
|
|
1179
|
+
interface RolloutProvenance {
|
|
1180
|
+
captured_at: string;
|
|
1181
|
+
capture: RolloutCapture;
|
|
1182
|
+
/**
|
|
1183
|
+
* Why this line is incomplete. Required when `messages` is empty (the
|
|
1184
|
+
* transcript could not be recovered); also set by interchange importers to
|
|
1185
|
+
* name a MISSING LABEL — an imported trajectory carries no verdict, so
|
|
1186
|
+
* `outcome.reward` is null and this says why.
|
|
1187
|
+
*/
|
|
1188
|
+
gap?: string;
|
|
1189
|
+
/**
|
|
1190
|
+
* Present only on a realness-gated line: the outcome fields the gate
|
|
1191
|
+
* relocated, kept so an auditor can still see WHY the run was gated and what
|
|
1192
|
+
* it claimed. Deliberately OUTSIDE `outcome`, because every training exporter
|
|
1193
|
+
* reads `outcome` and none reads `provenance`.
|
|
1194
|
+
*/
|
|
1195
|
+
gated_evidence?: GatedEvidence;
|
|
1196
|
+
}
|
|
1197
|
+
interface RolloutLine {
|
|
1198
|
+
schema: typeof ROLLOUT_SCHEMA;
|
|
1199
|
+
rollout_id: string;
|
|
1200
|
+
/** Spawning invocation within the same episode (worker → supervisor). */
|
|
1201
|
+
parent_rollout_id: string | null;
|
|
1202
|
+
run_id: string;
|
|
1203
|
+
/** Logical experiment grouping from `RunRecord.experimentId`; null = not recorded. */
|
|
1204
|
+
experiment_id: string | null;
|
|
1205
|
+
/** Stable candidate identity from `RunRecord.candidateId`; null = not recorded. */
|
|
1206
|
+
candidate_id: string | null;
|
|
1207
|
+
/** Improvement-loop generation (-1 = baseline); null = not an improvement loop. */
|
|
1208
|
+
generation: number | null;
|
|
1209
|
+
/** Improvement-loop candidate index (-1 = baseline); null = not an improvement loop. */
|
|
1210
|
+
candidate_index: number | null;
|
|
1211
|
+
role: RolloutRole;
|
|
1212
|
+
task: RolloutTask;
|
|
1213
|
+
policy: RolloutPolicy;
|
|
1214
|
+
/** Full transcript, inline. [] = gap line (see provenance.gap). */
|
|
1215
|
+
messages: ChatMessage[];
|
|
1216
|
+
tool_defs: ToolDef[];
|
|
1217
|
+
/** Trace-span projections, when minted from a trace. */
|
|
1218
|
+
steps?: RolloutStep[];
|
|
1219
|
+
outcome: RolloutOutcome;
|
|
1220
|
+
cost: RolloutCostBlock;
|
|
1221
|
+
artifacts: RolloutArtifacts;
|
|
1222
|
+
provenance: RolloutProvenance;
|
|
1223
|
+
}
|
|
1224
|
+
/**
|
|
1225
|
+
* Phantom property. `declare const` means it exists only in the type system:
|
|
1226
|
+
* nothing is written at runtime, so a branded line still serializes to exactly
|
|
1227
|
+
* the same JSON as a plain one.
|
|
1228
|
+
*/
|
|
1229
|
+
declare const MINTED_ROLLOUT: unique symbol;
|
|
1230
|
+
/**
|
|
1231
|
+
* A minted outcome states the gate verdict — it is not allowed to stay silent —
|
|
1232
|
+
* and, when that verdict is `true`, carries nothing else the reward was derived
|
|
1233
|
+
* from (`gateGamedOutcome` has run).
|
|
1234
|
+
*/
|
|
1235
|
+
interface MintedRolloutOutcome extends RolloutOutcome {
|
|
1236
|
+
realness_gated: boolean;
|
|
1237
|
+
}
|
|
1238
|
+
/**
|
|
1239
|
+
* A `RolloutLine` whose reward has been checked against the anti-Goodhart
|
|
1240
|
+
* invariant. The type every training-data exporter takes.
|
|
1241
|
+
*
|
|
1242
|
+
* Why a brand and not just the interface: `RolloutLine` is structural, so any
|
|
1243
|
+
* hand-built object literal of the right shape IS one — which is how a line
|
|
1244
|
+
* declaring `{reward: 0.95, realness_gated: true}` reached the exporters
|
|
1245
|
+
* despite them "only accepting a minted line". The phantom symbol makes the
|
|
1246
|
+
* type nominal: it cannot be produced by writing an object literal, only by
|
|
1247
|
+
* `mintRolloutRows` (which applies the gate), `readRolloutLedger` (which
|
|
1248
|
+
* validates every line off disk), or an explicit, greppable `assertMinted`.
|
|
1249
|
+
*
|
|
1250
|
+
* Belt and braces on purpose. The brand closes first-party call sites at
|
|
1251
|
+
* COMPILE time; `validateRolloutLine` closes data arriving at RUNTIME (ledger
|
|
1252
|
+
* files, foreign imports, JSON from another process) where types are absent.
|
|
1253
|
+
* Neither alone is enough.
|
|
1254
|
+
*
|
|
1255
|
+
* Assignable to `RolloutLine` in one direction only: readers, analysis, and
|
|
1256
|
+
* the ledger writer keep taking the plain type.
|
|
1257
|
+
*/
|
|
1258
|
+
type MintedRolloutLine = Omit<RolloutLine, 'outcome'> & {
|
|
1259
|
+
readonly [MINTED_ROLLOUT]: true;
|
|
1260
|
+
outcome: MintedRolloutOutcome;
|
|
1261
|
+
};
|
|
1262
|
+
|
|
1263
|
+
/**
|
|
1264
|
+
* Shared checks for trainer exports over canonical minted rollout lines.
|
|
1265
|
+
*
|
|
1266
|
+
* Exporters accept only `MintedRolloutLine[]`. Callers convert run records with
|
|
1267
|
+
* `mintRolloutRows` before deriving preferences or trainer files.
|
|
1268
|
+
*/
|
|
1269
|
+
|
|
1270
|
+
/**
|
|
1271
|
+
* The minted lines behind a LINE-LESS training artifact.
|
|
1272
|
+
*
|
|
1273
|
+
* `PreferenceTriple`, `PrmTrainingTriple` and `StepReward` all carry a bare
|
|
1274
|
+
* reward number plus run ids, and nothing that says whether those runs faked
|
|
1275
|
+
* their success. An exporter over them therefore has no way, from its input
|
|
1276
|
+
* alone, to learn that its chosen side is a run the gate flagged — it will
|
|
1277
|
+
* happily emit the gaming trajectory as the preferred one. Supplying the lines
|
|
1278
|
+
* is what gives it eyes.
|
|
1279
|
+
*/
|
|
1280
|
+
interface RolloutLineContext {
|
|
1281
|
+
/**
|
|
1282
|
+
* Minted lines for every INVOCATION the artifacts reference.
|
|
1283
|
+
*
|
|
1284
|
+
* Not "one line per run": `tangle.rollout.v1` models many invocations per
|
|
1285
|
+
* `run_id` — that is what `rollout_id` and `parent_rollout_id` are for, and
|
|
1286
|
+
* `supervisorRunRolloutLines` emits a supervisor node plus one per worker, all
|
|
1287
|
+
* sharing a single `run_id`. A reference is resolved against `rollout_id`
|
|
1288
|
+
* first and falls back to `run_id` only when that run has exactly one
|
|
1289
|
+
* invocation; see `resolveInvocation`.
|
|
1290
|
+
*/
|
|
1291
|
+
lines: MintedRolloutLine[];
|
|
1292
|
+
}
|
|
1293
|
+
/** How one exporter names itself and its context type in the failure messages. */
|
|
1294
|
+
interface LineContextRequirement {
|
|
1295
|
+
/** Exporter label, e.g. `'DPO export'`. */
|
|
1296
|
+
exporter: string;
|
|
1297
|
+
/** The context type the caller must pass, e.g. `'DpoLineContext'`. */
|
|
1298
|
+
contextType: string;
|
|
1299
|
+
/** Why this exporter cannot see the gate without lines. One sentence. */
|
|
1300
|
+
because: string;
|
|
1301
|
+
}
|
|
1302
|
+
|
|
1303
|
+
/**
|
|
1304
|
+
* Preference dataset extraction from canonical minted rollout lines.
|
|
943
1305
|
*
|
|
944
1306
|
* Production RLHF / DPO / KTO / SimPO pipelines need preference triples:
|
|
945
1307
|
* `(prompt, chosen, rejected)`. The campaign artifact already contains the
|
|
@@ -973,7 +1335,16 @@ declare function injectIrrelevantClause<S extends {
|
|
|
973
1335
|
* per scenario but biggest score gap per pair. Useful for early
|
|
974
1336
|
* bootstrapping when you have few variants.
|
|
975
1337
|
*
|
|
976
|
-
*
|
|
1338
|
+
* The output `PreferenceTriple` is *agent-eval-canonical* but trivially
|
|
1339
|
+
* mappable to TRL's `DPODataset` shape (`prompt`, `chosen`, `rejected`)
|
|
1340
|
+
* via the `toTRLFormat` helper, which resolves real prompt/completion text
|
|
1341
|
+
* through the same lookups `toDpoRows` takes (`./exporters` carries the
|
|
1342
|
+
* richer row with margin + metadata).
|
|
1343
|
+
*
|
|
1344
|
+
* Input discipline: the function accepts only `MintedRolloutLine[]`, whose
|
|
1345
|
+
* reward and authenticity fields have already been validated. `search` is the
|
|
1346
|
+
* default split; held-out pairing requires an explicit opt-in, while `dev` and
|
|
1347
|
+
* `canary` remain evaluation-only.
|
|
977
1348
|
*/
|
|
978
1349
|
|
|
979
1350
|
type PreferenceStrategy = 'paired-by-scenario-and-seed' | 'paired-by-scenario' | 'top-vs-bottom';
|
|
@@ -1000,7 +1371,7 @@ interface PreferenceTriple {
|
|
|
1000
1371
|
/** Tie-breaker — when multiple seeds match this scenario, the one used. */
|
|
1001
1372
|
seed?: number;
|
|
1002
1373
|
/**
|
|
1003
|
-
* Free-form metadata propagated from the
|
|
1374
|
+
* Free-form metadata propagated from the rollout lines, such as original
|
|
1004
1375
|
* prompt-hash, model, etc. Lets the RL trainer reconstruct the prompt.
|
|
1005
1376
|
*/
|
|
1006
1377
|
meta: {
|
|
@@ -1012,7 +1383,7 @@ interface PreferenceTriple {
|
|
|
1012
1383
|
rejectedModel: string;
|
|
1013
1384
|
};
|
|
1014
1385
|
}
|
|
1015
|
-
interface ExtractPreferencesOptions
|
|
1386
|
+
interface ExtractPreferencesOptions {
|
|
1016
1387
|
strategy?: PreferenceStrategy;
|
|
1017
1388
|
/**
|
|
1018
1389
|
* Minimum score gap required to admit a pair. Pairs below this are
|
|
@@ -1020,16 +1391,13 @@ interface ExtractPreferencesOptions extends TrainingRunSelectionOptions {
|
|
|
1020
1391
|
*/
|
|
1021
1392
|
minMargin?: number;
|
|
1022
1393
|
/**
|
|
1023
|
-
* Optional split
|
|
1024
|
-
* Holdout requires `allowHeldOutTrainingData: true`; dev
|
|
1394
|
+
* Optional split filter. Without one, only search is included.
|
|
1395
|
+
* Holdout requires `allowHeldOutTrainingData: true`; dev and canary are
|
|
1396
|
+
* evaluation-only.
|
|
1025
1397
|
*/
|
|
1026
|
-
|
|
1027
|
-
/**
|
|
1028
|
-
|
|
1029
|
-
* `outcome.searchScore`. Use to drive preferences off a verifiable
|
|
1030
|
-
* reward instead of the headline score.
|
|
1031
|
-
*/
|
|
1032
|
-
rewardOf?: (run: RunRecord) => number | null;
|
|
1398
|
+
split?: RolloutSplit;
|
|
1399
|
+
/** Named opt-in required before held-out lines may be paired. */
|
|
1400
|
+
allowHeldOutTrainingData?: boolean;
|
|
1033
1401
|
}
|
|
1034
1402
|
interface PreferenceExtractionReport {
|
|
1035
1403
|
pairs: PreferenceTriple[];
|
|
@@ -1041,26 +1409,64 @@ interface PreferenceExtractionReport {
|
|
|
1041
1409
|
cellsSingleton: number;
|
|
1042
1410
|
/** Strategy used. */
|
|
1043
1411
|
strategy: PreferenceStrategy;
|
|
1412
|
+
/**
|
|
1413
|
+
* Lines dropped before pairing because they carry no `candidate_id`. A
|
|
1414
|
+
* preference is a statement about two candidates, so a line that names none
|
|
1415
|
+
* cannot be paired.
|
|
1416
|
+
*/
|
|
1417
|
+
linesWithoutCandidateId: number;
|
|
1044
1418
|
}
|
|
1045
1419
|
/**
|
|
1046
|
-
* Convert
|
|
1420
|
+
* Convert rollout lines to preference triples for RL training.
|
|
1047
1421
|
*
|
|
1048
1422
|
* Returns a structured report so callers can see how much data was
|
|
1049
1423
|
* dropped and why (low-margin pairs, singleton cells). For production
|
|
1050
1424
|
* pipelines, you usually want to:
|
|
1051
1425
|
*
|
|
1052
1426
|
* 1. Run a campaign producing 5–10 variants × 50–200 scenarios × 3 seeds
|
|
1053
|
-
* 2.
|
|
1054
|
-
*
|
|
1055
|
-
* 3. Pass `report.pairs` to `toDpoRows`
|
|
1427
|
+
* 2. Mint the runs with `mintRolloutRows` and call this with
|
|
1428
|
+
* `strategy: 'paired-by-scenario-and-seed'`
|
|
1429
|
+
* 3. Pass `report.pairs` to `toDpoRows` (or `toTRLFormat`) with
|
|
1430
|
+
* prompt/completion resolvers and pipe to your DPO trainer
|
|
1431
|
+
*
|
|
1432
|
+
* The gate is what makes a preference dataset safe: ordered on an ungated
|
|
1433
|
+
* score, a gamed run with an inflated number becomes the `chosen` side and DPO
|
|
1434
|
+
* is trained to prefer the gaming trajectory over its honest sibling. A gated
|
|
1435
|
+
* line arrives here already scored 0, so it sinks to `rejected`.
|
|
1436
|
+
*/
|
|
1437
|
+
declare function extractPreferences(lines: MintedRolloutLine[], opts?: ExtractPreferencesOptions): PreferenceExtractionReport;
|
|
1438
|
+
/**
|
|
1439
|
+
* TRL-compatible export. TRL's `DPODataset` is `{ prompt, chosen, rejected }`
|
|
1440
|
+
* where `chosen`/`rejected` are completion TEXT — a trainer fed prompt hashes
|
|
1441
|
+
* would optimize the policy toward emitting hex digests. Neither the prompt
|
|
1442
|
+
* nor the completions live on the triple (it carries only run ids and hashes),
|
|
1443
|
+
* so the caller supplies the same `promptOf`/`completionOf` lookups `toDpoRows`
|
|
1444
|
+
* takes, keyed by run id, and this function resolves real text.
|
|
1445
|
+
*
|
|
1446
|
+
* The chosen and rejected sides of a valid pair share one prompt; resolving
|
|
1447
|
+
* both and comparing catches lookup bugs (a stale map keyed by the wrong id)
|
|
1448
|
+
* before they ship a row whose prompt does not match its rejected completion.
|
|
1449
|
+
*
|
|
1450
|
+
* `context` is REQUIRED: this is the third exporter over the identical
|
|
1451
|
+
* line-less input class, and the round that hardened `toPrmRows` while leaving
|
|
1452
|
+
* `toDpoRows` open is why every one of them now takes the same argument and
|
|
1453
|
+
* runs the same admission rule.
|
|
1056
1454
|
*/
|
|
1057
|
-
declare function
|
|
1455
|
+
declare function toTRLFormat(triples: PreferenceTriple[], lookups: DpoLookups, context: RolloutLineContext): Promise<Array<{
|
|
1456
|
+
prompt: string;
|
|
1457
|
+
chosen: string;
|
|
1458
|
+
rejected: string;
|
|
1459
|
+
}>>;
|
|
1058
1460
|
/**
|
|
1059
1461
|
* Anthropic finetuning JSONL export — `{ system, user, assistant_chosen, assistant_rejected }`
|
|
1060
1462
|
* shape. Same caveat as TRL: prompt + outputs are content the caller has
|
|
1061
1463
|
* to map back from the run record / raw event log.
|
|
1464
|
+
*
|
|
1465
|
+
* `context` is REQUIRED — see `toTRLFormat`. The emitted `margin` is a number
|
|
1466
|
+
* derived from the two runs' rewards, so this row is training signal even
|
|
1467
|
+
* though it ships no completion text.
|
|
1062
1468
|
*/
|
|
1063
|
-
declare function toAnthropicFormat(triples: PreferenceTriple[]): Array<{
|
|
1469
|
+
declare function toAnthropicFormat(triples: PreferenceTriple[], context: RolloutLineContext): Array<{
|
|
1064
1470
|
scenarioId: string;
|
|
1065
1471
|
chosenRunId: string;
|
|
1066
1472
|
rejectedRunId: string;
|
|
@@ -1239,7 +1645,7 @@ declare function prmTrainingPairs(stepRewardsByRun: Map<string, StepReward[]>, o
|
|
|
1239
1645
|
/**
|
|
1240
1646
|
* Trainer-format exporters.
|
|
1241
1647
|
*
|
|
1242
|
-
* agent-eval produces canonical artifacts (`
|
|
1648
|
+
* agent-eval produces canonical artifacts (`MintedRolloutLine[]`, `PreferenceTriple[]`,
|
|
1243
1649
|
* `StepReward[]`, `PrmTrainingTriple[]`). RL training pipelines consume
|
|
1244
1650
|
* different shapes — Hugging Face TRL, Prime Intellect's prime-rl, OpenAI
|
|
1245
1651
|
* fine-tuning, Anthropic finetuning, OpenRLHF, verl. Each has its own
|
|
@@ -1259,7 +1665,7 @@ declare function prmTrainingPairs(stepRewardsByRun: Map<string, StepReward[]>, o
|
|
|
1259
1665
|
* Consumed by Lightman-style PRM trainers and prime-rl's PRM mode.
|
|
1260
1666
|
*
|
|
1261
1667
|
* Why ship this in agent-eval rather than a separate adapter package: the
|
|
1262
|
-
* canonical artifacts (`
|
|
1668
|
+
* canonical artifacts (`MintedRolloutLine[]`, `PreferenceTriple[]`, etc.) are
|
|
1263
1669
|
* agent-eval's contract; without first-party exporters consumers reverse-
|
|
1264
1670
|
* engineer the mapping every release. The exporters codify it.
|
|
1265
1671
|
*
|
|
@@ -1267,6 +1673,10 @@ declare function prmTrainingPairs(stepRewardsByRun: Map<string, StepReward[]>, o
|
|
|
1267
1673
|
* artifact (specifically: prompt + completion text, since the package
|
|
1268
1674
|
* stores only their hashes by design — full text is the consumer's
|
|
1269
1675
|
* trace store / raw event log).
|
|
1676
|
+
*
|
|
1677
|
+
* Every exporter that produces a training row accepts canonical minted rollout
|
|
1678
|
+
* lines. Convert run records once with `mintRolloutRows`; downstream transforms
|
|
1679
|
+
* then share one reward, split, and authenticity contract.
|
|
1270
1680
|
*/
|
|
1271
1681
|
|
|
1272
1682
|
interface DpoLookups {
|
|
@@ -1284,25 +1694,43 @@ interface DpoExportRow {
|
|
|
1284
1694
|
/** Free-form metadata for downstream filtering / sharding. */
|
|
1285
1695
|
meta?: Record<string, unknown>;
|
|
1286
1696
|
}
|
|
1697
|
+
/** The minted lines for the runs a `PreferenceTriple` names on each side. */
|
|
1698
|
+
type DpoLineContext = RolloutLineContext;
|
|
1699
|
+
declare const DPO_CONTEXT_REQUIREMENT: LineContextRequirement;
|
|
1287
1700
|
/**
|
|
1288
1701
|
* Convert preference triples to TRL-compatible DPO rows. The shape
|
|
1289
1702
|
* `{prompt, chosen, rejected}` is the canonical HuggingFace DPODataset
|
|
1290
1703
|
* entry; every major DPO trainer accepts it.
|
|
1704
|
+
*
|
|
1705
|
+
* `context` is REQUIRED, and for the same reason it is required on the sibling
|
|
1706
|
+
* `toPrmRows`: a triple is a line-less artifact. It names two run ids and a
|
|
1707
|
+
* margin, and nothing on it says whether either run was flagged as gamed —
|
|
1708
|
+
* so a two-argument call applied NO gate at all and emitted the row verbatim,
|
|
1709
|
+
* reachable straight through the published bundle builder
|
|
1710
|
+
* (`buildRlDataset(lines, lookups, {formats:['dpo']}, {triples, lookups})`).
|
|
1711
|
+
* Triples whose chosen or rejected side is realness-gated are dropped; a triple
|
|
1712
|
+
* naming a run with no supplied line is refused. See `admitUngatedByInvocation` for
|
|
1713
|
+
* why dropping, not zeroing, is the right disposition for a preference pair.
|
|
1291
1714
|
*/
|
|
1292
|
-
declare function toDpoRows(triples: PreferenceTriple[], lookups: DpoLookups): Promise<DpoExportRow[]>;
|
|
1715
|
+
declare function toDpoRows(triples: PreferenceTriple[], lookups: DpoLookups, context: DpoLineContext): Promise<DpoExportRow[]>;
|
|
1293
1716
|
/** Serialize DPO rows as JSONL. One line per row. */
|
|
1294
1717
|
declare function toDpoJsonl(rows: DpoExportRow[]): string;
|
|
1295
|
-
interface
|
|
1718
|
+
interface TrainingLineSelectionOptions {
|
|
1296
1719
|
/** Include held-out evaluation data in training output. Default false. */
|
|
1297
1720
|
allowHeldOutTrainingData?: boolean;
|
|
1298
1721
|
/** Require quality to be strictly greater than this value. Default 0. */
|
|
1299
1722
|
minimumQualityExclusive?: number;
|
|
1723
|
+
/**
|
|
1724
|
+
* Explicit split selection, replacing the default trainable-split rule.
|
|
1725
|
+
* Use this only when producing a deliberately named non-training slice.
|
|
1726
|
+
*/
|
|
1727
|
+
splitFilter?: RolloutSplit[];
|
|
1300
1728
|
}
|
|
1301
|
-
interface GrpoLookups extends
|
|
1729
|
+
interface GrpoLookups extends Pick<TrainingLineSelectionOptions, 'allowHeldOutTrainingData' | 'splitFilter'> {
|
|
1730
|
+
/** Resolve the prompt text for a rollout, keyed by `line.run_id`. */
|
|
1302
1731
|
promptOf: (runId: string) => string | Promise<string>;
|
|
1732
|
+
/** Resolve the assistant completion text for a rollout. */
|
|
1303
1733
|
completionOf: (runId: string) => string | Promise<string>;
|
|
1304
|
-
/** Optional: derive a custom reward from the run. Defaults to score. */
|
|
1305
|
-
rewardOf?: (run: RunRecord) => number | null;
|
|
1306
1734
|
}
|
|
1307
1735
|
interface GrpoExportRow {
|
|
1308
1736
|
prompt: string;
|
|
@@ -1313,23 +1741,35 @@ interface GrpoExportRow {
|
|
|
1313
1741
|
meta?: Record<string, unknown>;
|
|
1314
1742
|
}
|
|
1315
1743
|
/**
|
|
1316
|
-
* Convert
|
|
1317
|
-
*
|
|
1744
|
+
* Convert rollout lines grouped by `task.instance_id` into GRPO offline rows —
|
|
1745
|
+
* one row per scenario, with one completion per rollout on that scenario.
|
|
1746
|
+
* A scenario with fewer than two rewarded completions emits no row because a
|
|
1747
|
+
* group of one has no relative baseline.
|
|
1318
1748
|
*
|
|
1319
1749
|
* GRPO (Shao et al. 2024 / DeepSeek-R1) trains on relative advantages
|
|
1320
1750
|
* within a group of completions for the same prompt; this is the
|
|
1321
|
-
* canonical input format.
|
|
1322
|
-
*
|
|
1751
|
+
* canonical input format. That relative baseline is exactly why the gate has
|
|
1752
|
+
* to hold here: one gamed sibling exporting at full reward shifts the advantage
|
|
1753
|
+
* of every honest run beside it.
|
|
1754
|
+
*
|
|
1755
|
+
* On the line path a realness-gated line stays in its group at reward 0 rather
|
|
1756
|
+
* than being dropped. 0 is the honest label for a faked success and is usable
|
|
1757
|
+
* signal; removing the line would also move the group's baseline, just in the
|
|
1758
|
+
* other direction. (SFT differs — see `toSftRows`.)
|
|
1323
1759
|
*/
|
|
1324
|
-
declare function toGrpoRows(
|
|
1760
|
+
declare function toGrpoRows(lines: MintedRolloutLine[], lookups: GrpoLookups): Promise<GrpoExportRow[]>;
|
|
1325
1761
|
declare function toGrpoJsonl(rows: GrpoExportRow[]): string;
|
|
1326
|
-
interface SftLookups extends
|
|
1762
|
+
interface SftLookups extends TrainingLineSelectionOptions {
|
|
1763
|
+
/** Resolve the prompt text for a rollout, keyed by `line.run_id`. */
|
|
1327
1764
|
promptOf: (runId: string) => string | Promise<string>;
|
|
1765
|
+
/** Resolve the assistant completion text for a rollout. */
|
|
1328
1766
|
completionOf: (runId: string) => string | Promise<string>;
|
|
1329
1767
|
/** Optional system message. Default omits. */
|
|
1330
|
-
systemOf?: (
|
|
1331
|
-
/**
|
|
1332
|
-
include?: (
|
|
1768
|
+
systemOf?: (line: MintedRolloutLine) => string | null | undefined;
|
|
1769
|
+
/** Extra filter on top of the realness gate (e.g., low score, failed cases). */
|
|
1770
|
+
include?: (line: MintedRolloutLine) => boolean;
|
|
1771
|
+
/** Include held-out lines under the default split rule. Default false. */
|
|
1772
|
+
allowHeldOutTrainingData?: boolean;
|
|
1333
1773
|
}
|
|
1334
1774
|
interface SftExportRow {
|
|
1335
1775
|
messages: Array<{
|
|
@@ -1339,11 +1779,24 @@ interface SftExportRow {
|
|
|
1339
1779
|
meta?: Record<string, unknown>;
|
|
1340
1780
|
}
|
|
1341
1781
|
/**
|
|
1342
|
-
* Convert
|
|
1343
|
-
* conversational SFT rows. By default
|
|
1344
|
-
*
|
|
1782
|
+
* Convert rollout lines into Hugging Face / OpenAI / Anthropic-style
|
|
1783
|
+
* conversational SFT rows. By default every qualifying line becomes one row;
|
|
1784
|
+
* pass `include` to filter further (e.g., keep only `reward >= 0.8` for
|
|
1785
|
+
* rejection-sampling SFT).
|
|
1786
|
+
*
|
|
1787
|
+
* Realness-gated lines are dropped outright, not zeroed. SFT is imitation
|
|
1788
|
+
* learning: unlike GRPO, where a 0 reward teaches "this trajectory was bad",
|
|
1789
|
+
* every row here is a target to copy, so a gamed trajectory must not be in the
|
|
1790
|
+
* file at all. Mirrors the waist filter in `rollout/exporters.toSftRows`.
|
|
1791
|
+
*
|
|
1792
|
+
* The exporter is fail-closed on the split, same rule as
|
|
1793
|
+
* `rollout/exporters.toSftRows` (`isSplitEligible`): `search` ships by
|
|
1794
|
+
* default, held-out lines need `allowHeldOutTrainingData: true`, `dev` and
|
|
1795
|
+
* `canary` never pass the default rule. A non-training bundle that wants an
|
|
1796
|
+
* explicit slice (e.g. a holdout-only eval bundle) names it with
|
|
1797
|
+
* `splitFilter: ['holdout']` — explicit selection replaces the default rule.
|
|
1345
1798
|
*/
|
|
1346
|
-
declare function toSftRows(
|
|
1799
|
+
declare function toSftRows(lines: MintedRolloutLine[], lookups: SftLookups): Promise<SftExportRow[]>;
|
|
1347
1800
|
declare function toSftJsonl(rows: SftExportRow[]): string;
|
|
1348
1801
|
interface PrmLookups {
|
|
1349
1802
|
/** Resolve the prompt text for a run. */
|
|
@@ -1365,11 +1818,48 @@ interface PrmExportRow {
|
|
|
1365
1818
|
marginScore: number;
|
|
1366
1819
|
meta?: Record<string, unknown>;
|
|
1367
1820
|
}
|
|
1821
|
+
interface PrmLineContext extends RolloutLineContext {
|
|
1822
|
+
/**
|
|
1823
|
+
* The `maxSteps` cap the lines were minted with, if any.
|
|
1824
|
+
*
|
|
1825
|
+
* `mintRolloutRows` drops the MIDDLE of an over-long trajectory and leaves no
|
|
1826
|
+
* marker on the line, so a capped trajectory is indistinguishable from a
|
|
1827
|
+
* short one. Declaring the cap lets this exporter refuse any line sitting at
|
|
1828
|
+
* it — a process-reward model trained on a trajectory with a hole in it
|
|
1829
|
+
* learns credit assignment that never happened.
|
|
1830
|
+
*/
|
|
1831
|
+
mintedWithMaxSteps?: number;
|
|
1832
|
+
}
|
|
1368
1833
|
/**
|
|
1369
1834
|
* Convert PRM training triples to JSONL rows. Caller's `stepTextOf`
|
|
1370
1835
|
* callback resolves span text from the consumer's trace store.
|
|
1836
|
+
*
|
|
1837
|
+
* Every referenced run is checked against its minted line before any row is
|
|
1838
|
+
* emitted, and the export FAILS LOUD on a trajectory that was never fully
|
|
1839
|
+
* captured (see `assertPrmTrainableLine`). Triples whose chosen or rejected
|
|
1840
|
+
* side is realness-gated are dropped instead: a capture defect is the caller's
|
|
1841
|
+
* mint configuration and must be fixed, whereas a gamed run is exactly the
|
|
1842
|
+
* condition the gate exists to filter.
|
|
1843
|
+
*
|
|
1844
|
+
* `context` is REQUIRED. A two-argument call used to be accepted and produced
|
|
1845
|
+
* rows with no gate applied at all — a `PrmTrainingTriple` carries a bare
|
|
1846
|
+
* `chosenReward` number and nothing that says which run it came from is honest,
|
|
1847
|
+
* so with no lines this exporter has no way to learn that its chosen step is a
|
|
1848
|
+
* step from a run that faked its success. It now throws: fail closed, because
|
|
1849
|
+
* the alternative is a process-reward model taught to prefer the gaming move at
|
|
1850
|
+
* the exact step the gaming happened.
|
|
1851
|
+
*/
|
|
1852
|
+
declare function toPrmRows(triples: PrmTrainingTriple[], lookups: PrmLookups, context: PrmLineContext): Promise<PrmExportRow[]>;
|
|
1853
|
+
/**
|
|
1854
|
+
* Refuse to build a process-reward row from a trajectory we do not fully have.
|
|
1855
|
+
*
|
|
1856
|
+
* PRM training assigns credit step by step, so a missing or silently shortened
|
|
1857
|
+
* step list is not degraded data — it is data about a trajectory that never
|
|
1858
|
+
* existed. Every condition below throws rather than filters, because each one
|
|
1859
|
+
* means the CALLER's capture or mint configuration is wrong.
|
|
1371
1860
|
*/
|
|
1372
|
-
declare function
|
|
1861
|
+
declare function assertPrmTrainableLine(line: MintedRolloutLine, mintedWithMaxSteps?: number): void;
|
|
1862
|
+
declare const PRM_CONTEXT_REQUIREMENT: LineContextRequirement;
|
|
1373
1863
|
declare function toPrmJsonl(rows: PrmExportRow[]): string;
|
|
1374
1864
|
interface StepRewardJsonlRow {
|
|
1375
1865
|
runId: string;
|
|
@@ -1379,8 +1869,18 @@ interface StepRewardJsonlRow {
|
|
|
1379
1869
|
determinism: 'deterministic' | 'probabilistic';
|
|
1380
1870
|
weight: number;
|
|
1381
1871
|
}
|
|
1382
|
-
declare
|
|
1383
|
-
|
|
1872
|
+
declare const STEP_REWARD_CONTEXT_REQUIREMENT: LineContextRequirement;
|
|
1873
|
+
/**
|
|
1874
|
+
* Step-level reward rows as JSONL.
|
|
1875
|
+
*
|
|
1876
|
+
* `context` is REQUIRED for the same reason it is on `toDpoRows` and
|
|
1877
|
+
* `toPrmRows`: this is a line-less input carrying a reward number. Steps
|
|
1878
|
+
* belonging to a realness-gated run are dropped rather than zeroed — a
|
|
1879
|
+
* per-step reward of 0 across a whole trajectory is a claim that every step was
|
|
1880
|
+
* bad, which is a different (and false) statement from "this run's success was
|
|
1881
|
+
* fabricated, so its step-level credit assignment is meaningless".
|
|
1882
|
+
*/
|
|
1883
|
+
declare function stepRewardsToJsonl(stepRewards: StepReward[], context: RolloutLineContext): string;
|
|
1384
1884
|
|
|
1385
1885
|
/**
|
|
1386
1886
|
* RL dataset packaging + datasheet — the publishable, sellable bundle.
|
|
@@ -1392,13 +1892,19 @@ declare function isTrainingRunEligible(run: RunRecord, quality: number | null |
|
|
|
1392
1892
|
* reward was derived (deterministic verifiable vs probabilistic judge — the
|
|
1393
1893
|
* credibility axis a buyer checks first), the split discipline, the reward
|
|
1394
1894
|
* distribution, the quality gates, the license, and the intended/out-of-scope
|
|
1395
|
-
* uses. This module computes those facts from
|
|
1895
|
+
* uses. This module computes those facts from `MintedRolloutLine[]` and renders a
|
|
1396
1896
|
* "Datasheet for Datasets" (Gebru et al. 2018) card alongside the format files.
|
|
1397
1897
|
*
|
|
1398
1898
|
* It composes the existing `rl/exporters` — it does not reimplement any trainer
|
|
1399
1899
|
* format. The renderers token-identity step (DeepSeek/Kimi/Qwen tokenization
|
|
1400
1900
|
* with per-token loss masks) is a downstream Python stage that consumes the
|
|
1401
1901
|
* `messages`/`completions` this bundle emits.
|
|
1902
|
+
*
|
|
1903
|
+
* Input discipline: the bundle is built from `MintedRolloutLine[]`. The datasheet's
|
|
1904
|
+
* reward distribution ships INSIDE the published artifact, so it has to be
|
|
1905
|
+
* derived from exactly the same gated number as the rows it describes —
|
|
1906
|
+
* otherwise the provenance a buyer checks first is a lie. Taking the same
|
|
1907
|
+
* gated input as the exporters is what guarantees that.
|
|
1402
1908
|
*/
|
|
1403
1909
|
|
|
1404
1910
|
type RewardKind = 'deterministic' | 'probabilistic' | 'mixed';
|
|
@@ -1446,10 +1952,10 @@ interface RewardStats {
|
|
|
1446
1952
|
}
|
|
1447
1953
|
interface RlDatasetStats {
|
|
1448
1954
|
records: number;
|
|
1449
|
-
/**
|
|
1955
|
+
/** Rollouts carrying an explicit task-quality score. */
|
|
1450
1956
|
scoredRecords: number;
|
|
1451
|
-
/**
|
|
1452
|
-
splits: Record<
|
|
1957
|
+
/** Rollout count per split. */
|
|
1958
|
+
splits: Record<RolloutSplit, number>;
|
|
1453
1959
|
reward: RewardStats;
|
|
1454
1960
|
/** Distinct snapshot-pinned models that produced the trajectories. */
|
|
1455
1961
|
models: string[];
|
|
@@ -1461,6 +1967,12 @@ interface RlDatasetStats {
|
|
|
1461
1967
|
output: number;
|
|
1462
1968
|
};
|
|
1463
1969
|
totalCostUsd: number;
|
|
1970
|
+
/**
|
|
1971
|
+
* Rollouts whose USD cost was never captured (`cost.usd === null`). When
|
|
1972
|
+
* non-zero, `totalCostUsd` is a floor, not the bill — a published dataset
|
|
1973
|
+
* must not present an unbilled run as a $0 one.
|
|
1974
|
+
*/
|
|
1975
|
+
rolloutsWithoutCost: number;
|
|
1464
1976
|
}
|
|
1465
1977
|
interface RlDatasetManifest extends RlDatasetConfig {
|
|
1466
1978
|
formats: DatasetFormat[];
|
|
@@ -1473,13 +1985,13 @@ interface RlDatasetBundle {
|
|
|
1473
1985
|
files: Record<string, string>;
|
|
1474
1986
|
}
|
|
1475
1987
|
/**
|
|
1476
|
-
* Package graded
|
|
1988
|
+
* Package graded rollout lines into a publishable RL dataset bundle: the
|
|
1477
1989
|
* trainer-format JSONL files + a manifest + a datasheet. DPO requires
|
|
1478
1990
|
* pre-extracted preference triples (pass `preferences`); GRPO/SFT derive from
|
|
1479
|
-
* the
|
|
1991
|
+
* the lines directly via the supplied lookups. Throws on an empty corpus —
|
|
1480
1992
|
* an empty dataset must never be published.
|
|
1481
1993
|
*/
|
|
1482
|
-
declare function buildRlDataset(
|
|
1994
|
+
declare function buildRlDataset(lines: MintedRolloutLine[], lookups: GrpoLookups & SftLookups, config: RlDatasetConfig, preferences?: {
|
|
1483
1995
|
triples: PreferenceTriple[];
|
|
1484
1996
|
lookups: DpoLookups;
|
|
1485
1997
|
}): Promise<RlDatasetBundle>;
|
|
@@ -1539,6 +2051,12 @@ interface HarvestOptions {
|
|
|
1539
2051
|
* missing either are excluded (a graded score with no trajectory can't train).
|
|
1540
2052
|
* Optionally filters by score / split. Throws (via buildRlDataset) if nothing
|
|
1541
2053
|
* survives — an empty dataset must never be published.
|
|
2054
|
+
*
|
|
2055
|
+
* `minScore` is applied to the GATED reward (`trainingScore`), so a gamed run
|
|
2056
|
+
* cannot buy its way into the published bundle with its claimed score —
|
|
2057
|
+
* `minScore` is exactly the door a reward-hacked run would otherwise clear for
|
|
2058
|
+
* SFT. Unscored records are dropped before packaging: a missing label is not a
|
|
2059
|
+
* zero, and it is not publishable either.
|
|
1542
2060
|
*/
|
|
1543
2061
|
declare function buildDatasetFromCorpus(corpusPath: string, config: RlDatasetConfig, opts?: HarvestOptions): Promise<RlDatasetBundle>;
|
|
1544
2062
|
|
|
@@ -1611,20 +2129,12 @@ interface OffPolicyTrajectory {
|
|
|
1611
2129
|
* values must come from a model cross-fitted or trained outside this row.
|
|
1612
2130
|
*/
|
|
1613
2131
|
vHatTarget?: number | null;
|
|
1614
|
-
/**
|
|
1615
|
-
* @deprecated Use `qHatChosen` and `vHatTarget` together. When the new pair
|
|
1616
|
-
* is absent, this scalar is used as both terms to preserve existing results.
|
|
1617
|
-
* When the new pair is present, this field is ignored.
|
|
1618
|
-
*/
|
|
1619
|
-
qHat?: number | null;
|
|
1620
2132
|
}
|
|
1621
2133
|
interface OffPolicyContributionCounts {
|
|
1622
2134
|
/** Contributions using the contextual-bandit doubly-robust formula. */
|
|
1623
2135
|
dr: number;
|
|
1624
2136
|
/** Contributions using exact IPS because no reward-model estimate was supplied. */
|
|
1625
2137
|
ipsFallback: number;
|
|
1626
|
-
/** Contributions using the deprecated single-scalar formula. */
|
|
1627
|
-
legacyScalar: number;
|
|
1628
2138
|
}
|
|
1629
2139
|
interface OffPolicyEstimate {
|
|
1630
2140
|
/** Estimated value of the target policy. */
|
|
@@ -1683,9 +2193,8 @@ declare function selfNormalizedImportanceWeighting(trajectories: OffPolicyTrajec
|
|
|
1683
2193
|
* default in production OPE pipelines.
|
|
1684
2194
|
*
|
|
1685
2195
|
* `qHatChosen` and `vHatTarget` must be supplied together. Rows with neither
|
|
1686
|
-
* use the exact IPS contribution.
|
|
1687
|
-
*
|
|
1688
|
-
* `contributionCounts` makes the mix explicit in the result.
|
|
2196
|
+
* use the exact IPS contribution. `contributionCounts` makes the mix explicit
|
|
2197
|
+
* in the result.
|
|
1689
2198
|
* Callers must cross-fit the Q-function or train it on independent rows;
|
|
1690
2199
|
* fitting and evaluating Q on the same outcomes leaks the answer.
|
|
1691
2200
|
*/
|
|
@@ -1899,6 +2408,13 @@ interface GateEvidence {
|
|
|
1899
2408
|
/** Median per-task USD cost across the baseline runs, for
|
|
1900
2409
|
* symmetric reporting. */
|
|
1901
2410
|
medianBaselineCost: number | null;
|
|
2411
|
+
/**
|
|
2412
|
+
* Runs (candidate + baseline) dropped before pairing because the
|
|
2413
|
+
* authenticity gate flagged them as gamed. Surfaced rather than silent: a
|
|
2414
|
+
* promotion decision computed over a shrunken pool has to say by how much,
|
|
2415
|
+
* and a nonzero count here is itself the finding.
|
|
2416
|
+
*/
|
|
2417
|
+
realnessGatedRuns: number;
|
|
1902
2418
|
}
|
|
1903
2419
|
interface GateDecision$1 {
|
|
1904
2420
|
/** Final promote/no-promote verdict. */
|
|
@@ -2268,12 +2784,34 @@ interface VerifiableReward {
|
|
|
2268
2784
|
*/
|
|
2269
2785
|
components: Record<string, number>;
|
|
2270
2786
|
/**
|
|
2271
|
-
*
|
|
2272
|
-
*
|
|
2273
|
-
*
|
|
2274
|
-
* the
|
|
2787
|
+
* The run carries `outcome.realness.gated` — the authenticity gate flagged
|
|
2788
|
+
* its success signal as faked.
|
|
2789
|
+
*
|
|
2790
|
+
* With the gate applied (the default) `value` and every `components` entry
|
|
2791
|
+
* are 0 on such a run; with `applyRealnessGate: false` the observed numbers
|
|
2792
|
+
* come back untouched and this flag is the only marker that they are not to
|
|
2793
|
+
* be trusted. Either way it distinguishes "measured a genuine failure" from
|
|
2794
|
+
* "claimed a success we refuse to believe", which a bare 0 cannot.
|
|
2275
2795
|
*/
|
|
2276
|
-
|
|
2796
|
+
realnessGated?: boolean;
|
|
2797
|
+
/**
|
|
2798
|
+
* Whether an authenticity screen COULD run on this reward at all — the same
|
|
2799
|
+
* distinction `RolloutOutcome.realness_screened` draws, for the same reason.
|
|
2800
|
+
*
|
|
2801
|
+
* `false` on every reward from `extractVerifiableReward`, because a
|
|
2802
|
+
* `VerificationReport` carries layer scores and nothing else: there is no
|
|
2803
|
+
* `outcome.realness` to consult, so no gate has run, and `realnessGated`
|
|
2804
|
+
* being absent there means "unknown", NOT "clean". Absent on the
|
|
2805
|
+
* `RunRecord` path when the record itself carries no realness verdict.
|
|
2806
|
+
*
|
|
2807
|
+
* This matters most exactly where it is easiest to miss: a report whose
|
|
2808
|
+
* deterministic layers all passed yields `determinism: 'deterministic'`,
|
|
2809
|
+
* `confidence: 1` — the highest-credibility reward this module can emit —
|
|
2810
|
+
* and a stubbed integration reporting green is precisely what a gamed run
|
|
2811
|
+
* looks like. Consumers driving training off this shape must screen the run
|
|
2812
|
+
* themselves; the flag is what tells them nobody has.
|
|
2813
|
+
*/
|
|
2814
|
+
realnessScreened?: boolean;
|
|
2277
2815
|
}
|
|
2278
2816
|
interface VerifiableRewardExtractionOptions {
|
|
2279
2817
|
/**
|
|
@@ -2300,6 +2838,18 @@ interface VerifiableRewardExtractionOptions {
|
|
|
2300
2838
|
* doesn't report one. Default `0.7`.
|
|
2301
2839
|
*/
|
|
2302
2840
|
judgeConfidenceFloor?: number;
|
|
2841
|
+
/**
|
|
2842
|
+
* Whether the anti-Goodhart realness gate applies. Default `true`, and the
|
|
2843
|
+
* default is the one every training path must keep.
|
|
2844
|
+
*
|
|
2845
|
+
* Set `false` ONLY for detection and analysis. `rl/reward-hacking.ts` does,
|
|
2846
|
+
* for the same reason it reads `observedScore` for its proxy: it measures the
|
|
2847
|
+
* DIVERGENCE between the judge signal and the deterministic one, and a
|
|
2848
|
+
* deterministic reward that another gate already forced to 0 manufactures
|
|
2849
|
+
* exactly that divergence on exactly the gamed population. The detector would
|
|
2850
|
+
* then be re-reporting a verdict it was supposed to reach independently.
|
|
2851
|
+
*/
|
|
2852
|
+
applyRealnessGate?: boolean;
|
|
2303
2853
|
}
|
|
2304
2854
|
/**
|
|
2305
2855
|
* Extract a `VerifiableReward` from a `VerificationReport`.
|
|
@@ -2308,6 +2858,11 @@ interface VerifiableRewardExtractionOptions {
|
|
|
2308
2858
|
* schema → sandbox), fall back to the judge layer if `fallbackToJudge` is
|
|
2309
2859
|
* true, return `null` if no signal qualifies. When multiple deterministic
|
|
2310
2860
|
* layers contribute, return a `'composite'` source with a weighted blend.
|
|
2861
|
+
*
|
|
2862
|
+
* NO realness gate is applied and none can be: a `VerificationReport` carries
|
|
2863
|
+
* layer scores and nothing about whether the run faked them — `realness` lives
|
|
2864
|
+
* on the `RunRecord`. Use `extractVerifiableRewardsFromRecords` for anything
|
|
2865
|
+
* that becomes training data; this signature is for scoring a report in hand.
|
|
2311
2866
|
*/
|
|
2312
2867
|
declare function extractVerifiableReward(report: VerificationReport, opts?: VerifiableRewardExtractionOptions): VerifiableReward | null;
|
|
2313
2868
|
/**
|
|
@@ -2321,12 +2876,30 @@ declare function extractVerifiableReward(report: VerificationReport, opts?: Veri
|
|
|
2321
2876
|
* verifiable reward becomes a training datum, every record that doesn't
|
|
2322
2877
|
* gets filtered out (or kept with `'probabilistic'` determinism for
|
|
2323
2878
|
* separate downstream handling).
|
|
2879
|
+
*
|
|
2880
|
+
* The realness gate applies to EVERY channel here, and to the deterministic one
|
|
2881
|
+
* MOST. It is tempting to reason that a decidable signal cannot be gamed, so
|
|
2882
|
+
* the gate is redundant on it — that reasoning is backwards. `realness.gated`
|
|
2883
|
+
* means the run's success signal was FAKED, and a test suite reporting green on
|
|
2884
|
+
* a stubbed integration is precisely what that looks like: the deterministic
|
|
2885
|
+
* layer is the thing that got faked. Exporting it ungated hands a trainer the
|
|
2886
|
+
* highest-credibility reward the module can emit (`determinism: 'deterministic'`,
|
|
2887
|
+
* `confidence: 1`) for the one population the gate exists to catch. Pass
|
|
2888
|
+
* `applyRealnessGate: false` only to look at the ungated numbers for detection.
|
|
2324
2889
|
*/
|
|
2325
2890
|
declare function extractVerifiableRewardsFromRecords(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
|
|
2326
2891
|
runId: string;
|
|
2327
2892
|
reward: VerifiableReward | null;
|
|
2328
2893
|
}>;
|
|
2329
|
-
/**
|
|
2894
|
+
/**
|
|
2895
|
+
* Filter `RunRecord[]` to those with deterministic verifiable rewards.
|
|
2896
|
+
*
|
|
2897
|
+
* A realness-gated run is KEPT, at reward 0 with `realnessGated: true` — the
|
|
2898
|
+
* same rule GRPO uses on a gated line. 0 is the honest label for a faked
|
|
2899
|
+
* success and is usable signal, whereas dropping the run would move a group
|
|
2900
|
+
* baseline without saying so. (SFT differs: there every row is a target to
|
|
2901
|
+
* imitate, so a gated row is removed outright.)
|
|
2902
|
+
*/
|
|
2330
2903
|
declare function filterDeterministicallyRewarded(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
|
|
2331
2904
|
run: RunRecord;
|
|
2332
2905
|
reward: VerifiableReward;
|
|
@@ -2574,9 +3147,8 @@ type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
|
|
|
2574
3147
|
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
2575
3148
|
* )
|
|
2576
3149
|
*
|
|
2577
|
-
*
|
|
2578
|
-
*
|
|
2579
|
-
* that need free-form text use `callLlm` and parse output themselves.
|
|
3150
|
+
* `createChatClient` wraps this implementation for provider-neutral package
|
|
3151
|
+
* entry points. Direct callers can use `callLlm` or `callLlmJson`.
|
|
2580
3152
|
*/
|
|
2581
3153
|
|
|
2582
3154
|
type LlmThinkingMode = 'enabled' | 'disabled';
|
|
@@ -2654,8 +3226,8 @@ interface LlmClientOptions {
|
|
|
2654
3226
|
* total attempts × `timeoutMs`.
|
|
2655
3227
|
*/
|
|
2656
3228
|
deadlineMs?: number;
|
|
2657
|
-
/** Total provider attempts.
|
|
2658
|
-
|
|
3229
|
+
/** Total provider attempts. Default 3. */
|
|
3230
|
+
maximumAttempts?: number;
|
|
2659
3231
|
/** Token rates used when the provider omits cost or package pricing does not cover the model. */
|
|
2660
3232
|
customTokenPricing?: CustomTokenPricing;
|
|
2661
3233
|
/**
|
|
@@ -3474,7 +4046,7 @@ interface InterimReleaseConfidence {
|
|
|
3474
4046
|
*
|
|
3475
4047
|
* Wires:
|
|
3476
4048
|
* 1. `runEvalCampaign` for the matrix run (capture, integrity, hooks)
|
|
3477
|
-
* 2. `
|
|
4049
|
+
* 2. `extractVerifiableRewardsFromRecords` over the runs, separating deterministic
|
|
3478
4050
|
* from probabilistic reward sources for the trainer
|
|
3479
4051
|
* 3. `extractPreferences` to produce DPO/PPO/KTO triples
|
|
3480
4052
|
* 4. `evaluateInterimReleaseConfidence` over paired deltas (anytime-valid)
|
|
@@ -3484,8 +4056,9 @@ interface InterimReleaseConfidence {
|
|
|
3484
4056
|
*
|
|
3485
4057
|
* The output `RLCampaignResult` is a single, audit-ready artifact: every
|
|
3486
4058
|
* stage's output is in there. The consumer's downstream fits in a single
|
|
3487
|
-
* line: pass `result.preferences` to
|
|
3488
|
-
* to GRPO, `result.runs` plus
|
|
4059
|
+
* line: pass `result.preferences.pairs` to a DPO trainer,
|
|
4060
|
+
* `result.trainerRows.grpo` to GRPO, or `result.campaign.runs` plus
|
|
4061
|
+
* `result.rewardSignals` to a custom RL loop.
|
|
3489
4062
|
*/
|
|
3490
4063
|
|
|
3491
4064
|
interface RunRLCampaignOptions<V> extends EvalCampaignOptions<V> {
|
|
@@ -3512,7 +4085,7 @@ interface RunRLCampaignOptions<V> extends EvalCampaignOptions<V> {
|
|
|
3512
4085
|
sft?: SftLookups;
|
|
3513
4086
|
};
|
|
3514
4087
|
}
|
|
3515
|
-
interface RLCampaignResult
|
|
4088
|
+
interface RLCampaignResult {
|
|
3516
4089
|
campaign: EvalCampaignResult;
|
|
3517
4090
|
/** Per-run verifiable reward (deterministic when available, probabilistic fallback otherwise). */
|
|
3518
4091
|
rewardSignals: Array<{
|
|
@@ -3541,9 +4114,8 @@ interface RLCampaignResult<V> {
|
|
|
3541
4114
|
* Convenience type-tag — consumers can branch on `result.kind`.
|
|
3542
4115
|
*/
|
|
3543
4116
|
kind: 'agent-eval-rl-campaign';
|
|
3544
|
-
unusedVariant?: V;
|
|
3545
4117
|
}
|
|
3546
|
-
declare function runRLCampaign<V>(opts: RunRLCampaignOptions<V>): Promise<RLCampaignResult
|
|
4118
|
+
declare function runRLCampaign<V>(opts: RunRLCampaignOptions<V>): Promise<RLCampaignResult>;
|
|
3547
4119
|
|
|
3548
4120
|
/**
|
|
3549
4121
|
* Pass A substrate types — `runCampaign` is the one primitive every
|
|
@@ -3582,7 +4154,7 @@ interface CampaignScenarioIdentity extends Pick<Scenario, 'id' | 'kind'> {
|
|
|
3582
4154
|
/** The canonical judge verdict shape — one declaration, shared by campaign
|
|
3583
4155
|
* judges and the multishot judge runner (which re-exports this type).
|
|
3584
4156
|
*
|
|
3585
|
-
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
|
|
4157
|
+
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
|
|
3586
4158
|
* multishot runner emits 0-10. Cross-scale comparison must go through
|
|
3587
4159
|
* `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
|
|
3588
4160
|
* promotion-policy) — never renormalize a producer's values in place, as
|
|
@@ -3736,8 +4308,6 @@ interface CampaignAggregates {
|
|
|
3736
4308
|
byScenario: Record<string, ScenarioAggregate>;
|
|
3737
4309
|
/** Canonical campaign accounting, including worker and judge calls. */
|
|
3738
4310
|
cost: CostLedgerSummary;
|
|
3739
|
-
/** Compatibility alias of `cost.totalCostUsd`. */
|
|
3740
|
-
totalCostUsd: number;
|
|
3741
4311
|
/** Cells whose dispatch completed, including cells whose later judge failed. */
|
|
3742
4312
|
cellsExecuted: number;
|
|
3743
4313
|
cellsSkipped: number;
|
|
@@ -3748,7 +4318,7 @@ interface CampaignAggregates {
|
|
|
3748
4318
|
cellsDispatchFailed?: number;
|
|
3749
4319
|
/** Present on results that record failure stages. */
|
|
3750
4320
|
cellsJudgeFailed?: number;
|
|
3751
|
-
/**
|
|
4321
|
+
/** Failures whose stage could not be classified. */
|
|
3752
4322
|
cellsUnclassifiedFailed?: number;
|
|
3753
4323
|
}
|
|
3754
4324
|
interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scenario> {
|
|
@@ -4113,4 +4683,4 @@ interface BuildPairwiseFromCampaignInput {
|
|
|
4113
4683
|
}
|
|
4114
4684
|
declare function buildPairwiseFromCampaign(input: BuildPairwiseFromCampaignInput): PairwiseOutcome[];
|
|
4115
4685
|
|
|
4116
|
-
export { ABSENT_CATEGORY, type AdaptationCurve, type AdaptationPoint, type AdaptationRunner, type AdapterContext, type AdversarialMutation, type BehaviorFeatures, type BradleyTerryFit, type BradleyTerryRating, type BuildPairwiseFromCampaignInput, type CellObservation, type CompareCurvesResult, type ComputeBestOfNOptions, type ComputeBestOfNResult, type ComputeCurve, type ComputeCurveBudget, type ComputeCurvePoint, type ContaminationProbeInput, type ContaminationProbeOptions, type ContaminationProbeReport, type CorpusAppendResult, type CorpusRecord, type CurriculumAllocation, DEFAULT_MIN_N_PER_FEATURE, DEFAULT_QUANTILE_BUCKETS, type DatasetFormat, type DeploymentOutcome, type DetectRewardHackingInput, type DpoExportRow, type DpoLookups, type EasyModeOptions, type EasyModeReport, type EloOptions, type ExtractPreferencesOptions, type ExtractStepRewardsOptions, type FeatureDivergence, type FeatureShift, type FidelityReport, type FidelityVerdict, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type GrpoExportRow, type GrpoLookups, type HarvestOptions, InMemoryOutcomeStore, type OffPolicyContributionCounts, type OffPolicyEstimate, type OffPolicyOptions, type OffPolicyTrajectory, type OutcomeStore, type PairwiseOutcome, type ParetoPointInput, PredictiveValidityResearcher, type PredictiveValidityResearcherOptions, type PreferenceExtractionReport, type PreferenceStrategy, type PreferenceTriple, type PrmExportRow, type PrmLookups, type PrmTrainingTriple, REPRESENTATIVE_MIN_FIDELITY, type RLCampaignResult, type RewardHackingFinding, type RewardHackingReport, type RewardHackingSignal, type RewardKind, type RewardStats, type RlDatasetBundle, type RlDatasetConfig, type RlDatasetManifest, type RlDatasetStats, type RunAdaptationCurveOptions, type RunComputeCurveOptions, type RunRLCampaignOptions, type RunwiseStepSummary, type ScenarioPerturbation, type ScenarioPerturbationKind, type SelfConsistencyOptions, type SelfConsistencyResult, type SftExportRow, type SftLookups, type SimFidelityOptions, type StepReward, type StepRewardJsonlRow, type StepScorer, type ThompsonCurriculumOptions, type
|
|
4686
|
+
export { ABSENT_CATEGORY, type AdaptationCurve, type AdaptationPoint, type AdaptationRunner, type AdapterContext, type AdversarialMutation, type BehaviorFeatures, type BradleyTerryFit, type BradleyTerryRating, type BuildPairwiseFromCampaignInput, type CellObservation, type CompareCurvesResult, type ComputeBestOfNOptions, type ComputeBestOfNResult, type ComputeCurve, type ComputeCurveBudget, type ComputeCurvePoint, type ContaminationProbeInput, type ContaminationProbeOptions, type ContaminationProbeReport, type CorpusAppendResult, type CorpusRecord, type CurriculumAllocation, DEFAULT_MIN_N_PER_FEATURE, DEFAULT_QUANTILE_BUCKETS, DPO_CONTEXT_REQUIREMENT, type DatasetFormat, type DeploymentOutcome, type DetectRewardHackingInput, type DpoExportRow, type DpoLineContext, type DpoLookups, type EasyModeOptions, type EasyModeReport, type EloOptions, type ExtractPreferencesOptions, type ExtractStepRewardsOptions, type FeatureDivergence, type FeatureShift, type FidelityReport, type FidelityVerdict, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type GrpoExportRow, type GrpoLookups, type HarvestOptions, InMemoryOutcomeStore, type OffPolicyContributionCounts, type OffPolicyEstimate, type OffPolicyOptions, type OffPolicyTrajectory, type OutcomeStore, PRM_CONTEXT_REQUIREMENT, type PairwiseOutcome, type ParetoPointInput, PredictiveValidityResearcher, type PredictiveValidityResearcherOptions, type PreferenceExtractionReport, type PreferenceStrategy, type PreferenceTriple, type PrmExportRow, type PrmLineContext, type PrmLookups, type PrmTrainingTriple, REPRESENTATIVE_MIN_FIDELITY, type RLCampaignResult, type RewardHackingFinding, type RewardHackingReport, type RewardHackingSignal, type RewardKind, type RewardStats, type RlDatasetBundle, type RlDatasetConfig, type RlDatasetManifest, type RlDatasetStats, type RolloutLineContext, type RunAdaptationCurveOptions, type RunComputeCurveOptions, type RunRLCampaignOptions, type RunwiseStepSummary, STEP_REWARD_CONTEXT_REQUIREMENT, type ScenarioPerturbation, type ScenarioPerturbationKind, type SelfConsistencyOptions, type SelfConsistencyResult, type SftExportRow, type SftLookups, type SimFidelityOptions, type StepReward, type StepRewardJsonlRow, type StepScorer, type ThompsonCurriculumOptions, type TrainingLineSelectionOptions, type VarianceCurriculumOptions, type VerifiableReward, type VerifiableRewardExtractionOptions, type VerifiableRewardSource, appendToCorpus, applyEloUpdate, assertPrmTrainableLine, bestOfN, bucketLabel, buildDatasetFromCorpus, buildPairwiseFromCampaign, buildRlDataset, campaignToRunRecords, compareAdaptationCurves, datasheetToMarkdown, defaultBehaviorFeatures, detectRewardHacking, doublyRobust, easyModeCheck, extractPreferences, extractStepRewards, extractVerifiableReward, extractVerifiableRewardsFromRecords, filterDeterministicallyRewarded, firstPassK, fitBradleyTerry, injectIrrelevantClause, inverseProbabilityWeighting, jsDivergence, observationsFromRunRecords, offPolicyEstimateAll, paretoFrontier, prmTrainingPairs, quantileEdges, readCorpus, renameVariables, runAdaptationCurve, runComputeCurve, runContaminationProbe, runEvalCampaign, runRLCampaign, runwiseStepRewardSummary, selfConsistency, selfNormalizedImportanceWeighting, shuffleOrder, simFidelityReport, stepRewardsToJsonl, thompsonCurriculum, toAnthropicFormat, toDpoJsonl, toDpoRows, toGrpoJsonl, toGrpoRows, toPrmJsonl, toPrmRows, toSftJsonl, toSftRows, toTRLFormat, validateDatasetFormats, varianceBasedCurriculum, verificationReportToRunRecord };
|