@tangle-network/agent-eval 0.120.1 → 0.120.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/package.json +1 -1
- package/dist/analyst/index.d.ts +0 -3111
- package/dist/analyst/index.js +0 -403
- package/dist/analyst/index.js.map +0 -1
- package/dist/authenticity/index.d.ts +0 -161
- package/dist/authenticity/index.js +0 -215
- package/dist/authenticity/index.js.map +0 -1
- package/dist/belief-state/index.d.ts +0 -1301
- package/dist/belief-state/index.js +0 -2152
- package/dist/belief-state/index.js.map +0 -1
- package/dist/benchmarks/index.d.ts +0 -974
- package/dist/benchmarks/index.js +0 -60
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/builder-eval/index.d.ts +0 -695
- package/dist/builder-eval/index.js +0 -366
- package/dist/builder-eval/index.js.map +0 -1
- package/dist/campaign/index.d.ts +0 -7454
- package/dist/campaign/index.js +0 -272
- package/dist/campaign/index.js.map +0 -1
- package/dist/chunk-32BZXMSO.js +0 -3878
- package/dist/chunk-32BZXMSO.js.map +0 -1
- package/dist/chunk-3A246TSA.js +0 -998
- package/dist/chunk-3A246TSA.js.map +0 -1
- package/dist/chunk-3RF76KTD.js +0 -84
- package/dist/chunk-3RF76KTD.js.map +0 -1
- package/dist/chunk-3YYRZDON.js +0 -45
- package/dist/chunk-3YYRZDON.js.map +0 -1
- package/dist/chunk-4I2E3LLO.js +0 -1030
- package/dist/chunk-4I2E3LLO.js.map +0 -1
- package/dist/chunk-ARU2PZFM.js +0 -312
- package/dist/chunk-ARU2PZFM.js.map +0 -1
- package/dist/chunk-BOD4O7OF.js +0 -40
- package/dist/chunk-BOD4O7OF.js.map +0 -1
- package/dist/chunk-DPZAEKA6.js +0 -880
- package/dist/chunk-DPZAEKA6.js.map +0 -1
- package/dist/chunk-DTJ6QUQB.js +0 -131
- package/dist/chunk-DTJ6QUQB.js.map +0 -1
- package/dist/chunk-GGE4NNQT.js +0 -65
- package/dist/chunk-GGE4NNQT.js.map +0 -1
- package/dist/chunk-H5UD2323.js +0 -286
- package/dist/chunk-H5UD2323.js.map +0 -1
- package/dist/chunk-HHWE3POT.js +0 -94
- package/dist/chunk-HHWE3POT.js.map +0 -1
- package/dist/chunk-HKUCJ437.js +0 -787
- package/dist/chunk-HKUCJ437.js.map +0 -1
- package/dist/chunk-JHCHEVET.js +0 -274
- package/dist/chunk-JHCHEVET.js.map +0 -1
- package/dist/chunk-JHOJHHU7.js +0 -867
- package/dist/chunk-JHOJHHU7.js.map +0 -1
- package/dist/chunk-JM2SKQMS.js +0 -750
- package/dist/chunk-JM2SKQMS.js.map +0 -1
- package/dist/chunk-JN2FCO5W.js +0 -7958
- package/dist/chunk-JN2FCO5W.js.map +0 -1
- package/dist/chunk-K4DBDHLK.js +0 -158
- package/dist/chunk-K4DBDHLK.js.map +0 -1
- package/dist/chunk-K6N6XJJX.js +0 -306
- package/dist/chunk-K6N6XJJX.js.map +0 -1
- package/dist/chunk-MA6HLL3S.js +0 -65
- package/dist/chunk-MA6HLL3S.js.map +0 -1
- package/dist/chunk-MAZ26DC7.js +0 -99
- package/dist/chunk-MAZ26DC7.js.map +0 -1
- package/dist/chunk-MOXWMGPC.js +0 -577
- package/dist/chunk-MOXWMGPC.js.map +0 -1
- package/dist/chunk-NJC7U437.js +0 -626
- package/dist/chunk-NJC7U437.js.map +0 -1
- package/dist/chunk-NPCTHQIO.js +0 -91
- package/dist/chunk-NPCTHQIO.js.map +0 -1
- package/dist/chunk-ONWEPEDO.js +0 -57
- package/dist/chunk-ONWEPEDO.js.map +0 -1
- package/dist/chunk-OYZAPX5G.js +0 -1526
- package/dist/chunk-OYZAPX5G.js.map +0 -1
- package/dist/chunk-PC4UYEBM.js +0 -166
- package/dist/chunk-PC4UYEBM.js.map +0 -1
- package/dist/chunk-PICTDURQ.js +0 -766
- package/dist/chunk-PICTDURQ.js.map +0 -1
- package/dist/chunk-PJQFMIOX.js +0 -1182
- package/dist/chunk-PJQFMIOX.js.map +0 -1
- package/dist/chunk-PXD6ZFNY.js +0 -1107
- package/dist/chunk-PXD6ZFNY.js.map +0 -1
- package/dist/chunk-PXE2VKMX.js +0 -140
- package/dist/chunk-PXE2VKMX.js.map +0 -1
- package/dist/chunk-PZ5AY32C.js +0 -10
- package/dist/chunk-PZ5AY32C.js.map +0 -1
- package/dist/chunk-QBRSJK47.js +0 -622
- package/dist/chunk-QBRSJK47.js.map +0 -1
- package/dist/chunk-QWMPPZ3X.js +0 -550
- package/dist/chunk-QWMPPZ3X.js.map +0 -1
- package/dist/chunk-S3UZOQ5Y.js +0 -328
- package/dist/chunk-S3UZOQ5Y.js.map +0 -1
- package/dist/chunk-S5TT5R3L.js +0 -2668
- package/dist/chunk-S5TT5R3L.js.map +0 -1
- package/dist/chunk-T4SQEITX.js +0 -95
- package/dist/chunk-T4SQEITX.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-U5CHZ5M3.js +0 -357
- package/dist/chunk-U5CHZ5M3.js.map +0 -1
- package/dist/chunk-ULOKLHIQ.js +0 -1937
- package/dist/chunk-ULOKLHIQ.js.map +0 -1
- package/dist/chunk-VI2UW6B6.js +0 -162
- package/dist/chunk-VI2UW6B6.js.map +0 -1
- package/dist/chunk-VQMK5FMP.js +0 -247
- package/dist/chunk-VQMK5FMP.js.map +0 -1
- package/dist/chunk-VSMTAMNK.js +0 -53
- package/dist/chunk-VSMTAMNK.js.map +0 -1
- package/dist/chunk-VZSRQ272.js +0 -149
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WW2A73HW.js +0 -159
- package/dist/chunk-WW2A73HW.js.map +0 -1
- package/dist/chunk-X4UCIOTZ.js +0 -136
- package/dist/chunk-X4UCIOTZ.js.map +0 -1
- package/dist/chunk-XDIRG3TO.js +0 -1266
- package/dist/chunk-XDIRG3TO.js.map +0 -1
- package/dist/chunk-XJYR7XFV.js +0 -317
- package/dist/chunk-XJYR7XFV.js.map +0 -1
- package/dist/chunk-ZET2UAYW.js +0 -89
- package/dist/chunk-ZET2UAYW.js.map +0 -1
- package/dist/chunk-ZZUXHH3R.js +0 -99
- package/dist/chunk-ZZUXHH3R.js.map +0 -1
- package/dist/cli.d.ts +0 -1
- package/dist/cli.js +0 -112
- package/dist/cli.js.map +0 -1
- package/dist/contract/index.d.ts +0 -4972
- package/dist/contract/index.js +0 -1654
- package/dist/contract/index.js.map +0 -1
- package/dist/control.d.ts +0 -1013
- package/dist/control.js +0 -34
- package/dist/control.js.map +0 -1
- package/dist/fuzz.d.ts +0 -759
- package/dist/fuzz.js +0 -714
- package/dist/fuzz.js.map +0 -1
- package/dist/hosted/index.d.ts +0 -730
- package/dist/hosted/index.js +0 -14
- package/dist/hosted/index.js.map +0 -1
- package/dist/index.d.ts +0 -16780
- package/dist/index.js +0 -12168
- package/dist/index.js.map +0 -1
- package/dist/matrix/index.d.ts +0 -155
- package/dist/matrix/index.js +0 -8
- package/dist/matrix/index.js.map +0 -1
- package/dist/meta-eval/index.d.ts +0 -1030
- package/dist/meta-eval/index.js +0 -417
- package/dist/meta-eval/index.js.map +0 -1
- package/dist/multishot/index.d.ts +0 -579
- package/dist/multishot/index.js +0 -589
- package/dist/multishot/index.js.map +0 -1
- package/dist/openapi.json +0 -992
- package/dist/pipelines/index.d.ts +0 -567
- package/dist/pipelines/index.js +0 -515
- package/dist/pipelines/index.js.map +0 -1
- package/dist/reporting.d.ts +0 -1277
- package/dist/reporting.js +0 -48
- package/dist/reporting.js.map +0 -1
- package/dist/rl.d.ts +0 -4092
- package/dist/rl.js +0 -1724
- package/dist/rl.js.map +0 -1
- package/dist/run-campaign-HNFPJET4.js +0 -14
- package/dist/run-campaign-HNFPJET4.js.map +0 -1
- package/dist/storyboard/index.d.ts +0 -279
- package/dist/storyboard/index.js +0 -767
- package/dist/storyboard/index.js.map +0 -1
- package/dist/trace-attributes.d.ts +0 -52
- package/dist/trace-attributes.js +0 -62
- package/dist/trace-attributes.js.map +0 -1
- package/dist/traces.d.ts +0 -2343
- package/dist/traces.js +0 -249
- package/dist/traces.js.map +0 -1
- package/dist/wire/index.d.ts +0 -1252
- package/dist/wire/index.js +0 -81
- package/dist/wire/index.js.map +0 -1
package/dist/rl.js
DELETED
|
@@ -1,1724 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
doublyRobust,
|
|
3
|
-
inverseProbabilityWeighting,
|
|
4
|
-
offPolicyEstimateAll,
|
|
5
|
-
selfNormalizedImportanceWeighting
|
|
6
|
-
} from "./chunk-DTJ6QUQB.js";
|
|
7
|
-
import {
|
|
8
|
-
FileSystemOutcomeStore,
|
|
9
|
-
InMemoryOutcomeStore
|
|
10
|
-
} from "./chunk-3RF76KTD.js";
|
|
11
|
-
import {
|
|
12
|
-
runEvalCampaign
|
|
13
|
-
} from "./chunk-U5CHZ5M3.js";
|
|
14
|
-
import {
|
|
15
|
-
detectRewardHacking,
|
|
16
|
-
extractVerifiableReward,
|
|
17
|
-
extractVerifiableRewardsFromRecords,
|
|
18
|
-
filterDeterministicallyRewarded
|
|
19
|
-
} from "./chunk-ARU2PZFM.js";
|
|
20
|
-
import "./chunk-NJC7U437.js";
|
|
21
|
-
import {
|
|
22
|
-
rubricPredictiveValidity
|
|
23
|
-
} from "./chunk-X4UCIOTZ.js";
|
|
24
|
-
import {
|
|
25
|
-
evaluateInterimReleaseConfidence
|
|
26
|
-
} from "./chunk-MAZ26DC7.js";
|
|
27
|
-
import "./chunk-DPZAEKA6.js";
|
|
28
|
-
import {
|
|
29
|
-
benjaminiHochberg,
|
|
30
|
-
wilcoxonSignedRank
|
|
31
|
-
} from "./chunk-PJQFMIOX.js";
|
|
32
|
-
import {
|
|
33
|
-
observationsFromRunRecords,
|
|
34
|
-
thompsonCurriculum,
|
|
35
|
-
varianceBasedCurriculum
|
|
36
|
-
} from "./chunk-VZSRQ272.js";
|
|
37
|
-
import "./chunk-TT4KNT67.js";
|
|
38
|
-
import "./chunk-PC4UYEBM.js";
|
|
39
|
-
import "./chunk-VQMK5FMP.js";
|
|
40
|
-
import "./chunk-S3UZOQ5Y.js";
|
|
41
|
-
import "./chunk-MA6HLL3S.js";
|
|
42
|
-
import "./chunk-XJYR7XFV.js";
|
|
43
|
-
import "./chunk-VSMTAMNK.js";
|
|
44
|
-
import {
|
|
45
|
-
ValidationError
|
|
46
|
-
} from "./chunk-ONWEPEDO.js";
|
|
47
|
-
import "./chunk-PZ5AY32C.js";
|
|
48
|
-
|
|
49
|
-
// src/rl/adaptation-eval.ts
|
|
50
|
-
async function runAdaptationCurve(opts) {
|
|
51
|
-
const ks = opts.ks ?? [0, 1, 2, 4, 8, 16];
|
|
52
|
-
const reps = opts.reps ?? 3;
|
|
53
|
-
const passThreshold = opts.passThreshold ?? 0.5;
|
|
54
|
-
const sortedKs = [...ks].sort((a, b) => a - b);
|
|
55
|
-
const points = [];
|
|
56
|
-
for (const k of sortedKs) {
|
|
57
|
-
const perScenario = [];
|
|
58
|
-
const allScores = [];
|
|
59
|
-
let totalPasses = 0;
|
|
60
|
-
let totalAttempts = 0;
|
|
61
|
-
for (const scenario of opts.scenarios) {
|
|
62
|
-
const sid = scenario.scenarioId ?? `scenario-${opts.scenarios.indexOf(scenario)}`;
|
|
63
|
-
const scores = [];
|
|
64
|
-
let passes = 0;
|
|
65
|
-
for (let r = 0; r < reps; r++) {
|
|
66
|
-
const score = await opts.runner.run({ scenario, k, rep: r });
|
|
67
|
-
scores.push(score);
|
|
68
|
-
if (score >= passThreshold) passes++;
|
|
69
|
-
allScores.push(score);
|
|
70
|
-
if (score >= passThreshold) totalPasses++;
|
|
71
|
-
totalAttempts++;
|
|
72
|
-
}
|
|
73
|
-
const meanS = scores.reduce((s, v) => s + v, 0) / scores.length;
|
|
74
|
-
perScenario.push({ scenarioId: sid, meanScore: meanS, passes, total: scores.length });
|
|
75
|
-
}
|
|
76
|
-
const meanScore = allScores.reduce((s, v) => s + v, 0) / Math.max(1, allScores.length);
|
|
77
|
-
const variance = allScores.length < 2 ? 0 : allScores.reduce((s, v) => s + (v - meanScore) ** 2, 0) / (allScores.length - 1);
|
|
78
|
-
points.push({
|
|
79
|
-
k,
|
|
80
|
-
meanScore,
|
|
81
|
-
passRate: totalPasses / Math.max(1, totalAttempts),
|
|
82
|
-
std: Math.sqrt(variance),
|
|
83
|
-
n: allScores.length,
|
|
84
|
-
perScenario
|
|
85
|
-
});
|
|
86
|
-
}
|
|
87
|
-
const firstPassK2 = points.find((p) => p.passRate >= passThreshold)?.k ?? null;
|
|
88
|
-
const maxK = sortedKs[sortedKs.length - 1] ?? 1;
|
|
89
|
-
let area = 0;
|
|
90
|
-
for (let i = 1; i < points.length; i++) {
|
|
91
|
-
const x1 = points[i - 1].k;
|
|
92
|
-
const x2 = points[i].k;
|
|
93
|
-
const y1 = points[i - 1].meanScore;
|
|
94
|
-
const y2 = points[i].meanScore;
|
|
95
|
-
area += (y1 + y2) / 2 * (x2 - x1);
|
|
96
|
-
}
|
|
97
|
-
const adaptationArea = maxK === 0 ? 0 : area / maxK;
|
|
98
|
-
return { points, firstPassK: firstPassK2, adaptationArea };
|
|
99
|
-
}
|
|
100
|
-
function compareAdaptationCurves(a, b, opts = {}) {
|
|
101
|
-
const conf = opts.confidence ?? 0.95;
|
|
102
|
-
const resamples = opts.bootstrapResamples ?? 500;
|
|
103
|
-
const rng = makeRng(opts.seed);
|
|
104
|
-
const perK = [];
|
|
105
|
-
for (const ap of a.points) {
|
|
106
|
-
const bp = b.points.find((p) => p.k === ap.k);
|
|
107
|
-
if (!bp) continue;
|
|
108
|
-
const aMeans = ap.perScenario.map((s) => s.meanScore);
|
|
109
|
-
const bMeans = bp.perScenario.map((s) => s.meanScore);
|
|
110
|
-
const aCi = bootstrapMeanCi(aMeans, resamples, conf, rng);
|
|
111
|
-
const bCi = bootstrapMeanCi(bMeans, resamples, conf, rng);
|
|
112
|
-
perK.push({
|
|
113
|
-
k: ap.k,
|
|
114
|
-
deltaMean: ap.meanScore - bp.meanScore,
|
|
115
|
-
aLow: aCi.low,
|
|
116
|
-
aHigh: aCi.high,
|
|
117
|
-
bLow: bCi.low,
|
|
118
|
-
bHigh: bCi.high
|
|
119
|
-
});
|
|
120
|
-
}
|
|
121
|
-
const areaDelta = a.adaptationArea - b.adaptationArea;
|
|
122
|
-
const firstPassKDelta = a.firstPassK !== null && b.firstPassK !== null ? b.firstPassK - a.firstPassK : null;
|
|
123
|
-
const meanDelta = perK.reduce((s, p) => s + p.deltaMean, 0) / Math.max(1, perK.length);
|
|
124
|
-
let verdict;
|
|
125
|
-
if (Math.abs(meanDelta) < 0.02 && Math.abs(areaDelta) < 0.02) verdict = "similar";
|
|
126
|
-
else if (meanDelta > 0 && areaDelta > 0) verdict = "a_better";
|
|
127
|
-
else if (meanDelta < 0 && areaDelta < 0) verdict = "b_better";
|
|
128
|
-
else verdict = "similar";
|
|
129
|
-
const rationale = `mean per-k delta=${meanDelta.toFixed(3)}, area delta=${areaDelta.toFixed(3)}` + (firstPassKDelta !== null ? `, first-pass-k delta=${firstPassKDelta}` : "");
|
|
130
|
-
return { perK, areaDelta, firstPassKDelta, verdict, rationale };
|
|
131
|
-
}
|
|
132
|
-
function firstPassK(curve, threshold = 0.5) {
|
|
133
|
-
return curve.points.find((p) => p.passRate >= threshold)?.k ?? null;
|
|
134
|
-
}
|
|
135
|
-
function makeRng(seed) {
|
|
136
|
-
if (seed === void 0) return Math.random;
|
|
137
|
-
let s = seed >>> 0;
|
|
138
|
-
return () => {
|
|
139
|
-
s = s + 1831565813 >>> 0;
|
|
140
|
-
let t = s;
|
|
141
|
-
t = Math.imul(t ^ t >>> 15, t | 1);
|
|
142
|
-
t ^= t + Math.imul(t ^ t >>> 7, t | 61);
|
|
143
|
-
return ((t ^ t >>> 14) >>> 0) / 4294967296;
|
|
144
|
-
};
|
|
145
|
-
}
|
|
146
|
-
function bootstrapMeanCi(xs, resamples, confidence, rng) {
|
|
147
|
-
if (xs.length < 2) return { low: xs[0] ?? 0, high: xs[0] ?? 0 };
|
|
148
|
-
const samples = new Array(resamples);
|
|
149
|
-
for (let b = 0; b < resamples; b++) {
|
|
150
|
-
let sum = 0;
|
|
151
|
-
for (let i = 0; i < xs.length; i++) sum += xs[Math.floor(rng() * xs.length)];
|
|
152
|
-
samples[b] = sum / xs.length;
|
|
153
|
-
}
|
|
154
|
-
samples.sort((a, b) => a - b);
|
|
155
|
-
const alpha = 1 - confidence;
|
|
156
|
-
return {
|
|
157
|
-
low: samples[Math.floor(alpha / 2 * resamples)],
|
|
158
|
-
high: samples[Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)]
|
|
159
|
-
};
|
|
160
|
-
}
|
|
161
|
-
|
|
162
|
-
// src/rl/compute-curves.ts
|
|
163
|
-
async function runComputeCurve(opts) {
|
|
164
|
-
const points = [];
|
|
165
|
-
for (const budget of opts.budgets) {
|
|
166
|
-
const r = await opts.runAtBudget(budget);
|
|
167
|
-
points.push({
|
|
168
|
-
budgetId: budget.id,
|
|
169
|
-
cost: budget.cost,
|
|
170
|
-
score: r.score,
|
|
171
|
-
samples: r.samples,
|
|
172
|
-
std: r.std,
|
|
173
|
-
metrics: r.metrics
|
|
174
|
-
});
|
|
175
|
-
}
|
|
176
|
-
const sorted = [...points].sort((a, b) => a.cost - b.cost);
|
|
177
|
-
const logSlope = sorted.length >= 2 ? fitLogSlope(sorted) : null;
|
|
178
|
-
const best = points.reduce((a, b) => b.score > a.score ? b : a);
|
|
179
|
-
return { candidateId: opts.candidateId, points: sorted, logSlope, best };
|
|
180
|
-
}
|
|
181
|
-
async function bestOfN(opts) {
|
|
182
|
-
if (opts.n <= 0) throw new ValidationError("bestOfN: n must be > 0");
|
|
183
|
-
const rollouts = [];
|
|
184
|
-
const scores = [];
|
|
185
|
-
for (let i = 0; i < opts.n; i++) {
|
|
186
|
-
const r = await opts.sample(i);
|
|
187
|
-
rollouts.push(r);
|
|
188
|
-
scores.push(await opts.scoreFn(r));
|
|
189
|
-
}
|
|
190
|
-
let bestIndex = 0;
|
|
191
|
-
for (let i = 1; i < scores.length; i++) if (scores[i] > scores[bestIndex]) bestIndex = i;
|
|
192
|
-
const meanScore = scores.reduce((s, x) => s + x, 0) / scores.length;
|
|
193
|
-
return {
|
|
194
|
-
best: rollouts[bestIndex],
|
|
195
|
-
bestScore: scores[bestIndex],
|
|
196
|
-
scores,
|
|
197
|
-
meanScore,
|
|
198
|
-
bestIndex
|
|
199
|
-
};
|
|
200
|
-
}
|
|
201
|
-
async function selfConsistency(opts) {
|
|
202
|
-
if (opts.n <= 0) throw new ValidationError("selfConsistency: n must be > 0");
|
|
203
|
-
const rollouts = [];
|
|
204
|
-
const histogram = {};
|
|
205
|
-
for (let i = 0; i < opts.n; i++) {
|
|
206
|
-
const r = await opts.sample(i);
|
|
207
|
-
rollouts.push(r);
|
|
208
|
-
const key = opts.answerKey(r);
|
|
209
|
-
histogram[key] = (histogram[key] ?? 0) + 1;
|
|
210
|
-
}
|
|
211
|
-
let answer = "";
|
|
212
|
-
let max = -1;
|
|
213
|
-
for (const [k, v] of Object.entries(histogram)) {
|
|
214
|
-
if (v > max) {
|
|
215
|
-
max = v;
|
|
216
|
-
answer = k;
|
|
217
|
-
}
|
|
218
|
-
}
|
|
219
|
-
const representative = rollouts.find((r) => opts.answerKey(r) === answer) ?? rollouts[0];
|
|
220
|
-
return {
|
|
221
|
-
answer,
|
|
222
|
-
agreement: max / opts.n,
|
|
223
|
-
histogram,
|
|
224
|
-
representative,
|
|
225
|
-
rollouts
|
|
226
|
-
};
|
|
227
|
-
}
|
|
228
|
-
function paretoFrontier(points) {
|
|
229
|
-
const onFrontier = [];
|
|
230
|
-
for (const p of points) {
|
|
231
|
-
const dominated = points.some(
|
|
232
|
-
(q) => q !== p && q.cost <= p.cost && q.score >= p.score && (q.cost < p.cost || q.score > p.score)
|
|
233
|
-
);
|
|
234
|
-
if (!dominated) onFrontier.push(p);
|
|
235
|
-
}
|
|
236
|
-
return onFrontier.sort((a, b) => a.cost - b.cost);
|
|
237
|
-
}
|
|
238
|
-
function fitLogSlope(points) {
|
|
239
|
-
const xs = points.map((p) => Math.log(Math.max(1e-12, p.cost)));
|
|
240
|
-
const ys = points.map((p) => p.score);
|
|
241
|
-
const n = xs.length;
|
|
242
|
-
const mx = xs.reduce((s, x) => s + x, 0) / n;
|
|
243
|
-
const my = ys.reduce((s, y) => s + y, 0) / n;
|
|
244
|
-
let num = 0;
|
|
245
|
-
let den = 0;
|
|
246
|
-
for (let i = 0; i < n; i++) {
|
|
247
|
-
num += (xs[i] - mx) * (ys[i] - my);
|
|
248
|
-
den += (xs[i] - mx) ** 2;
|
|
249
|
-
}
|
|
250
|
-
return den === 0 ? 0 : num / den;
|
|
251
|
-
}
|
|
252
|
-
|
|
253
|
-
// src/rl/contamination.ts
|
|
254
|
-
async function runContaminationProbe(input, opts = {}) {
|
|
255
|
-
const fdr = opts.fdr ?? 0.05;
|
|
256
|
-
const minMedianDrop = opts.minMedianDrop ?? 0.05;
|
|
257
|
-
const floor = opts.scoreFloor ?? 0;
|
|
258
|
-
if (!input.perturbed && !input.perturbation) {
|
|
259
|
-
throw new ValidationError(
|
|
260
|
-
"runContaminationProbe: must supply either `perturbed` or `perturbation`."
|
|
261
|
-
);
|
|
262
|
-
}
|
|
263
|
-
const perturbed = input.perturbed ?? await Promise.all(input.originals.map((s) => input.perturbation.apply(s)));
|
|
264
|
-
if (perturbed.length !== input.originals.length) {
|
|
265
|
-
throw new ValidationError(
|
|
266
|
-
`runContaminationProbe: perturbed length ${perturbed.length} \u2260 originals ${input.originals.length}`
|
|
267
|
-
);
|
|
268
|
-
}
|
|
269
|
-
const origScores = await Promise.all(input.originals.map((s) => input.scoreFn(s)));
|
|
270
|
-
const pertScores = await Promise.all(perturbed.map((s) => input.scoreFn(s)));
|
|
271
|
-
const perScenario = input.originals.map((s, i) => ({
|
|
272
|
-
scenarioId: input.scenarioId(s),
|
|
273
|
-
originalScore: origScores[i],
|
|
274
|
-
perturbedScore: pertScores[i],
|
|
275
|
-
delta: pertScores[i] - origScores[i],
|
|
276
|
-
qValue: NaN
|
|
277
|
-
}));
|
|
278
|
-
const valid = perScenario.filter((p) => p.originalScore >= floor && p.perturbedScore >= floor);
|
|
279
|
-
if (valid.length < 4) {
|
|
280
|
-
return {
|
|
281
|
-
perScenario,
|
|
282
|
-
pairedTest: { w: 0, p: 1 },
|
|
283
|
-
medianDelta: 0,
|
|
284
|
-
meanDelta: 0,
|
|
285
|
-
contaminationSuspected: false,
|
|
286
|
-
reason: `insufficient valid scenarios (n=${valid.length}, need \u2265 4)`,
|
|
287
|
-
n: valid.length
|
|
288
|
-
};
|
|
289
|
-
}
|
|
290
|
-
const origValid = valid.map((p) => p.originalScore);
|
|
291
|
-
const pertValid = valid.map((p) => p.perturbedScore);
|
|
292
|
-
const pairedTest = wilcoxonSignedRank(origValid, pertValid);
|
|
293
|
-
const deltas = valid.map((p) => p.delta);
|
|
294
|
-
const sortedDeltas = [...deltas].sort((a, b) => a - b);
|
|
295
|
-
const median = sortedDeltas[Math.floor(sortedDeltas.length / 2)];
|
|
296
|
-
const mean = deltas.reduce((s, d) => s + d, 0) / deltas.length;
|
|
297
|
-
const pseudoP = valid.map((p) => Math.min(1, Math.max(1e-6, 1 - Math.abs(p.delta) / 1)));
|
|
298
|
-
const { qValues } = benjaminiHochberg(pseudoP, fdr);
|
|
299
|
-
for (let i = 0; i < valid.length; i++) {
|
|
300
|
-
const v = valid[i];
|
|
301
|
-
const idx = perScenario.findIndex((p) => p.scenarioId === v.scenarioId);
|
|
302
|
-
if (idx >= 0) perScenario[idx].qValue = qValues[i];
|
|
303
|
-
}
|
|
304
|
-
const contaminationSuspected = pairedTest.p < fdr && median <= -minMedianDrop;
|
|
305
|
-
const reason = contaminationSuspected ? `paired p=${pairedTest.p.toFixed(4)} < ${fdr} and median drop ${median.toFixed(4)} \u2265 ${minMedianDrop}` : pairedTest.p >= fdr ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})` : `significant but small effect (median delta ${median.toFixed(4)})`;
|
|
306
|
-
return {
|
|
307
|
-
perScenario,
|
|
308
|
-
pairedTest,
|
|
309
|
-
medianDelta: median,
|
|
310
|
-
meanDelta: mean,
|
|
311
|
-
contaminationSuspected,
|
|
312
|
-
reason,
|
|
313
|
-
n: valid.length
|
|
314
|
-
};
|
|
315
|
-
}
|
|
316
|
-
function renameVariables(identifiers, rename = (n, i) => `${n}_${(i % 26 + 10).toString(36)}`) {
|
|
317
|
-
return {
|
|
318
|
-
kind: "rename_variables",
|
|
319
|
-
apply(scenario) {
|
|
320
|
-
let prompt = scenario.prompt;
|
|
321
|
-
identifiers.forEach((id, i) => {
|
|
322
|
-
const replacement = rename(id, i);
|
|
323
|
-
const re = new RegExp(`\\b${escapeRegex(id)}\\b`, "g");
|
|
324
|
-
prompt = prompt.replace(re, replacement);
|
|
325
|
-
});
|
|
326
|
-
return { ...scenario, prompt };
|
|
327
|
-
}
|
|
328
|
-
};
|
|
329
|
-
}
|
|
330
|
-
function shuffleOrder(shuffleSection, seed) {
|
|
331
|
-
let s = seed >>> 0;
|
|
332
|
-
const rng = () => {
|
|
333
|
-
s = s + 1831565813 >>> 0;
|
|
334
|
-
let t = s;
|
|
335
|
-
t = Math.imul(t ^ t >>> 15, t | 1);
|
|
336
|
-
t ^= t + Math.imul(t ^ t >>> 7, t | 61);
|
|
337
|
-
return ((t ^ t >>> 14) >>> 0) / 4294967296;
|
|
338
|
-
};
|
|
339
|
-
return {
|
|
340
|
-
kind: "shuffle_order",
|
|
341
|
-
apply(scenario) {
|
|
342
|
-
const newPrompt = shuffleSection(scenario.prompt, rng);
|
|
343
|
-
return { ...scenario, prompt: newPrompt };
|
|
344
|
-
}
|
|
345
|
-
};
|
|
346
|
-
}
|
|
347
|
-
function injectIrrelevantClause(clause, position = "prefix") {
|
|
348
|
-
return {
|
|
349
|
-
kind: "inject_irrelevant_clause",
|
|
350
|
-
apply(scenario) {
|
|
351
|
-
const prompt = position === "prefix" ? `${clause} ${scenario.prompt}` : `${scenario.prompt} ${clause}`;
|
|
352
|
-
return { ...scenario, prompt };
|
|
353
|
-
}
|
|
354
|
-
};
|
|
355
|
-
}
|
|
356
|
-
function escapeRegex(s) {
|
|
357
|
-
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
358
|
-
}
|
|
359
|
-
|
|
360
|
-
// src/rl/corpus.ts
|
|
361
|
-
import { appendFileSync, existsSync, mkdirSync, readFileSync } from "fs";
|
|
362
|
-
import { dirname } from "path";
|
|
363
|
-
|
|
364
|
-
// src/rl/exporters.ts
|
|
365
|
-
async function toDpoRows(triples, lookups) {
|
|
366
|
-
const out = [];
|
|
367
|
-
for (const t of triples) {
|
|
368
|
-
const [prompt, chosen, rejected] = await Promise.all([
|
|
369
|
-
Promise.resolve(lookups.promptOf(t.chosenRunId)),
|
|
370
|
-
Promise.resolve(lookups.completionOf(t.chosenRunId)),
|
|
371
|
-
Promise.resolve(lookups.completionOf(t.rejectedRunId))
|
|
372
|
-
]);
|
|
373
|
-
out.push({
|
|
374
|
-
prompt,
|
|
375
|
-
chosen,
|
|
376
|
-
rejected,
|
|
377
|
-
margin: t.marginScore,
|
|
378
|
-
meta: {
|
|
379
|
-
scenarioId: t.scenarioId,
|
|
380
|
-
chosenVariantId: t.chosenVariantId,
|
|
381
|
-
rejectedVariantId: t.rejectedVariantId,
|
|
382
|
-
chosenRunId: t.chosenRunId,
|
|
383
|
-
rejectedRunId: t.rejectedRunId,
|
|
384
|
-
chosenModel: t.meta.chosenModel,
|
|
385
|
-
rejectedModel: t.meta.rejectedModel
|
|
386
|
-
}
|
|
387
|
-
});
|
|
388
|
-
}
|
|
389
|
-
return out;
|
|
390
|
-
}
|
|
391
|
-
function toDpoJsonl(rows) {
|
|
392
|
-
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
393
|
-
}
|
|
394
|
-
async function toGrpoRows(runs, lookups) {
|
|
395
|
-
const rewardOf2 = lookups.rewardOf ?? defaultReward;
|
|
396
|
-
const grouped = /* @__PURE__ */ new Map();
|
|
397
|
-
for (const r of runs) {
|
|
398
|
-
const sid = r.scenarioId ?? r.experimentId;
|
|
399
|
-
const arr = grouped.get(sid) ?? [];
|
|
400
|
-
arr.push(r);
|
|
401
|
-
grouped.set(sid, arr);
|
|
402
|
-
}
|
|
403
|
-
const rows = [];
|
|
404
|
-
for (const [scenarioId, group] of grouped.entries()) {
|
|
405
|
-
if (group.length === 0) continue;
|
|
406
|
-
const prompt = await Promise.resolve(lookups.promptOf(group[0].runId));
|
|
407
|
-
const completions = [];
|
|
408
|
-
const rewards = [];
|
|
409
|
-
const runIds = [];
|
|
410
|
-
for (const r of group) {
|
|
411
|
-
const reward2 = rewardOf2(r);
|
|
412
|
-
if (reward2 === null) continue;
|
|
413
|
-
const completion = await Promise.resolve(lookups.completionOf(r.runId));
|
|
414
|
-
completions.push(completion);
|
|
415
|
-
rewards.push(reward2);
|
|
416
|
-
runIds.push(r.runId);
|
|
417
|
-
}
|
|
418
|
-
if (completions.length === 0) continue;
|
|
419
|
-
rows.push({
|
|
420
|
-
prompt,
|
|
421
|
-
completions,
|
|
422
|
-
rewards,
|
|
423
|
-
runIds,
|
|
424
|
-
meta: {
|
|
425
|
-
scenarioId,
|
|
426
|
-
n: completions.length,
|
|
427
|
-
meanReward: rewards.reduce((s, x) => s + x, 0) / rewards.length
|
|
428
|
-
}
|
|
429
|
-
});
|
|
430
|
-
}
|
|
431
|
-
return rows;
|
|
432
|
-
}
|
|
433
|
-
function toGrpoJsonl(rows) {
|
|
434
|
-
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
435
|
-
}
|
|
436
|
-
async function toSftRows(runs, lookups) {
|
|
437
|
-
const include = lookups.include ?? (() => true);
|
|
438
|
-
const rows = [];
|
|
439
|
-
for (const r of runs) {
|
|
440
|
-
if (!include(r)) continue;
|
|
441
|
-
const system = lookups.systemOf?.(r);
|
|
442
|
-
const [prompt, completion] = await Promise.all([
|
|
443
|
-
Promise.resolve(lookups.promptOf(r.runId)),
|
|
444
|
-
Promise.resolve(lookups.completionOf(r.runId))
|
|
445
|
-
]);
|
|
446
|
-
const messages = [];
|
|
447
|
-
if (system) messages.push({ role: "system", content: system });
|
|
448
|
-
messages.push({ role: "user", content: prompt });
|
|
449
|
-
messages.push({ role: "assistant", content: completion });
|
|
450
|
-
rows.push({
|
|
451
|
-
messages,
|
|
452
|
-
meta: {
|
|
453
|
-
runId: r.runId,
|
|
454
|
-
candidateId: r.candidateId,
|
|
455
|
-
scenarioId: r.scenarioId,
|
|
456
|
-
score: r.outcome.holdoutScore ?? r.outcome.searchScore,
|
|
457
|
-
model: r.model
|
|
458
|
-
}
|
|
459
|
-
});
|
|
460
|
-
}
|
|
461
|
-
return rows;
|
|
462
|
-
}
|
|
463
|
-
function toSftJsonl(rows) {
|
|
464
|
-
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
465
|
-
}
|
|
466
|
-
async function toPrmRows(triples, lookups) {
|
|
467
|
-
const rows = [];
|
|
468
|
-
for (const t of triples) {
|
|
469
|
-
const prompt = await Promise.resolve(lookups.promptOf(t.prefixRunId));
|
|
470
|
-
const prefixSpanIds = lookups.prefixOf ? await Promise.resolve(lookups.prefixOf(t.prefixRunId, t.prefixStepIndex)) : [];
|
|
471
|
-
const prefixStepText = [];
|
|
472
|
-
for (const spanId of prefixSpanIds) {
|
|
473
|
-
prefixStepText.push(await Promise.resolve(lookups.stepTextOf(t.prefixRunId, spanId)));
|
|
474
|
-
}
|
|
475
|
-
const chosenStep = await Promise.resolve(lookups.stepTextOf(t.prefixRunId, t.chosenSpanId));
|
|
476
|
-
const rejectedStep = await Promise.resolve(
|
|
477
|
-
lookups.stepTextOf(t.rejectedRunId, t.rejectedSpanId)
|
|
478
|
-
);
|
|
479
|
-
rows.push({
|
|
480
|
-
prompt,
|
|
481
|
-
prefixSpanIds,
|
|
482
|
-
prefixStepText,
|
|
483
|
-
chosenStep,
|
|
484
|
-
rejectedStep,
|
|
485
|
-
chosenReward: t.chosenReward,
|
|
486
|
-
rejectedReward: t.rejectedReward,
|
|
487
|
-
marginScore: t.marginScore,
|
|
488
|
-
meta: {
|
|
489
|
-
prefixRunId: t.prefixRunId,
|
|
490
|
-
rejectedRunId: t.rejectedRunId,
|
|
491
|
-
prefixStepIndex: t.prefixStepIndex
|
|
492
|
-
}
|
|
493
|
-
});
|
|
494
|
-
}
|
|
495
|
-
return rows;
|
|
496
|
-
}
|
|
497
|
-
function toPrmJsonl(rows) {
|
|
498
|
-
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
499
|
-
}
|
|
500
|
-
function stepRewardsToJsonl(stepRewards) {
|
|
501
|
-
const rows = stepRewards.map((s) => ({
|
|
502
|
-
runId: s.runId,
|
|
503
|
-
spanId: s.spanId,
|
|
504
|
-
stepIndex: s.stepIndex,
|
|
505
|
-
reward: s.reward,
|
|
506
|
-
determinism: s.determinism,
|
|
507
|
-
weight: s.weight ?? 1
|
|
508
|
-
}));
|
|
509
|
-
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
510
|
-
}
|
|
511
|
-
function defaultReward(run) {
|
|
512
|
-
const v = run.outcome.holdoutScore ?? run.outcome.searchScore;
|
|
513
|
-
return typeof v === "number" && Number.isFinite(v) ? v : null;
|
|
514
|
-
}
|
|
515
|
-
|
|
516
|
-
// src/rl/dataset.ts
|
|
517
|
-
function reward(r) {
|
|
518
|
-
const v = r.outcome.holdoutScore ?? r.outcome.searchScore;
|
|
519
|
-
return typeof v === "number" && Number.isFinite(v) ? v : null;
|
|
520
|
-
}
|
|
521
|
-
function distinct(xs) {
|
|
522
|
-
return [...new Set(xs)].sort();
|
|
523
|
-
}
|
|
524
|
-
function computeRewardStats(values) {
|
|
525
|
-
if (values.length === 0) return { n: 0, mean: 0, median: 0, min: 0, max: 0, std: 0 };
|
|
526
|
-
const sorted = [...values].sort((a, b) => a - b);
|
|
527
|
-
const n = sorted.length;
|
|
528
|
-
const mean = sorted.reduce((s, x) => s + x, 0) / n;
|
|
529
|
-
const mid = Math.floor(n / 2);
|
|
530
|
-
const median = n % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid];
|
|
531
|
-
const variance = sorted.reduce((s, x) => s + (x - mean) ** 2, 0) / n;
|
|
532
|
-
return { n, mean, median, min: sorted[0], max: sorted[n - 1], std: Math.sqrt(variance) };
|
|
533
|
-
}
|
|
534
|
-
function computeStats(records) {
|
|
535
|
-
const splits = { search: 0, dev: 0, holdout: 0 };
|
|
536
|
-
let inTok = 0;
|
|
537
|
-
let outTok = 0;
|
|
538
|
-
let cost = 0;
|
|
539
|
-
const rewards = [];
|
|
540
|
-
for (const r of records) {
|
|
541
|
-
splits[r.splitTag] = (splits[r.splitTag] ?? 0) + 1;
|
|
542
|
-
inTok += r.tokenUsage.input;
|
|
543
|
-
outTok += r.tokenUsage.output;
|
|
544
|
-
cost += r.costUsd;
|
|
545
|
-
const rw = reward(r);
|
|
546
|
-
if (rw !== null) rewards.push(rw);
|
|
547
|
-
}
|
|
548
|
-
return {
|
|
549
|
-
records: records.length,
|
|
550
|
-
splits,
|
|
551
|
-
reward: computeRewardStats(rewards),
|
|
552
|
-
models: distinct(records.map((r) => r.model)),
|
|
553
|
-
promptHashes: distinct(records.map((r) => r.promptHash)),
|
|
554
|
-
commitShas: distinct(records.map((r) => r.commitSha)),
|
|
555
|
-
totalTokens: { input: inTok, output: outTok },
|
|
556
|
-
totalCostUsd: cost
|
|
557
|
-
};
|
|
558
|
-
}
|
|
559
|
-
async function buildRlDataset(records, lookups, config, preferences) {
|
|
560
|
-
if (records.length === 0) {
|
|
561
|
-
throw new Error("buildRlDataset: no records \u2014 refusing to package an empty dataset");
|
|
562
|
-
}
|
|
563
|
-
const formats = config.formats ?? ["grpo", "sft"];
|
|
564
|
-
const files = {};
|
|
565
|
-
const rowCounts = {};
|
|
566
|
-
if (formats.includes("grpo")) {
|
|
567
|
-
const rows = await toGrpoRows(records, lookups);
|
|
568
|
-
files["train.grpo.jsonl"] = toGrpoJsonl(rows);
|
|
569
|
-
rowCounts.grpo = rows.length;
|
|
570
|
-
}
|
|
571
|
-
if (formats.includes("sft")) {
|
|
572
|
-
const rows = await toSftRows(records, lookups);
|
|
573
|
-
files["train.sft.jsonl"] = toSftJsonl(rows);
|
|
574
|
-
rowCounts.sft = rows.length;
|
|
575
|
-
}
|
|
576
|
-
if (formats.includes("dpo")) {
|
|
577
|
-
if (!preferences) {
|
|
578
|
-
throw new Error("buildRlDataset: format 'dpo' requires `preferences` (triples + lookups)");
|
|
579
|
-
}
|
|
580
|
-
const rows = await toDpoRows(preferences.triples, preferences.lookups);
|
|
581
|
-
files["train.dpo.jsonl"] = toDpoJsonl(rows);
|
|
582
|
-
rowCounts.dpo = rows.length;
|
|
583
|
-
}
|
|
584
|
-
const manifest = {
|
|
585
|
-
...config,
|
|
586
|
-
formats,
|
|
587
|
-
rowCounts,
|
|
588
|
-
stats: computeStats(records)
|
|
589
|
-
};
|
|
590
|
-
files["manifest.json"] = `${JSON.stringify(manifest, null, 2)}
|
|
591
|
-
`;
|
|
592
|
-
files["DATASHEET.md"] = datasheetToMarkdown(manifest);
|
|
593
|
-
return { manifest, files };
|
|
594
|
-
}
|
|
595
|
-
function pct(x) {
|
|
596
|
-
return `${(x * 100).toFixed(1)}%`;
|
|
597
|
-
}
|
|
598
|
-
function datasheetToMarkdown(m) {
|
|
599
|
-
const s = m.stats;
|
|
600
|
-
const total = s.records || 1;
|
|
601
|
-
const splitLines = ["search", "dev", "holdout"].map((k) => ` - \`${k}\`: ${s.splits[k]} (${pct(s.splits[k] / total)})`).join("\n");
|
|
602
|
-
const deterministic = m.reward.kind === "deterministic";
|
|
603
|
-
return [
|
|
604
|
-
`# Dataset: ${m.name} \`v${m.version}\``,
|
|
605
|
-
"",
|
|
606
|
-
`**Domain:** ${m.domain} \xB7 **Created:** ${m.createdAtIso} \xB7 **License:** ${m.license}`,
|
|
607
|
-
"",
|
|
608
|
-
"## Reward provenance",
|
|
609
|
-
`- **Kind:** ${m.reward.kind}${deterministic ? " \u2705 (decidable \u2014 not judge-noise)" : ""}`,
|
|
610
|
-
`- **Source:** ${m.reward.source}`,
|
|
611
|
-
`- **Description:** ${m.reward.description}`,
|
|
612
|
-
"",
|
|
613
|
-
"## Composition",
|
|
614
|
-
`- **Records (trajectories):** ${s.records}`,
|
|
615
|
-
`- **Formats:** ${m.formats.map((f) => `${f} (${m.rowCounts[f] ?? 0} rows)`).join(", ")}`,
|
|
616
|
-
"- **Splits:**",
|
|
617
|
-
splitLines,
|
|
618
|
-
"",
|
|
619
|
-
"## Reward distribution",
|
|
620
|
-
`- n=${s.reward.n} \xB7 mean=${s.reward.mean.toFixed(3)} \xB7 median=${s.reward.median.toFixed(3)} \xB7 min=${s.reward.min.toFixed(3)} \xB7 max=${s.reward.max.toFixed(3)} \xB7 std=${s.reward.std.toFixed(3)}`,
|
|
621
|
-
"",
|
|
622
|
-
"## Provenance",
|
|
623
|
-
`- **Models:** ${s.models.join(", ")}`,
|
|
624
|
-
`- **Prompt/agent versions (sha256):** ${s.promptHashes.length} distinct`,
|
|
625
|
-
`- **Commits:** ${s.commitShas.join(", ")}`,
|
|
626
|
-
`- **Tokens:** ${s.totalTokens.input} in / ${s.totalTokens.output} out \xB7 **Cost:** $${s.totalCostUsd.toFixed(2)}`,
|
|
627
|
-
"",
|
|
628
|
-
"## Quality gates",
|
|
629
|
-
`- Contamination probe: ${m.qualityGates?.contaminationProbe ?? "not-run"}`,
|
|
630
|
-
`- Dedup: ${m.qualityGates?.dedup ? "yes" : "no"} \xB7 Verifiable-reward filter: ${m.qualityGates?.verifiableRewardFilter ? "yes" : "no"}`,
|
|
631
|
-
"",
|
|
632
|
-
"## Recommended uses",
|
|
633
|
-
m.intendedUse,
|
|
634
|
-
"",
|
|
635
|
-
"## Out of scope",
|
|
636
|
-
m.outOfScope,
|
|
637
|
-
"",
|
|
638
|
-
"## Limitations",
|
|
639
|
-
m.limitations,
|
|
640
|
-
"",
|
|
641
|
-
"## Token rendering",
|
|
642
|
-
"For RL/SFT training, tokenize with the per-model renderer (DeepSeek-V3 / Kimi-K2 / Qwen3) to preserve token identity and per-token loss masks across tool-call turns \u2014 see `renderers` (PrimeIntellect). The `messages` / `completions` here are the renderer input.",
|
|
643
|
-
""
|
|
644
|
-
].join("\n");
|
|
645
|
-
}
|
|
646
|
-
|
|
647
|
-
// src/rl/corpus.ts
|
|
648
|
-
function appendToCorpus(records, corpusPath) {
|
|
649
|
-
mkdirSync(dirname(corpusPath), { recursive: true });
|
|
650
|
-
const existing = existsSync(corpusPath) ? readCorpus(corpusPath) : [];
|
|
651
|
-
const seen = new Set(existing.map((r) => r.runId));
|
|
652
|
-
const lines = [];
|
|
653
|
-
let appended = 0;
|
|
654
|
-
let skipped = 0;
|
|
655
|
-
for (const r of records) {
|
|
656
|
-
if (seen.has(r.runId)) {
|
|
657
|
-
skipped++;
|
|
658
|
-
continue;
|
|
659
|
-
}
|
|
660
|
-
seen.add(r.runId);
|
|
661
|
-
lines.push(JSON.stringify(r));
|
|
662
|
-
appended++;
|
|
663
|
-
}
|
|
664
|
-
if (lines.length > 0) appendFileSync(corpusPath, `${lines.join("\n")}
|
|
665
|
-
`);
|
|
666
|
-
return { appended, skipped, total: existing.length + appended };
|
|
667
|
-
}
|
|
668
|
-
function readCorpus(corpusPath) {
|
|
669
|
-
if (!existsSync(corpusPath)) return [];
|
|
670
|
-
const out = [];
|
|
671
|
-
for (const line of readFileSync(corpusPath, "utf8").split("\n")) {
|
|
672
|
-
if (line.trim()) out.push(JSON.parse(line));
|
|
673
|
-
}
|
|
674
|
-
return out;
|
|
675
|
-
}
|
|
676
|
-
function rewardOf(r) {
|
|
677
|
-
const v = r.outcome.holdoutScore ?? r.outcome.searchScore;
|
|
678
|
-
return typeof v === "number" && Number.isFinite(v) ? v : 0;
|
|
679
|
-
}
|
|
680
|
-
async function buildDatasetFromCorpus(corpusPath, config, opts = {}) {
|
|
681
|
-
let records = readCorpus(corpusPath).filter(
|
|
682
|
-
(r) => typeof r.prompt === "string" && typeof r.completion === "string"
|
|
683
|
-
);
|
|
684
|
-
if (opts.splits) records = records.filter((r) => opts.splits.includes(r.splitTag));
|
|
685
|
-
if (opts.minScore != null) records = records.filter((r) => rewardOf(r) >= opts.minScore);
|
|
686
|
-
const text = new Map(
|
|
687
|
-
records.map((r) => [r.runId, { prompt: r.prompt, completion: r.completion }])
|
|
688
|
-
);
|
|
689
|
-
const lookups = {
|
|
690
|
-
promptOf: (id) => text.get(id)?.prompt ?? "",
|
|
691
|
-
completionOf: (id) => text.get(id)?.completion ?? ""
|
|
692
|
-
};
|
|
693
|
-
return buildRlDataset(records, lookups, config);
|
|
694
|
-
}
|
|
695
|
-
|
|
696
|
-
// src/rl/predictive-validity-researcher.ts
|
|
697
|
-
var PredictiveValidityResearcher = class {
|
|
698
|
-
opts;
|
|
699
|
-
lastReport = null;
|
|
700
|
-
constructor(opts) {
|
|
701
|
-
this.opts = opts;
|
|
702
|
-
}
|
|
703
|
-
async inspectFailures(runs) {
|
|
704
|
-
const threshold = this.opts.failureThreshold ?? 0.5;
|
|
705
|
-
const failures = [];
|
|
706
|
-
const failingRuns = runs.filter((r) => {
|
|
707
|
-
const score = r.outcome.holdoutScore ?? r.outcome.searchScore;
|
|
708
|
-
return typeof score === "number" && score < threshold;
|
|
709
|
-
});
|
|
710
|
-
if (failingRuns.length === 0) return failures;
|
|
711
|
-
const grouped = /* @__PURE__ */ new Map();
|
|
712
|
-
for (const r of failingRuns) {
|
|
713
|
-
const arr = grouped.get(r.candidateId) ?? [];
|
|
714
|
-
arr.push(r);
|
|
715
|
-
grouped.set(r.candidateId, arr);
|
|
716
|
-
}
|
|
717
|
-
for (const [candidateId, group] of grouped.entries()) {
|
|
718
|
-
const meanScore = group.reduce((s, r) => {
|
|
719
|
-
const x = r.outcome.holdoutScore ?? r.outcome.searchScore ?? 0;
|
|
720
|
-
return s + x;
|
|
721
|
-
}, 0) / group.length;
|
|
722
|
-
failures.push({
|
|
723
|
-
code: `low-score-${candidateId}`,
|
|
724
|
-
description: `${candidateId} scored < ${threshold} on ${group.length} run(s) (mean ${meanScore.toFixed(3)})`,
|
|
725
|
-
evidence: {
|
|
726
|
-
runIds: group.slice(0, 8).map((r) => r.runId),
|
|
727
|
-
samples: group.length
|
|
728
|
-
}
|
|
729
|
-
});
|
|
730
|
-
}
|
|
731
|
-
return failures;
|
|
732
|
-
}
|
|
733
|
-
async proposeChange(failures) {
|
|
734
|
-
if (failures.length === 0) return [];
|
|
735
|
-
if (this.lastReport === null) {
|
|
736
|
-
return [
|
|
737
|
-
{
|
|
738
|
-
kind: "threshold",
|
|
739
|
-
payload: { directive: "researcher.collect-more-outcomes" },
|
|
740
|
-
rationale: "predictive-validity researcher has no prior report; cannot recommend rubric reweighting until at least one report exists"
|
|
741
|
-
}
|
|
742
|
-
];
|
|
743
|
-
}
|
|
744
|
-
const decorativeThreshold = this.opts.decorativeThreshold ?? 0.4;
|
|
745
|
-
const changes = [];
|
|
746
|
-
for (const ranking of this.lastReport.ranked) {
|
|
747
|
-
if (ranking.verdict === "load_bearing") continue;
|
|
748
|
-
if (Math.abs(ranking.spearman) >= decorativeThreshold) continue;
|
|
749
|
-
changes.push({
|
|
750
|
-
kind: "reviewer_prompt",
|
|
751
|
-
payload: {
|
|
752
|
-
rubric: ranking.rubric,
|
|
753
|
-
action: "down-weight",
|
|
754
|
-
spearman: ranking.spearman,
|
|
755
|
-
bestOutcome: ranking.bestOutcome
|
|
756
|
-
},
|
|
757
|
-
rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (decorative); recommend down-weighting`,
|
|
758
|
-
expectedDelta: -Math.max(0, 0.05 - Math.abs(ranking.spearman))
|
|
759
|
-
});
|
|
760
|
-
}
|
|
761
|
-
for (const ranking of this.lastReport.ranked.slice(0, 1)) {
|
|
762
|
-
if (ranking.verdict !== "load_bearing") continue;
|
|
763
|
-
changes.push({
|
|
764
|
-
kind: "reviewer_prompt",
|
|
765
|
-
payload: {
|
|
766
|
-
rubric: ranking.rubric,
|
|
767
|
-
action: "up-weight",
|
|
768
|
-
spearman: ranking.spearman,
|
|
769
|
-
bestOutcome: ranking.bestOutcome
|
|
770
|
-
},
|
|
771
|
-
rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (load-bearing); recommend up-weighting`,
|
|
772
|
-
expectedDelta: Math.max(0, Math.abs(ranking.spearman) - 0.5) * 0.1
|
|
773
|
-
});
|
|
774
|
-
}
|
|
775
|
-
return changes;
|
|
776
|
-
}
|
|
777
|
-
async applyChange(changes, baseline) {
|
|
778
|
-
return {
|
|
779
|
-
...baseline,
|
|
780
|
-
changes: [...baseline.changes, ...changes]
|
|
781
|
-
};
|
|
782
|
-
}
|
|
783
|
-
async evaluateChange(plan) {
|
|
784
|
-
const emptyGate = {
|
|
785
|
-
promote: false,
|
|
786
|
-
candidateId: plan.proposedCandidateId,
|
|
787
|
-
baselineId: plan.baselineCandidateId,
|
|
788
|
-
evidence: {
|
|
789
|
-
productiveRuns: 0,
|
|
790
|
-
medianPairedDelta: 0,
|
|
791
|
-
pairedCI: { low: 0, high: 0 },
|
|
792
|
-
pairedPValue: 1,
|
|
793
|
-
searchScore: 0,
|
|
794
|
-
holdoutScore: 0,
|
|
795
|
-
overfitGap: 0,
|
|
796
|
-
baselineOverfitGap: 0,
|
|
797
|
-
medianCandidateCost: Number.NaN,
|
|
798
|
-
medianBaselineCost: Number.NaN
|
|
799
|
-
},
|
|
800
|
-
reason: "predictive-validity researcher does not execute plans; the caller is expected to run the sweep and call rubricPredictiveValidity directly with the resulting RunRecord[].",
|
|
801
|
-
rejectionCode: "few_runs"
|
|
802
|
-
};
|
|
803
|
-
return {
|
|
804
|
-
plan,
|
|
805
|
-
runs: [],
|
|
806
|
-
gateDecision: emptyGate
|
|
807
|
-
};
|
|
808
|
-
}
|
|
809
|
-
/**
|
|
810
|
-
* Run the predictive-validity check explicitly against a fresh RunRecord
|
|
811
|
-
* set. Updates the researcher's cached report so subsequent
|
|
812
|
-
* `proposeChange` calls have evidence to draw from.
|
|
813
|
-
*/
|
|
814
|
-
async runValidityCheck(runs) {
|
|
815
|
-
const report = await rubricPredictiveValidity({
|
|
816
|
-
runs,
|
|
817
|
-
outcomes: this.opts.outcomes,
|
|
818
|
-
outcomeMetrics: this.opts.outcomeMetrics,
|
|
819
|
-
rubrics: this.opts.rubrics
|
|
820
|
-
});
|
|
821
|
-
if (this.opts.onReport) await this.opts.onReport(report);
|
|
822
|
-
this.lastReport = report;
|
|
823
|
-
return report;
|
|
824
|
-
}
|
|
825
|
-
/**
|
|
826
|
-
* Force-feed a predictive-validity report into the researcher state —
|
|
827
|
-
* useful when the consumer ran the report out-of-band and wants the
|
|
828
|
-
* researcher's later proposals informed by it.
|
|
829
|
-
*/
|
|
830
|
-
setReport(report) {
|
|
831
|
-
this.lastReport = report;
|
|
832
|
-
}
|
|
833
|
-
getLastReport() {
|
|
834
|
-
return this.lastReport;
|
|
835
|
-
}
|
|
836
|
-
};
|
|
837
|
-
|
|
838
|
-
// src/rl/preferences.ts
|
|
839
|
-
var SPLIT_TAG_DEFAULT = "holdout";
|
|
840
|
-
var DEFAULT_REWARD = (run) => {
|
|
841
|
-
const v = run.outcome.holdoutScore ?? run.outcome.searchScore;
|
|
842
|
-
return typeof v === "number" && Number.isFinite(v) ? v : null;
|
|
843
|
-
};
|
|
844
|
-
function extractPreferences(runs, opts = {}) {
|
|
845
|
-
const strategy = opts.strategy ?? "paired-by-scenario-and-seed";
|
|
846
|
-
const minMargin = opts.minMargin ?? 0.05;
|
|
847
|
-
const splitTag = opts.splitTag ?? SPLIT_TAG_DEFAULT;
|
|
848
|
-
const rewardOf2 = opts.rewardOf ?? DEFAULT_REWARD;
|
|
849
|
-
const filtered = runs.filter((r) => r.splitTag === splitTag);
|
|
850
|
-
const scoredEntries = [];
|
|
851
|
-
for (const run of filtered) {
|
|
852
|
-
const s = rewardOf2(run);
|
|
853
|
-
if (s === null) continue;
|
|
854
|
-
scoredEntries.push({ run, score: s });
|
|
855
|
-
}
|
|
856
|
-
const pairs = [];
|
|
857
|
-
let pairsBelowMargin = 0;
|
|
858
|
-
let cellsSingleton = 0;
|
|
859
|
-
let cellsInspected = 0;
|
|
860
|
-
if (strategy === "paired-by-scenario-and-seed") {
|
|
861
|
-
const groups = /* @__PURE__ */ new Map();
|
|
862
|
-
for (const e of scoredEntries) {
|
|
863
|
-
const sid = scenarioOf(e.run);
|
|
864
|
-
const key = `${sid}::${e.run.seed}`;
|
|
865
|
-
const arr = groups.get(key) ?? [];
|
|
866
|
-
arr.push(e);
|
|
867
|
-
groups.set(key, arr);
|
|
868
|
-
}
|
|
869
|
-
for (const [key, members] of groups.entries()) {
|
|
870
|
-
cellsInspected++;
|
|
871
|
-
if (members.length < 2) {
|
|
872
|
-
cellsSingleton++;
|
|
873
|
-
continue;
|
|
874
|
-
}
|
|
875
|
-
for (let i = 0; i < members.length; i++) {
|
|
876
|
-
for (let j = i + 1; j < members.length; j++) {
|
|
877
|
-
const a = members[i];
|
|
878
|
-
const b = members[j];
|
|
879
|
-
if (a.run.candidateId === b.run.candidateId) continue;
|
|
880
|
-
const result = makePair(a, b, key.split("::")[0], minMargin);
|
|
881
|
-
if (result.kind === "admit") pairs.push(result.pair);
|
|
882
|
-
else pairsBelowMargin++;
|
|
883
|
-
}
|
|
884
|
-
}
|
|
885
|
-
}
|
|
886
|
-
} else if (strategy === "paired-by-scenario") {
|
|
887
|
-
const byScenarioVariant = /* @__PURE__ */ new Map();
|
|
888
|
-
for (const e of scoredEntries) {
|
|
889
|
-
const sid = scenarioOf(e.run);
|
|
890
|
-
let perScenario = byScenarioVariant.get(sid);
|
|
891
|
-
if (!perScenario) {
|
|
892
|
-
perScenario = /* @__PURE__ */ new Map();
|
|
893
|
-
byScenarioVariant.set(sid, perScenario);
|
|
894
|
-
}
|
|
895
|
-
const cur = perScenario.get(e.run.candidateId);
|
|
896
|
-
if (cur) {
|
|
897
|
-
cur.sum += e.score;
|
|
898
|
-
cur.n++;
|
|
899
|
-
} else perScenario.set(e.run.candidateId, { run: e.run, sum: e.score, n: 1 });
|
|
900
|
-
}
|
|
901
|
-
for (const [sid, perVariant] of byScenarioVariant.entries()) {
|
|
902
|
-
cellsInspected++;
|
|
903
|
-
const arr = [...perVariant.entries()].map(([vid, agg]) => ({
|
|
904
|
-
run: agg.run,
|
|
905
|
-
score: agg.sum / agg.n,
|
|
906
|
-
variantId: vid
|
|
907
|
-
}));
|
|
908
|
-
if (arr.length < 2) {
|
|
909
|
-
cellsSingleton++;
|
|
910
|
-
continue;
|
|
911
|
-
}
|
|
912
|
-
for (let i = 0; i < arr.length; i++) {
|
|
913
|
-
for (let j = i + 1; j < arr.length; j++) {
|
|
914
|
-
const result = makePair(arr[i], arr[j], sid, minMargin);
|
|
915
|
-
if (result.kind === "admit") pairs.push(result.pair);
|
|
916
|
-
else pairsBelowMargin++;
|
|
917
|
-
}
|
|
918
|
-
}
|
|
919
|
-
}
|
|
920
|
-
} else {
|
|
921
|
-
const byScenario = /* @__PURE__ */ new Map();
|
|
922
|
-
for (const e of scoredEntries) {
|
|
923
|
-
const sid = scenarioOf(e.run);
|
|
924
|
-
const arr = byScenario.get(sid) ?? [];
|
|
925
|
-
arr.push(e);
|
|
926
|
-
byScenario.set(sid, arr);
|
|
927
|
-
}
|
|
928
|
-
for (const [sid, arr] of byScenario.entries()) {
|
|
929
|
-
cellsInspected++;
|
|
930
|
-
if (arr.length < 2) {
|
|
931
|
-
cellsSingleton++;
|
|
932
|
-
continue;
|
|
933
|
-
}
|
|
934
|
-
const sorted = [...arr].sort((a, b) => a.score - b.score);
|
|
935
|
-
const top = sorted[sorted.length - 1];
|
|
936
|
-
const bot = sorted[0];
|
|
937
|
-
if (top.run.candidateId === bot.run.candidateId) {
|
|
938
|
-
cellsSingleton++;
|
|
939
|
-
continue;
|
|
940
|
-
}
|
|
941
|
-
const result = makePair(bot, top, sid, minMargin);
|
|
942
|
-
if (result.kind === "admit") pairs.push(result.pair);
|
|
943
|
-
else pairsBelowMargin++;
|
|
944
|
-
}
|
|
945
|
-
}
|
|
946
|
-
return { pairs, cellsInspected, pairsBelowMargin, cellsSingleton, strategy };
|
|
947
|
-
}
|
|
948
|
-
function toTRLFormat(triples, promptOf) {
|
|
949
|
-
return triples.map((t) => ({
|
|
950
|
-
prompt: promptOf(t.meta.chosenPromptHash),
|
|
951
|
-
chosen: t.meta.chosenPromptHash,
|
|
952
|
-
// caller substitutes the model output via the runId map
|
|
953
|
-
rejected: t.meta.rejectedPromptHash
|
|
954
|
-
}));
|
|
955
|
-
}
|
|
956
|
-
function toAnthropicFormat(triples) {
|
|
957
|
-
return triples.map((t) => ({
|
|
958
|
-
scenarioId: t.scenarioId,
|
|
959
|
-
chosenRunId: t.chosenRunId,
|
|
960
|
-
rejectedRunId: t.rejectedRunId,
|
|
961
|
-
margin: t.marginScore
|
|
962
|
-
}));
|
|
963
|
-
}
|
|
964
|
-
function makePair(a, b, scenarioId, minMargin) {
|
|
965
|
-
const margin = Math.abs(a.score - b.score);
|
|
966
|
-
if (margin < minMargin) return { kind: "reject" };
|
|
967
|
-
const [chosen, rejected] = a.score > b.score ? [a, b] : [b, a];
|
|
968
|
-
return {
|
|
969
|
-
kind: "admit",
|
|
970
|
-
pair: {
|
|
971
|
-
scenarioId,
|
|
972
|
-
chosenRunId: chosen.run.runId,
|
|
973
|
-
rejectedRunId: rejected.run.runId,
|
|
974
|
-
chosenVariantId: chosen.run.candidateId,
|
|
975
|
-
rejectedVariantId: rejected.run.candidateId,
|
|
976
|
-
marginScore: chosen.score - rejected.score,
|
|
977
|
-
scores: { chosen: chosen.score, rejected: rejected.score },
|
|
978
|
-
seed: chosen.run.seed === rejected.run.seed ? chosen.run.seed : void 0,
|
|
979
|
-
meta: {
|
|
980
|
-
chosenPromptHash: chosen.run.promptHash,
|
|
981
|
-
rejectedPromptHash: rejected.run.promptHash,
|
|
982
|
-
chosenConfigHash: chosen.run.configHash,
|
|
983
|
-
rejectedConfigHash: rejected.run.configHash,
|
|
984
|
-
chosenModel: chosen.run.model,
|
|
985
|
-
rejectedModel: rejected.run.model
|
|
986
|
-
}
|
|
987
|
-
}
|
|
988
|
-
};
|
|
989
|
-
}
|
|
990
|
-
function scenarioOf(run) {
|
|
991
|
-
if (typeof run.scenarioId === "string" && run.scenarioId.length > 0) return run.scenarioId;
|
|
992
|
-
const fromRaw = run.outcome.raw.scenario_id;
|
|
993
|
-
if (typeof fromRaw === "number" && Number.isFinite(fromRaw)) return String(fromRaw);
|
|
994
|
-
if (typeof fromRaw === "string") return fromRaw;
|
|
995
|
-
return run.experimentId;
|
|
996
|
-
}
|
|
997
|
-
|
|
998
|
-
// src/rl/process-reward.ts
|
|
999
|
-
async function extractStepRewards(store, runId, opts) {
|
|
1000
|
-
const spans = await store.spans({ runId });
|
|
1001
|
-
const ordered = [...spans].sort((a, b) => a.startedAt - b.startedAt);
|
|
1002
|
-
const out = [];
|
|
1003
|
-
let idx = 0;
|
|
1004
|
-
for (const span of ordered) {
|
|
1005
|
-
if (opts.preFilter && !opts.preFilter(span)) continue;
|
|
1006
|
-
let scored = null;
|
|
1007
|
-
for (const s of opts.scorers) {
|
|
1008
|
-
if (!s.appliesTo.includes(span.kind)) continue;
|
|
1009
|
-
const r = await s.score(span);
|
|
1010
|
-
if (r) {
|
|
1011
|
-
scored = r;
|
|
1012
|
-
break;
|
|
1013
|
-
}
|
|
1014
|
-
}
|
|
1015
|
-
if (!scored) continue;
|
|
1016
|
-
out.push({
|
|
1017
|
-
spanId: span.spanId,
|
|
1018
|
-
runId,
|
|
1019
|
-
stepIndex: idx++,
|
|
1020
|
-
kind: span.kind,
|
|
1021
|
-
name: span.name,
|
|
1022
|
-
reward: scored.reward,
|
|
1023
|
-
determinism: scored.determinism,
|
|
1024
|
-
rationale: scored.rationale,
|
|
1025
|
-
weight: scored.weight
|
|
1026
|
-
});
|
|
1027
|
-
}
|
|
1028
|
-
return out;
|
|
1029
|
-
}
|
|
1030
|
-
function runwiseStepRewardSummary(stepRewards) {
|
|
1031
|
-
if (stepRewards.length === 0) {
|
|
1032
|
-
return {
|
|
1033
|
-
runId: "",
|
|
1034
|
-
totalSteps: 0,
|
|
1035
|
-
meanReward: 0,
|
|
1036
|
-
sumWeightedReward: 0,
|
|
1037
|
-
failureFraction: 0,
|
|
1038
|
-
worstStepDelta: 0,
|
|
1039
|
-
worstStepIndex: null
|
|
1040
|
-
};
|
|
1041
|
-
}
|
|
1042
|
-
const runId = stepRewards[0].runId;
|
|
1043
|
-
let sumW = 0;
|
|
1044
|
-
let sumWR = 0;
|
|
1045
|
-
let failures = 0;
|
|
1046
|
-
let worstDelta = 0;
|
|
1047
|
-
let worstIdx = null;
|
|
1048
|
-
let prev = stepRewards[0].reward;
|
|
1049
|
-
for (let i = 0; i < stepRewards.length; i++) {
|
|
1050
|
-
const s = stepRewards[i];
|
|
1051
|
-
const w = s.weight ?? 1;
|
|
1052
|
-
sumW += w;
|
|
1053
|
-
sumWR += w * s.reward;
|
|
1054
|
-
if (s.reward < 0.5) failures++;
|
|
1055
|
-
if (i > 0) {
|
|
1056
|
-
const delta = s.reward - prev;
|
|
1057
|
-
if (delta < worstDelta) {
|
|
1058
|
-
worstDelta = delta;
|
|
1059
|
-
worstIdx = i;
|
|
1060
|
-
}
|
|
1061
|
-
prev = s.reward;
|
|
1062
|
-
} else {
|
|
1063
|
-
prev = s.reward;
|
|
1064
|
-
}
|
|
1065
|
-
}
|
|
1066
|
-
return {
|
|
1067
|
-
runId,
|
|
1068
|
-
totalSteps: stepRewards.length,
|
|
1069
|
-
meanReward: sumW === 0 ? 0 : sumWR / sumW,
|
|
1070
|
-
sumWeightedReward: sumWR,
|
|
1071
|
-
failureFraction: failures / stepRewards.length,
|
|
1072
|
-
worstStepDelta: worstDelta,
|
|
1073
|
-
worstStepIndex: worstIdx
|
|
1074
|
-
};
|
|
1075
|
-
}
|
|
1076
|
-
function prmTrainingPairs(stepRewardsByRun, opts = {}) {
|
|
1077
|
-
const minMargin = opts.minMargin ?? 0.2;
|
|
1078
|
-
const minPrefix = opts.minPrefixLength ?? 1;
|
|
1079
|
-
const runs = [...stepRewardsByRun.entries()].map(([runId, steps]) => ({ runId, steps }));
|
|
1080
|
-
const triples = [];
|
|
1081
|
-
for (let i = 0; i < runs.length; i++) {
|
|
1082
|
-
for (let j = i + 1; j < runs.length; j++) {
|
|
1083
|
-
const a = runs[i];
|
|
1084
|
-
const b = runs[j];
|
|
1085
|
-
const minLen = Math.min(a.steps.length, b.steps.length);
|
|
1086
|
-
if (minLen < minPrefix + 1) continue;
|
|
1087
|
-
let divergenceIdx = -1;
|
|
1088
|
-
for (let k = 0; k < minLen; k++) {
|
|
1089
|
-
const sa = a.steps[k];
|
|
1090
|
-
const sb = b.steps[k];
|
|
1091
|
-
const structuralDivergence = sa.kind !== sb.kind || sa.name !== sb.name;
|
|
1092
|
-
const rewardGap = Math.abs(sa.reward - sb.reward);
|
|
1093
|
-
if (structuralDivergence || rewardGap >= minMargin) {
|
|
1094
|
-
divergenceIdx = k;
|
|
1095
|
-
break;
|
|
1096
|
-
}
|
|
1097
|
-
}
|
|
1098
|
-
if (divergenceIdx < 0) continue;
|
|
1099
|
-
if (divergenceIdx < minPrefix) continue;
|
|
1100
|
-
const aNext = a.steps[divergenceIdx];
|
|
1101
|
-
const bNext = b.steps[divergenceIdx];
|
|
1102
|
-
const margin = Math.abs(aNext.reward - bNext.reward);
|
|
1103
|
-
if (margin < minMargin) continue;
|
|
1104
|
-
const chosen = aNext.reward > bNext.reward ? aNext : bNext;
|
|
1105
|
-
const rejected = aNext.reward > bNext.reward ? bNext : aNext;
|
|
1106
|
-
const chosenRun = aNext.reward > bNext.reward ? a.runId : b.runId;
|
|
1107
|
-
const rejectedRun = aNext.reward > bNext.reward ? b.runId : a.runId;
|
|
1108
|
-
triples.push({
|
|
1109
|
-
prefixRunId: chosenRun,
|
|
1110
|
-
prefixStepIndex: divergenceIdx - 1,
|
|
1111
|
-
chosenSpanId: chosen.spanId,
|
|
1112
|
-
chosenReward: chosen.reward,
|
|
1113
|
-
rejectedSpanId: rejected.spanId,
|
|
1114
|
-
rejectedReward: rejected.reward,
|
|
1115
|
-
rejectedRunId: rejectedRun,
|
|
1116
|
-
marginScore: chosen.reward - rejected.reward
|
|
1117
|
-
});
|
|
1118
|
-
}
|
|
1119
|
-
}
|
|
1120
|
-
return triples;
|
|
1121
|
-
}
|
|
1122
|
-
|
|
1123
|
-
// src/rl/rl-campaign.ts
|
|
1124
|
-
async function runRLCampaign(opts) {
|
|
1125
|
-
const campaign = await runEvalCampaign(opts);
|
|
1126
|
-
const rewardSignals = extractVerifiableRewardsFromRecords(
|
|
1127
|
-
campaign.runs,
|
|
1128
|
-
opts.verifiableReward ?? {}
|
|
1129
|
-
);
|
|
1130
|
-
const preferences = extractPreferences(campaign.runs, {
|
|
1131
|
-
strategy: opts.preferences?.strategy ?? "paired-by-scenario-and-seed",
|
|
1132
|
-
minMargin: opts.preferences?.minMargin ?? 0.05,
|
|
1133
|
-
splitTag: opts.preferences?.splitTag ?? opts.splitTag ?? "holdout",
|
|
1134
|
-
rewardOf: opts.preferences?.rewardOf
|
|
1135
|
-
});
|
|
1136
|
-
let interimConfidence = null;
|
|
1137
|
-
if (opts.report?.comparator) {
|
|
1138
|
-
const comparator = opts.report.comparator;
|
|
1139
|
-
const deltaSeries = collectPairedDeltaSeries(campaign.runs, comparator);
|
|
1140
|
-
if (deltaSeries.some((s) => s.deltas.length > 0)) {
|
|
1141
|
-
interimConfidence = evaluateInterimReleaseConfidence({
|
|
1142
|
-
deltaSeries,
|
|
1143
|
-
alpha: opts.sequential?.alpha,
|
|
1144
|
-
bound: opts.sequential?.bound,
|
|
1145
|
-
rope: opts.sequential?.rope ?? opts.report?.rope
|
|
1146
|
-
});
|
|
1147
|
-
}
|
|
1148
|
-
}
|
|
1149
|
-
const rewardHacking = detectRewardHacking({
|
|
1150
|
-
runs: campaign.runs,
|
|
1151
|
-
verifiableRewardOptions: opts.verifiableReward
|
|
1152
|
-
});
|
|
1153
|
-
let predictiveValidity = null;
|
|
1154
|
-
if (opts.outcomeStore && opts.outcomeMetrics && opts.outcomeMetrics.length > 0) {
|
|
1155
|
-
predictiveValidity = await rubricPredictiveValidity({
|
|
1156
|
-
runs: campaign.runs,
|
|
1157
|
-
outcomes: opts.outcomeStore,
|
|
1158
|
-
outcomeMetrics: opts.outcomeMetrics
|
|
1159
|
-
});
|
|
1160
|
-
}
|
|
1161
|
-
const trainerRows = {};
|
|
1162
|
-
if (opts.trainerExport?.dpo) {
|
|
1163
|
-
trainerRows.dpo = await toDpoRows(preferences.pairs, opts.trainerExport.dpo);
|
|
1164
|
-
}
|
|
1165
|
-
if (opts.trainerExport?.grpo) {
|
|
1166
|
-
trainerRows.grpo = await toGrpoRows(campaign.runs, opts.trainerExport.grpo);
|
|
1167
|
-
}
|
|
1168
|
-
if (opts.trainerExport?.sft) {
|
|
1169
|
-
trainerRows.sft = await toSftRows(campaign.runs, opts.trainerExport.sft);
|
|
1170
|
-
}
|
|
1171
|
-
const summary = buildSummary({
|
|
1172
|
-
campaign,
|
|
1173
|
-
preferences,
|
|
1174
|
-
interimConfidence,
|
|
1175
|
-
rewardHacking,
|
|
1176
|
-
predictiveValidity
|
|
1177
|
-
});
|
|
1178
|
-
return {
|
|
1179
|
-
campaign,
|
|
1180
|
-
rewardSignals,
|
|
1181
|
-
preferences,
|
|
1182
|
-
interimConfidence,
|
|
1183
|
-
rewardHacking,
|
|
1184
|
-
predictiveValidity,
|
|
1185
|
-
trainerRows,
|
|
1186
|
-
summary,
|
|
1187
|
-
kind: "agent-eval-rl-campaign"
|
|
1188
|
-
};
|
|
1189
|
-
}
|
|
1190
|
-
function collectPairedDeltaSeries(runs, comparator) {
|
|
1191
|
-
const baseline = /* @__PURE__ */ new Map();
|
|
1192
|
-
for (const r of runs) {
|
|
1193
|
-
if (r.candidateId !== comparator) continue;
|
|
1194
|
-
const sid = r.scenarioId ?? r.experimentId;
|
|
1195
|
-
const score = r.outcome.holdoutScore ?? r.outcome.searchScore;
|
|
1196
|
-
if (typeof score !== "number" || !Number.isFinite(score)) continue;
|
|
1197
|
-
baseline.set(`${sid}::${r.seed}`, score);
|
|
1198
|
-
}
|
|
1199
|
-
const byCandidate = /* @__PURE__ */ new Map();
|
|
1200
|
-
for (const r of runs) {
|
|
1201
|
-
if (r.candidateId === comparator) continue;
|
|
1202
|
-
const sid = r.scenarioId ?? r.experimentId;
|
|
1203
|
-
const score = r.outcome.holdoutScore ?? r.outcome.searchScore;
|
|
1204
|
-
if (typeof score !== "number" || !Number.isFinite(score)) continue;
|
|
1205
|
-
const baseScore = baseline.get(`${sid}::${r.seed}`);
|
|
1206
|
-
if (typeof baseScore !== "number") continue;
|
|
1207
|
-
const arr = byCandidate.get(r.candidateId) ?? [];
|
|
1208
|
-
arr.push(score - baseScore);
|
|
1209
|
-
byCandidate.set(r.candidateId, arr);
|
|
1210
|
-
}
|
|
1211
|
-
return [...byCandidate.entries()].map(([candidateId, deltas]) => ({ candidateId, deltas }));
|
|
1212
|
-
}
|
|
1213
|
-
function buildSummary(args) {
|
|
1214
|
-
const c = args.campaign;
|
|
1215
|
-
const lines = [
|
|
1216
|
-
`${c.campaignId}: ${c.runs.length} successful runs / ${c.failedRuns.length} failed (fingerprint ${c.campaignFingerprint.slice(0, 12)}\u2026)`,
|
|
1217
|
-
`preferences: ${args.preferences.pairs.length} (${args.preferences.strategy}, ${args.preferences.pairsBelowMargin} below margin)`
|
|
1218
|
-
];
|
|
1219
|
-
if (args.interimConfidence) {
|
|
1220
|
-
lines.push(
|
|
1221
|
-
`sequential verdict: ${args.interimConfidence.recommendation.decision}` + (args.interimConfidence.recommendation.candidateId ? ` ${args.interimConfidence.recommendation.candidateId}` : "")
|
|
1222
|
-
);
|
|
1223
|
-
}
|
|
1224
|
-
lines.push(
|
|
1225
|
-
`reward-hacking: ${args.rewardHacking.verdict} (${args.rewardHacking.findings.length} signals checked)`
|
|
1226
|
-
);
|
|
1227
|
-
if (args.predictiveValidity) {
|
|
1228
|
-
const top = args.predictiveValidity.ranked[0];
|
|
1229
|
-
lines.push(
|
|
1230
|
-
`top-rubric: ${top?.rubric ?? "none"} \u03C1=${(top?.spearman ?? 0).toFixed(2)} (${top?.verdict ?? "no data"})`
|
|
1231
|
-
);
|
|
1232
|
-
}
|
|
1233
|
-
return lines.join(" | ");
|
|
1234
|
-
}
|
|
1235
|
-
|
|
1236
|
-
// src/rl/run-record-adapters.ts
|
|
1237
|
-
function campaignToRunRecords(campaign, ctx) {
|
|
1238
|
-
const splitTag = ctx.splitTag ?? "search";
|
|
1239
|
-
const candidateId = ctx.candidateId ?? campaign.manifestHash;
|
|
1240
|
-
return campaign.cells.map((cell) => {
|
|
1241
|
-
const composites = Object.values(cell.judgeScores).map((s) => s.composite);
|
|
1242
|
-
const score = composites.length > 0 ? composites.reduce((a, b) => a + b, 0) / composites.length : 0;
|
|
1243
|
-
const raw = { rep: cell.rep, duration_ms: cell.durationMs };
|
|
1244
|
-
for (const judge of Object.values(cell.judgeScores)) {
|
|
1245
|
-
for (const [dim, value] of Object.entries(judge.dimensions)) {
|
|
1246
|
-
if (Number.isFinite(value)) raw[`dim.${dim}`] = value;
|
|
1247
|
-
}
|
|
1248
|
-
}
|
|
1249
|
-
if (typeof cell.generation === "number") raw.generation = cell.generation;
|
|
1250
|
-
const outcome = { raw };
|
|
1251
|
-
if (splitTag === "holdout") outcome.holdoutScore = score;
|
|
1252
|
-
else outcome.searchScore = score;
|
|
1253
|
-
return {
|
|
1254
|
-
runId: cell.cellId,
|
|
1255
|
-
experimentId: ctx.experimentId,
|
|
1256
|
-
candidateId,
|
|
1257
|
-
seed: cell.seed,
|
|
1258
|
-
model: ctx.model,
|
|
1259
|
-
promptHash: ctx.promptHash,
|
|
1260
|
-
configHash: ctx.configHash,
|
|
1261
|
-
commitSha: ctx.commitSha,
|
|
1262
|
-
wallMs: cell.durationMs,
|
|
1263
|
-
costUsd: Number.isFinite(cell.costUsd) ? cell.costUsd : ctx.defaultCostUsd ?? 0,
|
|
1264
|
-
tokenUsage: { input: 0, output: 0 },
|
|
1265
|
-
outcome,
|
|
1266
|
-
failureMode: cell.error ? "cell_error" : void 0,
|
|
1267
|
-
splitTag,
|
|
1268
|
-
scenarioId: cell.scenarioId
|
|
1269
|
-
};
|
|
1270
|
-
});
|
|
1271
|
-
}
|
|
1272
|
-
function verificationReportToRunRecord(report, ctx, opts = {}) {
|
|
1273
|
-
const splitTag = ctx.splitTag ?? "search";
|
|
1274
|
-
const runId = opts.runId ?? `run-${ctx.candidateId}-${ctx.experimentId}-${report.startedAt}`;
|
|
1275
|
-
const raw = {
|
|
1276
|
-
pass_count: report.passCount,
|
|
1277
|
-
fail_count: report.failCount,
|
|
1278
|
-
error_count: report.errorCount,
|
|
1279
|
-
skipped_count: report.skippedCount,
|
|
1280
|
-
duration_ms: report.durationMs,
|
|
1281
|
-
blended_score: report.blendedScore
|
|
1282
|
-
};
|
|
1283
|
-
for (const layer of report.layers) {
|
|
1284
|
-
if (typeof layer.score === "number") raw[`layer.${layer.layer}`] = layer.score;
|
|
1285
|
-
raw[`layer_${layer.layer}_pass`] = layer.status === "pass" ? 1 : 0;
|
|
1286
|
-
if (layer.diagnostics) {
|
|
1287
|
-
for (const [k, v] of Object.entries(layer.diagnostics)) {
|
|
1288
|
-
if (typeof v === "number" && Number.isFinite(v)) raw[`layer.${layer.layer}.${k}`] = v;
|
|
1289
|
-
}
|
|
1290
|
-
}
|
|
1291
|
-
}
|
|
1292
|
-
const firstFail = report.layers.find((l) => l.status === "fail" || l.status === "error");
|
|
1293
|
-
const outcome = { raw };
|
|
1294
|
-
if (splitTag === "holdout") outcome.holdoutScore = report.blendedScore;
|
|
1295
|
-
else outcome.searchScore = report.blendedScore;
|
|
1296
|
-
return {
|
|
1297
|
-
runId,
|
|
1298
|
-
experimentId: ctx.experimentId,
|
|
1299
|
-
candidateId: ctx.candidateId,
|
|
1300
|
-
seed: 0,
|
|
1301
|
-
model: ctx.model,
|
|
1302
|
-
promptHash: ctx.promptHash,
|
|
1303
|
-
configHash: ctx.configHash,
|
|
1304
|
-
commitSha: ctx.commitSha,
|
|
1305
|
-
wallMs: report.durationMs,
|
|
1306
|
-
costUsd: ctx.defaultCostUsd ?? 0,
|
|
1307
|
-
tokenUsage: { input: 0, output: 0 },
|
|
1308
|
-
outcome,
|
|
1309
|
-
failureMode: firstFail ? failureModeFromLayer(firstFail) : void 0,
|
|
1310
|
-
splitTag,
|
|
1311
|
-
scenarioId: ctx.scenarioId
|
|
1312
|
-
};
|
|
1313
|
-
}
|
|
1314
|
-
function failureModeFromLayer(layer) {
|
|
1315
|
-
if (layer.status === "error") return `layer_${layer.layer}_error`;
|
|
1316
|
-
if (layer.status === "fail") return `layer_${layer.layer}_fail`;
|
|
1317
|
-
if (layer.status === "timeout") return `layer_${layer.layer}_timeout`;
|
|
1318
|
-
return `layer_${layer.layer}_${layer.status}`;
|
|
1319
|
-
}
|
|
1320
|
-
|
|
1321
|
-
// src/rl/sim-fidelity.ts
|
|
1322
|
-
var ABSENT_CATEGORY = "(absent)";
|
|
1323
|
-
var DEFAULT_MIN_N_PER_FEATURE = 20;
|
|
1324
|
-
var DEFAULT_QUANTILE_BUCKETS = 4;
|
|
1325
|
-
var REPRESENTATIVE_MIN_FIDELITY = 0.8;
|
|
1326
|
-
var TOP_SHIFT_COUNT = 5;
|
|
1327
|
-
var defaultBehaviorFeatures = (record) => {
|
|
1328
|
-
const raw = record.outcome?.raw ?? {};
|
|
1329
|
-
const toolErrors = finiteOrNull(raw.tool_errors);
|
|
1330
|
-
const turnsAborted = finiteOrNull(raw.turns_aborted);
|
|
1331
|
-
const completion = record.completion;
|
|
1332
|
-
return {
|
|
1333
|
-
score: finiteOrNull(record.outcome?.holdoutScore) ?? finiteOrNull(record.outcome?.searchScore),
|
|
1334
|
-
failure_class: record.failureClass ?? null,
|
|
1335
|
-
wall_ms: finiteOrNull(record.wallMs),
|
|
1336
|
-
output_tokens: finiteOrNull(record.tokenUsage?.output),
|
|
1337
|
-
turn_count: finiteOrNull(raw.turns_completed) ?? finiteOrNull(raw.assistant_messages),
|
|
1338
|
-
tool_errors: toolErrors,
|
|
1339
|
-
tool_error_recovery: toolErrorRecovery(toolErrors, turnsAborted, record.failureClass),
|
|
1340
|
-
completion_length: typeof completion === "string" ? completion.length : null
|
|
1341
|
-
};
|
|
1342
|
-
};
|
|
1343
|
-
function toolErrorRecovery(toolErrors, turnsAborted, failureClass) {
|
|
1344
|
-
if (toolErrors === null) return null;
|
|
1345
|
-
if (toolErrors === 0) return "no-tool-errors";
|
|
1346
|
-
const failed = (turnsAborted ?? 0) > 0 || failureClass !== void 0 && failureClass !== "success";
|
|
1347
|
-
return failed ? "unrecovered" : "recovered";
|
|
1348
|
-
}
|
|
1349
|
-
function finiteOrNull(value) {
|
|
1350
|
-
return typeof value === "number" && Number.isFinite(value) ? value : null;
|
|
1351
|
-
}
|
|
1352
|
-
function jsDivergence(p, q) {
|
|
1353
|
-
const keys = /* @__PURE__ */ new Set([...Object.keys(p), ...Object.keys(q)]);
|
|
1354
|
-
if (keys.size === 0) {
|
|
1355
|
-
throw new ValidationError("jsDivergence: both histograms are empty");
|
|
1356
|
-
}
|
|
1357
|
-
let pSum = 0;
|
|
1358
|
-
let qSum = 0;
|
|
1359
|
-
for (const key of keys) {
|
|
1360
|
-
const pv = p[key] ?? 0;
|
|
1361
|
-
const qv = q[key] ?? 0;
|
|
1362
|
-
if (!Number.isFinite(pv) || !Number.isFinite(qv) || pv < 0 || qv < 0) {
|
|
1363
|
-
throw new ValidationError(`jsDivergence: negative or non-finite count for category "${key}"`);
|
|
1364
|
-
}
|
|
1365
|
-
pSum += pv;
|
|
1366
|
-
qSum += qv;
|
|
1367
|
-
}
|
|
1368
|
-
if (pSum === 0 || qSum === 0) {
|
|
1369
|
-
throw new ValidationError("jsDivergence: a histogram with zero total mass has no distribution");
|
|
1370
|
-
}
|
|
1371
|
-
let divergence = 0;
|
|
1372
|
-
for (const key of keys) {
|
|
1373
|
-
const pp = (p[key] ?? 0) / pSum;
|
|
1374
|
-
const qp = (q[key] ?? 0) / qSum;
|
|
1375
|
-
const m = (pp + qp) / 2;
|
|
1376
|
-
if (pp > 0) divergence += 0.5 * pp * Math.log2(pp / m);
|
|
1377
|
-
if (qp > 0) divergence += 0.5 * qp * Math.log2(qp / m);
|
|
1378
|
-
}
|
|
1379
|
-
return Math.min(1, Math.max(0, divergence));
|
|
1380
|
-
}
|
|
1381
|
-
function quantileEdges(values, bucketCount = DEFAULT_QUANTILE_BUCKETS) {
|
|
1382
|
-
if (values.length === 0) {
|
|
1383
|
-
throw new ValidationError("quantileEdges: requires at least one value");
|
|
1384
|
-
}
|
|
1385
|
-
if (!Number.isInteger(bucketCount) || bucketCount < 2) {
|
|
1386
|
-
throw new ValidationError(
|
|
1387
|
-
`quantileEdges: bucketCount must be an integer >= 2, got ${bucketCount}`
|
|
1388
|
-
);
|
|
1389
|
-
}
|
|
1390
|
-
const sorted = [...values].sort((a, b) => a - b);
|
|
1391
|
-
const edges = [];
|
|
1392
|
-
for (let k = 1; k < bucketCount; k++) {
|
|
1393
|
-
const pos = k / bucketCount * (sorted.length - 1);
|
|
1394
|
-
const lo = sorted[Math.floor(pos)];
|
|
1395
|
-
const hi = sorted[Math.ceil(pos)];
|
|
1396
|
-
edges.push(lo + (pos - Math.floor(pos)) * (hi - lo));
|
|
1397
|
-
}
|
|
1398
|
-
return [...new Set(edges)];
|
|
1399
|
-
}
|
|
1400
|
-
function bucketLabel(value, edges) {
|
|
1401
|
-
let i = 0;
|
|
1402
|
-
while (i < edges.length && value >= edges[i]) i++;
|
|
1403
|
-
const lo = i === 0 ? "-inf" : String(edges[i - 1]);
|
|
1404
|
-
const hi = i === edges.length ? "+inf" : String(edges[i]);
|
|
1405
|
-
return `[${lo},${hi})`;
|
|
1406
|
-
}
|
|
1407
|
-
function simFidelityReport(simulated, production, opts = {}) {
|
|
1408
|
-
if (simulated.length === 0) {
|
|
1409
|
-
throw new ValidationError("simFidelityReport: simulated records are empty");
|
|
1410
|
-
}
|
|
1411
|
-
if (production.length === 0) {
|
|
1412
|
-
throw new ValidationError("simFidelityReport: production records are empty");
|
|
1413
|
-
}
|
|
1414
|
-
const extract = opts.features ?? defaultBehaviorFeatures;
|
|
1415
|
-
const minN = opts.minNPerFeature ?? DEFAULT_MIN_N_PER_FEATURE;
|
|
1416
|
-
const simMaps = simulated.map(extract);
|
|
1417
|
-
const prodMaps = production.map(extract);
|
|
1418
|
-
const featureNames = [];
|
|
1419
|
-
const seen = /* @__PURE__ */ new Set();
|
|
1420
|
-
for (const map of [...simMaps, ...prodMaps]) {
|
|
1421
|
-
for (const name of Object.keys(map)) {
|
|
1422
|
-
if (!seen.has(name)) {
|
|
1423
|
-
seen.add(name);
|
|
1424
|
-
featureNames.push(name);
|
|
1425
|
-
}
|
|
1426
|
-
}
|
|
1427
|
-
}
|
|
1428
|
-
const perDimension = [];
|
|
1429
|
-
const insufficientData = [];
|
|
1430
|
-
for (const feature of featureNames) {
|
|
1431
|
-
const simVals = simMaps.map((m) => m[feature] ?? null);
|
|
1432
|
-
const prodVals = prodMaps.map((m) => m[feature] ?? null);
|
|
1433
|
-
const nSim = simVals.filter((v) => v !== null).length;
|
|
1434
|
-
const nProd = prodVals.filter((v) => v !== null).length;
|
|
1435
|
-
if (nSim < minN || nProd < minN) {
|
|
1436
|
-
insufficientData.push(feature);
|
|
1437
|
-
continue;
|
|
1438
|
-
}
|
|
1439
|
-
const { sim, prod } = histograms(feature, simVals, prodVals);
|
|
1440
|
-
perDimension.push({
|
|
1441
|
-
feature,
|
|
1442
|
-
divergence: jsDivergence(sim, prod),
|
|
1443
|
-
topShifts: topShifts(sim, simVals.length, prod, prodVals.length),
|
|
1444
|
-
nSim,
|
|
1445
|
-
nProd
|
|
1446
|
-
});
|
|
1447
|
-
}
|
|
1448
|
-
if (perDimension.length === 0) {
|
|
1449
|
-
return { perDimension, fidelity: Number.NaN, insufficientData, verdict: "insufficient-data" };
|
|
1450
|
-
}
|
|
1451
|
-
const fidelity = 1 - perDimension.reduce((sum, d) => sum + d.divergence, 0) / perDimension.length;
|
|
1452
|
-
return {
|
|
1453
|
-
perDimension,
|
|
1454
|
-
fidelity,
|
|
1455
|
-
insufficientData,
|
|
1456
|
-
verdict: fidelity >= REPRESENTATIVE_MIN_FIDELITY ? "representative" : "skewed"
|
|
1457
|
-
};
|
|
1458
|
-
}
|
|
1459
|
-
function histograms(feature, simVals, prodVals) {
|
|
1460
|
-
const kinds = /* @__PURE__ */ new Set();
|
|
1461
|
-
for (const v of [...simVals, ...prodVals]) {
|
|
1462
|
-
if (v !== null) kinds.add(typeof v);
|
|
1463
|
-
}
|
|
1464
|
-
if (kinds.size > 1) {
|
|
1465
|
-
throw new ValidationError(
|
|
1466
|
-
`simFidelityReport: feature "${feature}" mixes string and number values \u2014 an extractor must return one kind per feature`
|
|
1467
|
-
);
|
|
1468
|
-
}
|
|
1469
|
-
let toCategory;
|
|
1470
|
-
if (kinds.has("number")) {
|
|
1471
|
-
const union = [];
|
|
1472
|
-
for (const v of [...simVals, ...prodVals]) {
|
|
1473
|
-
if (v !== null) union.push(v);
|
|
1474
|
-
}
|
|
1475
|
-
const edges = quantileEdges(union);
|
|
1476
|
-
toCategory = (v) => bucketLabel(v, edges);
|
|
1477
|
-
} else {
|
|
1478
|
-
toCategory = (v) => v;
|
|
1479
|
-
}
|
|
1480
|
-
const count = (vals) => {
|
|
1481
|
-
const hist = {};
|
|
1482
|
-
for (const v of vals) {
|
|
1483
|
-
const key = v === null ? ABSENT_CATEGORY : toCategory(v);
|
|
1484
|
-
hist[key] = (hist[key] ?? 0) + 1;
|
|
1485
|
-
}
|
|
1486
|
-
return hist;
|
|
1487
|
-
};
|
|
1488
|
-
return { sim: count(simVals), prod: count(prodVals) };
|
|
1489
|
-
}
|
|
1490
|
-
function topShifts(sim, simTotal, prod, prodTotal) {
|
|
1491
|
-
const keys = [.../* @__PURE__ */ new Set([...Object.keys(sim), ...Object.keys(prod)])];
|
|
1492
|
-
const shifts = keys.map((value) => ({
|
|
1493
|
-
value,
|
|
1494
|
-
pSim: (sim[value] ?? 0) / simTotal,
|
|
1495
|
-
pProd: (prod[value] ?? 0) / prodTotal
|
|
1496
|
-
}));
|
|
1497
|
-
shifts.sort((a, b) => {
|
|
1498
|
-
const delta = Math.abs(b.pSim - b.pProd) - Math.abs(a.pSim - a.pProd);
|
|
1499
|
-
return delta !== 0 ? delta : a.value.localeCompare(b.value);
|
|
1500
|
-
});
|
|
1501
|
-
return shifts.slice(0, TOP_SHIFT_COUNT);
|
|
1502
|
-
}
|
|
1503
|
-
function easyModeCheck(simulated, production, opts = {}) {
|
|
1504
|
-
if (simulated.length === 0) {
|
|
1505
|
-
throw new ValidationError("easyModeCheck: simulated records are empty");
|
|
1506
|
-
}
|
|
1507
|
-
if (production.length === 0) {
|
|
1508
|
-
throw new ValidationError("easyModeCheck: production records are empty");
|
|
1509
|
-
}
|
|
1510
|
-
const threshold = opts.passThreshold ?? 0.5;
|
|
1511
|
-
const tolerance = opts.inflationTolerance ?? 0.1;
|
|
1512
|
-
const passRate = (records, side) => {
|
|
1513
|
-
let passes = 0;
|
|
1514
|
-
for (const r of records) {
|
|
1515
|
-
const score = finiteOrNull(r.outcome?.holdoutScore) ?? finiteOrNull(r.outcome?.searchScore);
|
|
1516
|
-
if (score === null) {
|
|
1517
|
-
throw new ValidationError(
|
|
1518
|
-
`easyModeCheck: ${side} run "${r.runId}" carries neither holdoutScore nor searchScore`
|
|
1519
|
-
);
|
|
1520
|
-
}
|
|
1521
|
-
if (score >= threshold) passes++;
|
|
1522
|
-
}
|
|
1523
|
-
return passes / records.length;
|
|
1524
|
-
};
|
|
1525
|
-
const simPassRate = passRate(simulated, "simulated");
|
|
1526
|
-
const prodPassRate = passRate(production, "production");
|
|
1527
|
-
const gap = simPassRate - prodPassRate;
|
|
1528
|
-
return { simPassRate, prodPassRate, gap, inflated: gap > tolerance };
|
|
1529
|
-
}
|
|
1530
|
-
|
|
1531
|
-
// src/rl/tournament.ts
|
|
1532
|
-
function fitBradleyTerry(outcomes, opts = {}) {
|
|
1533
|
-
const tol = opts.tolerance ?? 1e-6;
|
|
1534
|
-
const maxIter = opts.maxIterations ?? 256;
|
|
1535
|
-
const smoothing = opts.smoothing ?? 0.1;
|
|
1536
|
-
const candidates = /* @__PURE__ */ new Set();
|
|
1537
|
-
for (const o of outcomes) {
|
|
1538
|
-
candidates.add(o.winner);
|
|
1539
|
-
candidates.add(o.loser);
|
|
1540
|
-
}
|
|
1541
|
-
const ids = [...candidates].sort();
|
|
1542
|
-
const idx = new Map(ids.map((id, i) => [id, i]));
|
|
1543
|
-
const n = ids.length;
|
|
1544
|
-
if (n === 0) return { ratings: [], iterations: 0, finalDelta: 0, converged: true };
|
|
1545
|
-
if (n === 1) {
|
|
1546
|
-
return {
|
|
1547
|
-
ratings: [{ candidateId: ids[0], strength: 1, logStrength: 0, n: 0, wins: 0 }],
|
|
1548
|
-
iterations: 0,
|
|
1549
|
-
finalDelta: 0,
|
|
1550
|
-
converged: true
|
|
1551
|
-
};
|
|
1552
|
-
}
|
|
1553
|
-
const W = Array.from({ length: n }, () => new Array(n).fill(0));
|
|
1554
|
-
const N = Array.from({ length: n }, () => new Array(n).fill(0));
|
|
1555
|
-
for (const o of outcomes) {
|
|
1556
|
-
const i = idx.get(o.winner);
|
|
1557
|
-
const j = idx.get(o.loser);
|
|
1558
|
-
const w = o.weight ?? 1;
|
|
1559
|
-
if (o.draw) {
|
|
1560
|
-
W[i][j] += 0.5 * w;
|
|
1561
|
-
W[j][i] += 0.5 * w;
|
|
1562
|
-
} else {
|
|
1563
|
-
W[i][j] += w;
|
|
1564
|
-
}
|
|
1565
|
-
N[i][j] += w;
|
|
1566
|
-
N[j][i] += w;
|
|
1567
|
-
}
|
|
1568
|
-
const winsTotal = new Array(n).fill(0);
|
|
1569
|
-
for (let i = 0; i < n; i++) {
|
|
1570
|
-
for (let j = 0; j < n; j++) winsTotal[i] += W[i][j];
|
|
1571
|
-
winsTotal[i] += smoothing;
|
|
1572
|
-
}
|
|
1573
|
-
const compsTotal = new Array(n).fill(0);
|
|
1574
|
-
for (let i = 0; i < n; i++) {
|
|
1575
|
-
for (let j = 0; j < n; j++) compsTotal[i] += N[i][j];
|
|
1576
|
-
}
|
|
1577
|
-
let theta = new Array(n).fill(1);
|
|
1578
|
-
let iter = 0;
|
|
1579
|
-
let delta = Infinity;
|
|
1580
|
-
for (; iter < maxIter; iter++) {
|
|
1581
|
-
const newTheta = new Array(n);
|
|
1582
|
-
for (let i = 0; i < n; i++) {
|
|
1583
|
-
let denom = 0;
|
|
1584
|
-
for (let j = 0; j < n; j++) {
|
|
1585
|
-
if (j === i) continue;
|
|
1586
|
-
if (N[i][j] === 0) continue;
|
|
1587
|
-
denom += N[i][j] / (theta[i] + theta[j]);
|
|
1588
|
-
}
|
|
1589
|
-
newTheta[i] = denom === 0 ? theta[i] : winsTotal[i] / denom;
|
|
1590
|
-
}
|
|
1591
|
-
let logSum = 0;
|
|
1592
|
-
for (let i = 0; i < n; i++) logSum += Math.log(Math.max(1e-300, newTheta[i]));
|
|
1593
|
-
const norm = Math.exp(logSum / n);
|
|
1594
|
-
for (let i = 0; i < n; i++) newTheta[i] = newTheta[i] / norm;
|
|
1595
|
-
delta = 0;
|
|
1596
|
-
for (let i = 0; i < n; i++) {
|
|
1597
|
-
const d = Math.abs(newTheta[i] - theta[i]) / Math.max(1e-12, theta[i]);
|
|
1598
|
-
if (d > delta) delta = d;
|
|
1599
|
-
}
|
|
1600
|
-
theta = newTheta;
|
|
1601
|
-
if (delta < tol) break;
|
|
1602
|
-
}
|
|
1603
|
-
const minLog = Math.min(...theta.map((t) => Math.log(Math.max(1e-300, t))));
|
|
1604
|
-
const ratings = ids.map((id, i) => ({
|
|
1605
|
-
candidateId: id,
|
|
1606
|
-
strength: theta[i],
|
|
1607
|
-
logStrength: Math.log(Math.max(1e-300, theta[i])) - minLog,
|
|
1608
|
-
n: compsTotal[i],
|
|
1609
|
-
wins: winsTotal[i] - smoothing
|
|
1610
|
-
}));
|
|
1611
|
-
return {
|
|
1612
|
-
ratings: ratings.sort((a, b) => b.strength - a.strength),
|
|
1613
|
-
iterations: iter,
|
|
1614
|
-
finalDelta: delta,
|
|
1615
|
-
converged: delta < tol
|
|
1616
|
-
};
|
|
1617
|
-
}
|
|
1618
|
-
function applyEloUpdate(ratings, outcome, opts = {}) {
|
|
1619
|
-
const defaultRating = opts.defaultRating ?? 1500;
|
|
1620
|
-
const k = opts.kFactor ?? 32;
|
|
1621
|
-
const rW = ratings.get(outcome.winner) ?? defaultRating;
|
|
1622
|
-
const rL = ratings.get(outcome.loser) ?? defaultRating;
|
|
1623
|
-
const expectedW = 1 / (1 + 10 ** ((rL - rW) / 400));
|
|
1624
|
-
const scoreW = outcome.draw ? 0.5 : 1;
|
|
1625
|
-
const scoreL = outcome.draw ? 0.5 : 0;
|
|
1626
|
-
const w = outcome.weight ?? 1;
|
|
1627
|
-
const winnerDelta = k * w * (scoreW - expectedW);
|
|
1628
|
-
const loserDelta = k * w * (scoreL - (1 - expectedW));
|
|
1629
|
-
ratings.set(outcome.winner, rW + winnerDelta);
|
|
1630
|
-
ratings.set(outcome.loser, rL + loserDelta);
|
|
1631
|
-
return { winnerDelta, loserDelta };
|
|
1632
|
-
}
|
|
1633
|
-
function buildPairwiseFromCampaign(input) {
|
|
1634
|
-
const drawMargin = input.drawMargin ?? 0;
|
|
1635
|
-
const byKey = /* @__PURE__ */ new Map();
|
|
1636
|
-
for (const r of input.runs) {
|
|
1637
|
-
const arr = byKey.get(r.matchKey) ?? [];
|
|
1638
|
-
arr.push({ candidateId: r.candidateId, score: r.score });
|
|
1639
|
-
byKey.set(r.matchKey, arr);
|
|
1640
|
-
}
|
|
1641
|
-
const outcomes = [];
|
|
1642
|
-
for (const arr of byKey.values()) {
|
|
1643
|
-
for (let i = 0; i < arr.length; i++) {
|
|
1644
|
-
for (let j = i + 1; j < arr.length; j++) {
|
|
1645
|
-
const a = arr[i];
|
|
1646
|
-
const b = arr[j];
|
|
1647
|
-
if (a.candidateId === b.candidateId) continue;
|
|
1648
|
-
const margin = Math.abs(a.score - b.score);
|
|
1649
|
-
if (margin <= drawMargin) {
|
|
1650
|
-
outcomes.push({ winner: a.candidateId, loser: b.candidateId, draw: true, weight: 1 });
|
|
1651
|
-
} else {
|
|
1652
|
-
const [winner, loser] = a.score > b.score ? [a, b] : [b, a];
|
|
1653
|
-
outcomes.push({ winner: winner.candidateId, loser: loser.candidateId, weight: margin });
|
|
1654
|
-
}
|
|
1655
|
-
}
|
|
1656
|
-
}
|
|
1657
|
-
}
|
|
1658
|
-
return outcomes;
|
|
1659
|
-
}
|
|
1660
|
-
export {
|
|
1661
|
-
ABSENT_CATEGORY,
|
|
1662
|
-
DEFAULT_MIN_N_PER_FEATURE,
|
|
1663
|
-
DEFAULT_QUANTILE_BUCKETS,
|
|
1664
|
-
FileSystemOutcomeStore,
|
|
1665
|
-
InMemoryOutcomeStore,
|
|
1666
|
-
PredictiveValidityResearcher,
|
|
1667
|
-
REPRESENTATIVE_MIN_FIDELITY,
|
|
1668
|
-
appendToCorpus,
|
|
1669
|
-
applyEloUpdate,
|
|
1670
|
-
bestOfN,
|
|
1671
|
-
bucketLabel,
|
|
1672
|
-
buildDatasetFromCorpus,
|
|
1673
|
-
buildPairwiseFromCampaign,
|
|
1674
|
-
buildRlDataset,
|
|
1675
|
-
campaignToRunRecords,
|
|
1676
|
-
compareAdaptationCurves,
|
|
1677
|
-
datasheetToMarkdown,
|
|
1678
|
-
defaultBehaviorFeatures,
|
|
1679
|
-
detectRewardHacking,
|
|
1680
|
-
doublyRobust,
|
|
1681
|
-
easyModeCheck,
|
|
1682
|
-
extractPreferences,
|
|
1683
|
-
extractStepRewards,
|
|
1684
|
-
extractVerifiableReward,
|
|
1685
|
-
extractVerifiableRewardsFromRecords,
|
|
1686
|
-
filterDeterministicallyRewarded,
|
|
1687
|
-
firstPassK,
|
|
1688
|
-
fitBradleyTerry,
|
|
1689
|
-
injectIrrelevantClause,
|
|
1690
|
-
inverseProbabilityWeighting,
|
|
1691
|
-
jsDivergence,
|
|
1692
|
-
observationsFromRunRecords,
|
|
1693
|
-
offPolicyEstimateAll,
|
|
1694
|
-
paretoFrontier,
|
|
1695
|
-
prmTrainingPairs,
|
|
1696
|
-
quantileEdges,
|
|
1697
|
-
readCorpus,
|
|
1698
|
-
renameVariables,
|
|
1699
|
-
runAdaptationCurve,
|
|
1700
|
-
runComputeCurve,
|
|
1701
|
-
runContaminationProbe,
|
|
1702
|
-
runEvalCampaign,
|
|
1703
|
-
runRLCampaign,
|
|
1704
|
-
runwiseStepRewardSummary,
|
|
1705
|
-
selfConsistency,
|
|
1706
|
-
selfNormalizedImportanceWeighting,
|
|
1707
|
-
shuffleOrder,
|
|
1708
|
-
simFidelityReport,
|
|
1709
|
-
stepRewardsToJsonl,
|
|
1710
|
-
thompsonCurriculum,
|
|
1711
|
-
toAnthropicFormat,
|
|
1712
|
-
toDpoJsonl,
|
|
1713
|
-
toDpoRows,
|
|
1714
|
-
toGrpoJsonl,
|
|
1715
|
-
toGrpoRows,
|
|
1716
|
-
toPrmJsonl,
|
|
1717
|
-
toPrmRows,
|
|
1718
|
-
toSftJsonl,
|
|
1719
|
-
toSftRows,
|
|
1720
|
-
toTRLFormat,
|
|
1721
|
-
varianceBasedCurriculum,
|
|
1722
|
-
verificationReportToRunRecord
|
|
1723
|
-
};
|
|
1724
|
-
//# sourceMappingURL=rl.js.map
|