@tangle-network/agent-eval 0.120.0 → 0.120.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/CHANGELOG.md +12 -0
  2. package/package.json +1 -1
  3. package/dist/analyst/index.d.ts +0 -3111
  4. package/dist/analyst/index.js +0 -403
  5. package/dist/analyst/index.js.map +0 -1
  6. package/dist/authenticity/index.d.ts +0 -161
  7. package/dist/authenticity/index.js +0 -215
  8. package/dist/authenticity/index.js.map +0 -1
  9. package/dist/belief-state/index.d.ts +0 -1301
  10. package/dist/belief-state/index.js +0 -2152
  11. package/dist/belief-state/index.js.map +0 -1
  12. package/dist/benchmarks/index.d.ts +0 -974
  13. package/dist/benchmarks/index.js +0 -60
  14. package/dist/benchmarks/index.js.map +0 -1
  15. package/dist/builder-eval/index.d.ts +0 -695
  16. package/dist/builder-eval/index.js +0 -366
  17. package/dist/builder-eval/index.js.map +0 -1
  18. package/dist/campaign/index.d.ts +0 -7454
  19. package/dist/campaign/index.js +0 -272
  20. package/dist/campaign/index.js.map +0 -1
  21. package/dist/chunk-32BZXMSO.js +0 -3878
  22. package/dist/chunk-32BZXMSO.js.map +0 -1
  23. package/dist/chunk-3A246TSA.js +0 -998
  24. package/dist/chunk-3A246TSA.js.map +0 -1
  25. package/dist/chunk-3RF76KTD.js +0 -84
  26. package/dist/chunk-3RF76KTD.js.map +0 -1
  27. package/dist/chunk-3YYRZDON.js +0 -45
  28. package/dist/chunk-3YYRZDON.js.map +0 -1
  29. package/dist/chunk-4I2E3LLO.js +0 -1030
  30. package/dist/chunk-4I2E3LLO.js.map +0 -1
  31. package/dist/chunk-ARU2PZFM.js +0 -312
  32. package/dist/chunk-ARU2PZFM.js.map +0 -1
  33. package/dist/chunk-BOD4O7OF.js +0 -40
  34. package/dist/chunk-BOD4O7OF.js.map +0 -1
  35. package/dist/chunk-DPZAEKA6.js +0 -880
  36. package/dist/chunk-DPZAEKA6.js.map +0 -1
  37. package/dist/chunk-DTJ6QUQB.js +0 -131
  38. package/dist/chunk-DTJ6QUQB.js.map +0 -1
  39. package/dist/chunk-GGE4NNQT.js +0 -65
  40. package/dist/chunk-GGE4NNQT.js.map +0 -1
  41. package/dist/chunk-H5UD2323.js +0 -286
  42. package/dist/chunk-H5UD2323.js.map +0 -1
  43. package/dist/chunk-HHWE3POT.js +0 -94
  44. package/dist/chunk-HHWE3POT.js.map +0 -1
  45. package/dist/chunk-HKUCJ437.js +0 -787
  46. package/dist/chunk-HKUCJ437.js.map +0 -1
  47. package/dist/chunk-JHCHEVET.js +0 -274
  48. package/dist/chunk-JHCHEVET.js.map +0 -1
  49. package/dist/chunk-JHOJHHU7.js +0 -867
  50. package/dist/chunk-JHOJHHU7.js.map +0 -1
  51. package/dist/chunk-JM2SKQMS.js +0 -750
  52. package/dist/chunk-JM2SKQMS.js.map +0 -1
  53. package/dist/chunk-JN2FCO5W.js +0 -7958
  54. package/dist/chunk-JN2FCO5W.js.map +0 -1
  55. package/dist/chunk-K4DBDHLK.js +0 -158
  56. package/dist/chunk-K4DBDHLK.js.map +0 -1
  57. package/dist/chunk-K6N6XJJX.js +0 -306
  58. package/dist/chunk-K6N6XJJX.js.map +0 -1
  59. package/dist/chunk-MA6HLL3S.js +0 -65
  60. package/dist/chunk-MA6HLL3S.js.map +0 -1
  61. package/dist/chunk-MAZ26DC7.js +0 -99
  62. package/dist/chunk-MAZ26DC7.js.map +0 -1
  63. package/dist/chunk-MOXWMGPC.js +0 -577
  64. package/dist/chunk-MOXWMGPC.js.map +0 -1
  65. package/dist/chunk-NJC7U437.js +0 -626
  66. package/dist/chunk-NJC7U437.js.map +0 -1
  67. package/dist/chunk-NPCTHQIO.js +0 -91
  68. package/dist/chunk-NPCTHQIO.js.map +0 -1
  69. package/dist/chunk-ONWEPEDO.js +0 -57
  70. package/dist/chunk-ONWEPEDO.js.map +0 -1
  71. package/dist/chunk-OYZAPX5G.js +0 -1526
  72. package/dist/chunk-OYZAPX5G.js.map +0 -1
  73. package/dist/chunk-PC4UYEBM.js +0 -166
  74. package/dist/chunk-PC4UYEBM.js.map +0 -1
  75. package/dist/chunk-PICTDURQ.js +0 -766
  76. package/dist/chunk-PICTDURQ.js.map +0 -1
  77. package/dist/chunk-PJQFMIOX.js +0 -1182
  78. package/dist/chunk-PJQFMIOX.js.map +0 -1
  79. package/dist/chunk-PXD6ZFNY.js +0 -1107
  80. package/dist/chunk-PXD6ZFNY.js.map +0 -1
  81. package/dist/chunk-PXE2VKMX.js +0 -140
  82. package/dist/chunk-PXE2VKMX.js.map +0 -1
  83. package/dist/chunk-PZ5AY32C.js +0 -10
  84. package/dist/chunk-PZ5AY32C.js.map +0 -1
  85. package/dist/chunk-QBRSJK47.js +0 -622
  86. package/dist/chunk-QBRSJK47.js.map +0 -1
  87. package/dist/chunk-QWMPPZ3X.js +0 -550
  88. package/dist/chunk-QWMPPZ3X.js.map +0 -1
  89. package/dist/chunk-S3UZOQ5Y.js +0 -328
  90. package/dist/chunk-S3UZOQ5Y.js.map +0 -1
  91. package/dist/chunk-S5TT5R3L.js +0 -2668
  92. package/dist/chunk-S5TT5R3L.js.map +0 -1
  93. package/dist/chunk-T4SQEITX.js +0 -95
  94. package/dist/chunk-T4SQEITX.js.map +0 -1
  95. package/dist/chunk-TT4KNT67.js +0 -124
  96. package/dist/chunk-TT4KNT67.js.map +0 -1
  97. package/dist/chunk-U5CHZ5M3.js +0 -357
  98. package/dist/chunk-U5CHZ5M3.js.map +0 -1
  99. package/dist/chunk-ULOKLHIQ.js +0 -1937
  100. package/dist/chunk-ULOKLHIQ.js.map +0 -1
  101. package/dist/chunk-VI2UW6B6.js +0 -162
  102. package/dist/chunk-VI2UW6B6.js.map +0 -1
  103. package/dist/chunk-VQMK5FMP.js +0 -247
  104. package/dist/chunk-VQMK5FMP.js.map +0 -1
  105. package/dist/chunk-VSMTAMNK.js +0 -53
  106. package/dist/chunk-VSMTAMNK.js.map +0 -1
  107. package/dist/chunk-VZSRQ272.js +0 -149
  108. package/dist/chunk-VZSRQ272.js.map +0 -1
  109. package/dist/chunk-WW2A73HW.js +0 -159
  110. package/dist/chunk-WW2A73HW.js.map +0 -1
  111. package/dist/chunk-X4UCIOTZ.js +0 -136
  112. package/dist/chunk-X4UCIOTZ.js.map +0 -1
  113. package/dist/chunk-XDIRG3TO.js +0 -1266
  114. package/dist/chunk-XDIRG3TO.js.map +0 -1
  115. package/dist/chunk-XJYR7XFV.js +0 -317
  116. package/dist/chunk-XJYR7XFV.js.map +0 -1
  117. package/dist/chunk-ZET2UAYW.js +0 -89
  118. package/dist/chunk-ZET2UAYW.js.map +0 -1
  119. package/dist/chunk-ZZUXHH3R.js +0 -99
  120. package/dist/chunk-ZZUXHH3R.js.map +0 -1
  121. package/dist/cli.d.ts +0 -1
  122. package/dist/cli.js +0 -112
  123. package/dist/cli.js.map +0 -1
  124. package/dist/contract/index.d.ts +0 -4969
  125. package/dist/contract/index.js +0 -1653
  126. package/dist/contract/index.js.map +0 -1
  127. package/dist/control.d.ts +0 -1013
  128. package/dist/control.js +0 -34
  129. package/dist/control.js.map +0 -1
  130. package/dist/fuzz.d.ts +0 -759
  131. package/dist/fuzz.js +0 -714
  132. package/dist/fuzz.js.map +0 -1
  133. package/dist/hosted/index.d.ts +0 -730
  134. package/dist/hosted/index.js +0 -14
  135. package/dist/hosted/index.js.map +0 -1
  136. package/dist/index.d.ts +0 -16780
  137. package/dist/index.js +0 -12168
  138. package/dist/index.js.map +0 -1
  139. package/dist/matrix/index.d.ts +0 -155
  140. package/dist/matrix/index.js +0 -8
  141. package/dist/matrix/index.js.map +0 -1
  142. package/dist/meta-eval/index.d.ts +0 -1030
  143. package/dist/meta-eval/index.js +0 -417
  144. package/dist/meta-eval/index.js.map +0 -1
  145. package/dist/multishot/index.d.ts +0 -579
  146. package/dist/multishot/index.js +0 -589
  147. package/dist/multishot/index.js.map +0 -1
  148. package/dist/openapi.json +0 -992
  149. package/dist/pipelines/index.d.ts +0 -567
  150. package/dist/pipelines/index.js +0 -515
  151. package/dist/pipelines/index.js.map +0 -1
  152. package/dist/reporting.d.ts +0 -1277
  153. package/dist/reporting.js +0 -48
  154. package/dist/reporting.js.map +0 -1
  155. package/dist/rl.d.ts +0 -4092
  156. package/dist/rl.js +0 -1724
  157. package/dist/rl.js.map +0 -1
  158. package/dist/run-campaign-HNFPJET4.js +0 -14
  159. package/dist/run-campaign-HNFPJET4.js.map +0 -1
  160. package/dist/storyboard/index.d.ts +0 -279
  161. package/dist/storyboard/index.js +0 -767
  162. package/dist/storyboard/index.js.map +0 -1
  163. package/dist/trace-attributes.d.ts +0 -52
  164. package/dist/trace-attributes.js +0 -62
  165. package/dist/trace-attributes.js.map +0 -1
  166. package/dist/traces.d.ts +0 -2343
  167. package/dist/traces.js +0 -249
  168. package/dist/traces.js.map +0 -1
  169. package/dist/wire/index.d.ts +0 -1252
  170. package/dist/wire/index.js +0 -81
  171. package/dist/wire/index.js.map +0 -1
package/dist/rl.js DELETED
@@ -1,1724 +0,0 @@
1
- import {
2
- doublyRobust,
3
- inverseProbabilityWeighting,
4
- offPolicyEstimateAll,
5
- selfNormalizedImportanceWeighting
6
- } from "./chunk-DTJ6QUQB.js";
7
- import {
8
- FileSystemOutcomeStore,
9
- InMemoryOutcomeStore
10
- } from "./chunk-3RF76KTD.js";
11
- import {
12
- runEvalCampaign
13
- } from "./chunk-U5CHZ5M3.js";
14
- import {
15
- detectRewardHacking,
16
- extractVerifiableReward,
17
- extractVerifiableRewardsFromRecords,
18
- filterDeterministicallyRewarded
19
- } from "./chunk-ARU2PZFM.js";
20
- import "./chunk-NJC7U437.js";
21
- import {
22
- rubricPredictiveValidity
23
- } from "./chunk-X4UCIOTZ.js";
24
- import {
25
- evaluateInterimReleaseConfidence
26
- } from "./chunk-MAZ26DC7.js";
27
- import "./chunk-DPZAEKA6.js";
28
- import {
29
- benjaminiHochberg,
30
- wilcoxonSignedRank
31
- } from "./chunk-PJQFMIOX.js";
32
- import {
33
- observationsFromRunRecords,
34
- thompsonCurriculum,
35
- varianceBasedCurriculum
36
- } from "./chunk-VZSRQ272.js";
37
- import "./chunk-TT4KNT67.js";
38
- import "./chunk-PC4UYEBM.js";
39
- import "./chunk-VQMK5FMP.js";
40
- import "./chunk-S3UZOQ5Y.js";
41
- import "./chunk-MA6HLL3S.js";
42
- import "./chunk-XJYR7XFV.js";
43
- import "./chunk-VSMTAMNK.js";
44
- import {
45
- ValidationError
46
- } from "./chunk-ONWEPEDO.js";
47
- import "./chunk-PZ5AY32C.js";
48
-
49
- // src/rl/adaptation-eval.ts
50
- async function runAdaptationCurve(opts) {
51
- const ks = opts.ks ?? [0, 1, 2, 4, 8, 16];
52
- const reps = opts.reps ?? 3;
53
- const passThreshold = opts.passThreshold ?? 0.5;
54
- const sortedKs = [...ks].sort((a, b) => a - b);
55
- const points = [];
56
- for (const k of sortedKs) {
57
- const perScenario = [];
58
- const allScores = [];
59
- let totalPasses = 0;
60
- let totalAttempts = 0;
61
- for (const scenario of opts.scenarios) {
62
- const sid = scenario.scenarioId ?? `scenario-${opts.scenarios.indexOf(scenario)}`;
63
- const scores = [];
64
- let passes = 0;
65
- for (let r = 0; r < reps; r++) {
66
- const score = await opts.runner.run({ scenario, k, rep: r });
67
- scores.push(score);
68
- if (score >= passThreshold) passes++;
69
- allScores.push(score);
70
- if (score >= passThreshold) totalPasses++;
71
- totalAttempts++;
72
- }
73
- const meanS = scores.reduce((s, v) => s + v, 0) / scores.length;
74
- perScenario.push({ scenarioId: sid, meanScore: meanS, passes, total: scores.length });
75
- }
76
- const meanScore = allScores.reduce((s, v) => s + v, 0) / Math.max(1, allScores.length);
77
- const variance = allScores.length < 2 ? 0 : allScores.reduce((s, v) => s + (v - meanScore) ** 2, 0) / (allScores.length - 1);
78
- points.push({
79
- k,
80
- meanScore,
81
- passRate: totalPasses / Math.max(1, totalAttempts),
82
- std: Math.sqrt(variance),
83
- n: allScores.length,
84
- perScenario
85
- });
86
- }
87
- const firstPassK2 = points.find((p) => p.passRate >= passThreshold)?.k ?? null;
88
- const maxK = sortedKs[sortedKs.length - 1] ?? 1;
89
- let area = 0;
90
- for (let i = 1; i < points.length; i++) {
91
- const x1 = points[i - 1].k;
92
- const x2 = points[i].k;
93
- const y1 = points[i - 1].meanScore;
94
- const y2 = points[i].meanScore;
95
- area += (y1 + y2) / 2 * (x2 - x1);
96
- }
97
- const adaptationArea = maxK === 0 ? 0 : area / maxK;
98
- return { points, firstPassK: firstPassK2, adaptationArea };
99
- }
100
- function compareAdaptationCurves(a, b, opts = {}) {
101
- const conf = opts.confidence ?? 0.95;
102
- const resamples = opts.bootstrapResamples ?? 500;
103
- const rng = makeRng(opts.seed);
104
- const perK = [];
105
- for (const ap of a.points) {
106
- const bp = b.points.find((p) => p.k === ap.k);
107
- if (!bp) continue;
108
- const aMeans = ap.perScenario.map((s) => s.meanScore);
109
- const bMeans = bp.perScenario.map((s) => s.meanScore);
110
- const aCi = bootstrapMeanCi(aMeans, resamples, conf, rng);
111
- const bCi = bootstrapMeanCi(bMeans, resamples, conf, rng);
112
- perK.push({
113
- k: ap.k,
114
- deltaMean: ap.meanScore - bp.meanScore,
115
- aLow: aCi.low,
116
- aHigh: aCi.high,
117
- bLow: bCi.low,
118
- bHigh: bCi.high
119
- });
120
- }
121
- const areaDelta = a.adaptationArea - b.adaptationArea;
122
- const firstPassKDelta = a.firstPassK !== null && b.firstPassK !== null ? b.firstPassK - a.firstPassK : null;
123
- const meanDelta = perK.reduce((s, p) => s + p.deltaMean, 0) / Math.max(1, perK.length);
124
- let verdict;
125
- if (Math.abs(meanDelta) < 0.02 && Math.abs(areaDelta) < 0.02) verdict = "similar";
126
- else if (meanDelta > 0 && areaDelta > 0) verdict = "a_better";
127
- else if (meanDelta < 0 && areaDelta < 0) verdict = "b_better";
128
- else verdict = "similar";
129
- const rationale = `mean per-k delta=${meanDelta.toFixed(3)}, area delta=${areaDelta.toFixed(3)}` + (firstPassKDelta !== null ? `, first-pass-k delta=${firstPassKDelta}` : "");
130
- return { perK, areaDelta, firstPassKDelta, verdict, rationale };
131
- }
132
- function firstPassK(curve, threshold = 0.5) {
133
- return curve.points.find((p) => p.passRate >= threshold)?.k ?? null;
134
- }
135
- function makeRng(seed) {
136
- if (seed === void 0) return Math.random;
137
- let s = seed >>> 0;
138
- return () => {
139
- s = s + 1831565813 >>> 0;
140
- let t = s;
141
- t = Math.imul(t ^ t >>> 15, t | 1);
142
- t ^= t + Math.imul(t ^ t >>> 7, t | 61);
143
- return ((t ^ t >>> 14) >>> 0) / 4294967296;
144
- };
145
- }
146
- function bootstrapMeanCi(xs, resamples, confidence, rng) {
147
- if (xs.length < 2) return { low: xs[0] ?? 0, high: xs[0] ?? 0 };
148
- const samples = new Array(resamples);
149
- for (let b = 0; b < resamples; b++) {
150
- let sum = 0;
151
- for (let i = 0; i < xs.length; i++) sum += xs[Math.floor(rng() * xs.length)];
152
- samples[b] = sum / xs.length;
153
- }
154
- samples.sort((a, b) => a - b);
155
- const alpha = 1 - confidence;
156
- return {
157
- low: samples[Math.floor(alpha / 2 * resamples)],
158
- high: samples[Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)]
159
- };
160
- }
161
-
162
- // src/rl/compute-curves.ts
163
- async function runComputeCurve(opts) {
164
- const points = [];
165
- for (const budget of opts.budgets) {
166
- const r = await opts.runAtBudget(budget);
167
- points.push({
168
- budgetId: budget.id,
169
- cost: budget.cost,
170
- score: r.score,
171
- samples: r.samples,
172
- std: r.std,
173
- metrics: r.metrics
174
- });
175
- }
176
- const sorted = [...points].sort((a, b) => a.cost - b.cost);
177
- const logSlope = sorted.length >= 2 ? fitLogSlope(sorted) : null;
178
- const best = points.reduce((a, b) => b.score > a.score ? b : a);
179
- return { candidateId: opts.candidateId, points: sorted, logSlope, best };
180
- }
181
- async function bestOfN(opts) {
182
- if (opts.n <= 0) throw new ValidationError("bestOfN: n must be > 0");
183
- const rollouts = [];
184
- const scores = [];
185
- for (let i = 0; i < opts.n; i++) {
186
- const r = await opts.sample(i);
187
- rollouts.push(r);
188
- scores.push(await opts.scoreFn(r));
189
- }
190
- let bestIndex = 0;
191
- for (let i = 1; i < scores.length; i++) if (scores[i] > scores[bestIndex]) bestIndex = i;
192
- const meanScore = scores.reduce((s, x) => s + x, 0) / scores.length;
193
- return {
194
- best: rollouts[bestIndex],
195
- bestScore: scores[bestIndex],
196
- scores,
197
- meanScore,
198
- bestIndex
199
- };
200
- }
201
- async function selfConsistency(opts) {
202
- if (opts.n <= 0) throw new ValidationError("selfConsistency: n must be > 0");
203
- const rollouts = [];
204
- const histogram = {};
205
- for (let i = 0; i < opts.n; i++) {
206
- const r = await opts.sample(i);
207
- rollouts.push(r);
208
- const key = opts.answerKey(r);
209
- histogram[key] = (histogram[key] ?? 0) + 1;
210
- }
211
- let answer = "";
212
- let max = -1;
213
- for (const [k, v] of Object.entries(histogram)) {
214
- if (v > max) {
215
- max = v;
216
- answer = k;
217
- }
218
- }
219
- const representative = rollouts.find((r) => opts.answerKey(r) === answer) ?? rollouts[0];
220
- return {
221
- answer,
222
- agreement: max / opts.n,
223
- histogram,
224
- representative,
225
- rollouts
226
- };
227
- }
228
- function paretoFrontier(points) {
229
- const onFrontier = [];
230
- for (const p of points) {
231
- const dominated = points.some(
232
- (q) => q !== p && q.cost <= p.cost && q.score >= p.score && (q.cost < p.cost || q.score > p.score)
233
- );
234
- if (!dominated) onFrontier.push(p);
235
- }
236
- return onFrontier.sort((a, b) => a.cost - b.cost);
237
- }
238
- function fitLogSlope(points) {
239
- const xs = points.map((p) => Math.log(Math.max(1e-12, p.cost)));
240
- const ys = points.map((p) => p.score);
241
- const n = xs.length;
242
- const mx = xs.reduce((s, x) => s + x, 0) / n;
243
- const my = ys.reduce((s, y) => s + y, 0) / n;
244
- let num = 0;
245
- let den = 0;
246
- for (let i = 0; i < n; i++) {
247
- num += (xs[i] - mx) * (ys[i] - my);
248
- den += (xs[i] - mx) ** 2;
249
- }
250
- return den === 0 ? 0 : num / den;
251
- }
252
-
253
- // src/rl/contamination.ts
254
- async function runContaminationProbe(input, opts = {}) {
255
- const fdr = opts.fdr ?? 0.05;
256
- const minMedianDrop = opts.minMedianDrop ?? 0.05;
257
- const floor = opts.scoreFloor ?? 0;
258
- if (!input.perturbed && !input.perturbation) {
259
- throw new ValidationError(
260
- "runContaminationProbe: must supply either `perturbed` or `perturbation`."
261
- );
262
- }
263
- const perturbed = input.perturbed ?? await Promise.all(input.originals.map((s) => input.perturbation.apply(s)));
264
- if (perturbed.length !== input.originals.length) {
265
- throw new ValidationError(
266
- `runContaminationProbe: perturbed length ${perturbed.length} \u2260 originals ${input.originals.length}`
267
- );
268
- }
269
- const origScores = await Promise.all(input.originals.map((s) => input.scoreFn(s)));
270
- const pertScores = await Promise.all(perturbed.map((s) => input.scoreFn(s)));
271
- const perScenario = input.originals.map((s, i) => ({
272
- scenarioId: input.scenarioId(s),
273
- originalScore: origScores[i],
274
- perturbedScore: pertScores[i],
275
- delta: pertScores[i] - origScores[i],
276
- qValue: NaN
277
- }));
278
- const valid = perScenario.filter((p) => p.originalScore >= floor && p.perturbedScore >= floor);
279
- if (valid.length < 4) {
280
- return {
281
- perScenario,
282
- pairedTest: { w: 0, p: 1 },
283
- medianDelta: 0,
284
- meanDelta: 0,
285
- contaminationSuspected: false,
286
- reason: `insufficient valid scenarios (n=${valid.length}, need \u2265 4)`,
287
- n: valid.length
288
- };
289
- }
290
- const origValid = valid.map((p) => p.originalScore);
291
- const pertValid = valid.map((p) => p.perturbedScore);
292
- const pairedTest = wilcoxonSignedRank(origValid, pertValid);
293
- const deltas = valid.map((p) => p.delta);
294
- const sortedDeltas = [...deltas].sort((a, b) => a - b);
295
- const median = sortedDeltas[Math.floor(sortedDeltas.length / 2)];
296
- const mean = deltas.reduce((s, d) => s + d, 0) / deltas.length;
297
- const pseudoP = valid.map((p) => Math.min(1, Math.max(1e-6, 1 - Math.abs(p.delta) / 1)));
298
- const { qValues } = benjaminiHochberg(pseudoP, fdr);
299
- for (let i = 0; i < valid.length; i++) {
300
- const v = valid[i];
301
- const idx = perScenario.findIndex((p) => p.scenarioId === v.scenarioId);
302
- if (idx >= 0) perScenario[idx].qValue = qValues[i];
303
- }
304
- const contaminationSuspected = pairedTest.p < fdr && median <= -minMedianDrop;
305
- const reason = contaminationSuspected ? `paired p=${pairedTest.p.toFixed(4)} < ${fdr} and median drop ${median.toFixed(4)} \u2265 ${minMedianDrop}` : pairedTest.p >= fdr ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})` : `significant but small effect (median delta ${median.toFixed(4)})`;
306
- return {
307
- perScenario,
308
- pairedTest,
309
- medianDelta: median,
310
- meanDelta: mean,
311
- contaminationSuspected,
312
- reason,
313
- n: valid.length
314
- };
315
- }
316
- function renameVariables(identifiers, rename = (n, i) => `${n}_${(i % 26 + 10).toString(36)}`) {
317
- return {
318
- kind: "rename_variables",
319
- apply(scenario) {
320
- let prompt = scenario.prompt;
321
- identifiers.forEach((id, i) => {
322
- const replacement = rename(id, i);
323
- const re = new RegExp(`\\b${escapeRegex(id)}\\b`, "g");
324
- prompt = prompt.replace(re, replacement);
325
- });
326
- return { ...scenario, prompt };
327
- }
328
- };
329
- }
330
- function shuffleOrder(shuffleSection, seed) {
331
- let s = seed >>> 0;
332
- const rng = () => {
333
- s = s + 1831565813 >>> 0;
334
- let t = s;
335
- t = Math.imul(t ^ t >>> 15, t | 1);
336
- t ^= t + Math.imul(t ^ t >>> 7, t | 61);
337
- return ((t ^ t >>> 14) >>> 0) / 4294967296;
338
- };
339
- return {
340
- kind: "shuffle_order",
341
- apply(scenario) {
342
- const newPrompt = shuffleSection(scenario.prompt, rng);
343
- return { ...scenario, prompt: newPrompt };
344
- }
345
- };
346
- }
347
- function injectIrrelevantClause(clause, position = "prefix") {
348
- return {
349
- kind: "inject_irrelevant_clause",
350
- apply(scenario) {
351
- const prompt = position === "prefix" ? `${clause} ${scenario.prompt}` : `${scenario.prompt} ${clause}`;
352
- return { ...scenario, prompt };
353
- }
354
- };
355
- }
356
- function escapeRegex(s) {
357
- return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
358
- }
359
-
360
- // src/rl/corpus.ts
361
- import { appendFileSync, existsSync, mkdirSync, readFileSync } from "fs";
362
- import { dirname } from "path";
363
-
364
- // src/rl/exporters.ts
365
- async function toDpoRows(triples, lookups) {
366
- const out = [];
367
- for (const t of triples) {
368
- const [prompt, chosen, rejected] = await Promise.all([
369
- Promise.resolve(lookups.promptOf(t.chosenRunId)),
370
- Promise.resolve(lookups.completionOf(t.chosenRunId)),
371
- Promise.resolve(lookups.completionOf(t.rejectedRunId))
372
- ]);
373
- out.push({
374
- prompt,
375
- chosen,
376
- rejected,
377
- margin: t.marginScore,
378
- meta: {
379
- scenarioId: t.scenarioId,
380
- chosenVariantId: t.chosenVariantId,
381
- rejectedVariantId: t.rejectedVariantId,
382
- chosenRunId: t.chosenRunId,
383
- rejectedRunId: t.rejectedRunId,
384
- chosenModel: t.meta.chosenModel,
385
- rejectedModel: t.meta.rejectedModel
386
- }
387
- });
388
- }
389
- return out;
390
- }
391
- function toDpoJsonl(rows) {
392
- return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
393
- }
394
- async function toGrpoRows(runs, lookups) {
395
- const rewardOf2 = lookups.rewardOf ?? defaultReward;
396
- const grouped = /* @__PURE__ */ new Map();
397
- for (const r of runs) {
398
- const sid = r.scenarioId ?? r.experimentId;
399
- const arr = grouped.get(sid) ?? [];
400
- arr.push(r);
401
- grouped.set(sid, arr);
402
- }
403
- const rows = [];
404
- for (const [scenarioId, group] of grouped.entries()) {
405
- if (group.length === 0) continue;
406
- const prompt = await Promise.resolve(lookups.promptOf(group[0].runId));
407
- const completions = [];
408
- const rewards = [];
409
- const runIds = [];
410
- for (const r of group) {
411
- const reward2 = rewardOf2(r);
412
- if (reward2 === null) continue;
413
- const completion = await Promise.resolve(lookups.completionOf(r.runId));
414
- completions.push(completion);
415
- rewards.push(reward2);
416
- runIds.push(r.runId);
417
- }
418
- if (completions.length === 0) continue;
419
- rows.push({
420
- prompt,
421
- completions,
422
- rewards,
423
- runIds,
424
- meta: {
425
- scenarioId,
426
- n: completions.length,
427
- meanReward: rewards.reduce((s, x) => s + x, 0) / rewards.length
428
- }
429
- });
430
- }
431
- return rows;
432
- }
433
- function toGrpoJsonl(rows) {
434
- return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
435
- }
436
- async function toSftRows(runs, lookups) {
437
- const include = lookups.include ?? (() => true);
438
- const rows = [];
439
- for (const r of runs) {
440
- if (!include(r)) continue;
441
- const system = lookups.systemOf?.(r);
442
- const [prompt, completion] = await Promise.all([
443
- Promise.resolve(lookups.promptOf(r.runId)),
444
- Promise.resolve(lookups.completionOf(r.runId))
445
- ]);
446
- const messages = [];
447
- if (system) messages.push({ role: "system", content: system });
448
- messages.push({ role: "user", content: prompt });
449
- messages.push({ role: "assistant", content: completion });
450
- rows.push({
451
- messages,
452
- meta: {
453
- runId: r.runId,
454
- candidateId: r.candidateId,
455
- scenarioId: r.scenarioId,
456
- score: r.outcome.holdoutScore ?? r.outcome.searchScore,
457
- model: r.model
458
- }
459
- });
460
- }
461
- return rows;
462
- }
463
- function toSftJsonl(rows) {
464
- return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
465
- }
466
- async function toPrmRows(triples, lookups) {
467
- const rows = [];
468
- for (const t of triples) {
469
- const prompt = await Promise.resolve(lookups.promptOf(t.prefixRunId));
470
- const prefixSpanIds = lookups.prefixOf ? await Promise.resolve(lookups.prefixOf(t.prefixRunId, t.prefixStepIndex)) : [];
471
- const prefixStepText = [];
472
- for (const spanId of prefixSpanIds) {
473
- prefixStepText.push(await Promise.resolve(lookups.stepTextOf(t.prefixRunId, spanId)));
474
- }
475
- const chosenStep = await Promise.resolve(lookups.stepTextOf(t.prefixRunId, t.chosenSpanId));
476
- const rejectedStep = await Promise.resolve(
477
- lookups.stepTextOf(t.rejectedRunId, t.rejectedSpanId)
478
- );
479
- rows.push({
480
- prompt,
481
- prefixSpanIds,
482
- prefixStepText,
483
- chosenStep,
484
- rejectedStep,
485
- chosenReward: t.chosenReward,
486
- rejectedReward: t.rejectedReward,
487
- marginScore: t.marginScore,
488
- meta: {
489
- prefixRunId: t.prefixRunId,
490
- rejectedRunId: t.rejectedRunId,
491
- prefixStepIndex: t.prefixStepIndex
492
- }
493
- });
494
- }
495
- return rows;
496
- }
497
- function toPrmJsonl(rows) {
498
- return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
499
- }
500
- function stepRewardsToJsonl(stepRewards) {
501
- const rows = stepRewards.map((s) => ({
502
- runId: s.runId,
503
- spanId: s.spanId,
504
- stepIndex: s.stepIndex,
505
- reward: s.reward,
506
- determinism: s.determinism,
507
- weight: s.weight ?? 1
508
- }));
509
- return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
510
- }
511
- function defaultReward(run) {
512
- const v = run.outcome.holdoutScore ?? run.outcome.searchScore;
513
- return typeof v === "number" && Number.isFinite(v) ? v : null;
514
- }
515
-
516
- // src/rl/dataset.ts
517
- function reward(r) {
518
- const v = r.outcome.holdoutScore ?? r.outcome.searchScore;
519
- return typeof v === "number" && Number.isFinite(v) ? v : null;
520
- }
521
- function distinct(xs) {
522
- return [...new Set(xs)].sort();
523
- }
524
- function computeRewardStats(values) {
525
- if (values.length === 0) return { n: 0, mean: 0, median: 0, min: 0, max: 0, std: 0 };
526
- const sorted = [...values].sort((a, b) => a - b);
527
- const n = sorted.length;
528
- const mean = sorted.reduce((s, x) => s + x, 0) / n;
529
- const mid = Math.floor(n / 2);
530
- const median = n % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid];
531
- const variance = sorted.reduce((s, x) => s + (x - mean) ** 2, 0) / n;
532
- return { n, mean, median, min: sorted[0], max: sorted[n - 1], std: Math.sqrt(variance) };
533
- }
534
- function computeStats(records) {
535
- const splits = { search: 0, dev: 0, holdout: 0 };
536
- let inTok = 0;
537
- let outTok = 0;
538
- let cost = 0;
539
- const rewards = [];
540
- for (const r of records) {
541
- splits[r.splitTag] = (splits[r.splitTag] ?? 0) + 1;
542
- inTok += r.tokenUsage.input;
543
- outTok += r.tokenUsage.output;
544
- cost += r.costUsd;
545
- const rw = reward(r);
546
- if (rw !== null) rewards.push(rw);
547
- }
548
- return {
549
- records: records.length,
550
- splits,
551
- reward: computeRewardStats(rewards),
552
- models: distinct(records.map((r) => r.model)),
553
- promptHashes: distinct(records.map((r) => r.promptHash)),
554
- commitShas: distinct(records.map((r) => r.commitSha)),
555
- totalTokens: { input: inTok, output: outTok },
556
- totalCostUsd: cost
557
- };
558
- }
559
- async function buildRlDataset(records, lookups, config, preferences) {
560
- if (records.length === 0) {
561
- throw new Error("buildRlDataset: no records \u2014 refusing to package an empty dataset");
562
- }
563
- const formats = config.formats ?? ["grpo", "sft"];
564
- const files = {};
565
- const rowCounts = {};
566
- if (formats.includes("grpo")) {
567
- const rows = await toGrpoRows(records, lookups);
568
- files["train.grpo.jsonl"] = toGrpoJsonl(rows);
569
- rowCounts.grpo = rows.length;
570
- }
571
- if (formats.includes("sft")) {
572
- const rows = await toSftRows(records, lookups);
573
- files["train.sft.jsonl"] = toSftJsonl(rows);
574
- rowCounts.sft = rows.length;
575
- }
576
- if (formats.includes("dpo")) {
577
- if (!preferences) {
578
- throw new Error("buildRlDataset: format 'dpo' requires `preferences` (triples + lookups)");
579
- }
580
- const rows = await toDpoRows(preferences.triples, preferences.lookups);
581
- files["train.dpo.jsonl"] = toDpoJsonl(rows);
582
- rowCounts.dpo = rows.length;
583
- }
584
- const manifest = {
585
- ...config,
586
- formats,
587
- rowCounts,
588
- stats: computeStats(records)
589
- };
590
- files["manifest.json"] = `${JSON.stringify(manifest, null, 2)}
591
- `;
592
- files["DATASHEET.md"] = datasheetToMarkdown(manifest);
593
- return { manifest, files };
594
- }
595
- function pct(x) {
596
- return `${(x * 100).toFixed(1)}%`;
597
- }
598
- function datasheetToMarkdown(m) {
599
- const s = m.stats;
600
- const total = s.records || 1;
601
- const splitLines = ["search", "dev", "holdout"].map((k) => ` - \`${k}\`: ${s.splits[k]} (${pct(s.splits[k] / total)})`).join("\n");
602
- const deterministic = m.reward.kind === "deterministic";
603
- return [
604
- `# Dataset: ${m.name} \`v${m.version}\``,
605
- "",
606
- `**Domain:** ${m.domain} \xB7 **Created:** ${m.createdAtIso} \xB7 **License:** ${m.license}`,
607
- "",
608
- "## Reward provenance",
609
- `- **Kind:** ${m.reward.kind}${deterministic ? " \u2705 (decidable \u2014 not judge-noise)" : ""}`,
610
- `- **Source:** ${m.reward.source}`,
611
- `- **Description:** ${m.reward.description}`,
612
- "",
613
- "## Composition",
614
- `- **Records (trajectories):** ${s.records}`,
615
- `- **Formats:** ${m.formats.map((f) => `${f} (${m.rowCounts[f] ?? 0} rows)`).join(", ")}`,
616
- "- **Splits:**",
617
- splitLines,
618
- "",
619
- "## Reward distribution",
620
- `- n=${s.reward.n} \xB7 mean=${s.reward.mean.toFixed(3)} \xB7 median=${s.reward.median.toFixed(3)} \xB7 min=${s.reward.min.toFixed(3)} \xB7 max=${s.reward.max.toFixed(3)} \xB7 std=${s.reward.std.toFixed(3)}`,
621
- "",
622
- "## Provenance",
623
- `- **Models:** ${s.models.join(", ")}`,
624
- `- **Prompt/agent versions (sha256):** ${s.promptHashes.length} distinct`,
625
- `- **Commits:** ${s.commitShas.join(", ")}`,
626
- `- **Tokens:** ${s.totalTokens.input} in / ${s.totalTokens.output} out \xB7 **Cost:** $${s.totalCostUsd.toFixed(2)}`,
627
- "",
628
- "## Quality gates",
629
- `- Contamination probe: ${m.qualityGates?.contaminationProbe ?? "not-run"}`,
630
- `- Dedup: ${m.qualityGates?.dedup ? "yes" : "no"} \xB7 Verifiable-reward filter: ${m.qualityGates?.verifiableRewardFilter ? "yes" : "no"}`,
631
- "",
632
- "## Recommended uses",
633
- m.intendedUse,
634
- "",
635
- "## Out of scope",
636
- m.outOfScope,
637
- "",
638
- "## Limitations",
639
- m.limitations,
640
- "",
641
- "## Token rendering",
642
- "For RL/SFT training, tokenize with the per-model renderer (DeepSeek-V3 / Kimi-K2 / Qwen3) to preserve token identity and per-token loss masks across tool-call turns \u2014 see `renderers` (PrimeIntellect). The `messages` / `completions` here are the renderer input.",
643
- ""
644
- ].join("\n");
645
- }
646
-
647
- // src/rl/corpus.ts
648
- function appendToCorpus(records, corpusPath) {
649
- mkdirSync(dirname(corpusPath), { recursive: true });
650
- const existing = existsSync(corpusPath) ? readCorpus(corpusPath) : [];
651
- const seen = new Set(existing.map((r) => r.runId));
652
- const lines = [];
653
- let appended = 0;
654
- let skipped = 0;
655
- for (const r of records) {
656
- if (seen.has(r.runId)) {
657
- skipped++;
658
- continue;
659
- }
660
- seen.add(r.runId);
661
- lines.push(JSON.stringify(r));
662
- appended++;
663
- }
664
- if (lines.length > 0) appendFileSync(corpusPath, `${lines.join("\n")}
665
- `);
666
- return { appended, skipped, total: existing.length + appended };
667
- }
668
- function readCorpus(corpusPath) {
669
- if (!existsSync(corpusPath)) return [];
670
- const out = [];
671
- for (const line of readFileSync(corpusPath, "utf8").split("\n")) {
672
- if (line.trim()) out.push(JSON.parse(line));
673
- }
674
- return out;
675
- }
676
- function rewardOf(r) {
677
- const v = r.outcome.holdoutScore ?? r.outcome.searchScore;
678
- return typeof v === "number" && Number.isFinite(v) ? v : 0;
679
- }
680
- async function buildDatasetFromCorpus(corpusPath, config, opts = {}) {
681
- let records = readCorpus(corpusPath).filter(
682
- (r) => typeof r.prompt === "string" && typeof r.completion === "string"
683
- );
684
- if (opts.splits) records = records.filter((r) => opts.splits.includes(r.splitTag));
685
- if (opts.minScore != null) records = records.filter((r) => rewardOf(r) >= opts.minScore);
686
- const text = new Map(
687
- records.map((r) => [r.runId, { prompt: r.prompt, completion: r.completion }])
688
- );
689
- const lookups = {
690
- promptOf: (id) => text.get(id)?.prompt ?? "",
691
- completionOf: (id) => text.get(id)?.completion ?? ""
692
- };
693
- return buildRlDataset(records, lookups, config);
694
- }
695
-
696
- // src/rl/predictive-validity-researcher.ts
697
- var PredictiveValidityResearcher = class {
698
- opts;
699
- lastReport = null;
700
- constructor(opts) {
701
- this.opts = opts;
702
- }
703
- async inspectFailures(runs) {
704
- const threshold = this.opts.failureThreshold ?? 0.5;
705
- const failures = [];
706
- const failingRuns = runs.filter((r) => {
707
- const score = r.outcome.holdoutScore ?? r.outcome.searchScore;
708
- return typeof score === "number" && score < threshold;
709
- });
710
- if (failingRuns.length === 0) return failures;
711
- const grouped = /* @__PURE__ */ new Map();
712
- for (const r of failingRuns) {
713
- const arr = grouped.get(r.candidateId) ?? [];
714
- arr.push(r);
715
- grouped.set(r.candidateId, arr);
716
- }
717
- for (const [candidateId, group] of grouped.entries()) {
718
- const meanScore = group.reduce((s, r) => {
719
- const x = r.outcome.holdoutScore ?? r.outcome.searchScore ?? 0;
720
- return s + x;
721
- }, 0) / group.length;
722
- failures.push({
723
- code: `low-score-${candidateId}`,
724
- description: `${candidateId} scored < ${threshold} on ${group.length} run(s) (mean ${meanScore.toFixed(3)})`,
725
- evidence: {
726
- runIds: group.slice(0, 8).map((r) => r.runId),
727
- samples: group.length
728
- }
729
- });
730
- }
731
- return failures;
732
- }
733
- async proposeChange(failures) {
734
- if (failures.length === 0) return [];
735
- if (this.lastReport === null) {
736
- return [
737
- {
738
- kind: "threshold",
739
- payload: { directive: "researcher.collect-more-outcomes" },
740
- rationale: "predictive-validity researcher has no prior report; cannot recommend rubric reweighting until at least one report exists"
741
- }
742
- ];
743
- }
744
- const decorativeThreshold = this.opts.decorativeThreshold ?? 0.4;
745
- const changes = [];
746
- for (const ranking of this.lastReport.ranked) {
747
- if (ranking.verdict === "load_bearing") continue;
748
- if (Math.abs(ranking.spearman) >= decorativeThreshold) continue;
749
- changes.push({
750
- kind: "reviewer_prompt",
751
- payload: {
752
- rubric: ranking.rubric,
753
- action: "down-weight",
754
- spearman: ranking.spearman,
755
- bestOutcome: ranking.bestOutcome
756
- },
757
- rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (decorative); recommend down-weighting`,
758
- expectedDelta: -Math.max(0, 0.05 - Math.abs(ranking.spearman))
759
- });
760
- }
761
- for (const ranking of this.lastReport.ranked.slice(0, 1)) {
762
- if (ranking.verdict !== "load_bearing") continue;
763
- changes.push({
764
- kind: "reviewer_prompt",
765
- payload: {
766
- rubric: ranking.rubric,
767
- action: "up-weight",
768
- spearman: ranking.spearman,
769
- bestOutcome: ranking.bestOutcome
770
- },
771
- rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (load-bearing); recommend up-weighting`,
772
- expectedDelta: Math.max(0, Math.abs(ranking.spearman) - 0.5) * 0.1
773
- });
774
- }
775
- return changes;
776
- }
777
- async applyChange(changes, baseline) {
778
- return {
779
- ...baseline,
780
- changes: [...baseline.changes, ...changes]
781
- };
782
- }
783
- async evaluateChange(plan) {
784
- const emptyGate = {
785
- promote: false,
786
- candidateId: plan.proposedCandidateId,
787
- baselineId: plan.baselineCandidateId,
788
- evidence: {
789
- productiveRuns: 0,
790
- medianPairedDelta: 0,
791
- pairedCI: { low: 0, high: 0 },
792
- pairedPValue: 1,
793
- searchScore: 0,
794
- holdoutScore: 0,
795
- overfitGap: 0,
796
- baselineOverfitGap: 0,
797
- medianCandidateCost: Number.NaN,
798
- medianBaselineCost: Number.NaN
799
- },
800
- reason: "predictive-validity researcher does not execute plans; the caller is expected to run the sweep and call rubricPredictiveValidity directly with the resulting RunRecord[].",
801
- rejectionCode: "few_runs"
802
- };
803
- return {
804
- plan,
805
- runs: [],
806
- gateDecision: emptyGate
807
- };
808
- }
809
- /**
810
- * Run the predictive-validity check explicitly against a fresh RunRecord
811
- * set. Updates the researcher's cached report so subsequent
812
- * `proposeChange` calls have evidence to draw from.
813
- */
814
- async runValidityCheck(runs) {
815
- const report = await rubricPredictiveValidity({
816
- runs,
817
- outcomes: this.opts.outcomes,
818
- outcomeMetrics: this.opts.outcomeMetrics,
819
- rubrics: this.opts.rubrics
820
- });
821
- if (this.opts.onReport) await this.opts.onReport(report);
822
- this.lastReport = report;
823
- return report;
824
- }
825
- /**
826
- * Force-feed a predictive-validity report into the researcher state —
827
- * useful when the consumer ran the report out-of-band and wants the
828
- * researcher's later proposals informed by it.
829
- */
830
- setReport(report) {
831
- this.lastReport = report;
832
- }
833
- getLastReport() {
834
- return this.lastReport;
835
- }
836
- };
837
-
838
- // src/rl/preferences.ts
839
- var SPLIT_TAG_DEFAULT = "holdout";
840
- var DEFAULT_REWARD = (run) => {
841
- const v = run.outcome.holdoutScore ?? run.outcome.searchScore;
842
- return typeof v === "number" && Number.isFinite(v) ? v : null;
843
- };
844
- function extractPreferences(runs, opts = {}) {
845
- const strategy = opts.strategy ?? "paired-by-scenario-and-seed";
846
- const minMargin = opts.minMargin ?? 0.05;
847
- const splitTag = opts.splitTag ?? SPLIT_TAG_DEFAULT;
848
- const rewardOf2 = opts.rewardOf ?? DEFAULT_REWARD;
849
- const filtered = runs.filter((r) => r.splitTag === splitTag);
850
- const scoredEntries = [];
851
- for (const run of filtered) {
852
- const s = rewardOf2(run);
853
- if (s === null) continue;
854
- scoredEntries.push({ run, score: s });
855
- }
856
- const pairs = [];
857
- let pairsBelowMargin = 0;
858
- let cellsSingleton = 0;
859
- let cellsInspected = 0;
860
- if (strategy === "paired-by-scenario-and-seed") {
861
- const groups = /* @__PURE__ */ new Map();
862
- for (const e of scoredEntries) {
863
- const sid = scenarioOf(e.run);
864
- const key = `${sid}::${e.run.seed}`;
865
- const arr = groups.get(key) ?? [];
866
- arr.push(e);
867
- groups.set(key, arr);
868
- }
869
- for (const [key, members] of groups.entries()) {
870
- cellsInspected++;
871
- if (members.length < 2) {
872
- cellsSingleton++;
873
- continue;
874
- }
875
- for (let i = 0; i < members.length; i++) {
876
- for (let j = i + 1; j < members.length; j++) {
877
- const a = members[i];
878
- const b = members[j];
879
- if (a.run.candidateId === b.run.candidateId) continue;
880
- const result = makePair(a, b, key.split("::")[0], minMargin);
881
- if (result.kind === "admit") pairs.push(result.pair);
882
- else pairsBelowMargin++;
883
- }
884
- }
885
- }
886
- } else if (strategy === "paired-by-scenario") {
887
- const byScenarioVariant = /* @__PURE__ */ new Map();
888
- for (const e of scoredEntries) {
889
- const sid = scenarioOf(e.run);
890
- let perScenario = byScenarioVariant.get(sid);
891
- if (!perScenario) {
892
- perScenario = /* @__PURE__ */ new Map();
893
- byScenarioVariant.set(sid, perScenario);
894
- }
895
- const cur = perScenario.get(e.run.candidateId);
896
- if (cur) {
897
- cur.sum += e.score;
898
- cur.n++;
899
- } else perScenario.set(e.run.candidateId, { run: e.run, sum: e.score, n: 1 });
900
- }
901
- for (const [sid, perVariant] of byScenarioVariant.entries()) {
902
- cellsInspected++;
903
- const arr = [...perVariant.entries()].map(([vid, agg]) => ({
904
- run: agg.run,
905
- score: agg.sum / agg.n,
906
- variantId: vid
907
- }));
908
- if (arr.length < 2) {
909
- cellsSingleton++;
910
- continue;
911
- }
912
- for (let i = 0; i < arr.length; i++) {
913
- for (let j = i + 1; j < arr.length; j++) {
914
- const result = makePair(arr[i], arr[j], sid, minMargin);
915
- if (result.kind === "admit") pairs.push(result.pair);
916
- else pairsBelowMargin++;
917
- }
918
- }
919
- }
920
- } else {
921
- const byScenario = /* @__PURE__ */ new Map();
922
- for (const e of scoredEntries) {
923
- const sid = scenarioOf(e.run);
924
- const arr = byScenario.get(sid) ?? [];
925
- arr.push(e);
926
- byScenario.set(sid, arr);
927
- }
928
- for (const [sid, arr] of byScenario.entries()) {
929
- cellsInspected++;
930
- if (arr.length < 2) {
931
- cellsSingleton++;
932
- continue;
933
- }
934
- const sorted = [...arr].sort((a, b) => a.score - b.score);
935
- const top = sorted[sorted.length - 1];
936
- const bot = sorted[0];
937
- if (top.run.candidateId === bot.run.candidateId) {
938
- cellsSingleton++;
939
- continue;
940
- }
941
- const result = makePair(bot, top, sid, minMargin);
942
- if (result.kind === "admit") pairs.push(result.pair);
943
- else pairsBelowMargin++;
944
- }
945
- }
946
- return { pairs, cellsInspected, pairsBelowMargin, cellsSingleton, strategy };
947
- }
948
- function toTRLFormat(triples, promptOf) {
949
- return triples.map((t) => ({
950
- prompt: promptOf(t.meta.chosenPromptHash),
951
- chosen: t.meta.chosenPromptHash,
952
- // caller substitutes the model output via the runId map
953
- rejected: t.meta.rejectedPromptHash
954
- }));
955
- }
956
- function toAnthropicFormat(triples) {
957
- return triples.map((t) => ({
958
- scenarioId: t.scenarioId,
959
- chosenRunId: t.chosenRunId,
960
- rejectedRunId: t.rejectedRunId,
961
- margin: t.marginScore
962
- }));
963
- }
964
- function makePair(a, b, scenarioId, minMargin) {
965
- const margin = Math.abs(a.score - b.score);
966
- if (margin < minMargin) return { kind: "reject" };
967
- const [chosen, rejected] = a.score > b.score ? [a, b] : [b, a];
968
- return {
969
- kind: "admit",
970
- pair: {
971
- scenarioId,
972
- chosenRunId: chosen.run.runId,
973
- rejectedRunId: rejected.run.runId,
974
- chosenVariantId: chosen.run.candidateId,
975
- rejectedVariantId: rejected.run.candidateId,
976
- marginScore: chosen.score - rejected.score,
977
- scores: { chosen: chosen.score, rejected: rejected.score },
978
- seed: chosen.run.seed === rejected.run.seed ? chosen.run.seed : void 0,
979
- meta: {
980
- chosenPromptHash: chosen.run.promptHash,
981
- rejectedPromptHash: rejected.run.promptHash,
982
- chosenConfigHash: chosen.run.configHash,
983
- rejectedConfigHash: rejected.run.configHash,
984
- chosenModel: chosen.run.model,
985
- rejectedModel: rejected.run.model
986
- }
987
- }
988
- };
989
- }
990
- function scenarioOf(run) {
991
- if (typeof run.scenarioId === "string" && run.scenarioId.length > 0) return run.scenarioId;
992
- const fromRaw = run.outcome.raw.scenario_id;
993
- if (typeof fromRaw === "number" && Number.isFinite(fromRaw)) return String(fromRaw);
994
- if (typeof fromRaw === "string") return fromRaw;
995
- return run.experimentId;
996
- }
997
-
998
- // src/rl/process-reward.ts
999
- async function extractStepRewards(store, runId, opts) {
1000
- const spans = await store.spans({ runId });
1001
- const ordered = [...spans].sort((a, b) => a.startedAt - b.startedAt);
1002
- const out = [];
1003
- let idx = 0;
1004
- for (const span of ordered) {
1005
- if (opts.preFilter && !opts.preFilter(span)) continue;
1006
- let scored = null;
1007
- for (const s of opts.scorers) {
1008
- if (!s.appliesTo.includes(span.kind)) continue;
1009
- const r = await s.score(span);
1010
- if (r) {
1011
- scored = r;
1012
- break;
1013
- }
1014
- }
1015
- if (!scored) continue;
1016
- out.push({
1017
- spanId: span.spanId,
1018
- runId,
1019
- stepIndex: idx++,
1020
- kind: span.kind,
1021
- name: span.name,
1022
- reward: scored.reward,
1023
- determinism: scored.determinism,
1024
- rationale: scored.rationale,
1025
- weight: scored.weight
1026
- });
1027
- }
1028
- return out;
1029
- }
1030
- function runwiseStepRewardSummary(stepRewards) {
1031
- if (stepRewards.length === 0) {
1032
- return {
1033
- runId: "",
1034
- totalSteps: 0,
1035
- meanReward: 0,
1036
- sumWeightedReward: 0,
1037
- failureFraction: 0,
1038
- worstStepDelta: 0,
1039
- worstStepIndex: null
1040
- };
1041
- }
1042
- const runId = stepRewards[0].runId;
1043
- let sumW = 0;
1044
- let sumWR = 0;
1045
- let failures = 0;
1046
- let worstDelta = 0;
1047
- let worstIdx = null;
1048
- let prev = stepRewards[0].reward;
1049
- for (let i = 0; i < stepRewards.length; i++) {
1050
- const s = stepRewards[i];
1051
- const w = s.weight ?? 1;
1052
- sumW += w;
1053
- sumWR += w * s.reward;
1054
- if (s.reward < 0.5) failures++;
1055
- if (i > 0) {
1056
- const delta = s.reward - prev;
1057
- if (delta < worstDelta) {
1058
- worstDelta = delta;
1059
- worstIdx = i;
1060
- }
1061
- prev = s.reward;
1062
- } else {
1063
- prev = s.reward;
1064
- }
1065
- }
1066
- return {
1067
- runId,
1068
- totalSteps: stepRewards.length,
1069
- meanReward: sumW === 0 ? 0 : sumWR / sumW,
1070
- sumWeightedReward: sumWR,
1071
- failureFraction: failures / stepRewards.length,
1072
- worstStepDelta: worstDelta,
1073
- worstStepIndex: worstIdx
1074
- };
1075
- }
1076
- function prmTrainingPairs(stepRewardsByRun, opts = {}) {
1077
- const minMargin = opts.minMargin ?? 0.2;
1078
- const minPrefix = opts.minPrefixLength ?? 1;
1079
- const runs = [...stepRewardsByRun.entries()].map(([runId, steps]) => ({ runId, steps }));
1080
- const triples = [];
1081
- for (let i = 0; i < runs.length; i++) {
1082
- for (let j = i + 1; j < runs.length; j++) {
1083
- const a = runs[i];
1084
- const b = runs[j];
1085
- const minLen = Math.min(a.steps.length, b.steps.length);
1086
- if (minLen < minPrefix + 1) continue;
1087
- let divergenceIdx = -1;
1088
- for (let k = 0; k < minLen; k++) {
1089
- const sa = a.steps[k];
1090
- const sb = b.steps[k];
1091
- const structuralDivergence = sa.kind !== sb.kind || sa.name !== sb.name;
1092
- const rewardGap = Math.abs(sa.reward - sb.reward);
1093
- if (structuralDivergence || rewardGap >= minMargin) {
1094
- divergenceIdx = k;
1095
- break;
1096
- }
1097
- }
1098
- if (divergenceIdx < 0) continue;
1099
- if (divergenceIdx < minPrefix) continue;
1100
- const aNext = a.steps[divergenceIdx];
1101
- const bNext = b.steps[divergenceIdx];
1102
- const margin = Math.abs(aNext.reward - bNext.reward);
1103
- if (margin < minMargin) continue;
1104
- const chosen = aNext.reward > bNext.reward ? aNext : bNext;
1105
- const rejected = aNext.reward > bNext.reward ? bNext : aNext;
1106
- const chosenRun = aNext.reward > bNext.reward ? a.runId : b.runId;
1107
- const rejectedRun = aNext.reward > bNext.reward ? b.runId : a.runId;
1108
- triples.push({
1109
- prefixRunId: chosenRun,
1110
- prefixStepIndex: divergenceIdx - 1,
1111
- chosenSpanId: chosen.spanId,
1112
- chosenReward: chosen.reward,
1113
- rejectedSpanId: rejected.spanId,
1114
- rejectedReward: rejected.reward,
1115
- rejectedRunId: rejectedRun,
1116
- marginScore: chosen.reward - rejected.reward
1117
- });
1118
- }
1119
- }
1120
- return triples;
1121
- }
1122
-
1123
- // src/rl/rl-campaign.ts
1124
- async function runRLCampaign(opts) {
1125
- const campaign = await runEvalCampaign(opts);
1126
- const rewardSignals = extractVerifiableRewardsFromRecords(
1127
- campaign.runs,
1128
- opts.verifiableReward ?? {}
1129
- );
1130
- const preferences = extractPreferences(campaign.runs, {
1131
- strategy: opts.preferences?.strategy ?? "paired-by-scenario-and-seed",
1132
- minMargin: opts.preferences?.minMargin ?? 0.05,
1133
- splitTag: opts.preferences?.splitTag ?? opts.splitTag ?? "holdout",
1134
- rewardOf: opts.preferences?.rewardOf
1135
- });
1136
- let interimConfidence = null;
1137
- if (opts.report?.comparator) {
1138
- const comparator = opts.report.comparator;
1139
- const deltaSeries = collectPairedDeltaSeries(campaign.runs, comparator);
1140
- if (deltaSeries.some((s) => s.deltas.length > 0)) {
1141
- interimConfidence = evaluateInterimReleaseConfidence({
1142
- deltaSeries,
1143
- alpha: opts.sequential?.alpha,
1144
- bound: opts.sequential?.bound,
1145
- rope: opts.sequential?.rope ?? opts.report?.rope
1146
- });
1147
- }
1148
- }
1149
- const rewardHacking = detectRewardHacking({
1150
- runs: campaign.runs,
1151
- verifiableRewardOptions: opts.verifiableReward
1152
- });
1153
- let predictiveValidity = null;
1154
- if (opts.outcomeStore && opts.outcomeMetrics && opts.outcomeMetrics.length > 0) {
1155
- predictiveValidity = await rubricPredictiveValidity({
1156
- runs: campaign.runs,
1157
- outcomes: opts.outcomeStore,
1158
- outcomeMetrics: opts.outcomeMetrics
1159
- });
1160
- }
1161
- const trainerRows = {};
1162
- if (opts.trainerExport?.dpo) {
1163
- trainerRows.dpo = await toDpoRows(preferences.pairs, opts.trainerExport.dpo);
1164
- }
1165
- if (opts.trainerExport?.grpo) {
1166
- trainerRows.grpo = await toGrpoRows(campaign.runs, opts.trainerExport.grpo);
1167
- }
1168
- if (opts.trainerExport?.sft) {
1169
- trainerRows.sft = await toSftRows(campaign.runs, opts.trainerExport.sft);
1170
- }
1171
- const summary = buildSummary({
1172
- campaign,
1173
- preferences,
1174
- interimConfidence,
1175
- rewardHacking,
1176
- predictiveValidity
1177
- });
1178
- return {
1179
- campaign,
1180
- rewardSignals,
1181
- preferences,
1182
- interimConfidence,
1183
- rewardHacking,
1184
- predictiveValidity,
1185
- trainerRows,
1186
- summary,
1187
- kind: "agent-eval-rl-campaign"
1188
- };
1189
- }
1190
- function collectPairedDeltaSeries(runs, comparator) {
1191
- const baseline = /* @__PURE__ */ new Map();
1192
- for (const r of runs) {
1193
- if (r.candidateId !== comparator) continue;
1194
- const sid = r.scenarioId ?? r.experimentId;
1195
- const score = r.outcome.holdoutScore ?? r.outcome.searchScore;
1196
- if (typeof score !== "number" || !Number.isFinite(score)) continue;
1197
- baseline.set(`${sid}::${r.seed}`, score);
1198
- }
1199
- const byCandidate = /* @__PURE__ */ new Map();
1200
- for (const r of runs) {
1201
- if (r.candidateId === comparator) continue;
1202
- const sid = r.scenarioId ?? r.experimentId;
1203
- const score = r.outcome.holdoutScore ?? r.outcome.searchScore;
1204
- if (typeof score !== "number" || !Number.isFinite(score)) continue;
1205
- const baseScore = baseline.get(`${sid}::${r.seed}`);
1206
- if (typeof baseScore !== "number") continue;
1207
- const arr = byCandidate.get(r.candidateId) ?? [];
1208
- arr.push(score - baseScore);
1209
- byCandidate.set(r.candidateId, arr);
1210
- }
1211
- return [...byCandidate.entries()].map(([candidateId, deltas]) => ({ candidateId, deltas }));
1212
- }
1213
- function buildSummary(args) {
1214
- const c = args.campaign;
1215
- const lines = [
1216
- `${c.campaignId}: ${c.runs.length} successful runs / ${c.failedRuns.length} failed (fingerprint ${c.campaignFingerprint.slice(0, 12)}\u2026)`,
1217
- `preferences: ${args.preferences.pairs.length} (${args.preferences.strategy}, ${args.preferences.pairsBelowMargin} below margin)`
1218
- ];
1219
- if (args.interimConfidence) {
1220
- lines.push(
1221
- `sequential verdict: ${args.interimConfidence.recommendation.decision}` + (args.interimConfidence.recommendation.candidateId ? ` ${args.interimConfidence.recommendation.candidateId}` : "")
1222
- );
1223
- }
1224
- lines.push(
1225
- `reward-hacking: ${args.rewardHacking.verdict} (${args.rewardHacking.findings.length} signals checked)`
1226
- );
1227
- if (args.predictiveValidity) {
1228
- const top = args.predictiveValidity.ranked[0];
1229
- lines.push(
1230
- `top-rubric: ${top?.rubric ?? "none"} \u03C1=${(top?.spearman ?? 0).toFixed(2)} (${top?.verdict ?? "no data"})`
1231
- );
1232
- }
1233
- return lines.join(" | ");
1234
- }
1235
-
1236
- // src/rl/run-record-adapters.ts
1237
- function campaignToRunRecords(campaign, ctx) {
1238
- const splitTag = ctx.splitTag ?? "search";
1239
- const candidateId = ctx.candidateId ?? campaign.manifestHash;
1240
- return campaign.cells.map((cell) => {
1241
- const composites = Object.values(cell.judgeScores).map((s) => s.composite);
1242
- const score = composites.length > 0 ? composites.reduce((a, b) => a + b, 0) / composites.length : 0;
1243
- const raw = { rep: cell.rep, duration_ms: cell.durationMs };
1244
- for (const judge of Object.values(cell.judgeScores)) {
1245
- for (const [dim, value] of Object.entries(judge.dimensions)) {
1246
- if (Number.isFinite(value)) raw[`dim.${dim}`] = value;
1247
- }
1248
- }
1249
- if (typeof cell.generation === "number") raw.generation = cell.generation;
1250
- const outcome = { raw };
1251
- if (splitTag === "holdout") outcome.holdoutScore = score;
1252
- else outcome.searchScore = score;
1253
- return {
1254
- runId: cell.cellId,
1255
- experimentId: ctx.experimentId,
1256
- candidateId,
1257
- seed: cell.seed,
1258
- model: ctx.model,
1259
- promptHash: ctx.promptHash,
1260
- configHash: ctx.configHash,
1261
- commitSha: ctx.commitSha,
1262
- wallMs: cell.durationMs,
1263
- costUsd: Number.isFinite(cell.costUsd) ? cell.costUsd : ctx.defaultCostUsd ?? 0,
1264
- tokenUsage: { input: 0, output: 0 },
1265
- outcome,
1266
- failureMode: cell.error ? "cell_error" : void 0,
1267
- splitTag,
1268
- scenarioId: cell.scenarioId
1269
- };
1270
- });
1271
- }
1272
- function verificationReportToRunRecord(report, ctx, opts = {}) {
1273
- const splitTag = ctx.splitTag ?? "search";
1274
- const runId = opts.runId ?? `run-${ctx.candidateId}-${ctx.experimentId}-${report.startedAt}`;
1275
- const raw = {
1276
- pass_count: report.passCount,
1277
- fail_count: report.failCount,
1278
- error_count: report.errorCount,
1279
- skipped_count: report.skippedCount,
1280
- duration_ms: report.durationMs,
1281
- blended_score: report.blendedScore
1282
- };
1283
- for (const layer of report.layers) {
1284
- if (typeof layer.score === "number") raw[`layer.${layer.layer}`] = layer.score;
1285
- raw[`layer_${layer.layer}_pass`] = layer.status === "pass" ? 1 : 0;
1286
- if (layer.diagnostics) {
1287
- for (const [k, v] of Object.entries(layer.diagnostics)) {
1288
- if (typeof v === "number" && Number.isFinite(v)) raw[`layer.${layer.layer}.${k}`] = v;
1289
- }
1290
- }
1291
- }
1292
- const firstFail = report.layers.find((l) => l.status === "fail" || l.status === "error");
1293
- const outcome = { raw };
1294
- if (splitTag === "holdout") outcome.holdoutScore = report.blendedScore;
1295
- else outcome.searchScore = report.blendedScore;
1296
- return {
1297
- runId,
1298
- experimentId: ctx.experimentId,
1299
- candidateId: ctx.candidateId,
1300
- seed: 0,
1301
- model: ctx.model,
1302
- promptHash: ctx.promptHash,
1303
- configHash: ctx.configHash,
1304
- commitSha: ctx.commitSha,
1305
- wallMs: report.durationMs,
1306
- costUsd: ctx.defaultCostUsd ?? 0,
1307
- tokenUsage: { input: 0, output: 0 },
1308
- outcome,
1309
- failureMode: firstFail ? failureModeFromLayer(firstFail) : void 0,
1310
- splitTag,
1311
- scenarioId: ctx.scenarioId
1312
- };
1313
- }
1314
- function failureModeFromLayer(layer) {
1315
- if (layer.status === "error") return `layer_${layer.layer}_error`;
1316
- if (layer.status === "fail") return `layer_${layer.layer}_fail`;
1317
- if (layer.status === "timeout") return `layer_${layer.layer}_timeout`;
1318
- return `layer_${layer.layer}_${layer.status}`;
1319
- }
1320
-
1321
- // src/rl/sim-fidelity.ts
1322
- var ABSENT_CATEGORY = "(absent)";
1323
- var DEFAULT_MIN_N_PER_FEATURE = 20;
1324
- var DEFAULT_QUANTILE_BUCKETS = 4;
1325
- var REPRESENTATIVE_MIN_FIDELITY = 0.8;
1326
- var TOP_SHIFT_COUNT = 5;
1327
- var defaultBehaviorFeatures = (record) => {
1328
- const raw = record.outcome?.raw ?? {};
1329
- const toolErrors = finiteOrNull(raw.tool_errors);
1330
- const turnsAborted = finiteOrNull(raw.turns_aborted);
1331
- const completion = record.completion;
1332
- return {
1333
- score: finiteOrNull(record.outcome?.holdoutScore) ?? finiteOrNull(record.outcome?.searchScore),
1334
- failure_class: record.failureClass ?? null,
1335
- wall_ms: finiteOrNull(record.wallMs),
1336
- output_tokens: finiteOrNull(record.tokenUsage?.output),
1337
- turn_count: finiteOrNull(raw.turns_completed) ?? finiteOrNull(raw.assistant_messages),
1338
- tool_errors: toolErrors,
1339
- tool_error_recovery: toolErrorRecovery(toolErrors, turnsAborted, record.failureClass),
1340
- completion_length: typeof completion === "string" ? completion.length : null
1341
- };
1342
- };
1343
- function toolErrorRecovery(toolErrors, turnsAborted, failureClass) {
1344
- if (toolErrors === null) return null;
1345
- if (toolErrors === 0) return "no-tool-errors";
1346
- const failed = (turnsAborted ?? 0) > 0 || failureClass !== void 0 && failureClass !== "success";
1347
- return failed ? "unrecovered" : "recovered";
1348
- }
1349
- function finiteOrNull(value) {
1350
- return typeof value === "number" && Number.isFinite(value) ? value : null;
1351
- }
1352
- function jsDivergence(p, q) {
1353
- const keys = /* @__PURE__ */ new Set([...Object.keys(p), ...Object.keys(q)]);
1354
- if (keys.size === 0) {
1355
- throw new ValidationError("jsDivergence: both histograms are empty");
1356
- }
1357
- let pSum = 0;
1358
- let qSum = 0;
1359
- for (const key of keys) {
1360
- const pv = p[key] ?? 0;
1361
- const qv = q[key] ?? 0;
1362
- if (!Number.isFinite(pv) || !Number.isFinite(qv) || pv < 0 || qv < 0) {
1363
- throw new ValidationError(`jsDivergence: negative or non-finite count for category "${key}"`);
1364
- }
1365
- pSum += pv;
1366
- qSum += qv;
1367
- }
1368
- if (pSum === 0 || qSum === 0) {
1369
- throw new ValidationError("jsDivergence: a histogram with zero total mass has no distribution");
1370
- }
1371
- let divergence = 0;
1372
- for (const key of keys) {
1373
- const pp = (p[key] ?? 0) / pSum;
1374
- const qp = (q[key] ?? 0) / qSum;
1375
- const m = (pp + qp) / 2;
1376
- if (pp > 0) divergence += 0.5 * pp * Math.log2(pp / m);
1377
- if (qp > 0) divergence += 0.5 * qp * Math.log2(qp / m);
1378
- }
1379
- return Math.min(1, Math.max(0, divergence));
1380
- }
1381
- function quantileEdges(values, bucketCount = DEFAULT_QUANTILE_BUCKETS) {
1382
- if (values.length === 0) {
1383
- throw new ValidationError("quantileEdges: requires at least one value");
1384
- }
1385
- if (!Number.isInteger(bucketCount) || bucketCount < 2) {
1386
- throw new ValidationError(
1387
- `quantileEdges: bucketCount must be an integer >= 2, got ${bucketCount}`
1388
- );
1389
- }
1390
- const sorted = [...values].sort((a, b) => a - b);
1391
- const edges = [];
1392
- for (let k = 1; k < bucketCount; k++) {
1393
- const pos = k / bucketCount * (sorted.length - 1);
1394
- const lo = sorted[Math.floor(pos)];
1395
- const hi = sorted[Math.ceil(pos)];
1396
- edges.push(lo + (pos - Math.floor(pos)) * (hi - lo));
1397
- }
1398
- return [...new Set(edges)];
1399
- }
1400
- function bucketLabel(value, edges) {
1401
- let i = 0;
1402
- while (i < edges.length && value >= edges[i]) i++;
1403
- const lo = i === 0 ? "-inf" : String(edges[i - 1]);
1404
- const hi = i === edges.length ? "+inf" : String(edges[i]);
1405
- return `[${lo},${hi})`;
1406
- }
1407
- function simFidelityReport(simulated, production, opts = {}) {
1408
- if (simulated.length === 0) {
1409
- throw new ValidationError("simFidelityReport: simulated records are empty");
1410
- }
1411
- if (production.length === 0) {
1412
- throw new ValidationError("simFidelityReport: production records are empty");
1413
- }
1414
- const extract = opts.features ?? defaultBehaviorFeatures;
1415
- const minN = opts.minNPerFeature ?? DEFAULT_MIN_N_PER_FEATURE;
1416
- const simMaps = simulated.map(extract);
1417
- const prodMaps = production.map(extract);
1418
- const featureNames = [];
1419
- const seen = /* @__PURE__ */ new Set();
1420
- for (const map of [...simMaps, ...prodMaps]) {
1421
- for (const name of Object.keys(map)) {
1422
- if (!seen.has(name)) {
1423
- seen.add(name);
1424
- featureNames.push(name);
1425
- }
1426
- }
1427
- }
1428
- const perDimension = [];
1429
- const insufficientData = [];
1430
- for (const feature of featureNames) {
1431
- const simVals = simMaps.map((m) => m[feature] ?? null);
1432
- const prodVals = prodMaps.map((m) => m[feature] ?? null);
1433
- const nSim = simVals.filter((v) => v !== null).length;
1434
- const nProd = prodVals.filter((v) => v !== null).length;
1435
- if (nSim < minN || nProd < minN) {
1436
- insufficientData.push(feature);
1437
- continue;
1438
- }
1439
- const { sim, prod } = histograms(feature, simVals, prodVals);
1440
- perDimension.push({
1441
- feature,
1442
- divergence: jsDivergence(sim, prod),
1443
- topShifts: topShifts(sim, simVals.length, prod, prodVals.length),
1444
- nSim,
1445
- nProd
1446
- });
1447
- }
1448
- if (perDimension.length === 0) {
1449
- return { perDimension, fidelity: Number.NaN, insufficientData, verdict: "insufficient-data" };
1450
- }
1451
- const fidelity = 1 - perDimension.reduce((sum, d) => sum + d.divergence, 0) / perDimension.length;
1452
- return {
1453
- perDimension,
1454
- fidelity,
1455
- insufficientData,
1456
- verdict: fidelity >= REPRESENTATIVE_MIN_FIDELITY ? "representative" : "skewed"
1457
- };
1458
- }
1459
- function histograms(feature, simVals, prodVals) {
1460
- const kinds = /* @__PURE__ */ new Set();
1461
- for (const v of [...simVals, ...prodVals]) {
1462
- if (v !== null) kinds.add(typeof v);
1463
- }
1464
- if (kinds.size > 1) {
1465
- throw new ValidationError(
1466
- `simFidelityReport: feature "${feature}" mixes string and number values \u2014 an extractor must return one kind per feature`
1467
- );
1468
- }
1469
- let toCategory;
1470
- if (kinds.has("number")) {
1471
- const union = [];
1472
- for (const v of [...simVals, ...prodVals]) {
1473
- if (v !== null) union.push(v);
1474
- }
1475
- const edges = quantileEdges(union);
1476
- toCategory = (v) => bucketLabel(v, edges);
1477
- } else {
1478
- toCategory = (v) => v;
1479
- }
1480
- const count = (vals) => {
1481
- const hist = {};
1482
- for (const v of vals) {
1483
- const key = v === null ? ABSENT_CATEGORY : toCategory(v);
1484
- hist[key] = (hist[key] ?? 0) + 1;
1485
- }
1486
- return hist;
1487
- };
1488
- return { sim: count(simVals), prod: count(prodVals) };
1489
- }
1490
- function topShifts(sim, simTotal, prod, prodTotal) {
1491
- const keys = [.../* @__PURE__ */ new Set([...Object.keys(sim), ...Object.keys(prod)])];
1492
- const shifts = keys.map((value) => ({
1493
- value,
1494
- pSim: (sim[value] ?? 0) / simTotal,
1495
- pProd: (prod[value] ?? 0) / prodTotal
1496
- }));
1497
- shifts.sort((a, b) => {
1498
- const delta = Math.abs(b.pSim - b.pProd) - Math.abs(a.pSim - a.pProd);
1499
- return delta !== 0 ? delta : a.value.localeCompare(b.value);
1500
- });
1501
- return shifts.slice(0, TOP_SHIFT_COUNT);
1502
- }
1503
- function easyModeCheck(simulated, production, opts = {}) {
1504
- if (simulated.length === 0) {
1505
- throw new ValidationError("easyModeCheck: simulated records are empty");
1506
- }
1507
- if (production.length === 0) {
1508
- throw new ValidationError("easyModeCheck: production records are empty");
1509
- }
1510
- const threshold = opts.passThreshold ?? 0.5;
1511
- const tolerance = opts.inflationTolerance ?? 0.1;
1512
- const passRate = (records, side) => {
1513
- let passes = 0;
1514
- for (const r of records) {
1515
- const score = finiteOrNull(r.outcome?.holdoutScore) ?? finiteOrNull(r.outcome?.searchScore);
1516
- if (score === null) {
1517
- throw new ValidationError(
1518
- `easyModeCheck: ${side} run "${r.runId}" carries neither holdoutScore nor searchScore`
1519
- );
1520
- }
1521
- if (score >= threshold) passes++;
1522
- }
1523
- return passes / records.length;
1524
- };
1525
- const simPassRate = passRate(simulated, "simulated");
1526
- const prodPassRate = passRate(production, "production");
1527
- const gap = simPassRate - prodPassRate;
1528
- return { simPassRate, prodPassRate, gap, inflated: gap > tolerance };
1529
- }
1530
-
1531
- // src/rl/tournament.ts
1532
- function fitBradleyTerry(outcomes, opts = {}) {
1533
- const tol = opts.tolerance ?? 1e-6;
1534
- const maxIter = opts.maxIterations ?? 256;
1535
- const smoothing = opts.smoothing ?? 0.1;
1536
- const candidates = /* @__PURE__ */ new Set();
1537
- for (const o of outcomes) {
1538
- candidates.add(o.winner);
1539
- candidates.add(o.loser);
1540
- }
1541
- const ids = [...candidates].sort();
1542
- const idx = new Map(ids.map((id, i) => [id, i]));
1543
- const n = ids.length;
1544
- if (n === 0) return { ratings: [], iterations: 0, finalDelta: 0, converged: true };
1545
- if (n === 1) {
1546
- return {
1547
- ratings: [{ candidateId: ids[0], strength: 1, logStrength: 0, n: 0, wins: 0 }],
1548
- iterations: 0,
1549
- finalDelta: 0,
1550
- converged: true
1551
- };
1552
- }
1553
- const W = Array.from({ length: n }, () => new Array(n).fill(0));
1554
- const N = Array.from({ length: n }, () => new Array(n).fill(0));
1555
- for (const o of outcomes) {
1556
- const i = idx.get(o.winner);
1557
- const j = idx.get(o.loser);
1558
- const w = o.weight ?? 1;
1559
- if (o.draw) {
1560
- W[i][j] += 0.5 * w;
1561
- W[j][i] += 0.5 * w;
1562
- } else {
1563
- W[i][j] += w;
1564
- }
1565
- N[i][j] += w;
1566
- N[j][i] += w;
1567
- }
1568
- const winsTotal = new Array(n).fill(0);
1569
- for (let i = 0; i < n; i++) {
1570
- for (let j = 0; j < n; j++) winsTotal[i] += W[i][j];
1571
- winsTotal[i] += smoothing;
1572
- }
1573
- const compsTotal = new Array(n).fill(0);
1574
- for (let i = 0; i < n; i++) {
1575
- for (let j = 0; j < n; j++) compsTotal[i] += N[i][j];
1576
- }
1577
- let theta = new Array(n).fill(1);
1578
- let iter = 0;
1579
- let delta = Infinity;
1580
- for (; iter < maxIter; iter++) {
1581
- const newTheta = new Array(n);
1582
- for (let i = 0; i < n; i++) {
1583
- let denom = 0;
1584
- for (let j = 0; j < n; j++) {
1585
- if (j === i) continue;
1586
- if (N[i][j] === 0) continue;
1587
- denom += N[i][j] / (theta[i] + theta[j]);
1588
- }
1589
- newTheta[i] = denom === 0 ? theta[i] : winsTotal[i] / denom;
1590
- }
1591
- let logSum = 0;
1592
- for (let i = 0; i < n; i++) logSum += Math.log(Math.max(1e-300, newTheta[i]));
1593
- const norm = Math.exp(logSum / n);
1594
- for (let i = 0; i < n; i++) newTheta[i] = newTheta[i] / norm;
1595
- delta = 0;
1596
- for (let i = 0; i < n; i++) {
1597
- const d = Math.abs(newTheta[i] - theta[i]) / Math.max(1e-12, theta[i]);
1598
- if (d > delta) delta = d;
1599
- }
1600
- theta = newTheta;
1601
- if (delta < tol) break;
1602
- }
1603
- const minLog = Math.min(...theta.map((t) => Math.log(Math.max(1e-300, t))));
1604
- const ratings = ids.map((id, i) => ({
1605
- candidateId: id,
1606
- strength: theta[i],
1607
- logStrength: Math.log(Math.max(1e-300, theta[i])) - minLog,
1608
- n: compsTotal[i],
1609
- wins: winsTotal[i] - smoothing
1610
- }));
1611
- return {
1612
- ratings: ratings.sort((a, b) => b.strength - a.strength),
1613
- iterations: iter,
1614
- finalDelta: delta,
1615
- converged: delta < tol
1616
- };
1617
- }
1618
- function applyEloUpdate(ratings, outcome, opts = {}) {
1619
- const defaultRating = opts.defaultRating ?? 1500;
1620
- const k = opts.kFactor ?? 32;
1621
- const rW = ratings.get(outcome.winner) ?? defaultRating;
1622
- const rL = ratings.get(outcome.loser) ?? defaultRating;
1623
- const expectedW = 1 / (1 + 10 ** ((rL - rW) / 400));
1624
- const scoreW = outcome.draw ? 0.5 : 1;
1625
- const scoreL = outcome.draw ? 0.5 : 0;
1626
- const w = outcome.weight ?? 1;
1627
- const winnerDelta = k * w * (scoreW - expectedW);
1628
- const loserDelta = k * w * (scoreL - (1 - expectedW));
1629
- ratings.set(outcome.winner, rW + winnerDelta);
1630
- ratings.set(outcome.loser, rL + loserDelta);
1631
- return { winnerDelta, loserDelta };
1632
- }
1633
- function buildPairwiseFromCampaign(input) {
1634
- const drawMargin = input.drawMargin ?? 0;
1635
- const byKey = /* @__PURE__ */ new Map();
1636
- for (const r of input.runs) {
1637
- const arr = byKey.get(r.matchKey) ?? [];
1638
- arr.push({ candidateId: r.candidateId, score: r.score });
1639
- byKey.set(r.matchKey, arr);
1640
- }
1641
- const outcomes = [];
1642
- for (const arr of byKey.values()) {
1643
- for (let i = 0; i < arr.length; i++) {
1644
- for (let j = i + 1; j < arr.length; j++) {
1645
- const a = arr[i];
1646
- const b = arr[j];
1647
- if (a.candidateId === b.candidateId) continue;
1648
- const margin = Math.abs(a.score - b.score);
1649
- if (margin <= drawMargin) {
1650
- outcomes.push({ winner: a.candidateId, loser: b.candidateId, draw: true, weight: 1 });
1651
- } else {
1652
- const [winner, loser] = a.score > b.score ? [a, b] : [b, a];
1653
- outcomes.push({ winner: winner.candidateId, loser: loser.candidateId, weight: margin });
1654
- }
1655
- }
1656
- }
1657
- }
1658
- return outcomes;
1659
- }
1660
- export {
1661
- ABSENT_CATEGORY,
1662
- DEFAULT_MIN_N_PER_FEATURE,
1663
- DEFAULT_QUANTILE_BUCKETS,
1664
- FileSystemOutcomeStore,
1665
- InMemoryOutcomeStore,
1666
- PredictiveValidityResearcher,
1667
- REPRESENTATIVE_MIN_FIDELITY,
1668
- appendToCorpus,
1669
- applyEloUpdate,
1670
- bestOfN,
1671
- bucketLabel,
1672
- buildDatasetFromCorpus,
1673
- buildPairwiseFromCampaign,
1674
- buildRlDataset,
1675
- campaignToRunRecords,
1676
- compareAdaptationCurves,
1677
- datasheetToMarkdown,
1678
- defaultBehaviorFeatures,
1679
- detectRewardHacking,
1680
- doublyRobust,
1681
- easyModeCheck,
1682
- extractPreferences,
1683
- extractStepRewards,
1684
- extractVerifiableReward,
1685
- extractVerifiableRewardsFromRecords,
1686
- filterDeterministicallyRewarded,
1687
- firstPassK,
1688
- fitBradleyTerry,
1689
- injectIrrelevantClause,
1690
- inverseProbabilityWeighting,
1691
- jsDivergence,
1692
- observationsFromRunRecords,
1693
- offPolicyEstimateAll,
1694
- paretoFrontier,
1695
- prmTrainingPairs,
1696
- quantileEdges,
1697
- readCorpus,
1698
- renameVariables,
1699
- runAdaptationCurve,
1700
- runComputeCurve,
1701
- runContaminationProbe,
1702
- runEvalCampaign,
1703
- runRLCampaign,
1704
- runwiseStepRewardSummary,
1705
- selfConsistency,
1706
- selfNormalizedImportanceWeighting,
1707
- shuffleOrder,
1708
- simFidelityReport,
1709
- stepRewardsToJsonl,
1710
- thompsonCurriculum,
1711
- toAnthropicFormat,
1712
- toDpoJsonl,
1713
- toDpoRows,
1714
- toGrpoJsonl,
1715
- toGrpoRows,
1716
- toPrmJsonl,
1717
- toPrmRows,
1718
- toSftJsonl,
1719
- toSftRows,
1720
- toTRLFormat,
1721
- varianceBasedCurriculum,
1722
- verificationReportToRunRecord
1723
- };
1724
- //# sourceMappingURL=rl.js.map