@tangle-network/agent-eval 0.94.0 → 0.95.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +32 -0
- package/README.md +44 -30
- package/dist/adapters/http.d.ts +8 -7
- package/dist/adapters/http.js.map +1 -1
- package/dist/adapters/langchain.d.ts +3 -2
- package/dist/adapters/otel.d.ts +5 -4
- package/dist/analyst/index.d.ts +11 -31
- package/dist/analyst/index.js +5 -65
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-B6Ljo_dI.d.ts → analyze-runs-DtT6F_6T.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +4 -3
- package/dist/benchmarks/index.d.ts +3 -2
- package/dist/campaign/index.d.ts +727 -616
- package/dist/campaign/index.js +1863 -1316
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-2K6UUZ7P.js → chunk-2T4EZACH.js} +1 -1
- package/dist/chunk-2T4EZACH.js.map +1 -0
- package/dist/{chunk-CTBHKLEU.js → chunk-77T4STFI.js} +59 -86
- package/dist/chunk-77T4STFI.js.map +1 -0
- package/dist/{chunk-EGPMSBEZ.js → chunk-7QTQKIDD.js} +178 -177
- package/dist/chunk-7QTQKIDD.js.map +1 -0
- package/dist/{chunk-MIFZUPEK.js → chunk-AQ5WQAIV.js} +21 -6
- package/dist/chunk-AQ5WQAIV.js.map +1 -0
- package/dist/chunk-DJWX3GVS.js +81 -0
- package/dist/chunk-DJWX3GVS.js.map +1 -0
- package/dist/{chunk-S6OZEZQK.js → chunk-HMA63UEO.js} +37 -9
- package/dist/{chunk-S6OZEZQK.js.map → chunk-HMA63UEO.js.map} +1 -1
- package/dist/{chunk-TBDR6PAI.js → chunk-IZCEK2HR.js} +2 -2
- package/dist/{chunk-SD2YFWQQ.js → chunk-KKWJD5E6.js} +20 -20
- package/dist/chunk-KKWJD5E6.js.map +1 -0
- package/dist/{chunk-KWRRMR3J.js → chunk-LO6IOIJ2.js} +10 -10
- package/dist/chunk-LO6IOIJ2.js.map +1 -0
- package/dist/{chunk-E4GH6USR.js → chunk-NZEQVRH5.js} +2 -2
- package/dist/chunk-NZEQVRH5.js.map +1 -0
- package/dist/{chunk-MPQWFX6Y.js → chunk-PSWWQXHF.js} +13 -88
- package/dist/chunk-PSWWQXHF.js.map +1 -0
- package/dist/{chunk-Q5LIB7BC.js → chunk-S4SYLDFX.js} +2 -2
- package/dist/chunk-S4SYLDFX.js.map +1 -0
- package/dist/{chunk-KW53MSA5.js → chunk-X74V6ESX.js} +2 -2
- package/dist/{chunk-QMUEXQJS.js → chunk-YBIGNSCZ.js} +81 -4
- package/dist/chunk-YBIGNSCZ.js.map +1 -0
- package/dist/{chunk-2KNZHH3P.js → chunk-Z6L6YSU6.js} +2 -2
- package/dist/{code-agent-session-BO8nCnv3.d.ts → code-agent-session-CPHRCb4-.d.ts} +1 -1
- package/dist/contract/index.d.ts +91 -43
- package/dist/contract/index.js +127 -17
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-D6qwHXIR.d.ts → control-Doncu-B_.d.ts} +2 -2
- package/dist/control.d.ts +3 -2
- package/dist/control.js +2 -2
- package/dist/{corpus-B8A4BDR3.d.ts → corpus-D4YW9UoJ.d.ts} +1 -1
- package/dist/{default-registry-6dhErQbs.d.ts → default-registry-GyE8X5SP.d.ts} +3 -3
- package/dist/diagnose.d.ts +4 -3
- package/dist/diagnose.js +1 -1
- package/dist/{run-improvement-loop-DBahB8Ax.d.ts → gepa-C1NCIZ9o.d.ts} +117 -130
- package/dist/hosted/index.d.ts +5 -4
- package/dist/{index-Bx3gZ8xl.d.ts → index-_Y4oNOOb.d.ts} +1 -1
- package/dist/index.d.ts +76 -81
- package/dist/index.js +66 -31
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DWl3z9tl.d.ts → insight-report-BnRjTibG.d.ts} +1 -1
- package/dist/{kind-factory-0BhLSI27.d.ts → kind-factory-X3eDYbKn.d.ts} +2 -3
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +3 -2
- package/dist/multishot/index.d.ts +4 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{pre-registration-mAnCugl9.d.ts → pre-registration-nfUdc9EQ.d.ts} +2 -42
- package/dist/{provenance-P-bCL2Fo.d.ts → provenance-CncDq9qE.d.ts} +26 -41
- package/dist/{release-report-BEbWmVYj.d.ts → release-report-pidWUMZ2.d.ts} +2 -2
- package/dist/reporting.d.ts +5 -4
- package/dist/{researcher-B0C2_fVO.d.ts → researcher-Jr8ME1dZ.d.ts} +2 -2
- package/dist/rl.d.ts +516 -515
- package/dist/rl.js +612 -612
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-Cy_W-hWZ.d.ts → rubric-predictive-validity-C2hDKM8Z.d.ts} +1 -1
- package/dist/{run-campaign-7WNXMDSN.js → run-campaign-WXY7KI67.js} +2 -2
- package/dist/{run-record-e7vj1uZQ.d.ts → run-record-CP2ObebC.d.ts} +14 -18
- package/dist/{runtime-trajectory-BDgfGZSr.d.ts → runtime-trajectory-BOUUjI0y.d.ts} +1 -1
- package/dist/{semantic-concept-judge-B9MgmBnM.d.ts → semantic-concept-judge-DSBB2Cfp.d.ts} +2 -2
- package/dist/{summary-report-BDOFevaT.d.ts → summary-report-CInXwsza.d.ts} +1 -1
- package/dist/testing-C21CHsq2.d.ts +20 -0
- package/dist/testing.d.ts +1 -0
- package/dist/testing.js +8 -0
- package/dist/testing.js.map +1 -0
- package/dist/traces.d.ts +26 -10
- package/dist/traces.js +41 -11
- package/dist/{types-Ce17tDlG.d.ts → types-B5x54y6n.d.ts} +1 -1
- package/dist/{types-mn5Aqk7x.d.ts → types-BUxNaJ8c.d.ts} +2 -4
- package/dist/{types-BU-7W85F.d.ts → types-DQRY8ZT-.d.ts} +60 -58
- package/dist/workflow/index.d.ts +5 -4
- package/dist/workflow/index.js +1 -1
- package/docs/campaign-proposers.md +170 -0
- package/docs/concepts.md +8 -4
- package/docs/customer-journeys.md +15 -13
- package/docs/design/loop-taxonomy.md +34 -66
- package/docs/distributed-driver.md +14 -14
- package/docs/feature-guide.md +1 -1
- package/docs/hosted-ingest-spec.md +2 -3
- package/docs/multi-shot-optimization.md +8 -8
- package/docs/product-eval-adoption.md +1 -1
- package/docs/self-improvement-map.md +33 -29
- package/package.json +8 -14
- package/dist/chunk-2K6UUZ7P.js.map +0 -1
- package/dist/chunk-CTBHKLEU.js.map +0 -1
- package/dist/chunk-E4GH6USR.js.map +0 -1
- package/dist/chunk-EGPMSBEZ.js.map +0 -1
- package/dist/chunk-KWRRMR3J.js.map +0 -1
- package/dist/chunk-MIFZUPEK.js.map +0 -1
- package/dist/chunk-MPQWFX6Y.js.map +0 -1
- package/dist/chunk-Q5LIB7BC.js.map +0 -1
- package/dist/chunk-QMUEXQJS.js.map +0 -1
- package/dist/chunk-SD2YFWQQ.js.map +0 -1
- package/docs/design/external-agent-wedge.md +0 -89
- package/docs/design/phase-d-rfc.md +0 -125
- package/docs/design/phase4-consumer-migration.md +0 -70
- package/docs/design/primitives-integration-spec.md +0 -393
- package/docs/design/product-self-improvement-loop.md +0 -146
- package/docs/design/self-improvement-engine.md +0 -140
- package/docs/design/self-improvement-protocol.md +0 -223
- package/docs/design/self-improvement-roadmap.md +0 -106
- package/docs/design/substrate-gaps.md +0 -118
- package/docs/phase-b-pairing-kit.md +0 -188
- package/docs/phase-b-runbook.md +0 -176
- package/docs/pilot/README.md +0 -62
- package/docs/pilot/customer-checklist.md +0 -90
- package/docs/pilot/integration-foreign-stack.md +0 -296
- package/docs/pilot/integration-tangle-stack.md +0 -248
- package/docs/pilot/one-pager.md +0 -161
- package/docs/pilot/sample-insight-report.json +0 -172
- package/docs/quickstart-external.md +0 -229
- package/docs/research/belief-state-agent-eval-roadmap.md +0 -593
- package/docs/research/research-roadmap.md +0 -205
- package/docs/specs/driver-honest-spec.md +0 -251
- package/docs/specs/hermes-self-improvement-audit.md +0 -93
- package/docs/specs/profile-versioning.md +0 -291
- package/docs/three-package-architecture.md +0 -168
- /package/dist/{chunk-TBDR6PAI.js.map → chunk-IZCEK2HR.js.map} +0 -0
- /package/dist/{chunk-KW53MSA5.js.map → chunk-X74V6ESX.js.map} +0 -0
- /package/dist/{chunk-2KNZHH3P.js.map → chunk-Z6L6YSU6.js.map} +0 -0
- /package/dist/{run-campaign-7WNXMDSN.js.map → run-campaign-WXY7KI67.js.map} +0 -0
package/dist/rl.js
CHANGED
|
@@ -16,7 +16,7 @@ import {
|
|
|
16
16
|
} from "./chunk-3RF76KTD.js";
|
|
17
17
|
import {
|
|
18
18
|
runEvalCampaign
|
|
19
|
-
} from "./chunk-
|
|
19
|
+
} from "./chunk-X74V6ESX.js";
|
|
20
20
|
import "./chunk-CWNP4DV4.js";
|
|
21
21
|
import {
|
|
22
22
|
rubricPredictiveValidity
|
|
@@ -36,7 +36,7 @@ import {
|
|
|
36
36
|
} from "./chunk-VZSRQ272.js";
|
|
37
37
|
import "./chunk-SBCB6VZY.js";
|
|
38
38
|
import "./chunk-PC4UYEBM.js";
|
|
39
|
-
import "./chunk-
|
|
39
|
+
import "./chunk-LO6IOIJ2.js";
|
|
40
40
|
import "./chunk-TVVP3ZZQ.js";
|
|
41
41
|
import "./chunk-VSMTAMNK.js";
|
|
42
42
|
import {
|
|
@@ -44,6 +44,197 @@ import {
|
|
|
44
44
|
} from "./chunk-3BFEG2F6.js";
|
|
45
45
|
import "./chunk-PZ5AY32C.js";
|
|
46
46
|
|
|
47
|
+
// src/rl/adaptation-eval.ts
|
|
48
|
+
async function runAdaptationCurve(opts) {
|
|
49
|
+
const ks = opts.ks ?? [0, 1, 2, 4, 8, 16];
|
|
50
|
+
const reps = opts.reps ?? 3;
|
|
51
|
+
const passThreshold = opts.passThreshold ?? 0.5;
|
|
52
|
+
const sortedKs = [...ks].sort((a, b) => a - b);
|
|
53
|
+
const points = [];
|
|
54
|
+
for (const k of sortedKs) {
|
|
55
|
+
const perScenario = [];
|
|
56
|
+
const allScores = [];
|
|
57
|
+
let totalPasses = 0;
|
|
58
|
+
let totalAttempts = 0;
|
|
59
|
+
for (const scenario of opts.scenarios) {
|
|
60
|
+
const sid = scenario.scenarioId ?? `scenario-${opts.scenarios.indexOf(scenario)}`;
|
|
61
|
+
const scores = [];
|
|
62
|
+
let passes = 0;
|
|
63
|
+
for (let r = 0; r < reps; r++) {
|
|
64
|
+
const score = await opts.runner.run({ scenario, k, rep: r });
|
|
65
|
+
scores.push(score);
|
|
66
|
+
if (score >= passThreshold) passes++;
|
|
67
|
+
allScores.push(score);
|
|
68
|
+
if (score >= passThreshold) totalPasses++;
|
|
69
|
+
totalAttempts++;
|
|
70
|
+
}
|
|
71
|
+
const meanS = scores.reduce((s, v) => s + v, 0) / scores.length;
|
|
72
|
+
perScenario.push({ scenarioId: sid, meanScore: meanS, passes, total: scores.length });
|
|
73
|
+
}
|
|
74
|
+
const meanScore = allScores.reduce((s, v) => s + v, 0) / Math.max(1, allScores.length);
|
|
75
|
+
const variance = allScores.length < 2 ? 0 : allScores.reduce((s, v) => s + (v - meanScore) ** 2, 0) / (allScores.length - 1);
|
|
76
|
+
points.push({
|
|
77
|
+
k,
|
|
78
|
+
meanScore,
|
|
79
|
+
passRate: totalPasses / Math.max(1, totalAttempts),
|
|
80
|
+
std: Math.sqrt(variance),
|
|
81
|
+
n: allScores.length,
|
|
82
|
+
perScenario
|
|
83
|
+
});
|
|
84
|
+
}
|
|
85
|
+
const firstPassK2 = points.find((p) => p.passRate >= passThreshold)?.k ?? null;
|
|
86
|
+
const maxK = sortedKs[sortedKs.length - 1] ?? 1;
|
|
87
|
+
let area = 0;
|
|
88
|
+
for (let i = 1; i < points.length; i++) {
|
|
89
|
+
const x1 = points[i - 1].k;
|
|
90
|
+
const x2 = points[i].k;
|
|
91
|
+
const y1 = points[i - 1].meanScore;
|
|
92
|
+
const y2 = points[i].meanScore;
|
|
93
|
+
area += (y1 + y2) / 2 * (x2 - x1);
|
|
94
|
+
}
|
|
95
|
+
const adaptationArea = maxK === 0 ? 0 : area / maxK;
|
|
96
|
+
return { points, firstPassK: firstPassK2, adaptationArea };
|
|
97
|
+
}
|
|
98
|
+
function compareAdaptationCurves(a, b, opts = {}) {
|
|
99
|
+
const conf = opts.confidence ?? 0.95;
|
|
100
|
+
const resamples = opts.bootstrapResamples ?? 500;
|
|
101
|
+
const rng = makeRng(opts.seed);
|
|
102
|
+
const perK = [];
|
|
103
|
+
for (const ap of a.points) {
|
|
104
|
+
const bp = b.points.find((p) => p.k === ap.k);
|
|
105
|
+
if (!bp) continue;
|
|
106
|
+
const aMeans = ap.perScenario.map((s) => s.meanScore);
|
|
107
|
+
const bMeans = bp.perScenario.map((s) => s.meanScore);
|
|
108
|
+
const aCi = bootstrapMeanCi(aMeans, resamples, conf, rng);
|
|
109
|
+
const bCi = bootstrapMeanCi(bMeans, resamples, conf, rng);
|
|
110
|
+
perK.push({
|
|
111
|
+
k: ap.k,
|
|
112
|
+
deltaMean: ap.meanScore - bp.meanScore,
|
|
113
|
+
aLow: aCi.low,
|
|
114
|
+
aHigh: aCi.high,
|
|
115
|
+
bLow: bCi.low,
|
|
116
|
+
bHigh: bCi.high
|
|
117
|
+
});
|
|
118
|
+
}
|
|
119
|
+
const areaDelta = a.adaptationArea - b.adaptationArea;
|
|
120
|
+
const firstPassKDelta = a.firstPassK !== null && b.firstPassK !== null ? b.firstPassK - a.firstPassK : null;
|
|
121
|
+
const meanDelta = perK.reduce((s, p) => s + p.deltaMean, 0) / Math.max(1, perK.length);
|
|
122
|
+
let verdict;
|
|
123
|
+
if (Math.abs(meanDelta) < 0.02 && Math.abs(areaDelta) < 0.02) verdict = "similar";
|
|
124
|
+
else if (meanDelta > 0 && areaDelta > 0) verdict = "a_better";
|
|
125
|
+
else if (meanDelta < 0 && areaDelta < 0) verdict = "b_better";
|
|
126
|
+
else verdict = "similar";
|
|
127
|
+
const rationale = `mean per-k delta=${meanDelta.toFixed(3)}, area delta=${areaDelta.toFixed(3)}` + (firstPassKDelta !== null ? `, first-pass-k delta=${firstPassKDelta}` : "");
|
|
128
|
+
return { perK, areaDelta, firstPassKDelta, verdict, rationale };
|
|
129
|
+
}
|
|
130
|
+
function firstPassK(curve, threshold = 0.5) {
|
|
131
|
+
return curve.points.find((p) => p.passRate >= threshold)?.k ?? null;
|
|
132
|
+
}
|
|
133
|
+
function makeRng(seed) {
|
|
134
|
+
if (seed === void 0) return Math.random;
|
|
135
|
+
let s = seed >>> 0;
|
|
136
|
+
return () => {
|
|
137
|
+
s = s + 1831565813 >>> 0;
|
|
138
|
+
let t = s;
|
|
139
|
+
t = Math.imul(t ^ t >>> 15, t | 1);
|
|
140
|
+
t ^= t + Math.imul(t ^ t >>> 7, t | 61);
|
|
141
|
+
return ((t ^ t >>> 14) >>> 0) / 4294967296;
|
|
142
|
+
};
|
|
143
|
+
}
|
|
144
|
+
function bootstrapMeanCi(xs, resamples, confidence, rng) {
|
|
145
|
+
if (xs.length < 2) return { low: xs[0] ?? 0, high: xs[0] ?? 0 };
|
|
146
|
+
const samples = new Array(resamples);
|
|
147
|
+
for (let b = 0; b < resamples; b++) {
|
|
148
|
+
let sum = 0;
|
|
149
|
+
for (let i = 0; i < xs.length; i++) sum += xs[Math.floor(rng() * xs.length)];
|
|
150
|
+
samples[b] = sum / xs.length;
|
|
151
|
+
}
|
|
152
|
+
samples.sort((a, b) => a - b);
|
|
153
|
+
const alpha = 1 - confidence;
|
|
154
|
+
return {
|
|
155
|
+
low: samples[Math.floor(alpha / 2 * resamples)],
|
|
156
|
+
high: samples[Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)]
|
|
157
|
+
};
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// src/rl/adversarial.ts
|
|
161
|
+
async function adversarialScenarioSearch(opts) {
|
|
162
|
+
const failureThreshold = opts.failureThreshold ?? 0.5;
|
|
163
|
+
const rounds = opts.rounds ?? 3;
|
|
164
|
+
const children = opts.childrenPerParent ?? 4;
|
|
165
|
+
const budget = opts.budget ?? Number.POSITIVE_INFINITY;
|
|
166
|
+
const seed = opts.seed ?? 1;
|
|
167
|
+
const rng = mulberry32(seed);
|
|
168
|
+
const scenarios = [];
|
|
169
|
+
const seen = /* @__PURE__ */ new Set();
|
|
170
|
+
let scoreCalls = 0;
|
|
171
|
+
for (const s of opts.seeds) {
|
|
172
|
+
const id = opts.mutateScenarioId(s);
|
|
173
|
+
if (seen.has(id)) continue;
|
|
174
|
+
seen.add(id);
|
|
175
|
+
if (scoreCalls >= budget) break;
|
|
176
|
+
const score = await opts.scoreFn(s);
|
|
177
|
+
scoreCalls++;
|
|
178
|
+
scenarios.push({
|
|
179
|
+
id,
|
|
180
|
+
generation: 0,
|
|
181
|
+
parentId: null,
|
|
182
|
+
scenario: s,
|
|
183
|
+
score,
|
|
184
|
+
mutationStrategy: null
|
|
185
|
+
});
|
|
186
|
+
}
|
|
187
|
+
for (let g = 1; g <= rounds; g++) {
|
|
188
|
+
if (scoreCalls >= budget) break;
|
|
189
|
+
const parents = scenarios.filter((s) => s.generation === g - 1);
|
|
190
|
+
for (const parent of parents) {
|
|
191
|
+
for (const mutation of opts.mutations) {
|
|
192
|
+
if (scoreCalls >= budget) break;
|
|
193
|
+
const produced = await mutation.mutate(parent.scenario, rng);
|
|
194
|
+
const childArr = Array.isArray(produced) ? produced : [produced];
|
|
195
|
+
for (let k = 0; k < Math.min(children, childArr.length); k++) {
|
|
196
|
+
if (scoreCalls >= budget) break;
|
|
197
|
+
const child = childArr[k];
|
|
198
|
+
const cid = opts.mutateScenarioId(child);
|
|
199
|
+
if (seen.has(cid)) continue;
|
|
200
|
+
seen.add(cid);
|
|
201
|
+
const cscore = await opts.scoreFn(child);
|
|
202
|
+
scoreCalls++;
|
|
203
|
+
scenarios.push({
|
|
204
|
+
id: cid,
|
|
205
|
+
generation: g,
|
|
206
|
+
parentId: parent.id,
|
|
207
|
+
scenario: child,
|
|
208
|
+
score: cscore,
|
|
209
|
+
mutationStrategy: mutation.id
|
|
210
|
+
});
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
const failures = scenarios.filter((s) => s.score !== null && s.score < failureThreshold).sort((a, b) => (a.score ?? 0) - (b.score ?? 0));
|
|
216
|
+
const byGeneration = [];
|
|
217
|
+
const maxGen = scenarios.reduce((m, s) => Math.max(m, s.generation), 0);
|
|
218
|
+
for (let g = 0; g <= maxGen; g++) {
|
|
219
|
+
const gens = scenarios.filter((s) => s.generation === g);
|
|
220
|
+
if (gens.length === 0) continue;
|
|
221
|
+
const fails = gens.filter((s) => s.score !== null && s.score < failureThreshold).length;
|
|
222
|
+
const meanScore = gens.reduce((sum, s) => sum + (s.score ?? 0), 0) / gens.length;
|
|
223
|
+
byGeneration.push({ generation: g, total: gens.length, failures: fails, meanScore });
|
|
224
|
+
}
|
|
225
|
+
return { scenarios, failures, byGeneration, scoreCalls };
|
|
226
|
+
}
|
|
227
|
+
function mulberry32(seed) {
|
|
228
|
+
let s = seed >>> 0;
|
|
229
|
+
return () => {
|
|
230
|
+
s = s + 1831565813 >>> 0;
|
|
231
|
+
let t = s;
|
|
232
|
+
t = Math.imul(t ^ t >>> 15, t | 1);
|
|
233
|
+
t ^= t + Math.imul(t ^ t >>> 7, t | 61);
|
|
234
|
+
return ((t ^ t >>> 14) >>> 0) / 4294967296;
|
|
235
|
+
};
|
|
236
|
+
}
|
|
237
|
+
|
|
47
238
|
// src/rl/compute-curves.ts
|
|
48
239
|
async function runComputeCurve(opts) {
|
|
49
240
|
const points = [];
|
|
@@ -182,631 +373,65 @@ async function runContaminationProbe(input, opts = {}) {
|
|
|
182
373
|
const pseudoP = valid.map((p) => Math.min(1, Math.max(1e-6, 1 - Math.abs(p.delta) / 1)));
|
|
183
374
|
const { qValues } = benjaminiHochberg(pseudoP, fdr);
|
|
184
375
|
for (let i = 0; i < valid.length; i++) {
|
|
185
|
-
const v = valid[i];
|
|
186
|
-
const idx = perScenario.findIndex((p) => p.scenarioId === v.scenarioId);
|
|
187
|
-
if (idx >= 0) perScenario[idx].qValue = qValues[i];
|
|
188
|
-
}
|
|
189
|
-
const contaminationSuspected = pairedTest.p < fdr && median <= -minMedianDrop;
|
|
190
|
-
const reason = contaminationSuspected ? `paired p=${pairedTest.p.toFixed(4)} < ${fdr} and median drop ${median.toFixed(4)} \u2265 ${minMedianDrop}` : pairedTest.p >= fdr ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})` : `significant but small effect (median delta ${median.toFixed(4)})`;
|
|
191
|
-
return {
|
|
192
|
-
perScenario,
|
|
193
|
-
pairedTest,
|
|
194
|
-
medianDelta: median,
|
|
195
|
-
meanDelta: mean,
|
|
196
|
-
contaminationSuspected,
|
|
197
|
-
reason,
|
|
198
|
-
n: valid.length
|
|
199
|
-
};
|
|
200
|
-
}
|
|
201
|
-
function renameVariables(identifiers, rename = (n, i) => `${n}_${(i % 26 + 10).toString(36)}`) {
|
|
202
|
-
return {
|
|
203
|
-
kind: "rename_variables",
|
|
204
|
-
apply(scenario) {
|
|
205
|
-
let prompt = scenario.prompt;
|
|
206
|
-
identifiers.forEach((id, i) => {
|
|
207
|
-
const replacement = rename(id, i);
|
|
208
|
-
const re = new RegExp(`\\b${escapeRegex(id)}\\b`, "g");
|
|
209
|
-
prompt = prompt.replace(re, replacement);
|
|
210
|
-
});
|
|
211
|
-
return { ...scenario, prompt };
|
|
212
|
-
}
|
|
213
|
-
};
|
|
214
|
-
}
|
|
215
|
-
function shuffleOrder(shuffleSection, seed) {
|
|
216
|
-
let s = seed >>> 0;
|
|
217
|
-
const rng = () => {
|
|
218
|
-
s = s + 1831565813 >>> 0;
|
|
219
|
-
let t = s;
|
|
220
|
-
t = Math.imul(t ^ t >>> 15, t | 1);
|
|
221
|
-
t ^= t + Math.imul(t ^ t >>> 7, t | 61);
|
|
222
|
-
return ((t ^ t >>> 14) >>> 0) / 4294967296;
|
|
223
|
-
};
|
|
224
|
-
return {
|
|
225
|
-
kind: "shuffle_order",
|
|
226
|
-
apply(scenario) {
|
|
227
|
-
const newPrompt = shuffleSection(scenario.prompt, rng);
|
|
228
|
-
return { ...scenario, prompt: newPrompt };
|
|
229
|
-
}
|
|
230
|
-
};
|
|
231
|
-
}
|
|
232
|
-
function injectIrrelevantClause(clause, position = "prefix") {
|
|
233
|
-
return {
|
|
234
|
-
kind: "inject_irrelevant_clause",
|
|
235
|
-
apply(scenario) {
|
|
236
|
-
const prompt = position === "prefix" ? `${clause} ${scenario.prompt}` : `${scenario.prompt} ${clause}`;
|
|
237
|
-
return { ...scenario, prompt };
|
|
238
|
-
}
|
|
239
|
-
};
|
|
240
|
-
}
|
|
241
|
-
function escapeRegex(s) {
|
|
242
|
-
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
243
|
-
}
|
|
244
|
-
|
|
245
|
-
// src/rl/preferences.ts
|
|
246
|
-
var SPLIT_TAG_DEFAULT = "holdout";
|
|
247
|
-
var DEFAULT_REWARD = (run) => {
|
|
248
|
-
const v = run.outcome.holdoutScore ?? run.outcome.searchScore;
|
|
249
|
-
return typeof v === "number" && Number.isFinite(v) ? v : null;
|
|
250
|
-
};
|
|
251
|
-
function extractPreferences(runs, opts = {}) {
|
|
252
|
-
const strategy = opts.strategy ?? "paired-by-scenario-and-seed";
|
|
253
|
-
const minMargin = opts.minMargin ?? 0.05;
|
|
254
|
-
const splitTag = opts.splitTag ?? SPLIT_TAG_DEFAULT;
|
|
255
|
-
const rewardOf2 = opts.rewardOf ?? DEFAULT_REWARD;
|
|
256
|
-
const filtered = runs.filter((r) => r.splitTag === splitTag);
|
|
257
|
-
const scoredEntries = [];
|
|
258
|
-
for (const run of filtered) {
|
|
259
|
-
const s = rewardOf2(run);
|
|
260
|
-
if (s === null) continue;
|
|
261
|
-
scoredEntries.push({ run, score: s });
|
|
262
|
-
}
|
|
263
|
-
const pairs = [];
|
|
264
|
-
let pairsBelowMargin = 0;
|
|
265
|
-
let cellsSingleton = 0;
|
|
266
|
-
let cellsInspected = 0;
|
|
267
|
-
if (strategy === "paired-by-scenario-and-seed") {
|
|
268
|
-
const groups = /* @__PURE__ */ new Map();
|
|
269
|
-
for (const e of scoredEntries) {
|
|
270
|
-
const sid = scenarioOf(e.run);
|
|
271
|
-
const key = `${sid}::${e.run.seed}`;
|
|
272
|
-
const arr = groups.get(key) ?? [];
|
|
273
|
-
arr.push(e);
|
|
274
|
-
groups.set(key, arr);
|
|
275
|
-
}
|
|
276
|
-
for (const [key, members] of groups.entries()) {
|
|
277
|
-
cellsInspected++;
|
|
278
|
-
if (members.length < 2) {
|
|
279
|
-
cellsSingleton++;
|
|
280
|
-
continue;
|
|
281
|
-
}
|
|
282
|
-
for (let i = 0; i < members.length; i++) {
|
|
283
|
-
for (let j = i + 1; j < members.length; j++) {
|
|
284
|
-
const a = members[i];
|
|
285
|
-
const b = members[j];
|
|
286
|
-
if (a.run.candidateId === b.run.candidateId) continue;
|
|
287
|
-
const result = makePair(a, b, key.split("::")[0], minMargin);
|
|
288
|
-
if (result.kind === "admit") pairs.push(result.pair);
|
|
289
|
-
else pairsBelowMargin++;
|
|
290
|
-
}
|
|
291
|
-
}
|
|
292
|
-
}
|
|
293
|
-
} else if (strategy === "paired-by-scenario") {
|
|
294
|
-
const byScenarioVariant = /* @__PURE__ */ new Map();
|
|
295
|
-
for (const e of scoredEntries) {
|
|
296
|
-
const sid = scenarioOf(e.run);
|
|
297
|
-
let perScenario = byScenarioVariant.get(sid);
|
|
298
|
-
if (!perScenario) {
|
|
299
|
-
perScenario = /* @__PURE__ */ new Map();
|
|
300
|
-
byScenarioVariant.set(sid, perScenario);
|
|
301
|
-
}
|
|
302
|
-
const cur = perScenario.get(e.run.candidateId);
|
|
303
|
-
if (cur) {
|
|
304
|
-
cur.sum += e.score;
|
|
305
|
-
cur.n++;
|
|
306
|
-
} else perScenario.set(e.run.candidateId, { run: e.run, sum: e.score, n: 1 });
|
|
307
|
-
}
|
|
308
|
-
for (const [sid, perVariant] of byScenarioVariant.entries()) {
|
|
309
|
-
cellsInspected++;
|
|
310
|
-
const arr = [...perVariant.entries()].map(([vid, agg]) => ({
|
|
311
|
-
run: agg.run,
|
|
312
|
-
score: agg.sum / agg.n,
|
|
313
|
-
variantId: vid
|
|
314
|
-
}));
|
|
315
|
-
if (arr.length < 2) {
|
|
316
|
-
cellsSingleton++;
|
|
317
|
-
continue;
|
|
318
|
-
}
|
|
319
|
-
for (let i = 0; i < arr.length; i++) {
|
|
320
|
-
for (let j = i + 1; j < arr.length; j++) {
|
|
321
|
-
const result = makePair(arr[i], arr[j], sid, minMargin);
|
|
322
|
-
if (result.kind === "admit") pairs.push(result.pair);
|
|
323
|
-
else pairsBelowMargin++;
|
|
324
|
-
}
|
|
325
|
-
}
|
|
326
|
-
}
|
|
327
|
-
} else {
|
|
328
|
-
const byScenario = /* @__PURE__ */ new Map();
|
|
329
|
-
for (const e of scoredEntries) {
|
|
330
|
-
const sid = scenarioOf(e.run);
|
|
331
|
-
const arr = byScenario.get(sid) ?? [];
|
|
332
|
-
arr.push(e);
|
|
333
|
-
byScenario.set(sid, arr);
|
|
334
|
-
}
|
|
335
|
-
for (const [sid, arr] of byScenario.entries()) {
|
|
336
|
-
cellsInspected++;
|
|
337
|
-
if (arr.length < 2) {
|
|
338
|
-
cellsSingleton++;
|
|
339
|
-
continue;
|
|
340
|
-
}
|
|
341
|
-
const sorted = [...arr].sort((a, b) => a.score - b.score);
|
|
342
|
-
const top = sorted[sorted.length - 1];
|
|
343
|
-
const bot = sorted[0];
|
|
344
|
-
if (top.run.candidateId === bot.run.candidateId) {
|
|
345
|
-
cellsSingleton++;
|
|
346
|
-
continue;
|
|
347
|
-
}
|
|
348
|
-
const result = makePair(bot, top, sid, minMargin);
|
|
349
|
-
if (result.kind === "admit") pairs.push(result.pair);
|
|
350
|
-
else pairsBelowMargin++;
|
|
351
|
-
}
|
|
352
|
-
}
|
|
353
|
-
return { pairs, cellsInspected, pairsBelowMargin, cellsSingleton, strategy };
|
|
354
|
-
}
|
|
355
|
-
function toTRLFormat(triples, promptOf) {
|
|
356
|
-
return triples.map((t) => ({
|
|
357
|
-
prompt: promptOf(t.meta.chosenPromptHash),
|
|
358
|
-
chosen: t.meta.chosenPromptHash,
|
|
359
|
-
// caller substitutes the model output via the runId map
|
|
360
|
-
rejected: t.meta.rejectedPromptHash
|
|
361
|
-
}));
|
|
362
|
-
}
|
|
363
|
-
function toAnthropicFormat(triples) {
|
|
364
|
-
return triples.map((t) => ({
|
|
365
|
-
scenarioId: t.scenarioId,
|
|
366
|
-
chosenRunId: t.chosenRunId,
|
|
367
|
-
rejectedRunId: t.rejectedRunId,
|
|
368
|
-
margin: t.marginScore
|
|
369
|
-
}));
|
|
370
|
-
}
|
|
371
|
-
function makePair(a, b, scenarioId, minMargin) {
|
|
372
|
-
const margin = Math.abs(a.score - b.score);
|
|
373
|
-
if (margin < minMargin) return { kind: "reject" };
|
|
374
|
-
const [chosen, rejected] = a.score > b.score ? [a, b] : [b, a];
|
|
375
|
-
return {
|
|
376
|
-
kind: "admit",
|
|
377
|
-
pair: {
|
|
378
|
-
scenarioId,
|
|
379
|
-
chosenRunId: chosen.run.runId,
|
|
380
|
-
rejectedRunId: rejected.run.runId,
|
|
381
|
-
chosenVariantId: chosen.run.candidateId,
|
|
382
|
-
rejectedVariantId: rejected.run.candidateId,
|
|
383
|
-
marginScore: chosen.score - rejected.score,
|
|
384
|
-
scores: { chosen: chosen.score, rejected: rejected.score },
|
|
385
|
-
seed: chosen.run.seed === rejected.run.seed ? chosen.run.seed : void 0,
|
|
386
|
-
meta: {
|
|
387
|
-
chosenPromptHash: chosen.run.promptHash,
|
|
388
|
-
rejectedPromptHash: rejected.run.promptHash,
|
|
389
|
-
chosenConfigHash: chosen.run.configHash,
|
|
390
|
-
rejectedConfigHash: rejected.run.configHash,
|
|
391
|
-
chosenModel: chosen.run.model,
|
|
392
|
-
rejectedModel: rejected.run.model
|
|
393
|
-
}
|
|
394
|
-
}
|
|
395
|
-
};
|
|
396
|
-
}
|
|
397
|
-
function scenarioOf(run) {
|
|
398
|
-
if (typeof run.scenarioId === "string" && run.scenarioId.length > 0) return run.scenarioId;
|
|
399
|
-
const fromRaw = run.outcome.raw.scenario_id;
|
|
400
|
-
if (typeof fromRaw === "number" && Number.isFinite(fromRaw)) return String(fromRaw);
|
|
401
|
-
if (typeof fromRaw === "string") return fromRaw;
|
|
402
|
-
return run.experimentId;
|
|
403
|
-
}
|
|
404
|
-
|
|
405
|
-
// src/rl/run-record-adapters.ts
|
|
406
|
-
function campaignToRunRecords(campaign, ctx) {
|
|
407
|
-
const splitTag = ctx.splitTag ?? "search";
|
|
408
|
-
const candidateId = ctx.candidateId ?? campaign.manifestHash;
|
|
409
|
-
return campaign.cells.map((cell) => {
|
|
410
|
-
const composites = Object.values(cell.judgeScores).map((s) => s.composite);
|
|
411
|
-
const score = composites.length > 0 ? composites.reduce((a, b) => a + b, 0) / composites.length : 0;
|
|
412
|
-
const raw = { rep: cell.rep, duration_ms: cell.durationMs };
|
|
413
|
-
for (const judge of Object.values(cell.judgeScores)) {
|
|
414
|
-
for (const [dim, value] of Object.entries(judge.dimensions)) {
|
|
415
|
-
if (Number.isFinite(value)) raw[`dim.${dim}`] = value;
|
|
416
|
-
}
|
|
417
|
-
}
|
|
418
|
-
if (typeof cell.generation === "number") raw.generation = cell.generation;
|
|
419
|
-
const outcome = { raw };
|
|
420
|
-
if (splitTag === "holdout") outcome.holdoutScore = score;
|
|
421
|
-
else outcome.searchScore = score;
|
|
422
|
-
return {
|
|
423
|
-
runId: cell.cellId,
|
|
424
|
-
experimentId: ctx.experimentId,
|
|
425
|
-
candidateId,
|
|
426
|
-
seed: cell.seed,
|
|
427
|
-
model: ctx.model,
|
|
428
|
-
promptHash: ctx.promptHash,
|
|
429
|
-
configHash: ctx.configHash,
|
|
430
|
-
commitSha: ctx.commitSha,
|
|
431
|
-
wallMs: cell.durationMs,
|
|
432
|
-
costUsd: Number.isFinite(cell.costUsd) ? cell.costUsd : ctx.defaultCostUsd ?? 0,
|
|
433
|
-
tokenUsage: { input: 0, output: 0 },
|
|
434
|
-
outcome,
|
|
435
|
-
failureMode: cell.error ? "cell_error" : void 0,
|
|
436
|
-
splitTag,
|
|
437
|
-
scenarioId: cell.scenarioId
|
|
438
|
-
};
|
|
439
|
-
});
|
|
440
|
-
}
|
|
441
|
-
function verificationReportToRunRecord(report, ctx, opts = {}) {
|
|
442
|
-
const splitTag = ctx.splitTag ?? "search";
|
|
443
|
-
const runId = opts.runId ?? `run-${ctx.candidateId}-${ctx.experimentId}-${report.startedAt}`;
|
|
444
|
-
const raw = {
|
|
445
|
-
pass_count: report.passCount,
|
|
446
|
-
fail_count: report.failCount,
|
|
447
|
-
error_count: report.errorCount,
|
|
448
|
-
skipped_count: report.skippedCount,
|
|
449
|
-
duration_ms: report.durationMs,
|
|
450
|
-
blended_score: report.blendedScore
|
|
451
|
-
};
|
|
452
|
-
for (const layer of report.layers) {
|
|
453
|
-
if (typeof layer.score === "number") raw[`layer.${layer.layer}`] = layer.score;
|
|
454
|
-
raw[`layer_${layer.layer}_pass`] = layer.status === "pass" ? 1 : 0;
|
|
455
|
-
if (layer.diagnostics) {
|
|
456
|
-
for (const [k, v] of Object.entries(layer.diagnostics)) {
|
|
457
|
-
if (typeof v === "number" && Number.isFinite(v)) raw[`layer.${layer.layer}.${k}`] = v;
|
|
458
|
-
}
|
|
459
|
-
}
|
|
460
|
-
}
|
|
461
|
-
const firstFail = report.layers.find((l) => l.status === "fail" || l.status === "error");
|
|
462
|
-
const outcome = { raw };
|
|
463
|
-
if (splitTag === "holdout") outcome.holdoutScore = report.blendedScore;
|
|
464
|
-
else outcome.searchScore = report.blendedScore;
|
|
465
|
-
return {
|
|
466
|
-
runId,
|
|
467
|
-
experimentId: ctx.experimentId,
|
|
468
|
-
candidateId: ctx.candidateId,
|
|
469
|
-
seed: 0,
|
|
470
|
-
model: ctx.model,
|
|
471
|
-
promptHash: ctx.promptHash,
|
|
472
|
-
configHash: ctx.configHash,
|
|
473
|
-
commitSha: ctx.commitSha,
|
|
474
|
-
wallMs: report.durationMs,
|
|
475
|
-
costUsd: ctx.defaultCostUsd ?? 0,
|
|
476
|
-
tokenUsage: { input: 0, output: 0 },
|
|
477
|
-
outcome,
|
|
478
|
-
failureMode: firstFail ? failureModeFromLayer(firstFail) : void 0,
|
|
479
|
-
splitTag,
|
|
480
|
-
scenarioId: ctx.scenarioId
|
|
481
|
-
};
|
|
482
|
-
}
|
|
483
|
-
function failureModeFromLayer(layer) {
|
|
484
|
-
if (layer.status === "error") return `layer_${layer.layer}_error`;
|
|
485
|
-
if (layer.status === "fail") return `layer_${layer.layer}_fail`;
|
|
486
|
-
if (layer.status === "timeout") return `layer_${layer.layer}_timeout`;
|
|
487
|
-
return `layer_${layer.layer}_${layer.status}`;
|
|
488
|
-
}
|
|
489
|
-
|
|
490
|
-
// src/rl/tournament.ts
|
|
491
|
-
function fitBradleyTerry(outcomes, opts = {}) {
|
|
492
|
-
const tol = opts.tolerance ?? 1e-6;
|
|
493
|
-
const maxIter = opts.maxIterations ?? 256;
|
|
494
|
-
const smoothing = opts.smoothing ?? 0.1;
|
|
495
|
-
const candidates = /* @__PURE__ */ new Set();
|
|
496
|
-
for (const o of outcomes) {
|
|
497
|
-
candidates.add(o.winner);
|
|
498
|
-
candidates.add(o.loser);
|
|
499
|
-
}
|
|
500
|
-
const ids = [...candidates].sort();
|
|
501
|
-
const idx = new Map(ids.map((id, i) => [id, i]));
|
|
502
|
-
const n = ids.length;
|
|
503
|
-
if (n === 0) return { ratings: [], iterations: 0, finalDelta: 0, converged: true };
|
|
504
|
-
if (n === 1) {
|
|
505
|
-
return {
|
|
506
|
-
ratings: [{ candidateId: ids[0], strength: 1, logStrength: 0, n: 0, wins: 0 }],
|
|
507
|
-
iterations: 0,
|
|
508
|
-
finalDelta: 0,
|
|
509
|
-
converged: true
|
|
510
|
-
};
|
|
511
|
-
}
|
|
512
|
-
const W = Array.from({ length: n }, () => new Array(n).fill(0));
|
|
513
|
-
const N = Array.from({ length: n }, () => new Array(n).fill(0));
|
|
514
|
-
for (const o of outcomes) {
|
|
515
|
-
const i = idx.get(o.winner);
|
|
516
|
-
const j = idx.get(o.loser);
|
|
517
|
-
const w = o.weight ?? 1;
|
|
518
|
-
if (o.draw) {
|
|
519
|
-
W[i][j] += 0.5 * w;
|
|
520
|
-
W[j][i] += 0.5 * w;
|
|
521
|
-
} else {
|
|
522
|
-
W[i][j] += w;
|
|
523
|
-
}
|
|
524
|
-
N[i][j] += w;
|
|
525
|
-
N[j][i] += w;
|
|
526
|
-
}
|
|
527
|
-
const winsTotal = new Array(n).fill(0);
|
|
528
|
-
for (let i = 0; i < n; i++) {
|
|
529
|
-
for (let j = 0; j < n; j++) winsTotal[i] += W[i][j];
|
|
530
|
-
winsTotal[i] += smoothing;
|
|
531
|
-
}
|
|
532
|
-
const compsTotal = new Array(n).fill(0);
|
|
533
|
-
for (let i = 0; i < n; i++) {
|
|
534
|
-
for (let j = 0; j < n; j++) compsTotal[i] += N[i][j];
|
|
535
|
-
}
|
|
536
|
-
let theta = new Array(n).fill(1);
|
|
537
|
-
let iter = 0;
|
|
538
|
-
let delta = Infinity;
|
|
539
|
-
for (; iter < maxIter; iter++) {
|
|
540
|
-
const newTheta = new Array(n);
|
|
541
|
-
for (let i = 0; i < n; i++) {
|
|
542
|
-
let denom = 0;
|
|
543
|
-
for (let j = 0; j < n; j++) {
|
|
544
|
-
if (j === i) continue;
|
|
545
|
-
if (N[i][j] === 0) continue;
|
|
546
|
-
denom += N[i][j] / (theta[i] + theta[j]);
|
|
547
|
-
}
|
|
548
|
-
newTheta[i] = denom === 0 ? theta[i] : winsTotal[i] / denom;
|
|
549
|
-
}
|
|
550
|
-
let logSum = 0;
|
|
551
|
-
for (let i = 0; i < n; i++) logSum += Math.log(Math.max(1e-300, newTheta[i]));
|
|
552
|
-
const norm = Math.exp(logSum / n);
|
|
553
|
-
for (let i = 0; i < n; i++) newTheta[i] = newTheta[i] / norm;
|
|
554
|
-
delta = 0;
|
|
555
|
-
for (let i = 0; i < n; i++) {
|
|
556
|
-
const d = Math.abs(newTheta[i] - theta[i]) / Math.max(1e-12, theta[i]);
|
|
557
|
-
if (d > delta) delta = d;
|
|
558
|
-
}
|
|
559
|
-
theta = newTheta;
|
|
560
|
-
if (delta < tol) break;
|
|
561
|
-
}
|
|
562
|
-
const minLog = Math.min(...theta.map((t) => Math.log(Math.max(1e-300, t))));
|
|
563
|
-
const ratings = ids.map((id, i) => ({
|
|
564
|
-
candidateId: id,
|
|
565
|
-
strength: theta[i],
|
|
566
|
-
logStrength: Math.log(Math.max(1e-300, theta[i])) - minLog,
|
|
567
|
-
n: compsTotal[i],
|
|
568
|
-
wins: winsTotal[i] - smoothing
|
|
569
|
-
}));
|
|
570
|
-
return {
|
|
571
|
-
ratings: ratings.sort((a, b) => b.strength - a.strength),
|
|
572
|
-
iterations: iter,
|
|
573
|
-
finalDelta: delta,
|
|
574
|
-
converged: delta < tol
|
|
575
|
-
};
|
|
576
|
-
}
|
|
577
|
-
function applyEloUpdate(ratings, outcome, opts = {}) {
|
|
578
|
-
const defaultRating = opts.defaultRating ?? 1500;
|
|
579
|
-
const k = opts.kFactor ?? 32;
|
|
580
|
-
const rW = ratings.get(outcome.winner) ?? defaultRating;
|
|
581
|
-
const rL = ratings.get(outcome.loser) ?? defaultRating;
|
|
582
|
-
const expectedW = 1 / (1 + 10 ** ((rL - rW) / 400));
|
|
583
|
-
const scoreW = outcome.draw ? 0.5 : 1;
|
|
584
|
-
const scoreL = outcome.draw ? 0.5 : 0;
|
|
585
|
-
const w = outcome.weight ?? 1;
|
|
586
|
-
const winnerDelta = k * w * (scoreW - expectedW);
|
|
587
|
-
const loserDelta = k * w * (scoreL - (1 - expectedW));
|
|
588
|
-
ratings.set(outcome.winner, rW + winnerDelta);
|
|
589
|
-
ratings.set(outcome.loser, rL + loserDelta);
|
|
590
|
-
return { winnerDelta, loserDelta };
|
|
591
|
-
}
|
|
592
|
-
function buildPairwiseFromCampaign(input) {
|
|
593
|
-
const drawMargin = input.drawMargin ?? 0;
|
|
594
|
-
const byKey = /* @__PURE__ */ new Map();
|
|
595
|
-
for (const r of input.runs) {
|
|
596
|
-
const arr = byKey.get(r.matchKey) ?? [];
|
|
597
|
-
arr.push({ candidateId: r.candidateId, score: r.score });
|
|
598
|
-
byKey.set(r.matchKey, arr);
|
|
599
|
-
}
|
|
600
|
-
const outcomes = [];
|
|
601
|
-
for (const arr of byKey.values()) {
|
|
602
|
-
for (let i = 0; i < arr.length; i++) {
|
|
603
|
-
for (let j = i + 1; j < arr.length; j++) {
|
|
604
|
-
const a = arr[i];
|
|
605
|
-
const b = arr[j];
|
|
606
|
-
if (a.candidateId === b.candidateId) continue;
|
|
607
|
-
const margin = Math.abs(a.score - b.score);
|
|
608
|
-
if (margin <= drawMargin) {
|
|
609
|
-
outcomes.push({ winner: a.candidateId, loser: b.candidateId, draw: true, weight: 1 });
|
|
610
|
-
} else {
|
|
611
|
-
const [winner, loser] = a.score > b.score ? [a, b] : [b, a];
|
|
612
|
-
outcomes.push({ winner: winner.candidateId, loser: loser.candidateId, weight: margin });
|
|
613
|
-
}
|
|
614
|
-
}
|
|
615
|
-
}
|
|
616
|
-
}
|
|
617
|
-
return outcomes;
|
|
618
|
-
}
|
|
619
|
-
|
|
620
|
-
// src/rl/adaptation-eval.ts
|
|
621
|
-
async function runAdaptationCurve(opts) {
|
|
622
|
-
const ks = opts.ks ?? [0, 1, 2, 4, 8, 16];
|
|
623
|
-
const reps = opts.reps ?? 3;
|
|
624
|
-
const passThreshold = opts.passThreshold ?? 0.5;
|
|
625
|
-
const sortedKs = [...ks].sort((a, b) => a - b);
|
|
626
|
-
const points = [];
|
|
627
|
-
for (const k of sortedKs) {
|
|
628
|
-
const perScenario = [];
|
|
629
|
-
const allScores = [];
|
|
630
|
-
let totalPasses = 0;
|
|
631
|
-
let totalAttempts = 0;
|
|
632
|
-
for (const scenario of opts.scenarios) {
|
|
633
|
-
const sid = scenario.scenarioId ?? `scenario-${opts.scenarios.indexOf(scenario)}`;
|
|
634
|
-
const scores = [];
|
|
635
|
-
let passes = 0;
|
|
636
|
-
for (let r = 0; r < reps; r++) {
|
|
637
|
-
const score = await opts.runner.run({ scenario, k, rep: r });
|
|
638
|
-
scores.push(score);
|
|
639
|
-
if (score >= passThreshold) passes++;
|
|
640
|
-
allScores.push(score);
|
|
641
|
-
if (score >= passThreshold) totalPasses++;
|
|
642
|
-
totalAttempts++;
|
|
643
|
-
}
|
|
644
|
-
const meanS = scores.reduce((s, v) => s + v, 0) / scores.length;
|
|
645
|
-
perScenario.push({ scenarioId: sid, meanScore: meanS, passes, total: scores.length });
|
|
646
|
-
}
|
|
647
|
-
const meanScore = allScores.reduce((s, v) => s + v, 0) / Math.max(1, allScores.length);
|
|
648
|
-
const variance = allScores.length < 2 ? 0 : allScores.reduce((s, v) => s + (v - meanScore) ** 2, 0) / (allScores.length - 1);
|
|
649
|
-
points.push({
|
|
650
|
-
k,
|
|
651
|
-
meanScore,
|
|
652
|
-
passRate: totalPasses / Math.max(1, totalAttempts),
|
|
653
|
-
std: Math.sqrt(variance),
|
|
654
|
-
n: allScores.length,
|
|
655
|
-
perScenario
|
|
656
|
-
});
|
|
657
|
-
}
|
|
658
|
-
const firstPassK2 = points.find((p) => p.passRate >= passThreshold)?.k ?? null;
|
|
659
|
-
const maxK = sortedKs[sortedKs.length - 1] ?? 1;
|
|
660
|
-
let area = 0;
|
|
661
|
-
for (let i = 1; i < points.length; i++) {
|
|
662
|
-
const x1 = points[i - 1].k;
|
|
663
|
-
const x2 = points[i].k;
|
|
664
|
-
const y1 = points[i - 1].meanScore;
|
|
665
|
-
const y2 = points[i].meanScore;
|
|
666
|
-
area += (y1 + y2) / 2 * (x2 - x1);
|
|
667
|
-
}
|
|
668
|
-
const adaptationArea = maxK === 0 ? 0 : area / maxK;
|
|
669
|
-
return { points, firstPassK: firstPassK2, adaptationArea };
|
|
670
|
-
}
|
|
671
|
-
function compareAdaptationCurves(a, b, opts = {}) {
|
|
672
|
-
const conf = opts.confidence ?? 0.95;
|
|
673
|
-
const resamples = opts.bootstrapResamples ?? 500;
|
|
674
|
-
const rng = makeRng(opts.seed);
|
|
675
|
-
const perK = [];
|
|
676
|
-
for (const ap of a.points) {
|
|
677
|
-
const bp = b.points.find((p) => p.k === ap.k);
|
|
678
|
-
if (!bp) continue;
|
|
679
|
-
const aMeans = ap.perScenario.map((s) => s.meanScore);
|
|
680
|
-
const bMeans = bp.perScenario.map((s) => s.meanScore);
|
|
681
|
-
const aCi = bootstrapMeanCi(aMeans, resamples, conf, rng);
|
|
682
|
-
const bCi = bootstrapMeanCi(bMeans, resamples, conf, rng);
|
|
683
|
-
perK.push({
|
|
684
|
-
k: ap.k,
|
|
685
|
-
deltaMean: ap.meanScore - bp.meanScore,
|
|
686
|
-
aLow: aCi.low,
|
|
687
|
-
aHigh: aCi.high,
|
|
688
|
-
bLow: bCi.low,
|
|
689
|
-
bHigh: bCi.high
|
|
690
|
-
});
|
|
691
|
-
}
|
|
692
|
-
const areaDelta = a.adaptationArea - b.adaptationArea;
|
|
693
|
-
const firstPassKDelta = a.firstPassK !== null && b.firstPassK !== null ? b.firstPassK - a.firstPassK : null;
|
|
694
|
-
const meanDelta = perK.reduce((s, p) => s + p.deltaMean, 0) / Math.max(1, perK.length);
|
|
695
|
-
let verdict;
|
|
696
|
-
if (Math.abs(meanDelta) < 0.02 && Math.abs(areaDelta) < 0.02) verdict = "similar";
|
|
697
|
-
else if (meanDelta > 0 && areaDelta > 0) verdict = "a_better";
|
|
698
|
-
else if (meanDelta < 0 && areaDelta < 0) verdict = "b_better";
|
|
699
|
-
else verdict = "similar";
|
|
700
|
-
const rationale = `mean per-k delta=${meanDelta.toFixed(3)}, area delta=${areaDelta.toFixed(3)}` + (firstPassKDelta !== null ? `, first-pass-k delta=${firstPassKDelta}` : "");
|
|
701
|
-
return { perK, areaDelta, firstPassKDelta, verdict, rationale };
|
|
376
|
+
const v = valid[i];
|
|
377
|
+
const idx = perScenario.findIndex((p) => p.scenarioId === v.scenarioId);
|
|
378
|
+
if (idx >= 0) perScenario[idx].qValue = qValues[i];
|
|
379
|
+
}
|
|
380
|
+
const contaminationSuspected = pairedTest.p < fdr && median <= -minMedianDrop;
|
|
381
|
+
const reason = contaminationSuspected ? `paired p=${pairedTest.p.toFixed(4)} < ${fdr} and median drop ${median.toFixed(4)} \u2265 ${minMedianDrop}` : pairedTest.p >= fdr ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})` : `significant but small effect (median delta ${median.toFixed(4)})`;
|
|
382
|
+
return {
|
|
383
|
+
perScenario,
|
|
384
|
+
pairedTest,
|
|
385
|
+
medianDelta: median,
|
|
386
|
+
meanDelta: mean,
|
|
387
|
+
contaminationSuspected,
|
|
388
|
+
reason,
|
|
389
|
+
n: valid.length
|
|
390
|
+
};
|
|
702
391
|
}
|
|
703
|
-
function
|
|
704
|
-
return
|
|
392
|
+
function renameVariables(identifiers, rename = (n, i) => `${n}_${(i % 26 + 10).toString(36)}`) {
|
|
393
|
+
return {
|
|
394
|
+
kind: "rename_variables",
|
|
395
|
+
apply(scenario) {
|
|
396
|
+
let prompt = scenario.prompt;
|
|
397
|
+
identifiers.forEach((id, i) => {
|
|
398
|
+
const replacement = rename(id, i);
|
|
399
|
+
const re = new RegExp(`\\b${escapeRegex(id)}\\b`, "g");
|
|
400
|
+
prompt = prompt.replace(re, replacement);
|
|
401
|
+
});
|
|
402
|
+
return { ...scenario, prompt };
|
|
403
|
+
}
|
|
404
|
+
};
|
|
705
405
|
}
|
|
706
|
-
function
|
|
707
|
-
if (seed === void 0) return Math.random;
|
|
406
|
+
function shuffleOrder(shuffleSection, seed) {
|
|
708
407
|
let s = seed >>> 0;
|
|
709
|
-
|
|
408
|
+
const rng = () => {
|
|
710
409
|
s = s + 1831565813 >>> 0;
|
|
711
410
|
let t = s;
|
|
712
411
|
t = Math.imul(t ^ t >>> 15, t | 1);
|
|
713
412
|
t ^= t + Math.imul(t ^ t >>> 7, t | 61);
|
|
714
413
|
return ((t ^ t >>> 14) >>> 0) / 4294967296;
|
|
715
414
|
};
|
|
716
|
-
}
|
|
717
|
-
function bootstrapMeanCi(xs, resamples, confidence, rng) {
|
|
718
|
-
if (xs.length < 2) return { low: xs[0] ?? 0, high: xs[0] ?? 0 };
|
|
719
|
-
const samples = new Array(resamples);
|
|
720
|
-
for (let b = 0; b < resamples; b++) {
|
|
721
|
-
let sum = 0;
|
|
722
|
-
for (let i = 0; i < xs.length; i++) sum += xs[Math.floor(rng() * xs.length)];
|
|
723
|
-
samples[b] = sum / xs.length;
|
|
724
|
-
}
|
|
725
|
-
samples.sort((a, b) => a - b);
|
|
726
|
-
const alpha = 1 - confidence;
|
|
727
415
|
return {
|
|
728
|
-
|
|
729
|
-
|
|
416
|
+
kind: "shuffle_order",
|
|
417
|
+
apply(scenario) {
|
|
418
|
+
const newPrompt = shuffleSection(scenario.prompt, rng);
|
|
419
|
+
return { ...scenario, prompt: newPrompt };
|
|
420
|
+
}
|
|
730
421
|
};
|
|
731
422
|
}
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
const budget = opts.budget ?? Number.POSITIVE_INFINITY;
|
|
739
|
-
const seed = opts.seed ?? 1;
|
|
740
|
-
const rng = mulberry32(seed);
|
|
741
|
-
const scenarios = [];
|
|
742
|
-
const seen = /* @__PURE__ */ new Set();
|
|
743
|
-
let scoreCalls = 0;
|
|
744
|
-
for (const s of opts.seeds) {
|
|
745
|
-
const id = opts.mutateScenarioId(s);
|
|
746
|
-
if (seen.has(id)) continue;
|
|
747
|
-
seen.add(id);
|
|
748
|
-
if (scoreCalls >= budget) break;
|
|
749
|
-
const score = await opts.scoreFn(s);
|
|
750
|
-
scoreCalls++;
|
|
751
|
-
scenarios.push({
|
|
752
|
-
id,
|
|
753
|
-
generation: 0,
|
|
754
|
-
parentId: null,
|
|
755
|
-
scenario: s,
|
|
756
|
-
score,
|
|
757
|
-
mutationStrategy: null
|
|
758
|
-
});
|
|
759
|
-
}
|
|
760
|
-
for (let g = 1; g <= rounds; g++) {
|
|
761
|
-
if (scoreCalls >= budget) break;
|
|
762
|
-
const parents = scenarios.filter((s) => s.generation === g - 1);
|
|
763
|
-
for (const parent of parents) {
|
|
764
|
-
for (const mutation of opts.mutations) {
|
|
765
|
-
if (scoreCalls >= budget) break;
|
|
766
|
-
const produced = await mutation.mutate(parent.scenario, rng);
|
|
767
|
-
const childArr = Array.isArray(produced) ? produced : [produced];
|
|
768
|
-
for (let k = 0; k < Math.min(children, childArr.length); k++) {
|
|
769
|
-
if (scoreCalls >= budget) break;
|
|
770
|
-
const child = childArr[k];
|
|
771
|
-
const cid = opts.mutateScenarioId(child);
|
|
772
|
-
if (seen.has(cid)) continue;
|
|
773
|
-
seen.add(cid);
|
|
774
|
-
const cscore = await opts.scoreFn(child);
|
|
775
|
-
scoreCalls++;
|
|
776
|
-
scenarios.push({
|
|
777
|
-
id: cid,
|
|
778
|
-
generation: g,
|
|
779
|
-
parentId: parent.id,
|
|
780
|
-
scenario: child,
|
|
781
|
-
score: cscore,
|
|
782
|
-
mutationStrategy: mutation.id
|
|
783
|
-
});
|
|
784
|
-
}
|
|
785
|
-
}
|
|
423
|
+
function injectIrrelevantClause(clause, position = "prefix") {
|
|
424
|
+
return {
|
|
425
|
+
kind: "inject_irrelevant_clause",
|
|
426
|
+
apply(scenario) {
|
|
427
|
+
const prompt = position === "prefix" ? `${clause} ${scenario.prompt}` : `${scenario.prompt} ${clause}`;
|
|
428
|
+
return { ...scenario, prompt };
|
|
786
429
|
}
|
|
787
|
-
}
|
|
788
|
-
const failures = scenarios.filter((s) => s.score !== null && s.score < failureThreshold).sort((a, b) => (a.score ?? 0) - (b.score ?? 0));
|
|
789
|
-
const byGeneration = [];
|
|
790
|
-
const maxGen = scenarios.reduce((m, s) => Math.max(m, s.generation), 0);
|
|
791
|
-
for (let g = 0; g <= maxGen; g++) {
|
|
792
|
-
const gens = scenarios.filter((s) => s.generation === g);
|
|
793
|
-
if (gens.length === 0) continue;
|
|
794
|
-
const fails = gens.filter((s) => s.score !== null && s.score < failureThreshold).length;
|
|
795
|
-
const meanScore = gens.reduce((sum, s) => sum + (s.score ?? 0), 0) / gens.length;
|
|
796
|
-
byGeneration.push({ generation: g, total: gens.length, failures: fails, meanScore });
|
|
797
|
-
}
|
|
798
|
-
return { scenarios, failures, byGeneration, scoreCalls };
|
|
799
|
-
}
|
|
800
|
-
function mulberry32(seed) {
|
|
801
|
-
let s = seed >>> 0;
|
|
802
|
-
return () => {
|
|
803
|
-
s = s + 1831565813 >>> 0;
|
|
804
|
-
let t = s;
|
|
805
|
-
t = Math.imul(t ^ t >>> 15, t | 1);
|
|
806
|
-
t ^= t + Math.imul(t ^ t >>> 7, t | 61);
|
|
807
|
-
return ((t ^ t >>> 14) >>> 0) / 4294967296;
|
|
808
430
|
};
|
|
809
431
|
}
|
|
432
|
+
function escapeRegex(s) {
|
|
433
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
434
|
+
}
|
|
810
435
|
|
|
811
436
|
// src/rl/corpus.ts
|
|
812
437
|
import { appendFileSync, existsSync, mkdirSync, readFileSync } from "fs";
|
|
@@ -1286,6 +911,166 @@ var PredictiveValidityResearcher = class {
|
|
|
1286
911
|
}
|
|
1287
912
|
};
|
|
1288
913
|
|
|
914
|
+
// src/rl/preferences.ts
|
|
915
|
+
var SPLIT_TAG_DEFAULT = "holdout";
|
|
916
|
+
var DEFAULT_REWARD = (run) => {
|
|
917
|
+
const v = run.outcome.holdoutScore ?? run.outcome.searchScore;
|
|
918
|
+
return typeof v === "number" && Number.isFinite(v) ? v : null;
|
|
919
|
+
};
|
|
920
|
+
function extractPreferences(runs, opts = {}) {
|
|
921
|
+
const strategy = opts.strategy ?? "paired-by-scenario-and-seed";
|
|
922
|
+
const minMargin = opts.minMargin ?? 0.05;
|
|
923
|
+
const splitTag = opts.splitTag ?? SPLIT_TAG_DEFAULT;
|
|
924
|
+
const rewardOf2 = opts.rewardOf ?? DEFAULT_REWARD;
|
|
925
|
+
const filtered = runs.filter((r) => r.splitTag === splitTag);
|
|
926
|
+
const scoredEntries = [];
|
|
927
|
+
for (const run of filtered) {
|
|
928
|
+
const s = rewardOf2(run);
|
|
929
|
+
if (s === null) continue;
|
|
930
|
+
scoredEntries.push({ run, score: s });
|
|
931
|
+
}
|
|
932
|
+
const pairs = [];
|
|
933
|
+
let pairsBelowMargin = 0;
|
|
934
|
+
let cellsSingleton = 0;
|
|
935
|
+
let cellsInspected = 0;
|
|
936
|
+
if (strategy === "paired-by-scenario-and-seed") {
|
|
937
|
+
const groups = /* @__PURE__ */ new Map();
|
|
938
|
+
for (const e of scoredEntries) {
|
|
939
|
+
const sid = scenarioOf(e.run);
|
|
940
|
+
const key = `${sid}::${e.run.seed}`;
|
|
941
|
+
const arr = groups.get(key) ?? [];
|
|
942
|
+
arr.push(e);
|
|
943
|
+
groups.set(key, arr);
|
|
944
|
+
}
|
|
945
|
+
for (const [key, members] of groups.entries()) {
|
|
946
|
+
cellsInspected++;
|
|
947
|
+
if (members.length < 2) {
|
|
948
|
+
cellsSingleton++;
|
|
949
|
+
continue;
|
|
950
|
+
}
|
|
951
|
+
for (let i = 0; i < members.length; i++) {
|
|
952
|
+
for (let j = i + 1; j < members.length; j++) {
|
|
953
|
+
const a = members[i];
|
|
954
|
+
const b = members[j];
|
|
955
|
+
if (a.run.candidateId === b.run.candidateId) continue;
|
|
956
|
+
const result = makePair(a, b, key.split("::")[0], minMargin);
|
|
957
|
+
if (result.kind === "admit") pairs.push(result.pair);
|
|
958
|
+
else pairsBelowMargin++;
|
|
959
|
+
}
|
|
960
|
+
}
|
|
961
|
+
}
|
|
962
|
+
} else if (strategy === "paired-by-scenario") {
|
|
963
|
+
const byScenarioVariant = /* @__PURE__ */ new Map();
|
|
964
|
+
for (const e of scoredEntries) {
|
|
965
|
+
const sid = scenarioOf(e.run);
|
|
966
|
+
let perScenario = byScenarioVariant.get(sid);
|
|
967
|
+
if (!perScenario) {
|
|
968
|
+
perScenario = /* @__PURE__ */ new Map();
|
|
969
|
+
byScenarioVariant.set(sid, perScenario);
|
|
970
|
+
}
|
|
971
|
+
const cur = perScenario.get(e.run.candidateId);
|
|
972
|
+
if (cur) {
|
|
973
|
+
cur.sum += e.score;
|
|
974
|
+
cur.n++;
|
|
975
|
+
} else perScenario.set(e.run.candidateId, { run: e.run, sum: e.score, n: 1 });
|
|
976
|
+
}
|
|
977
|
+
for (const [sid, perVariant] of byScenarioVariant.entries()) {
|
|
978
|
+
cellsInspected++;
|
|
979
|
+
const arr = [...perVariant.entries()].map(([vid, agg]) => ({
|
|
980
|
+
run: agg.run,
|
|
981
|
+
score: agg.sum / agg.n,
|
|
982
|
+
variantId: vid
|
|
983
|
+
}));
|
|
984
|
+
if (arr.length < 2) {
|
|
985
|
+
cellsSingleton++;
|
|
986
|
+
continue;
|
|
987
|
+
}
|
|
988
|
+
for (let i = 0; i < arr.length; i++) {
|
|
989
|
+
for (let j = i + 1; j < arr.length; j++) {
|
|
990
|
+
const result = makePair(arr[i], arr[j], sid, minMargin);
|
|
991
|
+
if (result.kind === "admit") pairs.push(result.pair);
|
|
992
|
+
else pairsBelowMargin++;
|
|
993
|
+
}
|
|
994
|
+
}
|
|
995
|
+
}
|
|
996
|
+
} else {
|
|
997
|
+
const byScenario = /* @__PURE__ */ new Map();
|
|
998
|
+
for (const e of scoredEntries) {
|
|
999
|
+
const sid = scenarioOf(e.run);
|
|
1000
|
+
const arr = byScenario.get(sid) ?? [];
|
|
1001
|
+
arr.push(e);
|
|
1002
|
+
byScenario.set(sid, arr);
|
|
1003
|
+
}
|
|
1004
|
+
for (const [sid, arr] of byScenario.entries()) {
|
|
1005
|
+
cellsInspected++;
|
|
1006
|
+
if (arr.length < 2) {
|
|
1007
|
+
cellsSingleton++;
|
|
1008
|
+
continue;
|
|
1009
|
+
}
|
|
1010
|
+
const sorted = [...arr].sort((a, b) => a.score - b.score);
|
|
1011
|
+
const top = sorted[sorted.length - 1];
|
|
1012
|
+
const bot = sorted[0];
|
|
1013
|
+
if (top.run.candidateId === bot.run.candidateId) {
|
|
1014
|
+
cellsSingleton++;
|
|
1015
|
+
continue;
|
|
1016
|
+
}
|
|
1017
|
+
const result = makePair(bot, top, sid, minMargin);
|
|
1018
|
+
if (result.kind === "admit") pairs.push(result.pair);
|
|
1019
|
+
else pairsBelowMargin++;
|
|
1020
|
+
}
|
|
1021
|
+
}
|
|
1022
|
+
return { pairs, cellsInspected, pairsBelowMargin, cellsSingleton, strategy };
|
|
1023
|
+
}
|
|
1024
|
+
function toTRLFormat(triples, promptOf) {
|
|
1025
|
+
return triples.map((t) => ({
|
|
1026
|
+
prompt: promptOf(t.meta.chosenPromptHash),
|
|
1027
|
+
chosen: t.meta.chosenPromptHash,
|
|
1028
|
+
// caller substitutes the model output via the runId map
|
|
1029
|
+
rejected: t.meta.rejectedPromptHash
|
|
1030
|
+
}));
|
|
1031
|
+
}
|
|
1032
|
+
function toAnthropicFormat(triples) {
|
|
1033
|
+
return triples.map((t) => ({
|
|
1034
|
+
scenarioId: t.scenarioId,
|
|
1035
|
+
chosenRunId: t.chosenRunId,
|
|
1036
|
+
rejectedRunId: t.rejectedRunId,
|
|
1037
|
+
margin: t.marginScore
|
|
1038
|
+
}));
|
|
1039
|
+
}
|
|
1040
|
+
function makePair(a, b, scenarioId, minMargin) {
|
|
1041
|
+
const margin = Math.abs(a.score - b.score);
|
|
1042
|
+
if (margin < minMargin) return { kind: "reject" };
|
|
1043
|
+
const [chosen, rejected] = a.score > b.score ? [a, b] : [b, a];
|
|
1044
|
+
return {
|
|
1045
|
+
kind: "admit",
|
|
1046
|
+
pair: {
|
|
1047
|
+
scenarioId,
|
|
1048
|
+
chosenRunId: chosen.run.runId,
|
|
1049
|
+
rejectedRunId: rejected.run.runId,
|
|
1050
|
+
chosenVariantId: chosen.run.candidateId,
|
|
1051
|
+
rejectedVariantId: rejected.run.candidateId,
|
|
1052
|
+
marginScore: chosen.score - rejected.score,
|
|
1053
|
+
scores: { chosen: chosen.score, rejected: rejected.score },
|
|
1054
|
+
seed: chosen.run.seed === rejected.run.seed ? chosen.run.seed : void 0,
|
|
1055
|
+
meta: {
|
|
1056
|
+
chosenPromptHash: chosen.run.promptHash,
|
|
1057
|
+
rejectedPromptHash: rejected.run.promptHash,
|
|
1058
|
+
chosenConfigHash: chosen.run.configHash,
|
|
1059
|
+
rejectedConfigHash: rejected.run.configHash,
|
|
1060
|
+
chosenModel: chosen.run.model,
|
|
1061
|
+
rejectedModel: rejected.run.model
|
|
1062
|
+
}
|
|
1063
|
+
}
|
|
1064
|
+
};
|
|
1065
|
+
}
|
|
1066
|
+
function scenarioOf(run) {
|
|
1067
|
+
if (typeof run.scenarioId === "string" && run.scenarioId.length > 0) return run.scenarioId;
|
|
1068
|
+
const fromRaw = run.outcome.raw.scenario_id;
|
|
1069
|
+
if (typeof fromRaw === "number" && Number.isFinite(fromRaw)) return String(fromRaw);
|
|
1070
|
+
if (typeof fromRaw === "string") return fromRaw;
|
|
1071
|
+
return run.experimentId;
|
|
1072
|
+
}
|
|
1073
|
+
|
|
1289
1074
|
// src/rl/process-reward.ts
|
|
1290
1075
|
async function extractStepRewards(store, runId, opts) {
|
|
1291
1076
|
const spans = await store.spans({ runId });
|
|
@@ -1524,6 +1309,91 @@ function buildSummary(args) {
|
|
|
1524
1309
|
return lines.join(" | ");
|
|
1525
1310
|
}
|
|
1526
1311
|
|
|
1312
|
+
// src/rl/run-record-adapters.ts
|
|
1313
|
+
function campaignToRunRecords(campaign, ctx) {
|
|
1314
|
+
const splitTag = ctx.splitTag ?? "search";
|
|
1315
|
+
const candidateId = ctx.candidateId ?? campaign.manifestHash;
|
|
1316
|
+
return campaign.cells.map((cell) => {
|
|
1317
|
+
const composites = Object.values(cell.judgeScores).map((s) => s.composite);
|
|
1318
|
+
const score = composites.length > 0 ? composites.reduce((a, b) => a + b, 0) / composites.length : 0;
|
|
1319
|
+
const raw = { rep: cell.rep, duration_ms: cell.durationMs };
|
|
1320
|
+
for (const judge of Object.values(cell.judgeScores)) {
|
|
1321
|
+
for (const [dim, value] of Object.entries(judge.dimensions)) {
|
|
1322
|
+
if (Number.isFinite(value)) raw[`dim.${dim}`] = value;
|
|
1323
|
+
}
|
|
1324
|
+
}
|
|
1325
|
+
if (typeof cell.generation === "number") raw.generation = cell.generation;
|
|
1326
|
+
const outcome = { raw };
|
|
1327
|
+
if (splitTag === "holdout") outcome.holdoutScore = score;
|
|
1328
|
+
else outcome.searchScore = score;
|
|
1329
|
+
return {
|
|
1330
|
+
runId: cell.cellId,
|
|
1331
|
+
experimentId: ctx.experimentId,
|
|
1332
|
+
candidateId,
|
|
1333
|
+
seed: cell.seed,
|
|
1334
|
+
model: ctx.model,
|
|
1335
|
+
promptHash: ctx.promptHash,
|
|
1336
|
+
configHash: ctx.configHash,
|
|
1337
|
+
commitSha: ctx.commitSha,
|
|
1338
|
+
wallMs: cell.durationMs,
|
|
1339
|
+
costUsd: Number.isFinite(cell.costUsd) ? cell.costUsd : ctx.defaultCostUsd ?? 0,
|
|
1340
|
+
tokenUsage: { input: 0, output: 0 },
|
|
1341
|
+
outcome,
|
|
1342
|
+
failureMode: cell.error ? "cell_error" : void 0,
|
|
1343
|
+
splitTag,
|
|
1344
|
+
scenarioId: cell.scenarioId
|
|
1345
|
+
};
|
|
1346
|
+
});
|
|
1347
|
+
}
|
|
1348
|
+
function verificationReportToRunRecord(report, ctx, opts = {}) {
|
|
1349
|
+
const splitTag = ctx.splitTag ?? "search";
|
|
1350
|
+
const runId = opts.runId ?? `run-${ctx.candidateId}-${ctx.experimentId}-${report.startedAt}`;
|
|
1351
|
+
const raw = {
|
|
1352
|
+
pass_count: report.passCount,
|
|
1353
|
+
fail_count: report.failCount,
|
|
1354
|
+
error_count: report.errorCount,
|
|
1355
|
+
skipped_count: report.skippedCount,
|
|
1356
|
+
duration_ms: report.durationMs,
|
|
1357
|
+
blended_score: report.blendedScore
|
|
1358
|
+
};
|
|
1359
|
+
for (const layer of report.layers) {
|
|
1360
|
+
if (typeof layer.score === "number") raw[`layer.${layer.layer}`] = layer.score;
|
|
1361
|
+
raw[`layer_${layer.layer}_pass`] = layer.status === "pass" ? 1 : 0;
|
|
1362
|
+
if (layer.diagnostics) {
|
|
1363
|
+
for (const [k, v] of Object.entries(layer.diagnostics)) {
|
|
1364
|
+
if (typeof v === "number" && Number.isFinite(v)) raw[`layer.${layer.layer}.${k}`] = v;
|
|
1365
|
+
}
|
|
1366
|
+
}
|
|
1367
|
+
}
|
|
1368
|
+
const firstFail = report.layers.find((l) => l.status === "fail" || l.status === "error");
|
|
1369
|
+
const outcome = { raw };
|
|
1370
|
+
if (splitTag === "holdout") outcome.holdoutScore = report.blendedScore;
|
|
1371
|
+
else outcome.searchScore = report.blendedScore;
|
|
1372
|
+
return {
|
|
1373
|
+
runId,
|
|
1374
|
+
experimentId: ctx.experimentId,
|
|
1375
|
+
candidateId: ctx.candidateId,
|
|
1376
|
+
seed: 0,
|
|
1377
|
+
model: ctx.model,
|
|
1378
|
+
promptHash: ctx.promptHash,
|
|
1379
|
+
configHash: ctx.configHash,
|
|
1380
|
+
commitSha: ctx.commitSha,
|
|
1381
|
+
wallMs: report.durationMs,
|
|
1382
|
+
costUsd: ctx.defaultCostUsd ?? 0,
|
|
1383
|
+
tokenUsage: { input: 0, output: 0 },
|
|
1384
|
+
outcome,
|
|
1385
|
+
failureMode: firstFail ? failureModeFromLayer(firstFail) : void 0,
|
|
1386
|
+
splitTag,
|
|
1387
|
+
scenarioId: ctx.scenarioId
|
|
1388
|
+
};
|
|
1389
|
+
}
|
|
1390
|
+
function failureModeFromLayer(layer) {
|
|
1391
|
+
if (layer.status === "error") return `layer_${layer.layer}_error`;
|
|
1392
|
+
if (layer.status === "fail") return `layer_${layer.layer}_fail`;
|
|
1393
|
+
if (layer.status === "timeout") return `layer_${layer.layer}_timeout`;
|
|
1394
|
+
return `layer_${layer.layer}_${layer.status}`;
|
|
1395
|
+
}
|
|
1396
|
+
|
|
1527
1397
|
// src/rl/sim-fidelity.ts
|
|
1528
1398
|
var ABSENT_CATEGORY = "(absent)";
|
|
1529
1399
|
var DEFAULT_MIN_N_PER_FEATURE = 20;
|
|
@@ -1733,6 +1603,136 @@ function easyModeCheck(simulated, production, opts = {}) {
|
|
|
1733
1603
|
const gap = simPassRate - prodPassRate;
|
|
1734
1604
|
return { simPassRate, prodPassRate, gap, inflated: gap > tolerance };
|
|
1735
1605
|
}
|
|
1606
|
+
|
|
1607
|
+
// src/rl/tournament.ts
|
|
1608
|
+
function fitBradleyTerry(outcomes, opts = {}) {
|
|
1609
|
+
const tol = opts.tolerance ?? 1e-6;
|
|
1610
|
+
const maxIter = opts.maxIterations ?? 256;
|
|
1611
|
+
const smoothing = opts.smoothing ?? 0.1;
|
|
1612
|
+
const candidates = /* @__PURE__ */ new Set();
|
|
1613
|
+
for (const o of outcomes) {
|
|
1614
|
+
candidates.add(o.winner);
|
|
1615
|
+
candidates.add(o.loser);
|
|
1616
|
+
}
|
|
1617
|
+
const ids = [...candidates].sort();
|
|
1618
|
+
const idx = new Map(ids.map((id, i) => [id, i]));
|
|
1619
|
+
const n = ids.length;
|
|
1620
|
+
if (n === 0) return { ratings: [], iterations: 0, finalDelta: 0, converged: true };
|
|
1621
|
+
if (n === 1) {
|
|
1622
|
+
return {
|
|
1623
|
+
ratings: [{ candidateId: ids[0], strength: 1, logStrength: 0, n: 0, wins: 0 }],
|
|
1624
|
+
iterations: 0,
|
|
1625
|
+
finalDelta: 0,
|
|
1626
|
+
converged: true
|
|
1627
|
+
};
|
|
1628
|
+
}
|
|
1629
|
+
const W = Array.from({ length: n }, () => new Array(n).fill(0));
|
|
1630
|
+
const N = Array.from({ length: n }, () => new Array(n).fill(0));
|
|
1631
|
+
for (const o of outcomes) {
|
|
1632
|
+
const i = idx.get(o.winner);
|
|
1633
|
+
const j = idx.get(o.loser);
|
|
1634
|
+
const w = o.weight ?? 1;
|
|
1635
|
+
if (o.draw) {
|
|
1636
|
+
W[i][j] += 0.5 * w;
|
|
1637
|
+
W[j][i] += 0.5 * w;
|
|
1638
|
+
} else {
|
|
1639
|
+
W[i][j] += w;
|
|
1640
|
+
}
|
|
1641
|
+
N[i][j] += w;
|
|
1642
|
+
N[j][i] += w;
|
|
1643
|
+
}
|
|
1644
|
+
const winsTotal = new Array(n).fill(0);
|
|
1645
|
+
for (let i = 0; i < n; i++) {
|
|
1646
|
+
for (let j = 0; j < n; j++) winsTotal[i] += W[i][j];
|
|
1647
|
+
winsTotal[i] += smoothing;
|
|
1648
|
+
}
|
|
1649
|
+
const compsTotal = new Array(n).fill(0);
|
|
1650
|
+
for (let i = 0; i < n; i++) {
|
|
1651
|
+
for (let j = 0; j < n; j++) compsTotal[i] += N[i][j];
|
|
1652
|
+
}
|
|
1653
|
+
let theta = new Array(n).fill(1);
|
|
1654
|
+
let iter = 0;
|
|
1655
|
+
let delta = Infinity;
|
|
1656
|
+
for (; iter < maxIter; iter++) {
|
|
1657
|
+
const newTheta = new Array(n);
|
|
1658
|
+
for (let i = 0; i < n; i++) {
|
|
1659
|
+
let denom = 0;
|
|
1660
|
+
for (let j = 0; j < n; j++) {
|
|
1661
|
+
if (j === i) continue;
|
|
1662
|
+
if (N[i][j] === 0) continue;
|
|
1663
|
+
denom += N[i][j] / (theta[i] + theta[j]);
|
|
1664
|
+
}
|
|
1665
|
+
newTheta[i] = denom === 0 ? theta[i] : winsTotal[i] / denom;
|
|
1666
|
+
}
|
|
1667
|
+
let logSum = 0;
|
|
1668
|
+
for (let i = 0; i < n; i++) logSum += Math.log(Math.max(1e-300, newTheta[i]));
|
|
1669
|
+
const norm = Math.exp(logSum / n);
|
|
1670
|
+
for (let i = 0; i < n; i++) newTheta[i] = newTheta[i] / norm;
|
|
1671
|
+
delta = 0;
|
|
1672
|
+
for (let i = 0; i < n; i++) {
|
|
1673
|
+
const d = Math.abs(newTheta[i] - theta[i]) / Math.max(1e-12, theta[i]);
|
|
1674
|
+
if (d > delta) delta = d;
|
|
1675
|
+
}
|
|
1676
|
+
theta = newTheta;
|
|
1677
|
+
if (delta < tol) break;
|
|
1678
|
+
}
|
|
1679
|
+
const minLog = Math.min(...theta.map((t) => Math.log(Math.max(1e-300, t))));
|
|
1680
|
+
const ratings = ids.map((id, i) => ({
|
|
1681
|
+
candidateId: id,
|
|
1682
|
+
strength: theta[i],
|
|
1683
|
+
logStrength: Math.log(Math.max(1e-300, theta[i])) - minLog,
|
|
1684
|
+
n: compsTotal[i],
|
|
1685
|
+
wins: winsTotal[i] - smoothing
|
|
1686
|
+
}));
|
|
1687
|
+
return {
|
|
1688
|
+
ratings: ratings.sort((a, b) => b.strength - a.strength),
|
|
1689
|
+
iterations: iter,
|
|
1690
|
+
finalDelta: delta,
|
|
1691
|
+
converged: delta < tol
|
|
1692
|
+
};
|
|
1693
|
+
}
|
|
1694
|
+
function applyEloUpdate(ratings, outcome, opts = {}) {
|
|
1695
|
+
const defaultRating = opts.defaultRating ?? 1500;
|
|
1696
|
+
const k = opts.kFactor ?? 32;
|
|
1697
|
+
const rW = ratings.get(outcome.winner) ?? defaultRating;
|
|
1698
|
+
const rL = ratings.get(outcome.loser) ?? defaultRating;
|
|
1699
|
+
const expectedW = 1 / (1 + 10 ** ((rL - rW) / 400));
|
|
1700
|
+
const scoreW = outcome.draw ? 0.5 : 1;
|
|
1701
|
+
const scoreL = outcome.draw ? 0.5 : 0;
|
|
1702
|
+
const w = outcome.weight ?? 1;
|
|
1703
|
+
const winnerDelta = k * w * (scoreW - expectedW);
|
|
1704
|
+
const loserDelta = k * w * (scoreL - (1 - expectedW));
|
|
1705
|
+
ratings.set(outcome.winner, rW + winnerDelta);
|
|
1706
|
+
ratings.set(outcome.loser, rL + loserDelta);
|
|
1707
|
+
return { winnerDelta, loserDelta };
|
|
1708
|
+
}
|
|
1709
|
+
function buildPairwiseFromCampaign(input) {
|
|
1710
|
+
const drawMargin = input.drawMargin ?? 0;
|
|
1711
|
+
const byKey = /* @__PURE__ */ new Map();
|
|
1712
|
+
for (const r of input.runs) {
|
|
1713
|
+
const arr = byKey.get(r.matchKey) ?? [];
|
|
1714
|
+
arr.push({ candidateId: r.candidateId, score: r.score });
|
|
1715
|
+
byKey.set(r.matchKey, arr);
|
|
1716
|
+
}
|
|
1717
|
+
const outcomes = [];
|
|
1718
|
+
for (const arr of byKey.values()) {
|
|
1719
|
+
for (let i = 0; i < arr.length; i++) {
|
|
1720
|
+
for (let j = i + 1; j < arr.length; j++) {
|
|
1721
|
+
const a = arr[i];
|
|
1722
|
+
const b = arr[j];
|
|
1723
|
+
if (a.candidateId === b.candidateId) continue;
|
|
1724
|
+
const margin = Math.abs(a.score - b.score);
|
|
1725
|
+
if (margin <= drawMargin) {
|
|
1726
|
+
outcomes.push({ winner: a.candidateId, loser: b.candidateId, draw: true, weight: 1 });
|
|
1727
|
+
} else {
|
|
1728
|
+
const [winner, loser] = a.score > b.score ? [a, b] : [b, a];
|
|
1729
|
+
outcomes.push({ winner: winner.candidateId, loser: loser.candidateId, weight: margin });
|
|
1730
|
+
}
|
|
1731
|
+
}
|
|
1732
|
+
}
|
|
1733
|
+
}
|
|
1734
|
+
return outcomes;
|
|
1735
|
+
}
|
|
1736
1736
|
export {
|
|
1737
1737
|
ABSENT_CATEGORY,
|
|
1738
1738
|
DEFAULT_MIN_N_PER_FEATURE,
|