@tangle-network/agent-eval 0.94.0 → 0.95.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/CHANGELOG.md +32 -0
  2. package/README.md +44 -30
  3. package/dist/adapters/http.d.ts +8 -7
  4. package/dist/adapters/http.js.map +1 -1
  5. package/dist/adapters/langchain.d.ts +3 -2
  6. package/dist/adapters/otel.d.ts +5 -4
  7. package/dist/analyst/index.d.ts +11 -31
  8. package/dist/analyst/index.js +5 -65
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/{analyze-runs-B6Ljo_dI.d.ts → analyze-runs-DtT6F_6T.d.ts} +3 -3
  11. package/dist/belief-state/index.d.ts +4 -3
  12. package/dist/benchmarks/index.d.ts +3 -2
  13. package/dist/campaign/index.d.ts +727 -616
  14. package/dist/campaign/index.js +1863 -1316
  15. package/dist/campaign/index.js.map +1 -1
  16. package/dist/{chunk-2K6UUZ7P.js → chunk-2T4EZACH.js} +1 -1
  17. package/dist/chunk-2T4EZACH.js.map +1 -0
  18. package/dist/{chunk-CTBHKLEU.js → chunk-77T4STFI.js} +59 -86
  19. package/dist/chunk-77T4STFI.js.map +1 -0
  20. package/dist/{chunk-EGPMSBEZ.js → chunk-7QTQKIDD.js} +178 -177
  21. package/dist/chunk-7QTQKIDD.js.map +1 -0
  22. package/dist/{chunk-MIFZUPEK.js → chunk-AQ5WQAIV.js} +21 -6
  23. package/dist/chunk-AQ5WQAIV.js.map +1 -0
  24. package/dist/chunk-DJWX3GVS.js +81 -0
  25. package/dist/chunk-DJWX3GVS.js.map +1 -0
  26. package/dist/{chunk-S6OZEZQK.js → chunk-HMA63UEO.js} +37 -9
  27. package/dist/{chunk-S6OZEZQK.js.map → chunk-HMA63UEO.js.map} +1 -1
  28. package/dist/{chunk-TBDR6PAI.js → chunk-IZCEK2HR.js} +2 -2
  29. package/dist/{chunk-SD2YFWQQ.js → chunk-KKWJD5E6.js} +20 -20
  30. package/dist/chunk-KKWJD5E6.js.map +1 -0
  31. package/dist/{chunk-KWRRMR3J.js → chunk-LO6IOIJ2.js} +10 -10
  32. package/dist/chunk-LO6IOIJ2.js.map +1 -0
  33. package/dist/{chunk-E4GH6USR.js → chunk-NZEQVRH5.js} +2 -2
  34. package/dist/chunk-NZEQVRH5.js.map +1 -0
  35. package/dist/{chunk-MPQWFX6Y.js → chunk-PSWWQXHF.js} +13 -88
  36. package/dist/chunk-PSWWQXHF.js.map +1 -0
  37. package/dist/{chunk-Q5LIB7BC.js → chunk-S4SYLDFX.js} +2 -2
  38. package/dist/chunk-S4SYLDFX.js.map +1 -0
  39. package/dist/{chunk-KW53MSA5.js → chunk-X74V6ESX.js} +2 -2
  40. package/dist/{chunk-QMUEXQJS.js → chunk-YBIGNSCZ.js} +81 -4
  41. package/dist/chunk-YBIGNSCZ.js.map +1 -0
  42. package/dist/{chunk-2KNZHH3P.js → chunk-Z6L6YSU6.js} +2 -2
  43. package/dist/{code-agent-session-BO8nCnv3.d.ts → code-agent-session-CPHRCb4-.d.ts} +1 -1
  44. package/dist/contract/index.d.ts +91 -43
  45. package/dist/contract/index.js +127 -17
  46. package/dist/contract/index.js.map +1 -1
  47. package/dist/{control-D6qwHXIR.d.ts → control-Doncu-B_.d.ts} +2 -2
  48. package/dist/control.d.ts +3 -2
  49. package/dist/control.js +2 -2
  50. package/dist/{corpus-B8A4BDR3.d.ts → corpus-D4YW9UoJ.d.ts} +1 -1
  51. package/dist/{default-registry-6dhErQbs.d.ts → default-registry-GyE8X5SP.d.ts} +3 -3
  52. package/dist/diagnose.d.ts +4 -3
  53. package/dist/diagnose.js +1 -1
  54. package/dist/{run-improvement-loop-DBahB8Ax.d.ts → gepa-C1NCIZ9o.d.ts} +117 -130
  55. package/dist/hosted/index.d.ts +5 -4
  56. package/dist/{index-Bx3gZ8xl.d.ts → index-_Y4oNOOb.d.ts} +1 -1
  57. package/dist/index.d.ts +76 -81
  58. package/dist/index.js +66 -31
  59. package/dist/index.js.map +1 -1
  60. package/dist/{insight-report-DWl3z9tl.d.ts → insight-report-BnRjTibG.d.ts} +1 -1
  61. package/dist/{kind-factory-0BhLSI27.d.ts → kind-factory-X3eDYbKn.d.ts} +2 -3
  62. package/dist/matrix/index.d.ts +1 -1
  63. package/dist/meta-eval/index.d.ts +3 -2
  64. package/dist/multishot/index.d.ts +4 -4
  65. package/dist/multishot/index.js.map +1 -1
  66. package/dist/openapi.json +1 -1
  67. package/dist/{pre-registration-mAnCugl9.d.ts → pre-registration-nfUdc9EQ.d.ts} +2 -42
  68. package/dist/{provenance-P-bCL2Fo.d.ts → provenance-CncDq9qE.d.ts} +26 -41
  69. package/dist/{release-report-BEbWmVYj.d.ts → release-report-pidWUMZ2.d.ts} +2 -2
  70. package/dist/reporting.d.ts +5 -4
  71. package/dist/{researcher-B0C2_fVO.d.ts → researcher-Jr8ME1dZ.d.ts} +2 -2
  72. package/dist/rl.d.ts +516 -515
  73. package/dist/rl.js +612 -612
  74. package/dist/rl.js.map +1 -1
  75. package/dist/{rubric-predictive-validity-Cy_W-hWZ.d.ts → rubric-predictive-validity-C2hDKM8Z.d.ts} +1 -1
  76. package/dist/{run-campaign-7WNXMDSN.js → run-campaign-WXY7KI67.js} +2 -2
  77. package/dist/{run-record-e7vj1uZQ.d.ts → run-record-CP2ObebC.d.ts} +14 -18
  78. package/dist/{runtime-trajectory-BDgfGZSr.d.ts → runtime-trajectory-BOUUjI0y.d.ts} +1 -1
  79. package/dist/{semantic-concept-judge-B9MgmBnM.d.ts → semantic-concept-judge-DSBB2Cfp.d.ts} +2 -2
  80. package/dist/{summary-report-BDOFevaT.d.ts → summary-report-CInXwsza.d.ts} +1 -1
  81. package/dist/testing-C21CHsq2.d.ts +20 -0
  82. package/dist/testing.d.ts +1 -0
  83. package/dist/testing.js +8 -0
  84. package/dist/testing.js.map +1 -0
  85. package/dist/traces.d.ts +26 -10
  86. package/dist/traces.js +41 -11
  87. package/dist/{types-Ce17tDlG.d.ts → types-B5x54y6n.d.ts} +1 -1
  88. package/dist/{types-mn5Aqk7x.d.ts → types-BUxNaJ8c.d.ts} +2 -4
  89. package/dist/{types-BU-7W85F.d.ts → types-DQRY8ZT-.d.ts} +60 -58
  90. package/dist/workflow/index.d.ts +5 -4
  91. package/dist/workflow/index.js +1 -1
  92. package/docs/campaign-proposers.md +170 -0
  93. package/docs/concepts.md +8 -4
  94. package/docs/customer-journeys.md +15 -13
  95. package/docs/design/loop-taxonomy.md +34 -66
  96. package/docs/distributed-driver.md +14 -14
  97. package/docs/feature-guide.md +1 -1
  98. package/docs/hosted-ingest-spec.md +2 -3
  99. package/docs/multi-shot-optimization.md +8 -8
  100. package/docs/product-eval-adoption.md +1 -1
  101. package/docs/self-improvement-map.md +33 -29
  102. package/package.json +8 -14
  103. package/dist/chunk-2K6UUZ7P.js.map +0 -1
  104. package/dist/chunk-CTBHKLEU.js.map +0 -1
  105. package/dist/chunk-E4GH6USR.js.map +0 -1
  106. package/dist/chunk-EGPMSBEZ.js.map +0 -1
  107. package/dist/chunk-KWRRMR3J.js.map +0 -1
  108. package/dist/chunk-MIFZUPEK.js.map +0 -1
  109. package/dist/chunk-MPQWFX6Y.js.map +0 -1
  110. package/dist/chunk-Q5LIB7BC.js.map +0 -1
  111. package/dist/chunk-QMUEXQJS.js.map +0 -1
  112. package/dist/chunk-SD2YFWQQ.js.map +0 -1
  113. package/docs/design/external-agent-wedge.md +0 -89
  114. package/docs/design/phase-d-rfc.md +0 -125
  115. package/docs/design/phase4-consumer-migration.md +0 -70
  116. package/docs/design/primitives-integration-spec.md +0 -393
  117. package/docs/design/product-self-improvement-loop.md +0 -146
  118. package/docs/design/self-improvement-engine.md +0 -140
  119. package/docs/design/self-improvement-protocol.md +0 -223
  120. package/docs/design/self-improvement-roadmap.md +0 -106
  121. package/docs/design/substrate-gaps.md +0 -118
  122. package/docs/phase-b-pairing-kit.md +0 -188
  123. package/docs/phase-b-runbook.md +0 -176
  124. package/docs/pilot/README.md +0 -62
  125. package/docs/pilot/customer-checklist.md +0 -90
  126. package/docs/pilot/integration-foreign-stack.md +0 -296
  127. package/docs/pilot/integration-tangle-stack.md +0 -248
  128. package/docs/pilot/one-pager.md +0 -161
  129. package/docs/pilot/sample-insight-report.json +0 -172
  130. package/docs/quickstart-external.md +0 -229
  131. package/docs/research/belief-state-agent-eval-roadmap.md +0 -593
  132. package/docs/research/research-roadmap.md +0 -205
  133. package/docs/specs/driver-honest-spec.md +0 -251
  134. package/docs/specs/hermes-self-improvement-audit.md +0 -93
  135. package/docs/specs/profile-versioning.md +0 -291
  136. package/docs/three-package-architecture.md +0 -168
  137. /package/dist/{chunk-TBDR6PAI.js.map → chunk-IZCEK2HR.js.map} +0 -0
  138. /package/dist/{chunk-KW53MSA5.js.map → chunk-X74V6ESX.js.map} +0 -0
  139. /package/dist/{chunk-2KNZHH3P.js.map → chunk-Z6L6YSU6.js.map} +0 -0
  140. /package/dist/{run-campaign-7WNXMDSN.js.map → run-campaign-WXY7KI67.js.map} +0 -0
package/dist/rl.js CHANGED
@@ -16,7 +16,7 @@ import {
16
16
  } from "./chunk-3RF76KTD.js";
17
17
  import {
18
18
  runEvalCampaign
19
- } from "./chunk-KW53MSA5.js";
19
+ } from "./chunk-X74V6ESX.js";
20
20
  import "./chunk-CWNP4DV4.js";
21
21
  import {
22
22
  rubricPredictiveValidity
@@ -36,7 +36,7 @@ import {
36
36
  } from "./chunk-VZSRQ272.js";
37
37
  import "./chunk-SBCB6VZY.js";
38
38
  import "./chunk-PC4UYEBM.js";
39
- import "./chunk-KWRRMR3J.js";
39
+ import "./chunk-LO6IOIJ2.js";
40
40
  import "./chunk-TVVP3ZZQ.js";
41
41
  import "./chunk-VSMTAMNK.js";
42
42
  import {
@@ -44,6 +44,197 @@ import {
44
44
  } from "./chunk-3BFEG2F6.js";
45
45
  import "./chunk-PZ5AY32C.js";
46
46
 
47
+ // src/rl/adaptation-eval.ts
48
+ async function runAdaptationCurve(opts) {
49
+ const ks = opts.ks ?? [0, 1, 2, 4, 8, 16];
50
+ const reps = opts.reps ?? 3;
51
+ const passThreshold = opts.passThreshold ?? 0.5;
52
+ const sortedKs = [...ks].sort((a, b) => a - b);
53
+ const points = [];
54
+ for (const k of sortedKs) {
55
+ const perScenario = [];
56
+ const allScores = [];
57
+ let totalPasses = 0;
58
+ let totalAttempts = 0;
59
+ for (const scenario of opts.scenarios) {
60
+ const sid = scenario.scenarioId ?? `scenario-${opts.scenarios.indexOf(scenario)}`;
61
+ const scores = [];
62
+ let passes = 0;
63
+ for (let r = 0; r < reps; r++) {
64
+ const score = await opts.runner.run({ scenario, k, rep: r });
65
+ scores.push(score);
66
+ if (score >= passThreshold) passes++;
67
+ allScores.push(score);
68
+ if (score >= passThreshold) totalPasses++;
69
+ totalAttempts++;
70
+ }
71
+ const meanS = scores.reduce((s, v) => s + v, 0) / scores.length;
72
+ perScenario.push({ scenarioId: sid, meanScore: meanS, passes, total: scores.length });
73
+ }
74
+ const meanScore = allScores.reduce((s, v) => s + v, 0) / Math.max(1, allScores.length);
75
+ const variance = allScores.length < 2 ? 0 : allScores.reduce((s, v) => s + (v - meanScore) ** 2, 0) / (allScores.length - 1);
76
+ points.push({
77
+ k,
78
+ meanScore,
79
+ passRate: totalPasses / Math.max(1, totalAttempts),
80
+ std: Math.sqrt(variance),
81
+ n: allScores.length,
82
+ perScenario
83
+ });
84
+ }
85
+ const firstPassK2 = points.find((p) => p.passRate >= passThreshold)?.k ?? null;
86
+ const maxK = sortedKs[sortedKs.length - 1] ?? 1;
87
+ let area = 0;
88
+ for (let i = 1; i < points.length; i++) {
89
+ const x1 = points[i - 1].k;
90
+ const x2 = points[i].k;
91
+ const y1 = points[i - 1].meanScore;
92
+ const y2 = points[i].meanScore;
93
+ area += (y1 + y2) / 2 * (x2 - x1);
94
+ }
95
+ const adaptationArea = maxK === 0 ? 0 : area / maxK;
96
+ return { points, firstPassK: firstPassK2, adaptationArea };
97
+ }
98
+ function compareAdaptationCurves(a, b, opts = {}) {
99
+ const conf = opts.confidence ?? 0.95;
100
+ const resamples = opts.bootstrapResamples ?? 500;
101
+ const rng = makeRng(opts.seed);
102
+ const perK = [];
103
+ for (const ap of a.points) {
104
+ const bp = b.points.find((p) => p.k === ap.k);
105
+ if (!bp) continue;
106
+ const aMeans = ap.perScenario.map((s) => s.meanScore);
107
+ const bMeans = bp.perScenario.map((s) => s.meanScore);
108
+ const aCi = bootstrapMeanCi(aMeans, resamples, conf, rng);
109
+ const bCi = bootstrapMeanCi(bMeans, resamples, conf, rng);
110
+ perK.push({
111
+ k: ap.k,
112
+ deltaMean: ap.meanScore - bp.meanScore,
113
+ aLow: aCi.low,
114
+ aHigh: aCi.high,
115
+ bLow: bCi.low,
116
+ bHigh: bCi.high
117
+ });
118
+ }
119
+ const areaDelta = a.adaptationArea - b.adaptationArea;
120
+ const firstPassKDelta = a.firstPassK !== null && b.firstPassK !== null ? b.firstPassK - a.firstPassK : null;
121
+ const meanDelta = perK.reduce((s, p) => s + p.deltaMean, 0) / Math.max(1, perK.length);
122
+ let verdict;
123
+ if (Math.abs(meanDelta) < 0.02 && Math.abs(areaDelta) < 0.02) verdict = "similar";
124
+ else if (meanDelta > 0 && areaDelta > 0) verdict = "a_better";
125
+ else if (meanDelta < 0 && areaDelta < 0) verdict = "b_better";
126
+ else verdict = "similar";
127
+ const rationale = `mean per-k delta=${meanDelta.toFixed(3)}, area delta=${areaDelta.toFixed(3)}` + (firstPassKDelta !== null ? `, first-pass-k delta=${firstPassKDelta}` : "");
128
+ return { perK, areaDelta, firstPassKDelta, verdict, rationale };
129
+ }
130
+ function firstPassK(curve, threshold = 0.5) {
131
+ return curve.points.find((p) => p.passRate >= threshold)?.k ?? null;
132
+ }
133
+ function makeRng(seed) {
134
+ if (seed === void 0) return Math.random;
135
+ let s = seed >>> 0;
136
+ return () => {
137
+ s = s + 1831565813 >>> 0;
138
+ let t = s;
139
+ t = Math.imul(t ^ t >>> 15, t | 1);
140
+ t ^= t + Math.imul(t ^ t >>> 7, t | 61);
141
+ return ((t ^ t >>> 14) >>> 0) / 4294967296;
142
+ };
143
+ }
144
+ function bootstrapMeanCi(xs, resamples, confidence, rng) {
145
+ if (xs.length < 2) return { low: xs[0] ?? 0, high: xs[0] ?? 0 };
146
+ const samples = new Array(resamples);
147
+ for (let b = 0; b < resamples; b++) {
148
+ let sum = 0;
149
+ for (let i = 0; i < xs.length; i++) sum += xs[Math.floor(rng() * xs.length)];
150
+ samples[b] = sum / xs.length;
151
+ }
152
+ samples.sort((a, b) => a - b);
153
+ const alpha = 1 - confidence;
154
+ return {
155
+ low: samples[Math.floor(alpha / 2 * resamples)],
156
+ high: samples[Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)]
157
+ };
158
+ }
159
+
160
+ // src/rl/adversarial.ts
161
+ async function adversarialScenarioSearch(opts) {
162
+ const failureThreshold = opts.failureThreshold ?? 0.5;
163
+ const rounds = opts.rounds ?? 3;
164
+ const children = opts.childrenPerParent ?? 4;
165
+ const budget = opts.budget ?? Number.POSITIVE_INFINITY;
166
+ const seed = opts.seed ?? 1;
167
+ const rng = mulberry32(seed);
168
+ const scenarios = [];
169
+ const seen = /* @__PURE__ */ new Set();
170
+ let scoreCalls = 0;
171
+ for (const s of opts.seeds) {
172
+ const id = opts.mutateScenarioId(s);
173
+ if (seen.has(id)) continue;
174
+ seen.add(id);
175
+ if (scoreCalls >= budget) break;
176
+ const score = await opts.scoreFn(s);
177
+ scoreCalls++;
178
+ scenarios.push({
179
+ id,
180
+ generation: 0,
181
+ parentId: null,
182
+ scenario: s,
183
+ score,
184
+ mutationStrategy: null
185
+ });
186
+ }
187
+ for (let g = 1; g <= rounds; g++) {
188
+ if (scoreCalls >= budget) break;
189
+ const parents = scenarios.filter((s) => s.generation === g - 1);
190
+ for (const parent of parents) {
191
+ for (const mutation of opts.mutations) {
192
+ if (scoreCalls >= budget) break;
193
+ const produced = await mutation.mutate(parent.scenario, rng);
194
+ const childArr = Array.isArray(produced) ? produced : [produced];
195
+ for (let k = 0; k < Math.min(children, childArr.length); k++) {
196
+ if (scoreCalls >= budget) break;
197
+ const child = childArr[k];
198
+ const cid = opts.mutateScenarioId(child);
199
+ if (seen.has(cid)) continue;
200
+ seen.add(cid);
201
+ const cscore = await opts.scoreFn(child);
202
+ scoreCalls++;
203
+ scenarios.push({
204
+ id: cid,
205
+ generation: g,
206
+ parentId: parent.id,
207
+ scenario: child,
208
+ score: cscore,
209
+ mutationStrategy: mutation.id
210
+ });
211
+ }
212
+ }
213
+ }
214
+ }
215
+ const failures = scenarios.filter((s) => s.score !== null && s.score < failureThreshold).sort((a, b) => (a.score ?? 0) - (b.score ?? 0));
216
+ const byGeneration = [];
217
+ const maxGen = scenarios.reduce((m, s) => Math.max(m, s.generation), 0);
218
+ for (let g = 0; g <= maxGen; g++) {
219
+ const gens = scenarios.filter((s) => s.generation === g);
220
+ if (gens.length === 0) continue;
221
+ const fails = gens.filter((s) => s.score !== null && s.score < failureThreshold).length;
222
+ const meanScore = gens.reduce((sum, s) => sum + (s.score ?? 0), 0) / gens.length;
223
+ byGeneration.push({ generation: g, total: gens.length, failures: fails, meanScore });
224
+ }
225
+ return { scenarios, failures, byGeneration, scoreCalls };
226
+ }
227
+ function mulberry32(seed) {
228
+ let s = seed >>> 0;
229
+ return () => {
230
+ s = s + 1831565813 >>> 0;
231
+ let t = s;
232
+ t = Math.imul(t ^ t >>> 15, t | 1);
233
+ t ^= t + Math.imul(t ^ t >>> 7, t | 61);
234
+ return ((t ^ t >>> 14) >>> 0) / 4294967296;
235
+ };
236
+ }
237
+
47
238
  // src/rl/compute-curves.ts
48
239
  async function runComputeCurve(opts) {
49
240
  const points = [];
@@ -182,631 +373,65 @@ async function runContaminationProbe(input, opts = {}) {
182
373
  const pseudoP = valid.map((p) => Math.min(1, Math.max(1e-6, 1 - Math.abs(p.delta) / 1)));
183
374
  const { qValues } = benjaminiHochberg(pseudoP, fdr);
184
375
  for (let i = 0; i < valid.length; i++) {
185
- const v = valid[i];
186
- const idx = perScenario.findIndex((p) => p.scenarioId === v.scenarioId);
187
- if (idx >= 0) perScenario[idx].qValue = qValues[i];
188
- }
189
- const contaminationSuspected = pairedTest.p < fdr && median <= -minMedianDrop;
190
- const reason = contaminationSuspected ? `paired p=${pairedTest.p.toFixed(4)} < ${fdr} and median drop ${median.toFixed(4)} \u2265 ${minMedianDrop}` : pairedTest.p >= fdr ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})` : `significant but small effect (median delta ${median.toFixed(4)})`;
191
- return {
192
- perScenario,
193
- pairedTest,
194
- medianDelta: median,
195
- meanDelta: mean,
196
- contaminationSuspected,
197
- reason,
198
- n: valid.length
199
- };
200
- }
201
- function renameVariables(identifiers, rename = (n, i) => `${n}_${(i % 26 + 10).toString(36)}`) {
202
- return {
203
- kind: "rename_variables",
204
- apply(scenario) {
205
- let prompt = scenario.prompt;
206
- identifiers.forEach((id, i) => {
207
- const replacement = rename(id, i);
208
- const re = new RegExp(`\\b${escapeRegex(id)}\\b`, "g");
209
- prompt = prompt.replace(re, replacement);
210
- });
211
- return { ...scenario, prompt };
212
- }
213
- };
214
- }
215
- function shuffleOrder(shuffleSection, seed) {
216
- let s = seed >>> 0;
217
- const rng = () => {
218
- s = s + 1831565813 >>> 0;
219
- let t = s;
220
- t = Math.imul(t ^ t >>> 15, t | 1);
221
- t ^= t + Math.imul(t ^ t >>> 7, t | 61);
222
- return ((t ^ t >>> 14) >>> 0) / 4294967296;
223
- };
224
- return {
225
- kind: "shuffle_order",
226
- apply(scenario) {
227
- const newPrompt = shuffleSection(scenario.prompt, rng);
228
- return { ...scenario, prompt: newPrompt };
229
- }
230
- };
231
- }
232
- function injectIrrelevantClause(clause, position = "prefix") {
233
- return {
234
- kind: "inject_irrelevant_clause",
235
- apply(scenario) {
236
- const prompt = position === "prefix" ? `${clause} ${scenario.prompt}` : `${scenario.prompt} ${clause}`;
237
- return { ...scenario, prompt };
238
- }
239
- };
240
- }
241
- function escapeRegex(s) {
242
- return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
243
- }
244
-
245
- // src/rl/preferences.ts
246
- var SPLIT_TAG_DEFAULT = "holdout";
247
- var DEFAULT_REWARD = (run) => {
248
- const v = run.outcome.holdoutScore ?? run.outcome.searchScore;
249
- return typeof v === "number" && Number.isFinite(v) ? v : null;
250
- };
251
- function extractPreferences(runs, opts = {}) {
252
- const strategy = opts.strategy ?? "paired-by-scenario-and-seed";
253
- const minMargin = opts.minMargin ?? 0.05;
254
- const splitTag = opts.splitTag ?? SPLIT_TAG_DEFAULT;
255
- const rewardOf2 = opts.rewardOf ?? DEFAULT_REWARD;
256
- const filtered = runs.filter((r) => r.splitTag === splitTag);
257
- const scoredEntries = [];
258
- for (const run of filtered) {
259
- const s = rewardOf2(run);
260
- if (s === null) continue;
261
- scoredEntries.push({ run, score: s });
262
- }
263
- const pairs = [];
264
- let pairsBelowMargin = 0;
265
- let cellsSingleton = 0;
266
- let cellsInspected = 0;
267
- if (strategy === "paired-by-scenario-and-seed") {
268
- const groups = /* @__PURE__ */ new Map();
269
- for (const e of scoredEntries) {
270
- const sid = scenarioOf(e.run);
271
- const key = `${sid}::${e.run.seed}`;
272
- const arr = groups.get(key) ?? [];
273
- arr.push(e);
274
- groups.set(key, arr);
275
- }
276
- for (const [key, members] of groups.entries()) {
277
- cellsInspected++;
278
- if (members.length < 2) {
279
- cellsSingleton++;
280
- continue;
281
- }
282
- for (let i = 0; i < members.length; i++) {
283
- for (let j = i + 1; j < members.length; j++) {
284
- const a = members[i];
285
- const b = members[j];
286
- if (a.run.candidateId === b.run.candidateId) continue;
287
- const result = makePair(a, b, key.split("::")[0], minMargin);
288
- if (result.kind === "admit") pairs.push(result.pair);
289
- else pairsBelowMargin++;
290
- }
291
- }
292
- }
293
- } else if (strategy === "paired-by-scenario") {
294
- const byScenarioVariant = /* @__PURE__ */ new Map();
295
- for (const e of scoredEntries) {
296
- const sid = scenarioOf(e.run);
297
- let perScenario = byScenarioVariant.get(sid);
298
- if (!perScenario) {
299
- perScenario = /* @__PURE__ */ new Map();
300
- byScenarioVariant.set(sid, perScenario);
301
- }
302
- const cur = perScenario.get(e.run.candidateId);
303
- if (cur) {
304
- cur.sum += e.score;
305
- cur.n++;
306
- } else perScenario.set(e.run.candidateId, { run: e.run, sum: e.score, n: 1 });
307
- }
308
- for (const [sid, perVariant] of byScenarioVariant.entries()) {
309
- cellsInspected++;
310
- const arr = [...perVariant.entries()].map(([vid, agg]) => ({
311
- run: agg.run,
312
- score: agg.sum / agg.n,
313
- variantId: vid
314
- }));
315
- if (arr.length < 2) {
316
- cellsSingleton++;
317
- continue;
318
- }
319
- for (let i = 0; i < arr.length; i++) {
320
- for (let j = i + 1; j < arr.length; j++) {
321
- const result = makePair(arr[i], arr[j], sid, minMargin);
322
- if (result.kind === "admit") pairs.push(result.pair);
323
- else pairsBelowMargin++;
324
- }
325
- }
326
- }
327
- } else {
328
- const byScenario = /* @__PURE__ */ new Map();
329
- for (const e of scoredEntries) {
330
- const sid = scenarioOf(e.run);
331
- const arr = byScenario.get(sid) ?? [];
332
- arr.push(e);
333
- byScenario.set(sid, arr);
334
- }
335
- for (const [sid, arr] of byScenario.entries()) {
336
- cellsInspected++;
337
- if (arr.length < 2) {
338
- cellsSingleton++;
339
- continue;
340
- }
341
- const sorted = [...arr].sort((a, b) => a.score - b.score);
342
- const top = sorted[sorted.length - 1];
343
- const bot = sorted[0];
344
- if (top.run.candidateId === bot.run.candidateId) {
345
- cellsSingleton++;
346
- continue;
347
- }
348
- const result = makePair(bot, top, sid, minMargin);
349
- if (result.kind === "admit") pairs.push(result.pair);
350
- else pairsBelowMargin++;
351
- }
352
- }
353
- return { pairs, cellsInspected, pairsBelowMargin, cellsSingleton, strategy };
354
- }
355
- function toTRLFormat(triples, promptOf) {
356
- return triples.map((t) => ({
357
- prompt: promptOf(t.meta.chosenPromptHash),
358
- chosen: t.meta.chosenPromptHash,
359
- // caller substitutes the model output via the runId map
360
- rejected: t.meta.rejectedPromptHash
361
- }));
362
- }
363
- function toAnthropicFormat(triples) {
364
- return triples.map((t) => ({
365
- scenarioId: t.scenarioId,
366
- chosenRunId: t.chosenRunId,
367
- rejectedRunId: t.rejectedRunId,
368
- margin: t.marginScore
369
- }));
370
- }
371
- function makePair(a, b, scenarioId, minMargin) {
372
- const margin = Math.abs(a.score - b.score);
373
- if (margin < minMargin) return { kind: "reject" };
374
- const [chosen, rejected] = a.score > b.score ? [a, b] : [b, a];
375
- return {
376
- kind: "admit",
377
- pair: {
378
- scenarioId,
379
- chosenRunId: chosen.run.runId,
380
- rejectedRunId: rejected.run.runId,
381
- chosenVariantId: chosen.run.candidateId,
382
- rejectedVariantId: rejected.run.candidateId,
383
- marginScore: chosen.score - rejected.score,
384
- scores: { chosen: chosen.score, rejected: rejected.score },
385
- seed: chosen.run.seed === rejected.run.seed ? chosen.run.seed : void 0,
386
- meta: {
387
- chosenPromptHash: chosen.run.promptHash,
388
- rejectedPromptHash: rejected.run.promptHash,
389
- chosenConfigHash: chosen.run.configHash,
390
- rejectedConfigHash: rejected.run.configHash,
391
- chosenModel: chosen.run.model,
392
- rejectedModel: rejected.run.model
393
- }
394
- }
395
- };
396
- }
397
- function scenarioOf(run) {
398
- if (typeof run.scenarioId === "string" && run.scenarioId.length > 0) return run.scenarioId;
399
- const fromRaw = run.outcome.raw.scenario_id;
400
- if (typeof fromRaw === "number" && Number.isFinite(fromRaw)) return String(fromRaw);
401
- if (typeof fromRaw === "string") return fromRaw;
402
- return run.experimentId;
403
- }
404
-
405
- // src/rl/run-record-adapters.ts
406
- function campaignToRunRecords(campaign, ctx) {
407
- const splitTag = ctx.splitTag ?? "search";
408
- const candidateId = ctx.candidateId ?? campaign.manifestHash;
409
- return campaign.cells.map((cell) => {
410
- const composites = Object.values(cell.judgeScores).map((s) => s.composite);
411
- const score = composites.length > 0 ? composites.reduce((a, b) => a + b, 0) / composites.length : 0;
412
- const raw = { rep: cell.rep, duration_ms: cell.durationMs };
413
- for (const judge of Object.values(cell.judgeScores)) {
414
- for (const [dim, value] of Object.entries(judge.dimensions)) {
415
- if (Number.isFinite(value)) raw[`dim.${dim}`] = value;
416
- }
417
- }
418
- if (typeof cell.generation === "number") raw.generation = cell.generation;
419
- const outcome = { raw };
420
- if (splitTag === "holdout") outcome.holdoutScore = score;
421
- else outcome.searchScore = score;
422
- return {
423
- runId: cell.cellId,
424
- experimentId: ctx.experimentId,
425
- candidateId,
426
- seed: cell.seed,
427
- model: ctx.model,
428
- promptHash: ctx.promptHash,
429
- configHash: ctx.configHash,
430
- commitSha: ctx.commitSha,
431
- wallMs: cell.durationMs,
432
- costUsd: Number.isFinite(cell.costUsd) ? cell.costUsd : ctx.defaultCostUsd ?? 0,
433
- tokenUsage: { input: 0, output: 0 },
434
- outcome,
435
- failureMode: cell.error ? "cell_error" : void 0,
436
- splitTag,
437
- scenarioId: cell.scenarioId
438
- };
439
- });
440
- }
441
- function verificationReportToRunRecord(report, ctx, opts = {}) {
442
- const splitTag = ctx.splitTag ?? "search";
443
- const runId = opts.runId ?? `run-${ctx.candidateId}-${ctx.experimentId}-${report.startedAt}`;
444
- const raw = {
445
- pass_count: report.passCount,
446
- fail_count: report.failCount,
447
- error_count: report.errorCount,
448
- skipped_count: report.skippedCount,
449
- duration_ms: report.durationMs,
450
- blended_score: report.blendedScore
451
- };
452
- for (const layer of report.layers) {
453
- if (typeof layer.score === "number") raw[`layer.${layer.layer}`] = layer.score;
454
- raw[`layer_${layer.layer}_pass`] = layer.status === "pass" ? 1 : 0;
455
- if (layer.diagnostics) {
456
- for (const [k, v] of Object.entries(layer.diagnostics)) {
457
- if (typeof v === "number" && Number.isFinite(v)) raw[`layer.${layer.layer}.${k}`] = v;
458
- }
459
- }
460
- }
461
- const firstFail = report.layers.find((l) => l.status === "fail" || l.status === "error");
462
- const outcome = { raw };
463
- if (splitTag === "holdout") outcome.holdoutScore = report.blendedScore;
464
- else outcome.searchScore = report.blendedScore;
465
- return {
466
- runId,
467
- experimentId: ctx.experimentId,
468
- candidateId: ctx.candidateId,
469
- seed: 0,
470
- model: ctx.model,
471
- promptHash: ctx.promptHash,
472
- configHash: ctx.configHash,
473
- commitSha: ctx.commitSha,
474
- wallMs: report.durationMs,
475
- costUsd: ctx.defaultCostUsd ?? 0,
476
- tokenUsage: { input: 0, output: 0 },
477
- outcome,
478
- failureMode: firstFail ? failureModeFromLayer(firstFail) : void 0,
479
- splitTag,
480
- scenarioId: ctx.scenarioId
481
- };
482
- }
483
- function failureModeFromLayer(layer) {
484
- if (layer.status === "error") return `layer_${layer.layer}_error`;
485
- if (layer.status === "fail") return `layer_${layer.layer}_fail`;
486
- if (layer.status === "timeout") return `layer_${layer.layer}_timeout`;
487
- return `layer_${layer.layer}_${layer.status}`;
488
- }
489
-
490
- // src/rl/tournament.ts
491
- function fitBradleyTerry(outcomes, opts = {}) {
492
- const tol = opts.tolerance ?? 1e-6;
493
- const maxIter = opts.maxIterations ?? 256;
494
- const smoothing = opts.smoothing ?? 0.1;
495
- const candidates = /* @__PURE__ */ new Set();
496
- for (const o of outcomes) {
497
- candidates.add(o.winner);
498
- candidates.add(o.loser);
499
- }
500
- const ids = [...candidates].sort();
501
- const idx = new Map(ids.map((id, i) => [id, i]));
502
- const n = ids.length;
503
- if (n === 0) return { ratings: [], iterations: 0, finalDelta: 0, converged: true };
504
- if (n === 1) {
505
- return {
506
- ratings: [{ candidateId: ids[0], strength: 1, logStrength: 0, n: 0, wins: 0 }],
507
- iterations: 0,
508
- finalDelta: 0,
509
- converged: true
510
- };
511
- }
512
- const W = Array.from({ length: n }, () => new Array(n).fill(0));
513
- const N = Array.from({ length: n }, () => new Array(n).fill(0));
514
- for (const o of outcomes) {
515
- const i = idx.get(o.winner);
516
- const j = idx.get(o.loser);
517
- const w = o.weight ?? 1;
518
- if (o.draw) {
519
- W[i][j] += 0.5 * w;
520
- W[j][i] += 0.5 * w;
521
- } else {
522
- W[i][j] += w;
523
- }
524
- N[i][j] += w;
525
- N[j][i] += w;
526
- }
527
- const winsTotal = new Array(n).fill(0);
528
- for (let i = 0; i < n; i++) {
529
- for (let j = 0; j < n; j++) winsTotal[i] += W[i][j];
530
- winsTotal[i] += smoothing;
531
- }
532
- const compsTotal = new Array(n).fill(0);
533
- for (let i = 0; i < n; i++) {
534
- for (let j = 0; j < n; j++) compsTotal[i] += N[i][j];
535
- }
536
- let theta = new Array(n).fill(1);
537
- let iter = 0;
538
- let delta = Infinity;
539
- for (; iter < maxIter; iter++) {
540
- const newTheta = new Array(n);
541
- for (let i = 0; i < n; i++) {
542
- let denom = 0;
543
- for (let j = 0; j < n; j++) {
544
- if (j === i) continue;
545
- if (N[i][j] === 0) continue;
546
- denom += N[i][j] / (theta[i] + theta[j]);
547
- }
548
- newTheta[i] = denom === 0 ? theta[i] : winsTotal[i] / denom;
549
- }
550
- let logSum = 0;
551
- for (let i = 0; i < n; i++) logSum += Math.log(Math.max(1e-300, newTheta[i]));
552
- const norm = Math.exp(logSum / n);
553
- for (let i = 0; i < n; i++) newTheta[i] = newTheta[i] / norm;
554
- delta = 0;
555
- for (let i = 0; i < n; i++) {
556
- const d = Math.abs(newTheta[i] - theta[i]) / Math.max(1e-12, theta[i]);
557
- if (d > delta) delta = d;
558
- }
559
- theta = newTheta;
560
- if (delta < tol) break;
561
- }
562
- const minLog = Math.min(...theta.map((t) => Math.log(Math.max(1e-300, t))));
563
- const ratings = ids.map((id, i) => ({
564
- candidateId: id,
565
- strength: theta[i],
566
- logStrength: Math.log(Math.max(1e-300, theta[i])) - minLog,
567
- n: compsTotal[i],
568
- wins: winsTotal[i] - smoothing
569
- }));
570
- return {
571
- ratings: ratings.sort((a, b) => b.strength - a.strength),
572
- iterations: iter,
573
- finalDelta: delta,
574
- converged: delta < tol
575
- };
576
- }
577
- function applyEloUpdate(ratings, outcome, opts = {}) {
578
- const defaultRating = opts.defaultRating ?? 1500;
579
- const k = opts.kFactor ?? 32;
580
- const rW = ratings.get(outcome.winner) ?? defaultRating;
581
- const rL = ratings.get(outcome.loser) ?? defaultRating;
582
- const expectedW = 1 / (1 + 10 ** ((rL - rW) / 400));
583
- const scoreW = outcome.draw ? 0.5 : 1;
584
- const scoreL = outcome.draw ? 0.5 : 0;
585
- const w = outcome.weight ?? 1;
586
- const winnerDelta = k * w * (scoreW - expectedW);
587
- const loserDelta = k * w * (scoreL - (1 - expectedW));
588
- ratings.set(outcome.winner, rW + winnerDelta);
589
- ratings.set(outcome.loser, rL + loserDelta);
590
- return { winnerDelta, loserDelta };
591
- }
592
- function buildPairwiseFromCampaign(input) {
593
- const drawMargin = input.drawMargin ?? 0;
594
- const byKey = /* @__PURE__ */ new Map();
595
- for (const r of input.runs) {
596
- const arr = byKey.get(r.matchKey) ?? [];
597
- arr.push({ candidateId: r.candidateId, score: r.score });
598
- byKey.set(r.matchKey, arr);
599
- }
600
- const outcomes = [];
601
- for (const arr of byKey.values()) {
602
- for (let i = 0; i < arr.length; i++) {
603
- for (let j = i + 1; j < arr.length; j++) {
604
- const a = arr[i];
605
- const b = arr[j];
606
- if (a.candidateId === b.candidateId) continue;
607
- const margin = Math.abs(a.score - b.score);
608
- if (margin <= drawMargin) {
609
- outcomes.push({ winner: a.candidateId, loser: b.candidateId, draw: true, weight: 1 });
610
- } else {
611
- const [winner, loser] = a.score > b.score ? [a, b] : [b, a];
612
- outcomes.push({ winner: winner.candidateId, loser: loser.candidateId, weight: margin });
613
- }
614
- }
615
- }
616
- }
617
- return outcomes;
618
- }
619
-
620
- // src/rl/adaptation-eval.ts
621
- async function runAdaptationCurve(opts) {
622
- const ks = opts.ks ?? [0, 1, 2, 4, 8, 16];
623
- const reps = opts.reps ?? 3;
624
- const passThreshold = opts.passThreshold ?? 0.5;
625
- const sortedKs = [...ks].sort((a, b) => a - b);
626
- const points = [];
627
- for (const k of sortedKs) {
628
- const perScenario = [];
629
- const allScores = [];
630
- let totalPasses = 0;
631
- let totalAttempts = 0;
632
- for (const scenario of opts.scenarios) {
633
- const sid = scenario.scenarioId ?? `scenario-${opts.scenarios.indexOf(scenario)}`;
634
- const scores = [];
635
- let passes = 0;
636
- for (let r = 0; r < reps; r++) {
637
- const score = await opts.runner.run({ scenario, k, rep: r });
638
- scores.push(score);
639
- if (score >= passThreshold) passes++;
640
- allScores.push(score);
641
- if (score >= passThreshold) totalPasses++;
642
- totalAttempts++;
643
- }
644
- const meanS = scores.reduce((s, v) => s + v, 0) / scores.length;
645
- perScenario.push({ scenarioId: sid, meanScore: meanS, passes, total: scores.length });
646
- }
647
- const meanScore = allScores.reduce((s, v) => s + v, 0) / Math.max(1, allScores.length);
648
- const variance = allScores.length < 2 ? 0 : allScores.reduce((s, v) => s + (v - meanScore) ** 2, 0) / (allScores.length - 1);
649
- points.push({
650
- k,
651
- meanScore,
652
- passRate: totalPasses / Math.max(1, totalAttempts),
653
- std: Math.sqrt(variance),
654
- n: allScores.length,
655
- perScenario
656
- });
657
- }
658
- const firstPassK2 = points.find((p) => p.passRate >= passThreshold)?.k ?? null;
659
- const maxK = sortedKs[sortedKs.length - 1] ?? 1;
660
- let area = 0;
661
- for (let i = 1; i < points.length; i++) {
662
- const x1 = points[i - 1].k;
663
- const x2 = points[i].k;
664
- const y1 = points[i - 1].meanScore;
665
- const y2 = points[i].meanScore;
666
- area += (y1 + y2) / 2 * (x2 - x1);
667
- }
668
- const adaptationArea = maxK === 0 ? 0 : area / maxK;
669
- return { points, firstPassK: firstPassK2, adaptationArea };
670
- }
671
- function compareAdaptationCurves(a, b, opts = {}) {
672
- const conf = opts.confidence ?? 0.95;
673
- const resamples = opts.bootstrapResamples ?? 500;
674
- const rng = makeRng(opts.seed);
675
- const perK = [];
676
- for (const ap of a.points) {
677
- const bp = b.points.find((p) => p.k === ap.k);
678
- if (!bp) continue;
679
- const aMeans = ap.perScenario.map((s) => s.meanScore);
680
- const bMeans = bp.perScenario.map((s) => s.meanScore);
681
- const aCi = bootstrapMeanCi(aMeans, resamples, conf, rng);
682
- const bCi = bootstrapMeanCi(bMeans, resamples, conf, rng);
683
- perK.push({
684
- k: ap.k,
685
- deltaMean: ap.meanScore - bp.meanScore,
686
- aLow: aCi.low,
687
- aHigh: aCi.high,
688
- bLow: bCi.low,
689
- bHigh: bCi.high
690
- });
691
- }
692
- const areaDelta = a.adaptationArea - b.adaptationArea;
693
- const firstPassKDelta = a.firstPassK !== null && b.firstPassK !== null ? b.firstPassK - a.firstPassK : null;
694
- const meanDelta = perK.reduce((s, p) => s + p.deltaMean, 0) / Math.max(1, perK.length);
695
- let verdict;
696
- if (Math.abs(meanDelta) < 0.02 && Math.abs(areaDelta) < 0.02) verdict = "similar";
697
- else if (meanDelta > 0 && areaDelta > 0) verdict = "a_better";
698
- else if (meanDelta < 0 && areaDelta < 0) verdict = "b_better";
699
- else verdict = "similar";
700
- const rationale = `mean per-k delta=${meanDelta.toFixed(3)}, area delta=${areaDelta.toFixed(3)}` + (firstPassKDelta !== null ? `, first-pass-k delta=${firstPassKDelta}` : "");
701
- return { perK, areaDelta, firstPassKDelta, verdict, rationale };
376
+ const v = valid[i];
377
+ const idx = perScenario.findIndex((p) => p.scenarioId === v.scenarioId);
378
+ if (idx >= 0) perScenario[idx].qValue = qValues[i];
379
+ }
380
+ const contaminationSuspected = pairedTest.p < fdr && median <= -minMedianDrop;
381
+ const reason = contaminationSuspected ? `paired p=${pairedTest.p.toFixed(4)} < ${fdr} and median drop ${median.toFixed(4)} \u2265 ${minMedianDrop}` : pairedTest.p >= fdr ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})` : `significant but small effect (median delta ${median.toFixed(4)})`;
382
+ return {
383
+ perScenario,
384
+ pairedTest,
385
+ medianDelta: median,
386
+ meanDelta: mean,
387
+ contaminationSuspected,
388
+ reason,
389
+ n: valid.length
390
+ };
702
391
  }
703
- function firstPassK(curve, threshold = 0.5) {
704
- return curve.points.find((p) => p.passRate >= threshold)?.k ?? null;
392
+ function renameVariables(identifiers, rename = (n, i) => `${n}_${(i % 26 + 10).toString(36)}`) {
393
+ return {
394
+ kind: "rename_variables",
395
+ apply(scenario) {
396
+ let prompt = scenario.prompt;
397
+ identifiers.forEach((id, i) => {
398
+ const replacement = rename(id, i);
399
+ const re = new RegExp(`\\b${escapeRegex(id)}\\b`, "g");
400
+ prompt = prompt.replace(re, replacement);
401
+ });
402
+ return { ...scenario, prompt };
403
+ }
404
+ };
705
405
  }
706
- function makeRng(seed) {
707
- if (seed === void 0) return Math.random;
406
+ function shuffleOrder(shuffleSection, seed) {
708
407
  let s = seed >>> 0;
709
- return () => {
408
+ const rng = () => {
710
409
  s = s + 1831565813 >>> 0;
711
410
  let t = s;
712
411
  t = Math.imul(t ^ t >>> 15, t | 1);
713
412
  t ^= t + Math.imul(t ^ t >>> 7, t | 61);
714
413
  return ((t ^ t >>> 14) >>> 0) / 4294967296;
715
414
  };
716
- }
717
- function bootstrapMeanCi(xs, resamples, confidence, rng) {
718
- if (xs.length < 2) return { low: xs[0] ?? 0, high: xs[0] ?? 0 };
719
- const samples = new Array(resamples);
720
- for (let b = 0; b < resamples; b++) {
721
- let sum = 0;
722
- for (let i = 0; i < xs.length; i++) sum += xs[Math.floor(rng() * xs.length)];
723
- samples[b] = sum / xs.length;
724
- }
725
- samples.sort((a, b) => a - b);
726
- const alpha = 1 - confidence;
727
415
  return {
728
- low: samples[Math.floor(alpha / 2 * resamples)],
729
- high: samples[Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)]
416
+ kind: "shuffle_order",
417
+ apply(scenario) {
418
+ const newPrompt = shuffleSection(scenario.prompt, rng);
419
+ return { ...scenario, prompt: newPrompt };
420
+ }
730
421
  };
731
422
  }
732
-
733
- // src/rl/adversarial.ts
734
- async function adversarialScenarioSearch(opts) {
735
- const failureThreshold = opts.failureThreshold ?? 0.5;
736
- const rounds = opts.rounds ?? 3;
737
- const children = opts.childrenPerParent ?? 4;
738
- const budget = opts.budget ?? Number.POSITIVE_INFINITY;
739
- const seed = opts.seed ?? 1;
740
- const rng = mulberry32(seed);
741
- const scenarios = [];
742
- const seen = /* @__PURE__ */ new Set();
743
- let scoreCalls = 0;
744
- for (const s of opts.seeds) {
745
- const id = opts.mutateScenarioId(s);
746
- if (seen.has(id)) continue;
747
- seen.add(id);
748
- if (scoreCalls >= budget) break;
749
- const score = await opts.scoreFn(s);
750
- scoreCalls++;
751
- scenarios.push({
752
- id,
753
- generation: 0,
754
- parentId: null,
755
- scenario: s,
756
- score,
757
- mutationStrategy: null
758
- });
759
- }
760
- for (let g = 1; g <= rounds; g++) {
761
- if (scoreCalls >= budget) break;
762
- const parents = scenarios.filter((s) => s.generation === g - 1);
763
- for (const parent of parents) {
764
- for (const mutation of opts.mutations) {
765
- if (scoreCalls >= budget) break;
766
- const produced = await mutation.mutate(parent.scenario, rng);
767
- const childArr = Array.isArray(produced) ? produced : [produced];
768
- for (let k = 0; k < Math.min(children, childArr.length); k++) {
769
- if (scoreCalls >= budget) break;
770
- const child = childArr[k];
771
- const cid = opts.mutateScenarioId(child);
772
- if (seen.has(cid)) continue;
773
- seen.add(cid);
774
- const cscore = await opts.scoreFn(child);
775
- scoreCalls++;
776
- scenarios.push({
777
- id: cid,
778
- generation: g,
779
- parentId: parent.id,
780
- scenario: child,
781
- score: cscore,
782
- mutationStrategy: mutation.id
783
- });
784
- }
785
- }
423
+ function injectIrrelevantClause(clause, position = "prefix") {
424
+ return {
425
+ kind: "inject_irrelevant_clause",
426
+ apply(scenario) {
427
+ const prompt = position === "prefix" ? `${clause} ${scenario.prompt}` : `${scenario.prompt} ${clause}`;
428
+ return { ...scenario, prompt };
786
429
  }
787
- }
788
- const failures = scenarios.filter((s) => s.score !== null && s.score < failureThreshold).sort((a, b) => (a.score ?? 0) - (b.score ?? 0));
789
- const byGeneration = [];
790
- const maxGen = scenarios.reduce((m, s) => Math.max(m, s.generation), 0);
791
- for (let g = 0; g <= maxGen; g++) {
792
- const gens = scenarios.filter((s) => s.generation === g);
793
- if (gens.length === 0) continue;
794
- const fails = gens.filter((s) => s.score !== null && s.score < failureThreshold).length;
795
- const meanScore = gens.reduce((sum, s) => sum + (s.score ?? 0), 0) / gens.length;
796
- byGeneration.push({ generation: g, total: gens.length, failures: fails, meanScore });
797
- }
798
- return { scenarios, failures, byGeneration, scoreCalls };
799
- }
800
- function mulberry32(seed) {
801
- let s = seed >>> 0;
802
- return () => {
803
- s = s + 1831565813 >>> 0;
804
- let t = s;
805
- t = Math.imul(t ^ t >>> 15, t | 1);
806
- t ^= t + Math.imul(t ^ t >>> 7, t | 61);
807
- return ((t ^ t >>> 14) >>> 0) / 4294967296;
808
430
  };
809
431
  }
432
+ function escapeRegex(s) {
433
+ return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
434
+ }
810
435
 
811
436
  // src/rl/corpus.ts
812
437
  import { appendFileSync, existsSync, mkdirSync, readFileSync } from "fs";
@@ -1286,6 +911,166 @@ var PredictiveValidityResearcher = class {
1286
911
  }
1287
912
  };
1288
913
 
914
+ // src/rl/preferences.ts
915
+ var SPLIT_TAG_DEFAULT = "holdout";
916
+ var DEFAULT_REWARD = (run) => {
917
+ const v = run.outcome.holdoutScore ?? run.outcome.searchScore;
918
+ return typeof v === "number" && Number.isFinite(v) ? v : null;
919
+ };
920
+ function extractPreferences(runs, opts = {}) {
921
+ const strategy = opts.strategy ?? "paired-by-scenario-and-seed";
922
+ const minMargin = opts.minMargin ?? 0.05;
923
+ const splitTag = opts.splitTag ?? SPLIT_TAG_DEFAULT;
924
+ const rewardOf2 = opts.rewardOf ?? DEFAULT_REWARD;
925
+ const filtered = runs.filter((r) => r.splitTag === splitTag);
926
+ const scoredEntries = [];
927
+ for (const run of filtered) {
928
+ const s = rewardOf2(run);
929
+ if (s === null) continue;
930
+ scoredEntries.push({ run, score: s });
931
+ }
932
+ const pairs = [];
933
+ let pairsBelowMargin = 0;
934
+ let cellsSingleton = 0;
935
+ let cellsInspected = 0;
936
+ if (strategy === "paired-by-scenario-and-seed") {
937
+ const groups = /* @__PURE__ */ new Map();
938
+ for (const e of scoredEntries) {
939
+ const sid = scenarioOf(e.run);
940
+ const key = `${sid}::${e.run.seed}`;
941
+ const arr = groups.get(key) ?? [];
942
+ arr.push(e);
943
+ groups.set(key, arr);
944
+ }
945
+ for (const [key, members] of groups.entries()) {
946
+ cellsInspected++;
947
+ if (members.length < 2) {
948
+ cellsSingleton++;
949
+ continue;
950
+ }
951
+ for (let i = 0; i < members.length; i++) {
952
+ for (let j = i + 1; j < members.length; j++) {
953
+ const a = members[i];
954
+ const b = members[j];
955
+ if (a.run.candidateId === b.run.candidateId) continue;
956
+ const result = makePair(a, b, key.split("::")[0], minMargin);
957
+ if (result.kind === "admit") pairs.push(result.pair);
958
+ else pairsBelowMargin++;
959
+ }
960
+ }
961
+ }
962
+ } else if (strategy === "paired-by-scenario") {
963
+ const byScenarioVariant = /* @__PURE__ */ new Map();
964
+ for (const e of scoredEntries) {
965
+ const sid = scenarioOf(e.run);
966
+ let perScenario = byScenarioVariant.get(sid);
967
+ if (!perScenario) {
968
+ perScenario = /* @__PURE__ */ new Map();
969
+ byScenarioVariant.set(sid, perScenario);
970
+ }
971
+ const cur = perScenario.get(e.run.candidateId);
972
+ if (cur) {
973
+ cur.sum += e.score;
974
+ cur.n++;
975
+ } else perScenario.set(e.run.candidateId, { run: e.run, sum: e.score, n: 1 });
976
+ }
977
+ for (const [sid, perVariant] of byScenarioVariant.entries()) {
978
+ cellsInspected++;
979
+ const arr = [...perVariant.entries()].map(([vid, agg]) => ({
980
+ run: agg.run,
981
+ score: agg.sum / agg.n,
982
+ variantId: vid
983
+ }));
984
+ if (arr.length < 2) {
985
+ cellsSingleton++;
986
+ continue;
987
+ }
988
+ for (let i = 0; i < arr.length; i++) {
989
+ for (let j = i + 1; j < arr.length; j++) {
990
+ const result = makePair(arr[i], arr[j], sid, minMargin);
991
+ if (result.kind === "admit") pairs.push(result.pair);
992
+ else pairsBelowMargin++;
993
+ }
994
+ }
995
+ }
996
+ } else {
997
+ const byScenario = /* @__PURE__ */ new Map();
998
+ for (const e of scoredEntries) {
999
+ const sid = scenarioOf(e.run);
1000
+ const arr = byScenario.get(sid) ?? [];
1001
+ arr.push(e);
1002
+ byScenario.set(sid, arr);
1003
+ }
1004
+ for (const [sid, arr] of byScenario.entries()) {
1005
+ cellsInspected++;
1006
+ if (arr.length < 2) {
1007
+ cellsSingleton++;
1008
+ continue;
1009
+ }
1010
+ const sorted = [...arr].sort((a, b) => a.score - b.score);
1011
+ const top = sorted[sorted.length - 1];
1012
+ const bot = sorted[0];
1013
+ if (top.run.candidateId === bot.run.candidateId) {
1014
+ cellsSingleton++;
1015
+ continue;
1016
+ }
1017
+ const result = makePair(bot, top, sid, minMargin);
1018
+ if (result.kind === "admit") pairs.push(result.pair);
1019
+ else pairsBelowMargin++;
1020
+ }
1021
+ }
1022
+ return { pairs, cellsInspected, pairsBelowMargin, cellsSingleton, strategy };
1023
+ }
1024
+ function toTRLFormat(triples, promptOf) {
1025
+ return triples.map((t) => ({
1026
+ prompt: promptOf(t.meta.chosenPromptHash),
1027
+ chosen: t.meta.chosenPromptHash,
1028
+ // caller substitutes the model output via the runId map
1029
+ rejected: t.meta.rejectedPromptHash
1030
+ }));
1031
+ }
1032
+ function toAnthropicFormat(triples) {
1033
+ return triples.map((t) => ({
1034
+ scenarioId: t.scenarioId,
1035
+ chosenRunId: t.chosenRunId,
1036
+ rejectedRunId: t.rejectedRunId,
1037
+ margin: t.marginScore
1038
+ }));
1039
+ }
1040
+ function makePair(a, b, scenarioId, minMargin) {
1041
+ const margin = Math.abs(a.score - b.score);
1042
+ if (margin < minMargin) return { kind: "reject" };
1043
+ const [chosen, rejected] = a.score > b.score ? [a, b] : [b, a];
1044
+ return {
1045
+ kind: "admit",
1046
+ pair: {
1047
+ scenarioId,
1048
+ chosenRunId: chosen.run.runId,
1049
+ rejectedRunId: rejected.run.runId,
1050
+ chosenVariantId: chosen.run.candidateId,
1051
+ rejectedVariantId: rejected.run.candidateId,
1052
+ marginScore: chosen.score - rejected.score,
1053
+ scores: { chosen: chosen.score, rejected: rejected.score },
1054
+ seed: chosen.run.seed === rejected.run.seed ? chosen.run.seed : void 0,
1055
+ meta: {
1056
+ chosenPromptHash: chosen.run.promptHash,
1057
+ rejectedPromptHash: rejected.run.promptHash,
1058
+ chosenConfigHash: chosen.run.configHash,
1059
+ rejectedConfigHash: rejected.run.configHash,
1060
+ chosenModel: chosen.run.model,
1061
+ rejectedModel: rejected.run.model
1062
+ }
1063
+ }
1064
+ };
1065
+ }
1066
+ function scenarioOf(run) {
1067
+ if (typeof run.scenarioId === "string" && run.scenarioId.length > 0) return run.scenarioId;
1068
+ const fromRaw = run.outcome.raw.scenario_id;
1069
+ if (typeof fromRaw === "number" && Number.isFinite(fromRaw)) return String(fromRaw);
1070
+ if (typeof fromRaw === "string") return fromRaw;
1071
+ return run.experimentId;
1072
+ }
1073
+
1289
1074
  // src/rl/process-reward.ts
1290
1075
  async function extractStepRewards(store, runId, opts) {
1291
1076
  const spans = await store.spans({ runId });
@@ -1524,6 +1309,91 @@ function buildSummary(args) {
1524
1309
  return lines.join(" | ");
1525
1310
  }
1526
1311
 
1312
+ // src/rl/run-record-adapters.ts
1313
+ function campaignToRunRecords(campaign, ctx) {
1314
+ const splitTag = ctx.splitTag ?? "search";
1315
+ const candidateId = ctx.candidateId ?? campaign.manifestHash;
1316
+ return campaign.cells.map((cell) => {
1317
+ const composites = Object.values(cell.judgeScores).map((s) => s.composite);
1318
+ const score = composites.length > 0 ? composites.reduce((a, b) => a + b, 0) / composites.length : 0;
1319
+ const raw = { rep: cell.rep, duration_ms: cell.durationMs };
1320
+ for (const judge of Object.values(cell.judgeScores)) {
1321
+ for (const [dim, value] of Object.entries(judge.dimensions)) {
1322
+ if (Number.isFinite(value)) raw[`dim.${dim}`] = value;
1323
+ }
1324
+ }
1325
+ if (typeof cell.generation === "number") raw.generation = cell.generation;
1326
+ const outcome = { raw };
1327
+ if (splitTag === "holdout") outcome.holdoutScore = score;
1328
+ else outcome.searchScore = score;
1329
+ return {
1330
+ runId: cell.cellId,
1331
+ experimentId: ctx.experimentId,
1332
+ candidateId,
1333
+ seed: cell.seed,
1334
+ model: ctx.model,
1335
+ promptHash: ctx.promptHash,
1336
+ configHash: ctx.configHash,
1337
+ commitSha: ctx.commitSha,
1338
+ wallMs: cell.durationMs,
1339
+ costUsd: Number.isFinite(cell.costUsd) ? cell.costUsd : ctx.defaultCostUsd ?? 0,
1340
+ tokenUsage: { input: 0, output: 0 },
1341
+ outcome,
1342
+ failureMode: cell.error ? "cell_error" : void 0,
1343
+ splitTag,
1344
+ scenarioId: cell.scenarioId
1345
+ };
1346
+ });
1347
+ }
1348
+ function verificationReportToRunRecord(report, ctx, opts = {}) {
1349
+ const splitTag = ctx.splitTag ?? "search";
1350
+ const runId = opts.runId ?? `run-${ctx.candidateId}-${ctx.experimentId}-${report.startedAt}`;
1351
+ const raw = {
1352
+ pass_count: report.passCount,
1353
+ fail_count: report.failCount,
1354
+ error_count: report.errorCount,
1355
+ skipped_count: report.skippedCount,
1356
+ duration_ms: report.durationMs,
1357
+ blended_score: report.blendedScore
1358
+ };
1359
+ for (const layer of report.layers) {
1360
+ if (typeof layer.score === "number") raw[`layer.${layer.layer}`] = layer.score;
1361
+ raw[`layer_${layer.layer}_pass`] = layer.status === "pass" ? 1 : 0;
1362
+ if (layer.diagnostics) {
1363
+ for (const [k, v] of Object.entries(layer.diagnostics)) {
1364
+ if (typeof v === "number" && Number.isFinite(v)) raw[`layer.${layer.layer}.${k}`] = v;
1365
+ }
1366
+ }
1367
+ }
1368
+ const firstFail = report.layers.find((l) => l.status === "fail" || l.status === "error");
1369
+ const outcome = { raw };
1370
+ if (splitTag === "holdout") outcome.holdoutScore = report.blendedScore;
1371
+ else outcome.searchScore = report.blendedScore;
1372
+ return {
1373
+ runId,
1374
+ experimentId: ctx.experimentId,
1375
+ candidateId: ctx.candidateId,
1376
+ seed: 0,
1377
+ model: ctx.model,
1378
+ promptHash: ctx.promptHash,
1379
+ configHash: ctx.configHash,
1380
+ commitSha: ctx.commitSha,
1381
+ wallMs: report.durationMs,
1382
+ costUsd: ctx.defaultCostUsd ?? 0,
1383
+ tokenUsage: { input: 0, output: 0 },
1384
+ outcome,
1385
+ failureMode: firstFail ? failureModeFromLayer(firstFail) : void 0,
1386
+ splitTag,
1387
+ scenarioId: ctx.scenarioId
1388
+ };
1389
+ }
1390
+ function failureModeFromLayer(layer) {
1391
+ if (layer.status === "error") return `layer_${layer.layer}_error`;
1392
+ if (layer.status === "fail") return `layer_${layer.layer}_fail`;
1393
+ if (layer.status === "timeout") return `layer_${layer.layer}_timeout`;
1394
+ return `layer_${layer.layer}_${layer.status}`;
1395
+ }
1396
+
1527
1397
  // src/rl/sim-fidelity.ts
1528
1398
  var ABSENT_CATEGORY = "(absent)";
1529
1399
  var DEFAULT_MIN_N_PER_FEATURE = 20;
@@ -1733,6 +1603,136 @@ function easyModeCheck(simulated, production, opts = {}) {
1733
1603
  const gap = simPassRate - prodPassRate;
1734
1604
  return { simPassRate, prodPassRate, gap, inflated: gap > tolerance };
1735
1605
  }
1606
+
1607
+ // src/rl/tournament.ts
1608
+ function fitBradleyTerry(outcomes, opts = {}) {
1609
+ const tol = opts.tolerance ?? 1e-6;
1610
+ const maxIter = opts.maxIterations ?? 256;
1611
+ const smoothing = opts.smoothing ?? 0.1;
1612
+ const candidates = /* @__PURE__ */ new Set();
1613
+ for (const o of outcomes) {
1614
+ candidates.add(o.winner);
1615
+ candidates.add(o.loser);
1616
+ }
1617
+ const ids = [...candidates].sort();
1618
+ const idx = new Map(ids.map((id, i) => [id, i]));
1619
+ const n = ids.length;
1620
+ if (n === 0) return { ratings: [], iterations: 0, finalDelta: 0, converged: true };
1621
+ if (n === 1) {
1622
+ return {
1623
+ ratings: [{ candidateId: ids[0], strength: 1, logStrength: 0, n: 0, wins: 0 }],
1624
+ iterations: 0,
1625
+ finalDelta: 0,
1626
+ converged: true
1627
+ };
1628
+ }
1629
+ const W = Array.from({ length: n }, () => new Array(n).fill(0));
1630
+ const N = Array.from({ length: n }, () => new Array(n).fill(0));
1631
+ for (const o of outcomes) {
1632
+ const i = idx.get(o.winner);
1633
+ const j = idx.get(o.loser);
1634
+ const w = o.weight ?? 1;
1635
+ if (o.draw) {
1636
+ W[i][j] += 0.5 * w;
1637
+ W[j][i] += 0.5 * w;
1638
+ } else {
1639
+ W[i][j] += w;
1640
+ }
1641
+ N[i][j] += w;
1642
+ N[j][i] += w;
1643
+ }
1644
+ const winsTotal = new Array(n).fill(0);
1645
+ for (let i = 0; i < n; i++) {
1646
+ for (let j = 0; j < n; j++) winsTotal[i] += W[i][j];
1647
+ winsTotal[i] += smoothing;
1648
+ }
1649
+ const compsTotal = new Array(n).fill(0);
1650
+ for (let i = 0; i < n; i++) {
1651
+ for (let j = 0; j < n; j++) compsTotal[i] += N[i][j];
1652
+ }
1653
+ let theta = new Array(n).fill(1);
1654
+ let iter = 0;
1655
+ let delta = Infinity;
1656
+ for (; iter < maxIter; iter++) {
1657
+ const newTheta = new Array(n);
1658
+ for (let i = 0; i < n; i++) {
1659
+ let denom = 0;
1660
+ for (let j = 0; j < n; j++) {
1661
+ if (j === i) continue;
1662
+ if (N[i][j] === 0) continue;
1663
+ denom += N[i][j] / (theta[i] + theta[j]);
1664
+ }
1665
+ newTheta[i] = denom === 0 ? theta[i] : winsTotal[i] / denom;
1666
+ }
1667
+ let logSum = 0;
1668
+ for (let i = 0; i < n; i++) logSum += Math.log(Math.max(1e-300, newTheta[i]));
1669
+ const norm = Math.exp(logSum / n);
1670
+ for (let i = 0; i < n; i++) newTheta[i] = newTheta[i] / norm;
1671
+ delta = 0;
1672
+ for (let i = 0; i < n; i++) {
1673
+ const d = Math.abs(newTheta[i] - theta[i]) / Math.max(1e-12, theta[i]);
1674
+ if (d > delta) delta = d;
1675
+ }
1676
+ theta = newTheta;
1677
+ if (delta < tol) break;
1678
+ }
1679
+ const minLog = Math.min(...theta.map((t) => Math.log(Math.max(1e-300, t))));
1680
+ const ratings = ids.map((id, i) => ({
1681
+ candidateId: id,
1682
+ strength: theta[i],
1683
+ logStrength: Math.log(Math.max(1e-300, theta[i])) - minLog,
1684
+ n: compsTotal[i],
1685
+ wins: winsTotal[i] - smoothing
1686
+ }));
1687
+ return {
1688
+ ratings: ratings.sort((a, b) => b.strength - a.strength),
1689
+ iterations: iter,
1690
+ finalDelta: delta,
1691
+ converged: delta < tol
1692
+ };
1693
+ }
1694
+ function applyEloUpdate(ratings, outcome, opts = {}) {
1695
+ const defaultRating = opts.defaultRating ?? 1500;
1696
+ const k = opts.kFactor ?? 32;
1697
+ const rW = ratings.get(outcome.winner) ?? defaultRating;
1698
+ const rL = ratings.get(outcome.loser) ?? defaultRating;
1699
+ const expectedW = 1 / (1 + 10 ** ((rL - rW) / 400));
1700
+ const scoreW = outcome.draw ? 0.5 : 1;
1701
+ const scoreL = outcome.draw ? 0.5 : 0;
1702
+ const w = outcome.weight ?? 1;
1703
+ const winnerDelta = k * w * (scoreW - expectedW);
1704
+ const loserDelta = k * w * (scoreL - (1 - expectedW));
1705
+ ratings.set(outcome.winner, rW + winnerDelta);
1706
+ ratings.set(outcome.loser, rL + loserDelta);
1707
+ return { winnerDelta, loserDelta };
1708
+ }
1709
+ function buildPairwiseFromCampaign(input) {
1710
+ const drawMargin = input.drawMargin ?? 0;
1711
+ const byKey = /* @__PURE__ */ new Map();
1712
+ for (const r of input.runs) {
1713
+ const arr = byKey.get(r.matchKey) ?? [];
1714
+ arr.push({ candidateId: r.candidateId, score: r.score });
1715
+ byKey.set(r.matchKey, arr);
1716
+ }
1717
+ const outcomes = [];
1718
+ for (const arr of byKey.values()) {
1719
+ for (let i = 0; i < arr.length; i++) {
1720
+ for (let j = i + 1; j < arr.length; j++) {
1721
+ const a = arr[i];
1722
+ const b = arr[j];
1723
+ if (a.candidateId === b.candidateId) continue;
1724
+ const margin = Math.abs(a.score - b.score);
1725
+ if (margin <= drawMargin) {
1726
+ outcomes.push({ winner: a.candidateId, loser: b.candidateId, draw: true, weight: 1 });
1727
+ } else {
1728
+ const [winner, loser] = a.score > b.score ? [a, b] : [b, a];
1729
+ outcomes.push({ winner: winner.candidateId, loser: loser.candidateId, weight: margin });
1730
+ }
1731
+ }
1732
+ }
1733
+ }
1734
+ return outcomes;
1735
+ }
1736
1736
  export {
1737
1737
  ABSENT_CATEGORY,
1738
1738
  DEFAULT_MIN_N_PER_FEATURE,