@tangle-network/agent-eval 0.180.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. package/CHANGELOG.md +60 -0
  2. package/README.md +119 -159
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +3 -3
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +4 -7
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +10 -9
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  38. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  39. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  40. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  41. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  42. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  43. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  44. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  45. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  46. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  47. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  48. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  49. package/dist/experiment/index.d.ts +27 -477
  50. package/dist/experiment/index.d.ts.map +1 -1
  51. package/dist/experiment/index.js +95 -559
  52. package/dist/experiment/index.js.map +1 -1
  53. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  54. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  55. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  56. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  57. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  58. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  59. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  60. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  61. package/dist/hosted/index.d.ts +2 -2
  62. package/dist/hosted/index.d.ts.map +1 -1
  63. package/dist/hosted/index.js +1 -1
  64. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  65. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  66. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  67. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  68. package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
  69. package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  70. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  71. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  72. package/dist/index.d.ts +28 -28
  73. package/dist/index.js +24 -15
  74. package/dist/index.js.map +1 -1
  75. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  76. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  77. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  78. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  79. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  80. package/dist/journal-Cs9f7385.js.map +1 -0
  81. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  82. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  83. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  84. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  85. package/dist/ledger-core/index.d.ts +1 -1
  86. package/dist/ledger-core/index.js +1 -1
  87. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  88. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  89. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  90. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  91. package/dist/meta-eval/index.d.ts +138 -7
  92. package/dist/meta-eval/index.d.ts.map +1 -1
  93. package/dist/meta-eval/index.js +245 -97
  94. package/dist/meta-eval/index.js.map +1 -1
  95. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  96. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  97. package/dist/multishot/golden/index.d.ts +1 -1
  98. package/dist/multishot/index.d.ts +2 -2
  99. package/dist/openapi.json +1 -1
  100. package/dist/outcome-store-BXlkwMPR.js +131 -0
  101. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  102. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  103. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  104. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  105. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  106. package/dist/pipelines/index.js +1 -1
  107. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  108. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  109. package/dist/profile-cell.d.ts +1 -1
  110. package/dist/profile-cell.js +1 -1
  111. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  112. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  113. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  114. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  115. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  116. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  117. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  118. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  119. package/dist/reporting.d.ts +4 -4
  120. package/dist/reporting.js +3 -3
  121. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  122. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  123. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  124. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  125. package/dist/rl.d.ts +53 -99
  126. package/dist/rl.d.ts.map +1 -1
  127. package/dist/rl.js +182 -169
  128. package/dist/rl.js.map +1 -1
  129. package/dist/rollout/index.d.ts +1 -1
  130. package/dist/rollout/index.js +2 -2
  131. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  132. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  133. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  134. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  135. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  136. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  137. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  138. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  139. package/dist/run-record-Br-Yzt_k.js +464 -0
  140. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  141. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  142. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  143. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  144. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  145. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  146. package/dist/sequential-DAsyV2T9.js.map +1 -0
  147. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  148. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  149. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  150. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  151. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  152. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  153. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  154. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  155. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  156. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  157. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  158. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  159. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  160. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  161. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  162. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  163. package/dist/trace-repair/index.d.ts +2 -2
  164. package/dist/traces.d.ts +6 -6
  165. package/dist/traces.js +1 -1
  166. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  167. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  168. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  169. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  170. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  171. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  172. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  173. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  174. package/dist/wire/index.d.ts +2 -2
  175. package/docs/adapters-observability.md +14 -0
  176. package/docs/campaign-proposers.md +86 -128
  177. package/docs/charter.md +108 -112
  178. package/docs/concepts.md +157 -69
  179. package/docs/design/mlbenchmarks-book-review.md +440 -0
  180. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  181. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  182. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  183. package/docs/design/self-improvement-evidence-audit.md +263 -0
  184. package/docs/design.md +2 -1
  185. package/docs/eval-surface-map.md +95 -42
  186. package/docs/evaluation-integrity.md +220 -0
  187. package/docs/experiment.md +111 -55
  188. package/docs/feature-guide.md +5 -6
  189. package/docs/hosted-ingest-spec.md +4 -11
  190. package/docs/insight-report.md +187 -455
  191. package/docs/outcome-validity.md +182 -0
  192. package/docs/product-eval-adoption.md +1 -2
  193. package/docs/research-report-methodology.md +7 -7
  194. package/docs/search-history-receipts.md +8 -0
  195. package/docs/statistical-evidence.md +129 -0
  196. package/docs/verdicts.md +76 -49
  197. package/package.json +1 -1
  198. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  199. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  200. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  201. package/dist/client-BlLY6o2w.js.map +0 -1
  202. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  203. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  204. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  205. package/dist/index-BxWvILU8.d.ts.map +0 -1
  206. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  207. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  208. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  209. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  210. package/dist/outcome-store-ChBKlTd_.js +0 -75
  211. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  212. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  213. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  214. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  215. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  216. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  217. package/dist/run-record-CR63CpHK.js +0 -216
  218. package/dist/run-record-CR63CpHK.js.map +0 -1
  219. package/dist/sequential-B5gXgcyp.js.map +0 -1
  220. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  221. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
package/dist/rl.js CHANGED
@@ -1,22 +1,22 @@
1
1
  import { s as ValidationError } from "./errors-Dngq5h35.js";
2
2
  import { i as compareCodeUnits } from "./canonical-DPyQ_rpt.js";
3
- import { i as makeRng } from "./internal-BMFSR8Ns.js";
3
+ import { a as medianInPlace } from "./internal-BMFSR8Ns.js";
4
4
  import { t as mulberry32 } from "./random-Dn5fPWkt.js";
5
- import { o as benjaminiHochberg } from "./power-and-mde-B8F2RdcD.js";
6
5
  import { l as wilcoxonSignedRank } from "./paired-arms-D4aeIHUy.js";
6
+ import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
7
+ import { a as campaignCellToRunRecord, s as decidePairedPromotion } from "./run-record-Br-Yzt_k.js";
7
8
  import { r as observedSplitScore, s as trainingScore } from "./reward-nw2xZGZG.js";
8
- import { s as runTaskScore } from "./run-record-DQpSf7t-.js";
9
- import { a as campaignCellToRunRecord } from "./run-record-CR63CpHK.js";
9
+ import { s as runTaskScore } from "./run-record-DualPTn2.js";
10
10
  import { i as filterDeterministicallyRewarded, n as extractVerifiableReward, r as extractVerifiableRewardsFromRecords, t as detectRewardHacking } from "./reward-hacking-D0XwhVWE.js";
11
11
  import { n as InMemoryTraceStore } from "./store-DNe_Uv1Q.js";
12
- import { t as runEvalCampaign } from "./eval-campaign-Cs-7MiCs.js";
12
+ import { t as runEvalCampaign } from "./eval-campaign-aYdtjtJR.js";
13
13
  import { s as assertRewardGate } from "./schema-C1aaAxTf.js";
14
14
  import { t as isSplitEligible } from "./exporters-Df7TgHFv.js";
15
- import { t as mintRolloutRows } from "./mint-Cc1_zwRQ.js";
15
+ import { t as mintRolloutRows } from "./mint-ySIIkKlV.js";
16
16
  import { t as evaluateInterimReleaseConfidence } from "./sequential-CzK5DarL.js";
17
- import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-2D5Gw9z9.js";
17
+ import { n as assertUniqueObservationIds, s as validateOutcomeMetricSpecifications, t as rubricPredictiveValidity } from "./rubric-predictive-validity-CCK-1B7w.js";
18
18
  import { n as thompsonCurriculum, r as varianceBasedCurriculum, t as observationsFromRunRecords } from "./active-curriculum-OjIWgrUJ.js";
19
- import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "./outcome-store-ChBKlTd_.js";
19
+ import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "./outcome-store-BXlkwMPR.js";
20
20
  import { createHash } from "node:crypto";
21
21
  import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
22
22
  import { dirname, join } from "node:path";
@@ -24,29 +24,10 @@ import { dirname, join } from "node:path";
24
24
  /**
25
25
  * Sample-efficient adaptation evaluation.
26
26
  *
27
- * For foundation-model-based agents, the load-bearing capability isn't
28
- * raw end-state performance it's *how fast the agent reaches that
29
- * performance from cold start*. The same model with a worse prompt that
30
- * adapts in 5 demonstrations beats the same model with a better prompt
31
- * that needs 50. Standard meta-learning eval (Finn et al., MAML, RL² lit)
32
- * reports an *adaptation curve*: score after k=0, 1, 2, 4, 8, 16, …
33
- * in-context examples or fine-tune steps.
34
- *
35
- * This module ships:
36
- *
37
- * 1. `runAdaptationCurve` — given a runner that takes k demonstrations
38
- * and returns a score, produce the (k, score) curve.
39
- * 2. `compareAdaptationCurves` — paired comparison across two policies.
40
- * Returns per-k delta with bootstrap CIs and an "area-under-curve"
41
- * summary statistic.
42
- * 3. `firstPassK` — for pass/fail evaluation, the minimum k at which
43
- * the policy reliably passes (≥ pass-rate threshold over reps).
44
- *
45
- * Use cases:
46
- * - Compare two prompt designs that have similar end-state performance
47
- * but different in-context efficiency.
48
- * - Decide between fine-tuning and prompting based on adaptation cost.
49
- * - Detect when a policy "memorizes" k=0 inputs vs. genuinely adapts.
27
+ * An adaptation curve records scores after k demonstrations or training steps.
28
+ * Comparison pairs the same scenarios and resamples their whole curves.
29
+ * The normalized area summarizes performance over the observed k range.
30
+ * A first-pass k is descriptive and carries no separate reliability claim.
50
31
  */
51
32
  async function runAdaptationCurve(opts) {
52
33
  const ks = opts.ks ?? [
@@ -59,6 +40,10 @@ async function runAdaptationCurve(opts) {
59
40
  ];
60
41
  const reps = opts.reps ?? 3;
61
42
  const passThreshold = opts.passThreshold ?? .5;
43
+ assertKs(ks, "runAdaptationCurve");
44
+ assertScenarioIds(opts.scenarios, "runAdaptationCurve");
45
+ if (!Number.isInteger(reps) || reps < 1) throw new ValidationError("runAdaptationCurve: reps must be a positive integer");
46
+ if (!Number.isFinite(passThreshold) || passThreshold < 0 || passThreshold > 1) throw new ValidationError("runAdaptationCurve: passThreshold must be in [0,1]");
62
47
  const sortedKs = [...ks].sort((a, b) => a - b);
63
48
  const points = [];
64
49
  for (const k of sortedKs) {
@@ -67,7 +52,7 @@ async function runAdaptationCurve(opts) {
67
52
  let totalPasses = 0;
68
53
  let totalAttempts = 0;
69
54
  for (const scenario of opts.scenarios) {
70
- const sid = scenario.scenarioId ?? `scenario-${opts.scenarios.indexOf(scenario)}`;
55
+ const sid = scenario.scenarioId;
71
56
  const scores = [];
72
57
  let passes = 0;
73
58
  for (let r = 0; r < reps; r++) {
@@ -76,6 +61,7 @@ async function runAdaptationCurve(opts) {
76
61
  k,
77
62
  rep: r
78
63
  });
64
+ assertScore(score, `runAdaptationCurve: scenario '${sid}', k=${k}, rep=${r}`);
79
65
  scores.push(score);
80
66
  if (score >= passThreshold) passes++;
81
67
  allScores.push(score);
@@ -118,69 +104,106 @@ async function runAdaptationCurve(opts) {
118
104
  };
119
105
  }
120
106
  /**
121
- * Paired comparison of two adaptation curves. Per-k deltas with 95%
122
- * bootstrap CIs (constructed from each curve's `perScenario` per-k means
123
- * the bootstrap unit is the scenario, not the rep).
107
+ * Compare identical scenario cohorts on identical k grids, paired by scenarioId.
108
+ * Missing pairs, duplicate identities, and cohort changes across k are refused.
109
+ * The bootstrap resamples whole scenarios, preserving dependence across k.
110
+ * Per-k intervals describe the curve; shared paired area decisions determine the verdict.
111
+ * Bootstrap eligibility is necessary but does not establish scenario independence.
124
112
  */
125
113
  function compareAdaptationCurves(a, b, opts = {}) {
126
- const conf = opts.confidence ?? .95;
127
- const resamples = opts.bootstrapResamples ?? 500;
128
- const rng = makeRng(opts.seed, a.points.flatMap((point) => point.perScenario.map((cell) => cell.meanScore)), b.points.flatMap((point) => point.perScenario.map((cell) => cell.meanScore)));
129
- const perK = [];
130
- for (const ap of a.points) {
131
- const bp = b.points.find((p) => p.k === ap.k);
132
- if (!bp) continue;
133
- const aMeans = ap.perScenario.map((s) => s.meanScore);
134
- const bMeans = bp.perScenario.map((s) => s.meanScore);
135
- const aCi = bootstrapMeanCi(aMeans, resamples, conf, rng);
136
- const bCi = bootstrapMeanCi(bMeans, resamples, conf, rng);
137
- perK.push({
138
- k: ap.k,
139
- deltaMean: ap.meanScore - bp.meanScore,
140
- aLow: aCi.low,
141
- aHigh: aCi.high,
142
- bLow: bCi.low,
143
- bHigh: bCi.high
144
- });
145
- }
146
- const areaDelta = a.adaptationArea - b.adaptationArea;
147
- const firstPassKDelta = a.firstPassK !== null && b.firstPassK !== null ? b.firstPassK - a.firstPassK : null;
148
- const meanDelta = perK.reduce((s, p) => s + p.deltaMean, 0) / Math.max(1, perK.length);
114
+ const confidence = opts.confidence ?? .95;
115
+ const resamples = opts.bootstrapResamples ?? 2e3;
116
+ const minimumEffect = opts.minimumEffect ?? 0;
117
+ if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new ValidationError("compareAdaptationCurves: confidence must be in (0,1)");
118
+ if (!Number.isInteger(resamples) || resamples < 1) throw new ValidationError("compareAdaptationCurves: bootstrapResamples must be a positive integer");
119
+ if (!Number.isFinite(minimumEffect) || minimumEffect < 0 || minimumEffect > 1) throw new ValidationError("compareAdaptationCurves: minimumEffect must be in [0,1]");
120
+ const aPoints = indexCurve(a, "A");
121
+ const bPoints = indexCurve(b, "B");
122
+ const ks = [...aPoints.keys()].sort((x, y) => x - y);
123
+ const missingInA = [...bPoints.keys()].filter((k) => !aPoints.has(k));
124
+ const missingInB = ks.filter((k) => !bPoints.has(k));
125
+ if (missingInA.length > 0 || missingInB.length > 0) throw new ValidationError(`compareAdaptationCurves: k grids differ; missing in A=[${missingInA}], missing in B=[${missingInB}]`);
126
+ const scenarioIds = [...aPoints.get(ks[0]).keys()].sort();
127
+ const expectedIds = new Set(scenarioIds);
128
+ for (const [arm, points] of [["A", aPoints], ["B", bPoints]]) for (const [k, cells] of points) {
129
+ const missing = scenarioIds.filter((id) => !cells.has(id));
130
+ const extra = [...cells.keys()].filter((id) => !expectedIds.has(id));
131
+ if (missing.length > 0 || extra.length > 0) throw new ValidationError(`compareAdaptationCurves: scenario pairs differ in ${arm} at k=${k}; missing=[${missing}], unexpected=[${extra}]`);
132
+ }
133
+ const bootstrapOptions = {
134
+ confidence,
135
+ resamples,
136
+ statistic: "mean",
137
+ seed: opts.seed
138
+ };
139
+ const perK = ks.map((k) => ({
140
+ k,
141
+ delta: pairedBootstrap(scenarioIds.map((id) => bPoints.get(k).get(id)), scenarioIds.map((id) => aPoints.get(k).get(id)), bootstrapOptions)
142
+ }));
143
+ const aAreas = scenarioIds.map((id) => scenarioArea(ks, aPoints, id));
144
+ const bAreas = scenarioIds.map((id) => scenarioArea(ks, bPoints, id));
145
+ const decisionOptions = {
146
+ ...bootstrapOptions,
147
+ threshold: minimumEffect
148
+ };
149
+ const aImprovement = decidePairedPromotion(bAreas, aAreas, decisionOptions);
150
+ const bImprovement = decidePairedPromotion(aAreas, bAreas, decisionOptions);
151
+ const areaDelta = aImprovement.bootstrap ?? pairedBootstrap(bAreas, aAreas, bootstrapOptions);
149
152
  let verdict;
150
- if (Math.abs(meanDelta) < .02 && Math.abs(areaDelta) < .02) verdict = "similar";
151
- else if (meanDelta > 0 && areaDelta > 0) verdict = "a_better";
152
- else if (meanDelta < 0 && areaDelta < 0) verdict = "b_better";
153
- else verdict = "similar";
154
- const rationale = `mean per-k delta=${meanDelta.toFixed(3)}, area delta=${areaDelta.toFixed(3)}` + (firstPassKDelta !== null ? `, first-pass-k delta=${firstPassKDelta}` : "");
153
+ if (!aImprovement.sufficient || ks.length < 2) verdict = "insufficient_evidence";
154
+ else if (aImprovement.promote) verdict = "a_better";
155
+ else if (bImprovement.promote) verdict = "b_better";
156
+ else verdict = "inconclusive";
157
+ const rationale = `paired scenarios=${scenarioIds.length}, area delta=${areaDelta.mean.toFixed(3)}, ${confidence * 100}% ${aImprovement.statistic} interval=[${aImprovement.low.toFixed(3)}, ${aImprovement.high.toFixed(3)}], minimum effect=${minimumEffect}; ${verdict}`;
155
158
  return {
156
159
  perK,
157
160
  areaDelta,
158
- firstPassKDelta,
161
+ aImprovement,
162
+ bImprovement,
163
+ scenarioIds,
159
164
  verdict,
160
165
  rationale
161
166
  };
162
167
  }
163
- /** First k at which the curve's per-scenario pass rate reliably hits the threshold. */
168
+ /** First observed k whose pass rate reaches the threshold; this is a descriptive summary. */
164
169
  function firstPassK(curve, threshold = .5) {
165
170
  return curve.points.find((p) => p.passRate >= threshold)?.k ?? null;
166
171
  }
167
- function bootstrapMeanCi(xs, resamples, confidence, rng) {
168
- if (xs.length < 2) return {
169
- low: xs[0] ?? 0,
170
- high: xs[0] ?? 0
171
- };
172
- const samples = new Array(resamples);
173
- for (let b = 0; b < resamples; b++) {
174
- let sum = 0;
175
- for (let i = 0; i < xs.length; i++) sum += xs[Math.floor(rng() * xs.length)];
176
- samples[b] = sum / xs.length;
172
+ function assertKs(ks, where) {
173
+ if (ks.length === 0 || ks.some((k) => !Number.isInteger(k) || k < 0)) throw new ValidationError(`${where}: ks must contain nonnegative integers`);
174
+ if (new Set(ks).size !== ks.length) throw new ValidationError(`${where}: duplicate k values`);
175
+ }
176
+ function assertScenarioIds(cells, where) {
177
+ if (cells.length === 0 || cells.some((cell) => !cell.scenarioId?.trim())) throw new ValidationError(`${where}: scenarios must have explicit nonempty scenarioId values`);
178
+ const seen = /* @__PURE__ */ new Set();
179
+ for (const { scenarioId } of cells) {
180
+ if (seen.has(scenarioId)) throw new ValidationError(`${where}: duplicate scenarioId '${scenarioId}'`);
181
+ seen.add(scenarioId);
182
+ }
183
+ }
184
+ function assertScore(score, where) {
185
+ if (!Number.isFinite(score) || score < 0 || score > 1) throw new ValidationError(`${where}: score must be finite and in [0,1], got ${score}`);
186
+ }
187
+ function indexCurve(curve, arm) {
188
+ const where = `compareAdaptationCurves: ${arm}`;
189
+ assertKs(curve.points.map((point) => point.k), where);
190
+ return new Map(curve.points.map((point) => {
191
+ assertScenarioIds(point.perScenario, `${where} at k=${point.k}`);
192
+ return [point.k, new Map(point.perScenario.map((cell) => {
193
+ assertScore(cell.meanScore, `${where}: '${cell.scenarioId}' at k=${point.k}`);
194
+ return [cell.scenarioId, cell.meanScore];
195
+ }))];
196
+ }));
197
+ }
198
+ function scenarioArea(ks, points, id) {
199
+ let area = 0;
200
+ for (let i = 1; i < ks.length; i++) {
201
+ const left = ks[i - 1];
202
+ const right = ks[i];
203
+ area += (points.get(left).get(id) + points.get(right).get(id)) * (right - left) / 2;
177
204
  }
178
- samples.sort((a, b) => a - b);
179
- const alpha = 1 - confidence;
180
- return {
181
- low: samples[Math.floor(alpha / 2 * resamples)],
182
- high: samples[Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)]
183
- };
205
+ const maxK = ks[ks.length - 1];
206
+ return maxK === 0 ? 0 : area / maxK;
184
207
  }
185
208
  //#endregion
186
209
  //#region src/rl/compute-curves.ts
@@ -311,81 +334,58 @@ function fitLogSlope(points) {
311
334
  /**
312
335
  * Contamination probe — held-out perturbation tests.
313
336
  *
314
- * The bug class: once a benchmark scenario set is published, models train
315
- * on it, and your scores become invalid. SWE-Bench-Verified, GPQA, and
316
- * MMLU-Pro all exist because their predecessors got contaminated within
317
- * months. The right defense is to keep a held-out *perturbed* version of
318
- * every scenario same task, slightly different surface — and check
319
- * whether scores diverge significantly. Genuine capability transfers; rote
320
- * memorization doesn't.
321
- *
322
- * This module ships the probe contract:
323
- *
324
- * 1. A `ScenarioPerturbation` strategy type — function that produces a
325
- * perturbed scenario from an original.
326
- * 2. `runContaminationProbe({ originals, perturbed, scoreFn })` — runs
327
- * both halves and reports per-scenario score divergence + a global
328
- * contamination verdict via paired Wilcoxon.
329
- * 3. Several stock perturbations: `renameVariables`, `shuffleOrder`,
330
- * `paraphrasePrompt`, `injectIrrelevantClause`. Each preserves the
331
- * task's structural difficulty while breaking surface memorization.
332
- *
333
- * The verdict is conservative: if the perturbed-vs-original score
334
- * difference is statistically significant (BH-adjusted p < 0.05) AND
335
- * the median drop is > 5 percentage points, we flag *contamination
336
- * suspected*. False positives are possible (the perturbation might
337
- * actually be harder); the default is to flag for review, not to
338
- * autoreject.
337
+ * Score each scenario and its perturbation, then test the paired differences.
338
+ * A significant global Wilcoxon result plus a worthwhile median drop flags
339
+ * contamination for review. Perturbations may change difficulty, so the result
340
+ * does not identify contamination as the cause. Per-item differences have no
341
+ * calibrated sampling null and carry no p-values or q-values.
339
342
  */
340
343
  async function runContaminationProbe(input, opts = {}) {
341
- const fdr = opts.fdr ?? .05;
344
+ const alpha = opts.alpha ?? .05;
342
345
  const minMedianDrop = opts.minMedianDrop ?? .05;
343
346
  const floor = opts.scoreFloor ?? 0;
347
+ if (!Number.isFinite(alpha) || alpha <= 0 || alpha >= 1) throw new ValidationError("runContaminationProbe: alpha must be in (0,1)");
348
+ for (const [name, value] of [["scoreFloor", floor], ["minMedianDrop", minMedianDrop]]) if (!Number.isFinite(value) || value < 0 || value > 1) throw new ValidationError(`runContaminationProbe: ${name} must be in [0,1]`);
349
+ const ids = input.originals.map(input.scenarioId);
350
+ if (ids.some((id) => !id?.trim()) || new Set(ids).size !== ids.length) throw new ValidationError("runContaminationProbe: original scenario IDs must be nonempty and unique");
344
351
  if (!input.perturbed && !input.perturbation) throw new ValidationError("runContaminationProbe: must supply either `perturbed` or `perturbation`.");
345
352
  const perturbed = input.perturbed ?? await Promise.all(input.originals.map((s) => input.perturbation.apply(s)));
346
353
  if (perturbed.length !== input.originals.length) throw new ValidationError(`runContaminationProbe: perturbed length ${perturbed.length} ≠ originals ${input.originals.length}`);
347
354
  const origScores = await Promise.all(input.originals.map((s) => input.scoreFn(s)));
348
355
  const pertScores = await Promise.all(perturbed.map((s) => input.scoreFn(s)));
349
- const perScenario = input.originals.map((s, i) => ({
350
- scenarioId: input.scenarioId(s),
356
+ for (const score of [...origScores, ...pertScores]) if (!Number.isFinite(score) || score < 0 || score > 1) throw new ValidationError(`runContaminationProbe: scores must be finite and in [0,1], got ${score}`);
357
+ const perScenario = ids.map((scenarioId, i) => ({
358
+ scenarioId,
351
359
  originalScore: origScores[i],
352
360
  perturbedScore: pertScores[i],
353
- delta: pertScores[i] - origScores[i],
354
- qValue: NaN
361
+ delta: pertScores[i] - origScores[i]
355
362
  }));
356
363
  const valid = perScenario.filter((p) => p.originalScore >= floor && p.perturbedScore >= floor);
364
+ const excludedScenarioIds = perScenario.filter((p) => p.originalScore < floor || p.perturbedScore < floor).map((p) => p.scenarioId);
365
+ const deltas = valid.map((p) => p.delta);
366
+ const medianDelta = deltas.length === 0 ? null : medianInPlace(deltas);
367
+ const meanDelta = deltas.length === 0 ? null : deltas.reduce((sum, d) => sum + d, 0) / deltas.length;
357
368
  if (valid.length < 4) return {
358
369
  perScenario,
359
- pairedTest: {
360
- w: 0,
361
- p: 1
362
- },
363
- medianDelta: 0,
364
- meanDelta: 0,
370
+ pairedTest: null,
371
+ medianDelta,
372
+ meanDelta,
365
373
  contaminationSuspected: false,
366
374
  reason: `insufficient valid scenarios (n=${valid.length}, need ≥ 4)`,
367
- n: valid.length
375
+ n: valid.length,
376
+ excludedScenarioIds
368
377
  };
369
378
  const pairedTest = wilcoxonSignedRank(valid.map((p) => p.originalScore), valid.map((p) => p.perturbedScore));
370
- const deltas = valid.map((p) => p.delta);
371
- const sortedDeltas = [...deltas].sort((a, b) => a - b);
372
- const median = sortedDeltas[Math.floor(sortedDeltas.length / 2)];
373
- const mean = deltas.reduce((s, d) => s + d, 0) / deltas.length;
374
- const { qValues } = benjaminiHochberg(valid.map((p) => Math.min(1, Math.max(1e-6, 1 - Math.abs(p.delta) / 1))), fdr);
375
- for (let i = 0; i < valid.length; i++) {
376
- const v = valid[i];
377
- const idx = perScenario.findIndex((p) => p.scenarioId === v.scenarioId);
378
- if (idx >= 0) perScenario[idx].qValue = qValues[i];
379
- }
380
- const contaminationSuspected = pairedTest.p < fdr && median <= -minMedianDrop;
379
+ const contaminationSuspected = pairedTest.p < alpha && medianDelta <= -minMedianDrop;
381
380
  return {
382
381
  perScenario,
383
382
  pairedTest,
384
- medianDelta: median,
385
- meanDelta: mean,
383
+ medianDelta,
384
+ meanDelta,
386
385
  contaminationSuspected,
387
- reason: contaminationSuspected ? `paired p=${pairedTest.p.toFixed(4)} < ${fdr} and median drop ${median.toFixed(4)} ≥ ${minMedianDrop}` : pairedTest.p >= fdr ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})` : `significant but small effect (median delta ${median.toFixed(4)})`,
388
- n: valid.length
386
+ reason: contaminationSuspected ? `paired p=${pairedTest.p.toFixed(4)} < ${alpha} and median drop ${(-medianDelta).toFixed(4)} ≥ ${minMedianDrop}` : pairedTest.p >= alpha ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})` : `significant but no qualifying drop (median delta ${medianDelta.toFixed(4)})`,
387
+ n: valid.length,
388
+ excludedScenarioIds
389
389
  };
390
390
  }
391
391
  /**
@@ -1474,14 +1474,21 @@ function clamp(x, lo, hi) {
1474
1474
  //#endregion
1475
1475
  //#region src/rl/predictive-validity-researcher.ts
1476
1476
  /**
1477
- * Concrete `Researcher` driven by `rubricPredictiveValidity`. The brain:
1478
- * rubrics that don't predict deployment outcomes don't earn weight.
1477
+ * Proposes rubric experiments against one declared outcome.
1478
+ * A correlation supports a hypothesis; the caller must measure any resulting change.
1479
1479
  */
1480
1480
  var PredictiveValidityResearcher = class {
1481
1481
  opts;
1482
1482
  lastReport = null;
1483
1483
  constructor(opts) {
1484
- this.opts = opts;
1484
+ validateOutcomeMetricSpecifications([opts.targetOutcome]);
1485
+ if (opts.rubrics !== void 0) assertUniqueObservationIds(opts.rubrics, "rubric");
1486
+ if (opts.failureThreshold !== void 0 && !Number.isFinite(opts.failureThreshold)) throw new Error("failureThreshold must be finite");
1487
+ this.opts = {
1488
+ ...opts,
1489
+ targetOutcome: { ...opts.targetOutcome },
1490
+ rubrics: opts.rubrics === void 0 ? void 0 : [...opts.rubrics]
1491
+ };
1485
1492
  }
1486
1493
  async inspectFailures(runs) {
1487
1494
  const threshold = this.opts.failureThreshold ?? .5;
@@ -1521,35 +1528,34 @@ var PredictiveValidityResearcher = class {
1521
1528
  payload: { directive: "researcher.collect-more-outcomes" },
1522
1529
  rationale: "predictive-validity researcher has no prior report; cannot recommend rubric reweighting until at least one report exists"
1523
1530
  }];
1524
- const decorativeThreshold = this.opts.decorativeThreshold ?? .4;
1525
1531
  const changes = [];
1526
- for (const ranking of this.lastReport.ranked) {
1527
- if (ranking.verdict === "load_bearing") continue;
1528
- if (Math.abs(ranking.spearman) >= decorativeThreshold) continue;
1529
- changes.push({
1530
- kind: "reviewer_prompt",
1531
- payload: {
1532
- rubric: ranking.rubric,
1533
- action: "down-weight",
1534
- spearman: ranking.spearman,
1535
- bestOutcome: ranking.bestOutcome
1536
- },
1537
- rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (decorative); recommend down-weighting`,
1538
- expectedDelta: -Math.max(0, .05 - Math.abs(ranking.spearman))
1539
- });
1540
- }
1541
- for (const ranking of this.lastReport.ranked.slice(0, 1)) {
1542
- if (ranking.verdict !== "load_bearing") continue;
1532
+ const target = { ...this.opts.targetOutcome };
1533
+ const pairs = this.lastReport.pairs.filter((pair) => pair.outcome === target.id && pair.outcomeDirection === target.direction && (this.opts.rubrics === void 0 || this.opts.rubrics.includes(pair.rubric)));
1534
+ if (pairs.length === 0) return [{
1535
+ kind: "threshold",
1536
+ payload: {
1537
+ directive: "researcher.collect-more-outcomes",
1538
+ targetOutcome: target
1539
+ },
1540
+ rationale: `no estimable rubric association with ${target.id}; collect independent outcome observations before proposing weight changes`
1541
+ }];
1542
+ for (const pair of pairs) {
1543
+ const interval = pair.alignedSpearmanCi95;
1544
+ const aligned = pair.alignedSpearman >= .4 && interval !== null && interval.lower > 0;
1545
+ const inverse = pair.alignedSpearman <= -.4 && interval !== null && interval.upper < 0;
1546
+ const action = aligned ? "test-up-weight" : inverse ? "test-reverse-or-replace" : "collect-calibration-evidence";
1543
1547
  changes.push({
1544
1548
  kind: "reviewer_prompt",
1545
1549
  payload: {
1546
- rubric: ranking.rubric,
1547
- action: "up-weight",
1548
- spearman: ranking.spearman,
1549
- bestOutcome: ranking.bestOutcome
1550
+ rubric: pair.rubric,
1551
+ action,
1552
+ targetOutcome: target,
1553
+ spearman: pair.spearman,
1554
+ alignedSpearman: pair.alignedSpearman,
1555
+ alignedSpearmanCi95: interval === null ? null : { ...interval },
1556
+ samples: pair.n
1550
1557
  },
1551
- rationale: `predictive-validity Spearman=${ranking.spearman.toFixed(3)} vs ${ranking.bestOutcome} (load-bearing); recommend up-weighting`,
1552
- expectedDelta: Math.max(0, Math.abs(ranking.spearman) - .5) * .1
1558
+ rationale: aligned ? `higher ${pair.rubric} scores associate with better ${target.id}; test increased weight on fresh evidence before adopting it` : inverse ? `higher ${pair.rubric} scores associate with worse ${target.id}; test reversal or replacement on fresh evidence` : `the association of ${pair.rubric} with desired ${target.id} does not support a direction of change; collect calibration evidence`
1553
1559
  });
1554
1560
  }
1555
1561
  return changes;
@@ -1604,11 +1610,11 @@ var PredictiveValidityResearcher = class {
1604
1610
  const report = await rubricPredictiveValidity({
1605
1611
  runs,
1606
1612
  outcomes: this.opts.outcomes,
1607
- outcomeMetrics: this.opts.outcomeMetrics,
1613
+ outcomeMetrics: [this.opts.targetOutcome],
1608
1614
  rubrics: this.opts.rubrics
1609
1615
  });
1610
- if (this.opts.onReport) await this.opts.onReport(report);
1611
- this.lastReport = report;
1616
+ if (this.opts.onReport) await this.opts.onReport(structuredClone(report));
1617
+ this.setReport(report);
1612
1618
  return report;
1613
1619
  }
1614
1620
  /**
@@ -1617,10 +1623,11 @@ var PredictiveValidityResearcher = class {
1617
1623
  * researcher's later proposals informed by it.
1618
1624
  */
1619
1625
  setReport(report) {
1620
- this.lastReport = report;
1626
+ if (report.outcomeMetrics.find((metric) => metric.id === this.opts.targetOutcome.id)?.direction !== this.opts.targetOutcome.direction) throw new Error("predictive validity report does not match the declared target outcome and direction");
1627
+ this.lastReport = structuredClone(report);
1621
1628
  }
1622
1629
  getLastReport() {
1623
- return this.lastReport;
1630
+ return this.lastReport === null ? null : structuredClone(this.lastReport);
1624
1631
  }
1625
1632
  };
1626
1633
  /** Coverage of a split that was never dealt any work. */
@@ -2044,6 +2051,12 @@ function prmTrainingPairs(stepRewardsByRun, opts = {}) {
2044
2051
  * `result.rewardSignals` to a custom RL loop.
2045
2052
  */
2046
2053
  async function runRLCampaign(opts) {
2054
+ const outcomeStore = opts.outcomeStore;
2055
+ const outcomeMetrics = opts.outcomeMetrics?.map((metric) => ({ ...metric }));
2056
+ if (outcomeStore !== void 0 || outcomeMetrics !== void 0) {
2057
+ if (outcomeStore === void 0 || outcomeMetrics === void 0) throw new Error("runRLCampaign requires outcomeStore and outcomeMetrics together");
2058
+ validateOutcomeMetricSpecifications(outcomeMetrics);
2059
+ }
2047
2060
  const splitTag = opts.splitTag ?? "search";
2048
2061
  const campaign = await runEvalCampaign({
2049
2062
  ...opts,
@@ -2080,10 +2093,10 @@ async function runRLCampaign(opts) {
2080
2093
  verifiableRewardOptions: opts.verifiableReward
2081
2094
  });
2082
2095
  let predictiveValidity = null;
2083
- if (opts.outcomeStore && opts.outcomeMetrics && opts.outcomeMetrics.length > 0) predictiveValidity = await rubricPredictiveValidity({
2096
+ if (outcomeStore && outcomeMetrics) predictiveValidity = await rubricPredictiveValidity({
2084
2097
  runs: campaign.runs,
2085
- outcomes: opts.outcomeStore,
2086
- outcomeMetrics: opts.outcomeMetrics
2098
+ outcomes: outcomeStore,
2099
+ outcomeMetrics
2087
2100
  });
2088
2101
  const trainerRows = {};
2089
2102
  if (opts.trainerExport?.dpo) trainerRows.dpo = await toDpoRows(preferences.pairs, opts.trainerExport.dpo, { lines: rolloutLines });
@@ -2215,7 +2228,7 @@ function buildSummary(args) {
2215
2228
  lines.push(`reward-hacking: ${args.rewardHacking.verdict} (${args.rewardHacking.findings.length} signals checked)`);
2216
2229
  if (args.predictiveValidity) {
2217
2230
  const top = args.predictiveValidity.ranked[0];
2218
- lines.push(`top-rubric: ${top?.rubric ?? "none"} ρ=${(top?.spearman ?? 0).toFixed(2)} (${top?.verdict ?? "no data"})`);
2231
+ lines.push(top ? `top-rubric: ${top.rubric} aligned ρ=${top.alignedSpearman.toFixed(2)} vs ${top.bestOutcome} (${top.outcomeDirection}; ${top.verdict})` : "top-rubric: none (no estimable outcome associations)");
2219
2232
  }
2220
2233
  return lines.join(" | ");
2221
2234
  }