@tangle-network/agent-eval 0.179.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (236) hide show
  1. package/CHANGELOG.md +66 -0
  2. package/README.md +119 -146
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +4 -4
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-CY6Dg5t5.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-CY6Dg5t5.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +5 -8
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +11 -10
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{default-registry-BryMEmr8.js → default-registry-aL7xUrUz.js} +2 -2
  38. package/dist/{default-registry-BryMEmr8.js.map → default-registry-aL7xUrUz.js.map} +1 -1
  39. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  40. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  41. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  42. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  43. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  44. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  45. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  46. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  47. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  48. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  49. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  50. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  51. package/dist/experiment/index.d.ts +27 -477
  52. package/dist/experiment/index.d.ts.map +1 -1
  53. package/dist/experiment/index.js +95 -559
  54. package/dist/experiment/index.js.map +1 -1
  55. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  56. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  57. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  58. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  59. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  60. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  61. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  62. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  63. package/dist/hosted/index.d.ts +2 -2
  64. package/dist/hosted/index.d.ts.map +1 -1
  65. package/dist/hosted/index.js +1 -1
  66. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  67. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  68. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  69. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  70. package/dist/{index-CbLmrWCa.d.ts → index-DNntP4ch.d.ts} +8 -8
  71. package/dist/{index-CbLmrWCa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  72. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  73. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  74. package/dist/index.d.ts +28 -28
  75. package/dist/index.js +25 -16
  76. package/dist/index.js.map +1 -1
  77. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  78. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  79. package/dist/{integrity-DsHWCebQ.js → integrity-DH5ng72x.js} +2 -2
  80. package/dist/{integrity-DsHWCebQ.js.map → integrity-DH5ng72x.js.map} +1 -1
  81. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  82. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  83. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  84. package/dist/journal-Cs9f7385.js.map +1 -0
  85. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  86. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  87. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  88. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  89. package/dist/ledger-core/index.d.ts +1 -1
  90. package/dist/ledger-core/index.js +1 -1
  91. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  92. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  93. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  94. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  95. package/dist/meta-eval/index.d.ts +138 -7
  96. package/dist/meta-eval/index.d.ts.map +1 -1
  97. package/dist/meta-eval/index.js +245 -97
  98. package/dist/meta-eval/index.js.map +1 -1
  99. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  100. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  101. package/dist/multishot/golden/index.d.ts +1 -1
  102. package/dist/multishot/index.d.ts +2 -2
  103. package/dist/openapi.json +1 -1
  104. package/dist/outcome-store-BXlkwMPR.js +131 -0
  105. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  106. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  107. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  108. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  109. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  110. package/dist/pipelines/index.js +1 -1
  111. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  112. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  113. package/dist/profile-cell.d.ts +1 -1
  114. package/dist/profile-cell.js +1 -1
  115. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  116. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  117. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  118. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  119. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  120. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  121. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  122. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  123. package/dist/{report-command-DKlXfU5r.js → report-command-V1ecVgAv.js} +27 -3
  124. package/dist/report-command-V1ecVgAv.js.map +1 -0
  125. package/dist/reporting.d.ts +4 -4
  126. package/dist/reporting.js +3 -3
  127. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  128. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  130. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  131. package/dist/rl.d.ts +53 -99
  132. package/dist/rl.d.ts.map +1 -1
  133. package/dist/rl.js +182 -169
  134. package/dist/rl.js.map +1 -1
  135. package/dist/rollout/index.d.ts +1 -1
  136. package/dist/rollout/index.js +2 -2
  137. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  138. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  139. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  140. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  141. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  142. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  143. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  144. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  145. package/dist/run-record-Br-Yzt_k.js +464 -0
  146. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  147. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  148. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  149. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  150. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  151. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  152. package/dist/sequential-DAsyV2T9.js.map +1 -0
  153. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  154. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  155. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  156. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  157. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  158. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  159. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  160. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  161. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  162. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  163. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  164. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  165. package/dist/supervisor-run/index.d.ts +4 -2
  166. package/dist/supervisor-run/index.d.ts.map +1 -1
  167. package/dist/supervisor-run/index.js +3 -3
  168. package/dist/{terminal-record-Ce9_UjRz.js → terminal-record-BtPwKTSr.js} +58 -26
  169. package/dist/terminal-record-BtPwKTSr.js.map +1 -0
  170. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  171. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  172. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  173. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  174. package/dist/trace-repair/index.d.ts +2 -2
  175. package/dist/traces.d.ts +6 -6
  176. package/dist/traces.js +1 -1
  177. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  178. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  179. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  180. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  181. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  182. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  183. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  184. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  185. package/dist/{types-vUdAx2Cj.d.ts → types-lPkDQNqJ.d.ts} +20 -2
  186. package/dist/{types-vUdAx2Cj.d.ts.map → types-lPkDQNqJ.d.ts.map} +1 -1
  187. package/dist/wire/index.d.ts +2 -2
  188. package/docs/adapters-observability.md +14 -0
  189. package/docs/campaign-proposers.md +86 -128
  190. package/docs/charter.md +108 -112
  191. package/docs/concepts.md +157 -69
  192. package/docs/design/mlbenchmarks-book-review.md +440 -0
  193. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  194. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  195. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  196. package/docs/design/self-improvement-evidence-audit.md +263 -0
  197. package/docs/design.md +2 -1
  198. package/docs/eval-surface-map.md +95 -42
  199. package/docs/evaluation-integrity.md +220 -0
  200. package/docs/experiment.md +111 -55
  201. package/docs/feature-guide.md +5 -6
  202. package/docs/hosted-ingest-spec.md +4 -11
  203. package/docs/insight-report.md +187 -455
  204. package/docs/outcome-validity.md +182 -0
  205. package/docs/product-eval-adoption.md +1 -2
  206. package/docs/research-report-methodology.md +7 -7
  207. package/docs/search-history-receipts.md +8 -0
  208. package/docs/statistical-evidence.md +129 -0
  209. package/docs/verdicts.md +76 -49
  210. package/package.json +1 -1
  211. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  212. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  213. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  214. package/dist/client-BlLY6o2w.js.map +0 -1
  215. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  216. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  217. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  218. package/dist/index-BxWvILU8.d.ts.map +0 -1
  219. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  220. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  221. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  222. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  223. package/dist/outcome-store-ChBKlTd_.js +0 -75
  224. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  225. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  226. package/dist/report-command-DKlXfU5r.js.map +0 -1
  227. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  228. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  229. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  230. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  231. package/dist/run-record-CR63CpHK.js +0 -216
  232. package/dist/run-record-CR63CpHK.js.map +0 -1
  233. package/dist/sequential-B5gXgcyp.js.map +0 -1
  234. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  235. package/dist/terminal-record-Ce9_UjRz.js.map +0 -1
  236. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
@@ -0,0 +1 @@
1
+ {"version":3,"file":"rubric-predictive-validity-CCK-1B7w.js","names":[],"sources":["../src/meta-eval/outcome-observations.ts","../src/meta-eval/rubric-predictive-validity.ts"],"sourcesContent":["import { pearsonR, spearmanR } from '../statistics'\nimport { makeRng } from '../statistics/internal'\nimport type { DeploymentOutcome } from './outcome-store'\n\nexport type OutcomeReduction = 'latest' | 'mean' | 'max'\n\nexport interface CorrelationInterval {\n lower: number\n upper: number\n}\n\n/** Reduce observations of one named metric; unrelated keys cannot supply a value. */\nexport function reduceOutcomeMetric(\n outcomes: readonly DeploymentOutcome[],\n metric: string,\n reduction: OutcomeReduction,\n): number | null {\n const observations = outcomes.flatMap((outcome) => {\n const value = outcome.metrics[metric]\n return Number.isFinite(outcome.capturedAt) &&\n typeof value === 'number' &&\n Number.isFinite(value)\n ? [{ capturedAt: outcome.capturedAt, value }]\n : []\n })\n if (observations.length === 0) return null\n if (reduction === 'mean') {\n return observations.reduce(\n (sum, observation) => sum + observation.value / observations.length,\n 0,\n )\n }\n if (reduction === 'max') return Math.max(...observations.map((observation) => observation.value))\n return observations.reduce((a, b) => (b.capturedAt > a.capturedAt ? b : a)).value\n}\n\nexport function hasVariation(values: readonly number[]): boolean {\n return values.some((value) => value !== values[0])\n}\n\nexport function correlationSummary(\n xs: number[],\n ys: number[],\n iterations: number,\n seed: number | undefined,\n): {\n pearson: number\n spearman: number\n pearsonCi95: CorrelationInterval | null\n spearmanCi95: CorrelationInterval | null\n} {\n const rng = makeRng(seed, xs, ys)\n const pearsons: number[] = []\n const spearmans: number[] = []\n for (let b = 0; b < iterations; b++) {\n const rx: number[] = []\n const ry: number[] = []\n for (let i = 0; i < xs.length; i++) {\n const index = Math.floor(rng() * xs.length)\n rx.push(xs[index]!)\n ry.push(ys[index]!)\n }\n // A constant resample has no estimable correlation, even when both sides agree.\n if (!hasVariation(rx) || !hasVariation(ry)) continue\n const pearson = pearsonR(rx, ry)\n const spearman = spearmanR(rx, ry)\n if (Number.isFinite(pearson)) pearsons.push(pearson)\n if (Number.isFinite(spearman)) spearmans.push(spearman)\n }\n return {\n pearson: pearsonR(xs, ys),\n spearman: spearmanR(xs, ys),\n pearsonCi95: interval(pearsons),\n spearmanCi95: interval(spearmans),\n }\n}\n\nexport function validateObservationOptions(\n reduction: OutcomeReduction,\n iterations: number,\n seed: number | undefined,\n): void {\n if (!['latest', 'mean', 'max'].includes(reduction)) {\n throw new Error('outcome reduction must be latest, mean, or max')\n }\n if (!Number.isSafeInteger(iterations) || iterations < 1) {\n throw new Error('bootstrap iterations must be a positive safe integer')\n }\n if (seed !== undefined && !Number.isFinite(seed)) {\n throw new Error('bootstrap seed must be finite')\n }\n}\n\nexport function assertUniqueObservationIds(ids: readonly string[], name: string): void {\n if (ids.some((id) => typeof id !== 'string' || id.trim().length === 0)) {\n throw new Error(`${name} must be a nonempty string`)\n }\n if (new Set(ids).size !== ids.length) throw new Error(`duplicate ${name} in outcome study`)\n}\n\nexport function validateOutcomeMetricSpecifications(\n metrics: readonly { id: string; direction: string }[],\n): void {\n if (!Array.isArray(metrics) || metrics.length === 0) {\n throw new Error('outcomeMetrics must declare at least one outcome metric and direction')\n }\n for (const metric of metrics) {\n if (!metric || typeof metric.id !== 'string' || metric.id.trim().length === 0) {\n throw new Error('each outcome metric must have a nonempty id and explicit direction')\n }\n if (metric.direction !== 'higher-is-better' && metric.direction !== 'lower-is-better') {\n throw new Error(\n `outcome metric ${metric.id} must declare higher-is-better or lower-is-better`,\n )\n }\n }\n assertUniqueObservationIds(\n metrics.map((metric) => metric.id),\n 'outcome metric',\n )\n}\n\nfunction interval(values: number[]): CorrelationInterval | null {\n if (values.length === 0) return null\n values.sort((a, b) => a - b)\n return {\n lower: Math.max(-1, values[Math.floor(0.025 * values.length)]!),\n upper: Math.min(1, values[Math.min(values.length - 1, Math.floor(0.975 * values.length))]!),\n }\n}\n","/**\n * Join rubric scores to deployment outcomes and measure their association.\n * Higher rubric scores always mean better evaluated behavior.\n * Outcome directions are explicit because success rate and failure rate have opposite meanings.\n * These descriptive associations neither establish causation nor validate a change to rubric weights.\n */\n\nimport type { RunRecord } from '../run-record'\nimport {\n assertUniqueObservationIds,\n type CorrelationInterval,\n correlationSummary,\n hasVariation,\n reduceOutcomeMetric,\n validateObservationOptions,\n validateOutcomeMetricSpecifications,\n} from './outcome-observations'\nimport type { DeploymentOutcome, OutcomeStore } from './outcome-store'\n\nexport interface OutcomeMetricSpec {\n /** Exact key in DeploymentOutcome.metrics. */\n id: string\n direction: 'higher-is-better' | 'lower-is-better'\n}\n\nexport interface RubricPredictiveValidityInput {\n /** One record per independent run; rubric scores come from outcome.raw. */\n runs: RunRecord[]\n outcomes: OutcomeStore\n /** Declare desired directions before inspecting associations. */\n outcomeMetrics: readonly OutcomeMetricSpec[]\n /** Higher is better for each rubric. Omit to discover finite numeric outcome.raw keys. */\n rubrics?: readonly string[]\n /** Minimum joined runs for an estimate; an integer at least 3. Default 8. */\n minSamples?: number\n /** Bootstrap resamples for both correlation intervals. Default 500. */\n bootstrapResamples?: number\n /** Omit to derive a reproducible seed from the paired observations. */\n seed?: number\n /** Reduce finite observations of each named outcome within a run. Default latest. */\n reduction?: 'latest' | 'mean' | 'max'\n}\n\nexport interface RubricOutcomePair {\n rubric: string\n outcome: string\n outcomeDirection: OutcomeMetricSpec['direction']\n n: number\n /** Raw association with the recorded outcome, before direction alignment. */\n pearson: number\n spearman: number\n pearsonCi95: CorrelationInterval | null\n spearmanCi95: CorrelationInterval | null\n /** Positive values associate higher rubric scores with better outcomes. */\n alignedPearson: number\n alignedSpearman: number\n alignedSpearmanCi95: CorrelationInterval | null\n /** Descriptive buckets at aligned Spearman +/-0.4; no causal or release authority. */\n verdict: 'aligned' | 'inverse' | 'weak'\n}\n\nexport interface RubricRanking extends Omit<RubricOutcomePair, 'outcome'> {\n /** Outcome with the greatest direction-aligned Spearman for this rubric. */\n bestOutcome: string\n}\n\nexport interface RubricOutcomeExclusion {\n rubric: string\n outcome: string\n outcomeDirection: OutcomeMetricSpec['direction']\n /** Finite joined observations, including measured zeros. */\n n: number\n reason: 'insufficient_samples' | 'constant_rubric' | 'constant_outcome'\n}\n\nexport interface RubricPredictiveValidityReport {\n outcomeMetrics: OutcomeMetricSpec[]\n pairs: RubricOutcomePair[]\n /** All declared pairs lacking an estimate, with their usable observation count. */\n excludedPairs: RubricOutcomeExclusion[]\n /** Exploratory ordering by aligned Spearman; never use outcome selection as confirmatory evidence. */\n ranked: RubricRanking[]\n /** Runs contributing at least one finite pair, including pairs below minSamples. */\n joinedSamples: number\n /** Runs contributing no finite pair; joinedSamples + skippedRuns equals the input run count. */\n skippedRuns: number\n /** Declared rubrics with no finite score, distinct from too few outcomes or constant observations. */\n rubricsWithoutData: string[]\n}\n\nexport async function rubricPredictiveValidity(\n input: RubricPredictiveValidityInput,\n): Promise<RubricPredictiveValidityReport> {\n const minSamples = input.minSamples ?? 8\n const reduction = input.reduction ?? 'latest'\n const resamples = input.bootstrapResamples ?? 500\n const seed = input.seed\n if (!Number.isSafeInteger(minSamples) || minSamples < 3) {\n throw new Error('minSamples must be a safe integer at least 3')\n }\n validateObservationOptions(reduction, resamples, seed)\n validateOutcomeMetricSpecifications(input.outcomeMetrics)\n assertUniqueObservationIds(\n input.runs.map((run) => run.runId),\n 'runId',\n )\n\n const outcomeMetrics = input.outcomeMetrics.map((metric) => ({ ...metric }))\n const runs = input.runs.map((run) => ({ runId: run.runId, scores: { ...run.outcome.raw } }))\n const declaredRubrics = input.rubrics === undefined ? undefined : [...input.rubrics]\n if (declaredRubrics !== undefined) assertUniqueObservationIds(declaredRubrics, 'rubric')\n\n const outcomes = await input.outcomes.list()\n const outcomesByRun = new Map<string, DeploymentOutcome[]>()\n for (const outcome of outcomes) {\n const rows = outcomesByRun.get(outcome.runId) ?? []\n rows.push(outcome)\n outcomesByRun.set(outcome.runId, rows)\n }\n\n const observedRubrics = new Set<string>()\n for (const run of runs) {\n for (const [rubric, value] of Object.entries(run.scores)) {\n if (typeof value === 'number' && Number.isFinite(value)) observedRubrics.add(rubric)\n }\n }\n const rubrics = declaredRubrics ?? [...observedRubrics]\n const buckets = rubrics.flatMap((rubric) =>\n outcomeMetrics.map((outcome) => ({\n rubric,\n outcome,\n xs: [] as number[],\n ys: [] as number[],\n })),\n )\n\n let joined = 0\n for (const run of runs) {\n const rows = outcomesByRun.get(run.runId) ?? []\n let joinedThisRun = false\n for (const bucket of buckets) {\n const x = run.scores[bucket.rubric]\n if (typeof x !== 'number' || !Number.isFinite(x)) continue\n const y = reduceOutcomeMetric(rows, bucket.outcome.id, reduction)\n if (y === null) continue\n bucket.xs.push(x)\n bucket.ys.push(y)\n joinedThisRun = true\n }\n if (joinedThisRun) joined++\n }\n\n const pairs: RubricOutcomePair[] = []\n const excludedPairs: RubricOutcomeExclusion[] = []\n for (const bucket of buckets) {\n const identity = {\n rubric: bucket.rubric,\n outcome: bucket.outcome.id,\n outcomeDirection: bucket.outcome.direction,\n n: bucket.xs.length,\n }\n const reason =\n bucket.xs.length < minSamples\n ? 'insufficient_samples'\n : !hasVariation(bucket.xs)\n ? 'constant_rubric'\n : !hasVariation(bucket.ys)\n ? 'constant_outcome'\n : null\n if (reason !== null) {\n excludedPairs.push({ ...identity, reason })\n continue\n }\n const summary = correlationSummary(bucket.xs, bucket.ys, resamples, seed)\n const sign = bucket.outcome.direction === 'higher-is-better' ? 1 : -1\n const alignedSpearman = summary.spearman * sign\n const alignedSpearmanCi95 =\n summary.spearmanCi95 === null\n ? null\n : sign === 1\n ? summary.spearmanCi95\n : { lower: -summary.spearmanCi95.upper, upper: -summary.spearmanCi95.lower }\n pairs.push({\n ...identity,\n ...summary,\n alignedPearson: summary.pearson * sign,\n alignedSpearman,\n alignedSpearmanCi95,\n verdict: alignedSpearman >= 0.4 ? 'aligned' : alignedSpearman <= -0.4 ? 'inverse' : 'weak',\n })\n }\n\n const bestByRubric = new Map<string, RubricOutcomePair>()\n for (const pair of pairs) {\n const best = bestByRubric.get(pair.rubric)\n if (!best || pair.alignedSpearman > best.alignedSpearman) bestByRubric.set(pair.rubric, pair)\n }\n const ranked = [...bestByRubric.values()]\n .map(({ outcome, ...pair }) => ({ ...pair, bestOutcome: outcome }))\n .sort((a, b) => b.alignedSpearman - a.alignedSpearman)\n\n return {\n outcomeMetrics,\n pairs,\n excludedPairs,\n ranked,\n joinedSamples: joined,\n skippedRuns: runs.length - joined,\n rubricsWithoutData: rubrics.filter((rubric) => !observedRubrics.has(rubric)),\n }\n}\n"],"mappings":";;;;AAYA,SAAgB,oBACd,UACA,QACA,WACe;CACf,MAAM,eAAe,SAAS,SAAS,YAAY;EACjD,MAAM,QAAQ,QAAQ,QAAQ;EAC9B,OAAO,OAAO,SAAS,QAAQ,UAAU,KACvC,OAAO,UAAU,YACjB,OAAO,SAAS,KAAK,IACnB,CAAC;GAAE,YAAY,QAAQ;GAAY;EAAM,CAAC,IAC1C,CAAC;CACP,CAAC;CACD,IAAI,aAAa,WAAW,GAAG,OAAO;CACtC,IAAI,cAAc,QAChB,OAAO,aAAa,QACjB,KAAK,gBAAgB,MAAM,YAAY,QAAQ,aAAa,QAC7D,CACF;CAEF,IAAI,cAAc,OAAO,OAAO,KAAK,IAAI,GAAG,aAAa,KAAK,gBAAgB,YAAY,KAAK,CAAC;CAChG,OAAO,aAAa,QAAQ,GAAG,MAAO,EAAE,aAAa,EAAE,aAAa,IAAI,CAAE,CAAC,CAAC;AAC9E;AAEA,SAAgB,aAAa,QAAoC;CAC/D,OAAO,OAAO,MAAM,UAAU,UAAU,OAAO,EAAE;AACnD;AAEA,SAAgB,mBACd,IACA,IACA,YACA,MAMA;CACA,MAAM,MAAM,QAAQ,MAAM,IAAI,EAAE;CAChC,MAAM,WAAqB,CAAC;CAC5B,MAAM,YAAsB,CAAC;CAC7B,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,KAAK;EACnC,MAAM,KAAe,CAAC;EACtB,MAAM,KAAe,CAAC;EACtB,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,QAAQ,KAAK;GAClC,MAAM,QAAQ,KAAK,MAAM,IAAI,IAAI,GAAG,MAAM;GAC1C,GAAG,KAAK,GAAG,MAAO;GAClB,GAAG,KAAK,GAAG,MAAO;EACpB;EAEA,IAAI,CAAC,aAAa,EAAE,KAAK,CAAC,aAAa,EAAE,GAAG;EAC5C,MAAM,UAAU,SAAS,IAAI,EAAE;EAC/B,MAAM,WAAW,UAAU,IAAI,EAAE;EACjC,IAAI,OAAO,SAAS,OAAO,GAAG,SAAS,KAAK,OAAO;EACnD,IAAI,OAAO,SAAS,QAAQ,GAAG,UAAU,KAAK,QAAQ;CACxD;CACA,OAAO;EACL,SAAS,SAAS,IAAI,EAAE;EACxB,UAAU,UAAU,IAAI,EAAE;EAC1B,aAAa,SAAS,QAAQ;EAC9B,cAAc,SAAS,SAAS;CAClC;AACF;AAEA,SAAgB,2BACd,WACA,YACA,MACM;CACN,IAAI,CAAC;EAAC;EAAU;EAAQ;CAAK,CAAC,CAAC,SAAS,SAAS,GAC/C,MAAM,IAAI,MAAM,gDAAgD;CAElE,IAAI,CAAC,OAAO,cAAc,UAAU,KAAK,aAAa,GACpD,MAAM,IAAI,MAAM,sDAAsD;CAExE,IAAI,SAAS,KAAA,KAAa,CAAC,OAAO,SAAS,IAAI,GAC7C,MAAM,IAAI,MAAM,+BAA+B;AAEnD;AAEA,SAAgB,2BAA2B,KAAwB,MAAoB;CACrF,IAAI,IAAI,MAAM,OAAO,OAAO,OAAO,YAAY,GAAG,KAAK,CAAC,CAAC,WAAW,CAAC,GACnE,MAAM,IAAI,MAAM,GAAG,KAAK,2BAA2B;CAErD,IAAI,IAAI,IAAI,GAAG,CAAC,CAAC,SAAS,IAAI,QAAQ,MAAM,IAAI,MAAM,aAAa,KAAK,kBAAkB;AAC5F;AAEA,SAAgB,oCACd,SACM;CACN,IAAI,CAAC,MAAM,QAAQ,OAAO,KAAK,QAAQ,WAAW,GAChD,MAAM,IAAI,MAAM,uEAAuE;CAEzF,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,CAAC,UAAU,OAAO,OAAO,OAAO,YAAY,OAAO,GAAG,KAAK,CAAC,CAAC,WAAW,GAC1E,MAAM,IAAI,MAAM,oEAAoE;EAEtF,IAAI,OAAO,cAAc,sBAAsB,OAAO,cAAc,mBAClE,MAAM,IAAI,MACR,kBAAkB,OAAO,GAAG,kDAC9B;CAEJ;CACA,2BACE,QAAQ,KAAK,WAAW,OAAO,EAAE,GACjC,gBACF;AACF;AAEA,SAAS,SAAS,QAA8C;CAC9D,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,OAAO,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3B,OAAO;EACL,OAAO,KAAK,IAAI,IAAI,OAAO,KAAK,MAAM,OAAQ,OAAO,MAAM,EAAG;EAC9D,OAAO,KAAK,IAAI,GAAG,OAAO,KAAK,IAAI,OAAO,SAAS,GAAG,KAAK,MAAM,OAAQ,OAAO,MAAM,CAAC,EAAG;CAC5F;AACF;;;ACvCA,eAAsB,yBACpB,OACyC;CACzC,MAAM,aAAa,MAAM,cAAc;CACvC,MAAM,YAAY,MAAM,aAAa;CACrC,MAAM,YAAY,MAAM,sBAAsB;CAC9C,MAAM,OAAO,MAAM;CACnB,IAAI,CAAC,OAAO,cAAc,UAAU,KAAK,aAAa,GACpD,MAAM,IAAI,MAAM,8CAA8C;CAEhE,2BAA2B,WAAW,WAAW,IAAI;CACrD,oCAAoC,MAAM,cAAc;CACxD,2BACE,MAAM,KAAK,KAAK,QAAQ,IAAI,KAAK,GACjC,OACF;CAEA,MAAM,iBAAiB,MAAM,eAAe,KAAK,YAAY,EAAE,GAAG,OAAO,EAAE;CAC3E,MAAM,OAAO,MAAM,KAAK,KAAK,SAAS;EAAE,OAAO,IAAI;EAAO,QAAQ,EAAE,GAAG,IAAI,QAAQ,IAAI;CAAE,EAAE;CAC3F,MAAM,kBAAkB,MAAM,YAAY,KAAA,IAAY,KAAA,IAAY,CAAC,GAAG,MAAM,OAAO;CACnF,IAAI,oBAAoB,KAAA,GAAW,2BAA2B,iBAAiB,QAAQ;CAEvF,MAAM,WAAW,MAAM,MAAM,SAAS,KAAK;CAC3C,MAAM,gCAAgB,IAAI,IAAiC;CAC3D,KAAK,MAAM,WAAW,UAAU;EAC9B,MAAM,OAAO,cAAc,IAAI,QAAQ,KAAK,KAAK,CAAC;EAClD,KAAK,KAAK,OAAO;EACjB,cAAc,IAAI,QAAQ,OAAO,IAAI;CACvC;CAEA,MAAM,kCAAkB,IAAI,IAAY;CACxC,KAAK,MAAM,OAAO,MAChB,KAAK,MAAM,CAAC,QAAQ,UAAU,OAAO,QAAQ,IAAI,MAAM,GACrD,IAAI,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,GAAG,gBAAgB,IAAI,MAAM;CAGvF,MAAM,UAAU,mBAAmB,CAAC,GAAG,eAAe;CACtD,MAAM,UAAU,QAAQ,SAAS,WAC/B,eAAe,KAAK,aAAa;EAC/B;EACA;EACA,IAAI,CAAC;EACL,IAAI,CAAC;CACP,EAAE,CACJ;CAEA,IAAI,SAAS;CACb,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,OAAO,cAAc,IAAI,IAAI,KAAK,KAAK,CAAC;EAC9C,IAAI,gBAAgB;EACpB,KAAK,MAAM,UAAU,SAAS;GAC5B,MAAM,IAAI,IAAI,OAAO,OAAO;GAC5B,IAAI,OAAO,MAAM,YAAY,CAAC,OAAO,SAAS,CAAC,GAAG;GAClD,MAAM,IAAI,oBAAoB,MAAM,OAAO,QAAQ,IAAI,SAAS;GAChE,IAAI,MAAM,MAAM;GAChB,OAAO,GAAG,KAAK,CAAC;GAChB,OAAO,GAAG,KAAK,CAAC;GAChB,gBAAgB;EAClB;EACA,IAAI,eAAe;CACrB;CAEA,MAAM,QAA6B,CAAC;CACpC,MAAM,gBAA0C,CAAC;CACjD,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,WAAW;GACf,QAAQ,OAAO;GACf,SAAS,OAAO,QAAQ;GACxB,kBAAkB,OAAO,QAAQ;GACjC,GAAG,OAAO,GAAG;EACf;EACA,MAAM,SACJ,OAAO,GAAG,SAAS,aACf,yBACA,CAAC,aAAa,OAAO,EAAE,IACrB,oBACA,CAAC,aAAa,OAAO,EAAE,IACrB,qBACA;EACV,IAAI,WAAW,MAAM;GACnB,cAAc,KAAK;IAAE,GAAG;IAAU;GAAO,CAAC;GAC1C;EACF;EACA,MAAM,UAAU,mBAAmB,OAAO,IAAI,OAAO,IAAI,WAAW,IAAI;EACxE,MAAM,OAAO,OAAO,QAAQ,cAAc,qBAAqB,IAAI;EACnE,MAAM,kBAAkB,QAAQ,WAAW;EAC3C,MAAM,sBACJ,QAAQ,iBAAiB,OACrB,OACA,SAAS,IACP,QAAQ,eACR;GAAE,OAAO,CAAC,QAAQ,aAAa;GAAO,OAAO,CAAC,QAAQ,aAAa;EAAM;EACjF,MAAM,KAAK;GACT,GAAG;GACH,GAAG;GACH,gBAAgB,QAAQ,UAAU;GAClC;GACA;GACA,SAAS,mBAAmB,KAAM,YAAY,mBAAmB,MAAO,YAAY;EACtF,CAAC;CACH;CAEA,MAAM,+BAAe,IAAI,IAA+B;CACxD,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,OAAO,aAAa,IAAI,KAAK,MAAM;EACzC,IAAI,CAAC,QAAQ,KAAK,kBAAkB,KAAK,iBAAiB,aAAa,IAAI,KAAK,QAAQ,IAAI;CAC9F;CAKA,OAAO;EACL;EACA;EACA;EACA,QARa,CAAC,GAAG,aAAa,OAAO,CAAC,CAAC,CACtC,KAAK,EAAE,SAAS,GAAG,YAAY;GAAE,GAAG;GAAM,aAAa;EAAQ,EAAE,CAAC,CAClE,MAAM,GAAG,MAAM,EAAE,kBAAkB,EAAE,eAMjC;EACL,eAAe;EACf,aAAa,KAAK,SAAS;EAC3B,oBAAoB,QAAQ,QAAQ,WAAW,CAAC,gBAAgB,IAAI,MAAM,CAAC;CAC7E;AACF"}
@@ -1,6 +1,6 @@
1
1
  import { c as ValidationError } from "./errors-DEE6u6ot.js";
2
2
  import { p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
3
- import { r as AgentProfileCell } from "./agent-profile-cell-CTOZJUuE.js";
3
+ import { r as AgentProfileCell } from "./agent-profile-cell-s__adRnK.js";
4
4
  import { o as FailureClass } from "./schema-CR5cpjQ3.js";
5
5
  //#region src/run-record.d.ts
6
6
  /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
@@ -244,4 +244,4 @@ declare function roundTripRunRecord(record: RunRecord): RunRecord;
244
244
  declare function modelHasSnapshot(model: string): boolean;
245
245
  //#endregion
246
246
  export { validateRunRecord as _, RunRecord as a, RunTaskFailure as c, UNKNOWN_MODEL as d, isRunRecord as f, runTaskScore as g, roundTripRunRecord as h, RunOutcome as i, RunTerminalOutcome as l, parseRunRecordSafe as m, RunCostProvenance as n, RunRecordValidationError as o, modelHasSnapshot as p, RunJudgeMetadata as r, RunSplitTag as s, JudgeScoresRecord as t, RunTokenUsage as u };
247
- //# sourceMappingURL=run-record-DTv1MdjK.d.ts.map
247
+ //# sourceMappingURL=run-record-BiTWauyO.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"run-record-DTv1MdjK.d.ts","names":[],"sources":["../src/run-record.ts"],"mappings":";;;;;;;KAsCY;;;;;;;KAQA;;cAGC;UAEI;EACf;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;;KAIU,oBAAoB;UAEf;EACf;EACA;;;EAGA;;;EAGA;;;;;;;;;;;;;;;;;;;;;UAsBe;;EAEf,UAAU,eAAe;;EAEzB,YAAY;;;EAGZ;;;;EAIA;;;EAGA;;UAGe;;;EAGf;;;EAGA;;;;EAIA,KAAK;;;;;;EAML,cAAc;;;;;;;EAOd;IAAa;IAAe;IAAgB;;;;;;;;;;;;;;;;;;;;;;UAsB7B;;EAEf;;;EAGA;;;EAGA;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,gBAAgB;;EAEhB,YAAY;;EAEZ,iBAAiB;;;EAGjB;;EAEA,gBAAgB;;EAEhB,SAAS;;;;;EAKT,eAAe;;;EAGf;;EAEA,UAAU;;;;;EAKV;;;;;;;;EAQA,eAAe;;;;;;;;;KAUL;EACN;EAA0B;;EAC1B;EAAyB;;EAEzB,cAAc,QAAQ;EACtB;;;;;;;;;;iBAWU,aAAa,QAAQ;cAmCxB,iCAAiC;WACnC;EACT,YAAY,iBAAiB;;;;;;;iBAWf,kBAAkB,iBAAiB;;iBAgPnC,YAAY,iBAAiB,SAAS;;iBAUtC,mBACd;EACG;EAAU,OAAO;;EAAgB;EAAW,OAAO;;;iBAUxC,mBAAmB,QAAQ,YAAY;;;;;;;iBAuFvC,iBAAiB"}
1
+ {"version":3,"file":"run-record-BiTWauyO.d.ts","names":[],"sources":["../src/run-record.ts"],"mappings":";;;;;;;KAsCY;;;;;;;KAQA;;cAGC;UAEI;EACf;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;;KAIU,oBAAoB;UAEf;EACf;EACA;;;EAGA;;;EAGA;;;;;;;;;;;;;;;;;;;;;UAsBe;;EAEf,UAAU,eAAe;;EAEzB,YAAY;;;EAGZ;;;;EAIA;;;EAGA;;UAGe;;;EAGf;;;EAGA;;;;EAIA,KAAK;;;;;;EAML,cAAc;;;;;;;EAOd;IAAa;IAAe;IAAgB;;;;;;;;;;;;;;;;;;;;;;UAsB7B;;EAEf;;;EAGA;;;EAGA;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,gBAAgB;;EAEhB,YAAY;;EAEZ,iBAAiB;;;EAGjB;;EAEA,gBAAgB;;EAEhB,SAAS;;;;;EAKT,eAAe;;;EAGf;;EAEA,UAAU;;;;;EAKV;;;;;;;;EAQA,eAAe;;;;;;;;;KAUL;EACN;EAA0B;;EAC1B;EAAyB;;EAEzB,cAAc,QAAQ;EACtB;;;;;;;;;;iBAWU,aAAa,QAAQ;cAmCxB,iCAAiC;WACnC;EACT,YAAY,iBAAiB;;;;;;;iBAWf,kBAAkB,iBAAiB;;iBAgPnC,YAAY,iBAAiB,SAAS;;iBAUtC,mBACd;EACG;EAAU,OAAO;;EAAgB;EAAW,OAAO;;;iBAUxC,mBAAmB,QAAQ,YAAY;;;;;;;iBAuFvC,iBAAiB"}
@@ -0,0 +1,464 @@
1
+ import { g as pairedRiskDifferenceScore, h as pairedRiskDifferenceExact, p as pairedBinaryScale } from "./paired-arms-D4aeIHUy.js";
2
+ import { a as pairedSignTest, i as pairedDeltaTieFraction, r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
3
+ import { c as validateRunRecord } from "./run-record-DualPTn2.js";
4
+ //#region src/paired-delta-test.ts
5
+ /** Smallest all-positive sample that can clear a one-sided exact sign test. */
6
+ function minimumPairsForPairedDeltaTest(confidence = .95) {
7
+ if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`);
8
+ const oneSidedAlpha = (1 - confidence) / 2;
9
+ return Math.ceil(Math.log2(1 / oneSidedAlpha));
10
+ }
11
+ /**
12
+ * Tests whether a paired candidate-minus-baseline delta clears a threshold.
13
+ *
14
+ * At 20 or more pairs, the percentile bootstrap lower bound carries the
15
+ * decision. Below that point the interval is descriptive only, so the function
16
+ * switches to a pre-registered one-sided exact sign test. The exact path is
17
+ * deliberately conservative: it requires both a point estimate above the
18
+ * threshold and enough consistently positive paired differences.
19
+ *
20
+ * ## A zero-width interval is never significant
21
+ *
22
+ * When every paired delta is identical the resample distribution is a point
23
+ * mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n
24
+ * identical deltas of g. Neither says the effect is certain — both say the
25
+ * sample carries no information about how far the estimate could be wrong, and
26
+ * `low > threshold` then answers on the point estimate alone. It fails in both
27
+ * directions: `[0, 0]` clears every NEGATIVE threshold, which is how a
28
+ * tie-dominated pass/fail comparison laundered a regression into a
29
+ * noninferiority pass, and `[g, g]` clears every threshold below g with no
30
+ * spread behind it. Under a bounded asymmetric null whose true mean paired
31
+ * delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —
32
+ * every sample that misses the drop is exactly that shape, and deciding on
33
+ * `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.
34
+ *
35
+ * So `indeterminate` is reported and `significant` is false whenever the
36
+ * interval has zero width, on BOTH paths: at small n the exact sign test is a
37
+ * test of the MEDIAN and a zero-spread sample is precisely where it stops
38
+ * saying anything about the mean the caller is thresholding.
39
+ *
40
+ * `threshold` may be negative — that is a noninferiority margin, and it is the
41
+ * regime the zero-width hole is worst in. For a two-point (pass/fail) outcome
42
+ * the percentile bootstrap is not a valid interval at a nonzero margin at all;
43
+ * use {@link decidePairedPromotion}, which routes those to Tango's score
44
+ * interval, rather than thresholding this function's bootstrap directly.
45
+ */
46
+ function pairedDeltaTest(before, after, options = {}) {
47
+ const threshold = options.threshold ?? 0;
48
+ if (!Number.isFinite(threshold)) throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`);
49
+ const exactMinimum = minimumPairsForPairedDeltaTest(options.confidence ?? .95);
50
+ const requestedMinimum = options.minPairs ?? exactMinimum;
51
+ if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`);
52
+ const minimumPairs = Math.max(requestedMinimum, exactMinimum);
53
+ const bootstrap = pairedBootstrap(before, after, options);
54
+ const sufficient = bootstrap.n >= minimumPairs;
55
+ const indeterminate = bootstrap.n > 0 && (!Number.isFinite(bootstrap.low) || !Number.isFinite(bootstrap.high) || bootstrap.low === bootstrap.high);
56
+ if (bootstrap.gateEligible) return {
57
+ bootstrap,
58
+ method: "bootstrap-ci",
59
+ pValue: null,
60
+ minimumPairs,
61
+ sufficient,
62
+ indeterminate,
63
+ significant: sufficient && !indeterminate && bootstrap.low > threshold
64
+ };
65
+ const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
66
+ const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
67
+ return {
68
+ bootstrap,
69
+ method: "exact-sign",
70
+ pValue: exact.pValue,
71
+ minimumPairs,
72
+ sufficient,
73
+ indeterminate,
74
+ significant: sufficient && !indeterminate && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
75
+ };
76
+ }
77
+ //#endregion
78
+ //#region src/paired-promotion-decision.ts
79
+ /**
80
+ * @module
81
+ * ONE rule for "does this paired interval clear a promotion threshold".
82
+ *
83
+ * The rule below was derived on `HeldOutGate` (#479) after the same estimator
84
+ * bug shipped twice. It then turned out that a SECOND gate — the composable
85
+ * `heldOutGate`, plus everything else routed through `heldoutSignificance` —
86
+ * still carried the original defect, because the rule had been written into one
87
+ * gate's method body rather than into a shared function. Two copies of a
88
+ * statistical rule is how a defect survives in one of them, so there is now
89
+ * exactly one copy and both gates call it.
90
+ *
91
+ * Three things the rule does that a bare `pairedBootstrap(...).low > threshold`
92
+ * does not:
93
+ *
94
+ * 1. **Two-point (pass/fail) outcomes decide on Tango's SCORE interval.** On a
95
+ * pass/fail eval the paired delta vector is dominated by ties, so the
96
+ * bootstrap of the mean is a resample of a lattice with three atoms and its
97
+ * percentile interval is not valid at a nonzero margin. The score interval
98
+ * (`pairedRiskDifferenceScore`) re-estimates the nuisance loss rate under
99
+ * each hypothesised margin instead of fixing it at the observed value, which
100
+ * is the only construction that stays a confidence interval as the margin
101
+ * moves off zero — the regime every noninferiority threshold lives in.
102
+ * Measured on the composable gate before this change, at a true risk
103
+ * difference sitting exactly on the production caller's -0.05 margin and a
104
+ * nominal 5 %: 14.60 % false promotion at n = 40 and 10.10 % at n = 76.
105
+ * 2. **McNemar's exact test holds a VETO at every non-negative threshold.**
106
+ * Redundant with the interval by construction and kept anyway, so that
107
+ * swapping the estimator for one without that duality cannot silently
108
+ * reintroduce "promotes what the exact test refuses". Witness: n = 6, b = 5,
109
+ * c = 0 — no exact argument reaches alpha = 0.05 with 5 discordant pairs
110
+ * (two-sided floor 2/2^5 = 0.0625), whatever an interval says. A NEGATIVE
111
+ * threshold is a noninferiority question, which McNemar's test of "no
112
+ * difference" is not the right test for, so the veto does not apply there.
113
+ * 3. **A ZERO-WIDTH interval is refused, wherever it sits.** At [0, 0] it
114
+ * cannot tell a gain from a regression and clears every negative threshold.
115
+ * Away from zero it fails the opposite way: n identical positive deltas give
116
+ * [g, g], which clears threshold 0 on no spread at all. Both are an absence
117
+ * of evidence. Measured on the composable gate before this change, under a
118
+ * bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %
119
+ * false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.
120
+ *
121
+ * Eligibility follows the requested target. A continuous mean needs the
122
+ * bootstrap minimum; binary outcomes and explicit median targets use the
123
+ * exact-test minimum. A small-sample sign diagnostic cannot certify a mean
124
+ * effect. These floors establish estimator eligibility, not adequate power.
125
+ */
126
+ /**
127
+ * Which estimator {@link decidePairedPromotion} would use on this data, and the
128
+ * shape facts behind it — for callers that must report the shape on a path
129
+ * where no interval is computed at all (an early rejection, or zero pairs).
130
+ * Cheap: no bootstrap, no interval.
131
+ */
132
+ function pairedDecisionShape(before, after, statistic = "mean", declaredBinaryScale) {
133
+ if (declaredBinaryScale !== void 0) {
134
+ if (!Number.isFinite(declaredBinaryScale) || declaredBinaryScale <= 0) throw new Error("pairedDecisionShape: binaryScale must be finite and positive");
135
+ if (statistic === "median") throw new Error("pairedDecisionShape: binaryScale requires the mean statistic, not median");
136
+ for (const [name, arm] of [["before", before], ["after", after]]) for (let i = 0; i < arm.length; i++) {
137
+ const value = arm[i];
138
+ if (value !== 0 && value !== declaredBinaryScale) throw new Error(`pairedDecisionShape: ${name}[${i}] must be 0 or binaryScale (${declaredBinaryScale}); got ${value}`);
139
+ }
140
+ }
141
+ const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after);
142
+ if (statistic === "median") return {
143
+ statistic: "median_bootstrap",
144
+ binaryScale: null,
145
+ tieFraction
146
+ };
147
+ const binaryScale = declaredBinaryScale ?? pairedBinaryScale(before, after);
148
+ if (binaryScale !== null) return {
149
+ statistic: "paired_risk_difference",
150
+ binaryScale,
151
+ tieFraction
152
+ };
153
+ return {
154
+ statistic: "mean_bootstrap",
155
+ binaryScale: null,
156
+ tieFraction
157
+ };
158
+ }
159
+ /**
160
+ * Decide whether a paired candidate-minus-baseline delta clears a promotion
161
+ * threshold. `before` is the baseline arm, `after` the candidate arm, paired by
162
+ * position. Throws on unequal lengths.
163
+ */
164
+ function decidePairedPromotion(before, after, options = {}) {
165
+ if (before.length !== after.length) throw new Error(`decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`);
166
+ const threshold = options.threshold ?? 0;
167
+ if (!Number.isFinite(threshold)) throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`);
168
+ const confidence = options.confidence ?? .95;
169
+ const exactMinimum = minimumPairsForPairedDeltaTest(confidence);
170
+ const requestedMinimum = options.minPairs ?? exactMinimum;
171
+ if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`);
172
+ const { binaryScale, tieFraction } = pairedDecisionShape(before, after, options.statistic, options.binaryScale);
173
+ const estimatorMinimum = binaryScale === null && options.statistic !== "median" ? 20 : exactMinimum;
174
+ const minimumPairs = Math.max(requestedMinimum, exactMinimum, estimatorMinimum);
175
+ const n = before.length;
176
+ const sufficient = n >= minimumPairs;
177
+ let core;
178
+ if (binaryScale !== null) {
179
+ const unitControl = before.map((v) => v / binaryScale);
180
+ const unitTreatment = after.map((v) => v / binaryScale);
181
+ const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence);
182
+ const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence);
183
+ const low = score.lower * binaryScale;
184
+ core = {
185
+ statistic: "paired_risk_difference",
186
+ method: "score-interval",
187
+ delta: score.riskDifference * binaryScale,
188
+ low,
189
+ high: score.upper * binaryScale,
190
+ bootstrap: null,
191
+ mcnemar: {
192
+ b: exact.b,
193
+ c: exact.c,
194
+ nDiscordant: exact.nDiscordant,
195
+ pValue: exact.pValue
196
+ },
197
+ pValue: null,
198
+ clearsThreshold: low > threshold,
199
+ label: "success-rate",
200
+ methodDetail: ""
201
+ };
202
+ } else {
203
+ const bootstrapStatistic = options.statistic === "median" ? "median" : "mean";
204
+ const test = pairedDeltaTest(before, after, {
205
+ confidence,
206
+ resamples: options.resamples,
207
+ statistic: bootstrapStatistic,
208
+ seed: options.seed,
209
+ threshold,
210
+ minPairs: minimumPairs
211
+ });
212
+ const ci = test.bootstrap;
213
+ core = {
214
+ statistic: bootstrapStatistic === "mean" ? "mean_bootstrap" : "median_bootstrap",
215
+ method: test.method,
216
+ delta: bootstrapStatistic === "mean" ? ci.mean : ci.median,
217
+ low: ci.low,
218
+ high: ci.high,
219
+ bootstrap: ci,
220
+ mcnemar: null,
221
+ pValue: test.pValue,
222
+ clearsThreshold: test.significant,
223
+ label: bootstrapStatistic,
224
+ methodDetail: test.method === "exact-sign" ? bootstrapStatistic === "mean" ? ` The mean requires ${minimumPairs} pairs for bootstrap eligibility; the exact sign-test p=${fmt(test.pValue ?? 1)} does not establish a mean effect.` : ` Below 20 pairs the interval is descriptive only; the median decision uses the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.` : ""
225
+ };
226
+ }
227
+ const intervalTolerance = Math.max(Math.abs(core.low), Math.abs(core.high)) * Number.EPSILON * 8;
228
+ const indeterminate = !Number.isFinite(core.low) || !Number.isFinite(core.high) || core.high - core.low <= intervalTolerance;
229
+ const indeterminateCause = !indeterminate ? "" : tieFraction === 1 ? "every paired delta is an exact tie" : core.mcnemar !== null && core.mcnemar.nDiscordant === 0 ? "every pair is concordant (0 discordant pairs)" : `the ${core.label} CI collapsed to a point at ${fmt(core.low)}`;
230
+ const exactTestVetoes = core.mcnemar !== null && threshold >= 0 && !(core.mcnemar.pValue < 1 - confidence);
231
+ return {
232
+ n,
233
+ threshold,
234
+ confidence,
235
+ binaryScale,
236
+ tieFraction,
237
+ minimumPairs,
238
+ sufficient,
239
+ indeterminate,
240
+ indeterminateCause,
241
+ exactTestVetoes,
242
+ promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,
243
+ ...core
244
+ };
245
+ }
246
+ function fmt(x) {
247
+ return x.toFixed(4);
248
+ }
249
+ //#endregion
250
+ //#region src/campaign/run-record.ts
251
+ /**
252
+ * A campaign cell carried a judge score without a `dimensions` record.
253
+ *
254
+ * Two exported types share the name `JudgeScore`: the campaign verdict
255
+ * (`{ dimensions, composite, notes }` from `@tangle-network/agent-eval/campaign`)
256
+ * and the root export's flat per-dimension row (`{ judgeName, dimension, score }`
257
+ * from `@tangle-network/agent-eval`). Campaign aggregation accepts only the
258
+ * campaign shape; the flat shape previously crashed here with an opaque
259
+ * TypeError deep inside aggregation.
260
+ */
261
+ var CampaignJudgeScoreShapeError = class extends TypeError {
262
+ constructor(judgeName) {
263
+ super(`campaign cell judge '${judgeName}' carries a score without a 'dimensions' record. Use the campaign JudgeScore ({ dimensions, composite, notes }) from '@tangle-network/agent-eval/campaign'; the root export's JudgeScore ({ judgeName, dimension, score }) is a different type with the same name.`);
264
+ this.name = "CampaignJudgeScoreShapeError";
265
+ }
266
+ };
267
+ /**
268
+ * Project one campaign cell into the canonical run format.
269
+ *
270
+ * A dispatch error establishes terminal execution failure. A judge error only
271
+ * establishes that quality measurement failed after dispatch completed.
272
+ * Failures without a stage remain unknown. No failure becomes a zero-quality
273
+ * label.
274
+ */
275
+ function campaignCellToRunRecord(cell, options) {
276
+ const quality = projectCampaignCellQuality(cell);
277
+ const execution = campaignCellExecutionEvidence(cell);
278
+ const judgeErrorCount = Math.max(quality.raw.judge_error_count ?? 0, execution.judgeErrorCount ?? 0);
279
+ const cellCostProvenance = campaignCellCostProvenance(cell);
280
+ const costProvenance = cellCostProvenance.kind === "uncaptured" && options.defaultCostUsd !== void 0 ? {
281
+ kind: "estimated",
282
+ usd: options.defaultCostUsd
283
+ } : cellCostProvenance;
284
+ const costUsd = costProvenance.kind === "uncaptured" ? null : costProvenance.usd;
285
+ const raw = {
286
+ ...finiteMetrics(options.raw),
287
+ ...quality.raw,
288
+ rep: cell.rep,
289
+ duration_ms: cell.durationMs,
290
+ ...costUsd === null ? {} : { cost_usd: costUsd },
291
+ ...cellCostProvenance.kind === "uncaptured" ? { cost_known_subtotal_usd: cell.costUsd } : {},
292
+ cost_observed: costProvenance.kind === "observed" ? 1 : 0,
293
+ cost_estimated: costProvenance.kind === "estimated" ? 1 : 0,
294
+ cost_uncaptured: costProvenance.kind === "uncaptured" ? 1 : 0,
295
+ tokens_input: cell.tokenUsage.input,
296
+ tokens_output: cell.tokenUsage.output,
297
+ tokens_known: cell.tokenUsage.tokensKnown === false ? 0 : 1,
298
+ latency_ms: cell.durationMs,
299
+ ...execution.executionErrorCount === void 0 ? {} : { execution_error_count: execution.executionErrorCount },
300
+ ...judgeErrorCount > 0 ? { judge_error_count: judgeErrorCount } : {},
301
+ ...execution.unclassifiedErrorCount === void 0 ? {} : { unclassified_error_count: execution.unclassifiedErrorCount }
302
+ };
303
+ if (typeof cell.generation === "number") raw.generation = cell.generation;
304
+ if (cell.tokenUsage.reasoning !== void 0) raw.tokens_reasoning = cell.tokenUsage.reasoning;
305
+ if (cell.tokenUsage.cached !== void 0) raw.tokens_cached = cell.tokenUsage.cached;
306
+ if (cell.tokenUsage.cacheWrite !== void 0) raw.tokens_cache_write = cell.tokenUsage.cacheWrite;
307
+ if (cell.tokenUsage.tokensKnown !== false && costUsd !== null && costUsd > 0) raw.tokens_per_dollar = (cell.tokenUsage.input + cell.tokenUsage.output) / costUsd;
308
+ if (costUsd !== null && quality.score !== void 0 && quality.score > .01) raw.cost_per_quality = costUsd / quality.score;
309
+ const outcome = {
310
+ raw,
311
+ ...quality.judgeScores ? { judgeScores: quality.judgeScores } : {}
312
+ };
313
+ if (quality.score !== void 0) if (options.splitTag === "holdout") outcome.holdoutScore = quality.score;
314
+ else outcome.searchScore = quality.score;
315
+ return validateRunRecord({
316
+ runId: options.runId,
317
+ experimentId: options.experimentId,
318
+ candidateId: options.candidateId,
319
+ seed: options.seed ?? cell.seed,
320
+ model: options.model,
321
+ promptHash: options.promptHash,
322
+ configHash: options.configHash,
323
+ commitSha: options.commitSha,
324
+ wallMs: cell.durationMs,
325
+ costUsd,
326
+ costProvenance,
327
+ tokenUsage: { ...cell.tokenUsage },
328
+ terminalOutcome: execution.terminalOutcome,
329
+ ...execution.terminalFailureReason ? { terminalFailureReason: execution.terminalFailureReason } : {},
330
+ outcome,
331
+ splitTag: options.splitTag,
332
+ scenarioId: options.scenarioId ?? cell.scenarioId,
333
+ ...options.agentProfile ? { agentProfile: options.agentProfile } : {}
334
+ });
335
+ }
336
+ /**
337
+ * Validate the cost fields that cross campaign cache and RunRecord boundaries.
338
+ * `costUsd` is a known subtotal for uncaptured cells, but it must equal the
339
+ * authoritative total whenever that total is observed or estimated.
340
+ */
341
+ function campaignCellCostProvenance(cell) {
342
+ if (!Number.isFinite(cell.costUsd) || cell.costUsd < 0) throw new Error(`campaign cell '${cell.cellId}' has invalid costUsd`);
343
+ const provenance = cell.costProvenance;
344
+ if (!provenance || typeof provenance !== "object") throw new Error(`campaign cell '${cell.cellId}' has no costProvenance`);
345
+ if (provenance.kind === "uncaptured") {
346
+ if (provenance.usd !== null) throw new Error(`campaign cell '${cell.cellId}' has invalid uncaptured costProvenance`);
347
+ return {
348
+ kind: "uncaptured",
349
+ usd: null
350
+ };
351
+ }
352
+ if (provenance.kind !== "observed" && provenance.kind !== "estimated" || !Number.isFinite(provenance.usd) || provenance.usd < 0) throw new Error(`campaign cell '${cell.cellId}' has invalid costProvenance`);
353
+ if (provenance.usd !== cell.costUsd) throw new Error(`campaign cell '${cell.cellId}' has costUsd inconsistent with costProvenance`);
354
+ return {
355
+ kind: provenance.kind,
356
+ usd: provenance.usd
357
+ };
358
+ }
359
+ function campaignCellExecutionEvidence(cell) {
360
+ if (cell.errorStage === "dispatch") return {
361
+ terminalOutcome: "failed",
362
+ executionErrorCount: 1,
363
+ ...cell.error ? { terminalFailureReason: cell.error } : {}
364
+ };
365
+ if (cell.errorStage === "judge") return {
366
+ terminalOutcome: "succeeded",
367
+ executionErrorCount: 0,
368
+ judgeErrorCount: 1
369
+ };
370
+ if (!cell.error) return {
371
+ terminalOutcome: "succeeded",
372
+ executionErrorCount: 0
373
+ };
374
+ return {
375
+ terminalOutcome: "unknown",
376
+ unclassifiedErrorCount: 1
377
+ };
378
+ }
379
+ /**
380
+ * Produce the only task-quality view used by campaign aggregates and exports.
381
+ *
382
+ * Successful judge results remain available for diagnosis after another judge
383
+ * fails, but a task score exists only for an error-free cell whose reported
384
+ * judge values are all finite.
385
+ */
386
+ function projectCampaignCellQuality(cell) {
387
+ if (cell.errorStage === "dispatch") return {
388
+ successfulJudgeScores: {},
389
+ failedJudges: [],
390
+ raw: {}
391
+ };
392
+ const perJudge = {};
393
+ const successfulJudgeScores = {};
394
+ const dimensionValues = /* @__PURE__ */ new Map();
395
+ const composites = [];
396
+ const notes = [];
397
+ const failedJudges = new Set(cell.errorStage === "judge" ? [cell.errorJudge ?? "unknown-judge"] : []);
398
+ const raw = {};
399
+ for (const [judgeName, score] of Object.entries(cell.judgeScores)) {
400
+ const dimensionsShape = score.dimensions;
401
+ if (typeof dimensionsShape !== "object" || dimensionsShape === null || Array.isArray(dimensionsShape)) throw new CampaignJudgeScoreShapeError(judgeName);
402
+ const finiteDimensions = Object.values(score.dimensions).every(Number.isFinite);
403
+ if (score.failed || !Number.isFinite(score.composite) || !finiteDimensions) {
404
+ failedJudges.add(judgeName);
405
+ continue;
406
+ }
407
+ composites.push(score.composite);
408
+ successfulJudgeScores[judgeName] = score;
409
+ const dimensions = { ...score.dimensions };
410
+ perJudge[judgeName] = dimensions;
411
+ for (const [dimension, value] of Object.entries(dimensions)) {
412
+ raw[`${judgeName}.${dimension}`] = value;
413
+ const values = dimensionValues.get(dimension) ?? [];
414
+ values.push(value);
415
+ dimensionValues.set(dimension, values);
416
+ }
417
+ if (score.notes) notes.push(`${judgeName}: ${score.notes}`);
418
+ for (const failedJudge of score.failedJudges ?? []) failedJudges.add(`${judgeName}/${failedJudge}`);
419
+ }
420
+ if (failedJudges.size > 0) raw.judge_error_count = failedJudges.size;
421
+ const sortedFailedJudges = [...failedJudges].sort();
422
+ if (composites.length === 0) return {
423
+ successfulJudgeScores,
424
+ failedJudges: sortedFailedJudges,
425
+ raw
426
+ };
427
+ const composite = mean(composites);
428
+ const perDimMean = Object.fromEntries([...dimensionValues.entries()].map(([dimension, values]) => [dimension, mean(values)]));
429
+ const complete = cell.error === void 0 && cell.errorStage === void 0 && failedJudges.size === 0;
430
+ if (complete) raw.composite = composite;
431
+ return {
432
+ ...complete ? { score: composite } : {},
433
+ raw,
434
+ successfulJudgeScores,
435
+ failedJudges: sortedFailedJudges,
436
+ judgeScores: {
437
+ perJudge,
438
+ perDimMean,
439
+ composite,
440
+ ...sortedFailedJudges.length > 0 ? { failedJudges: sortedFailedJudges } : {},
441
+ ...notes.length > 0 ? { notes: notes.join(" | ") } : {}
442
+ }
443
+ };
444
+ }
445
+ /** Read the canonical task score without recomputing cell quality. */
446
+ function campaignCellTaskScore(cell) {
447
+ return projectCampaignCellQuality(cell).score;
448
+ }
449
+ /** Read canonical successful judge dimensions without recomputing cell quality. */
450
+ function campaignCellJudgeDimensions(cell) {
451
+ return projectCampaignCellQuality(cell).judgeScores?.perJudge ?? {};
452
+ }
453
+ function finiteMetrics(metrics) {
454
+ const finite = {};
455
+ for (const [key, value] of Object.entries(metrics ?? {})) if (Number.isFinite(value)) finite[key] = value;
456
+ return finite;
457
+ }
458
+ function mean(values) {
459
+ return values.reduce((sum, value) => sum + value, 0) / values.length;
460
+ }
461
+ //#endregion
462
+ export { campaignCellToRunRecord as a, pairedDecisionShape as c, campaignCellTaskScore as i, minimumPairsForPairedDeltaTest as l, campaignCellExecutionEvidence as n, projectCampaignCellQuality as o, campaignCellJudgeDimensions as r, decidePairedPromotion as s, campaignCellCostProvenance as t, pairedDeltaTest as u };
463
+
464
+ //# sourceMappingURL=run-record-Br-Yzt_k.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"run-record-Br-Yzt_k.js","names":[],"sources":["../src/paired-delta-test.ts","../src/paired-promotion-decision.ts","../src/campaign/run-record.ts"],"sourcesContent":["import {\n type PairedBootstrapOptions,\n type PairedBootstrapResult,\n pairedBootstrap,\n pairedSignTest,\n} from './statistics'\n\nexport interface PairedDeltaTestOptions extends PairedBootstrapOptions {\n /** Smallest candidate-minus-baseline delta that counts as improvement. Default 0. */\n threshold?: number\n /** Caller-required paired observations. The exact test may impose a higher minimum. */\n minPairs?: number\n}\n\nexport interface PairedDeltaTestResult {\n bootstrap: PairedBootstrapResult\n method: 'bootstrap-ci' | 'exact-sign'\n /** Exact one-sided p-value below the bootstrap minimum; otherwise null. */\n pValue: number | null\n /** Effective observation minimum after accounting for confidence. */\n minimumPairs: number\n sufficient: boolean\n /**\n * The bootstrap interval has zero width (or is non-finite), so it carries no\n * information about how far the estimate could be wrong and cannot support a\n * decision in either direction. See {@link pairedDeltaTest}.\n */\n indeterminate: boolean\n significant: boolean\n}\n\n/** Smallest all-positive sample that can clear a one-sided exact sign test. */\nexport function minimumPairsForPairedDeltaTest(confidence = 0.95): number {\n if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) {\n throw new Error(\n `minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`,\n )\n }\n const oneSidedAlpha = (1 - confidence) / 2\n return Math.ceil(Math.log2(1 / oneSidedAlpha))\n}\n\n/**\n * Tests whether a paired candidate-minus-baseline delta clears a threshold.\n *\n * At 20 or more pairs, the percentile bootstrap lower bound carries the\n * decision. Below that point the interval is descriptive only, so the function\n * switches to a pre-registered one-sided exact sign test. The exact path is\n * deliberately conservative: it requires both a point estimate above the\n * threshold and enough consistently positive paired differences.\n *\n * ## A zero-width interval is never significant\n *\n * When every paired delta is identical the resample distribution is a point\n * mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n\n * identical deltas of g. Neither says the effect is certain — both say the\n * sample carries no information about how far the estimate could be wrong, and\n * `low > threshold` then answers on the point estimate alone. It fails in both\n * directions: `[0, 0]` clears every NEGATIVE threshold, which is how a\n * tie-dominated pass/fail comparison laundered a regression into a\n * noninferiority pass, and `[g, g]` clears every threshold below g with no\n * spread behind it. Under a bounded asymmetric null whose true mean paired\n * delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —\n * every sample that misses the drop is exactly that shape, and deciding on\n * `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.\n *\n * So `indeterminate` is reported and `significant` is false whenever the\n * interval has zero width, on BOTH paths: at small n the exact sign test is a\n * test of the MEDIAN and a zero-spread sample is precisely where it stops\n * saying anything about the mean the caller is thresholding.\n *\n * `threshold` may be negative — that is a noninferiority margin, and it is the\n * regime the zero-width hole is worst in. For a two-point (pass/fail) outcome\n * the percentile bootstrap is not a valid interval at a nonzero margin at all;\n * use {@link decidePairedPromotion}, which routes those to Tango's score\n * interval, rather than thresholding this function's bootstrap directly.\n */\nexport function pairedDeltaTest(\n before: number[],\n after: number[],\n options: PairedDeltaTestOptions = {},\n): PairedDeltaTestResult {\n const threshold = options.threshold ?? 0\n if (!Number.isFinite(threshold)) {\n throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`)\n }\n const confidence = options.confidence ?? 0.95\n const exactMinimum = minimumPairsForPairedDeltaTest(confidence)\n const requestedMinimum = options.minPairs ?? exactMinimum\n if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) {\n throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`)\n }\n const minimumPairs = Math.max(requestedMinimum, exactMinimum)\n const bootstrap = pairedBootstrap(before, after, options)\n const sufficient = bootstrap.n >= minimumPairs\n // A point-mass resample distribution is an absence of evidence, not a\n // certainty, and it is the shape that clears every negative threshold. Both\n // paths refuse it: see the \"zero-width\" section of this function's docs.\n const indeterminate =\n bootstrap.n > 0 &&\n (!Number.isFinite(bootstrap.low) ||\n !Number.isFinite(bootstrap.high) ||\n bootstrap.low === bootstrap.high)\n\n if (bootstrap.gateEligible) {\n return {\n bootstrap,\n method: 'bootstrap-ci',\n pValue: null,\n minimumPairs,\n sufficient,\n indeterminate,\n significant: sufficient && !indeterminate && bootstrap.low > threshold,\n }\n }\n\n const differences = before.map((value, index) => after[index]! - value - threshold)\n const exact = pairedSignTest(differences, 'greater')\n const estimate = options.statistic === 'mean' ? bootstrap.mean : bootstrap.median\n return {\n bootstrap,\n method: 'exact-sign',\n pValue: exact.pValue,\n minimumPairs,\n sufficient,\n indeterminate,\n significant:\n sufficient &&\n !indeterminate &&\n estimate > threshold &&\n exact.pValue <= (1 - bootstrap.confidence) / 2,\n }\n}\n","/**\n * @module\n * ONE rule for \"does this paired interval clear a promotion threshold\".\n *\n * The rule below was derived on `HeldOutGate` (#479) after the same estimator\n * bug shipped twice. It then turned out that a SECOND gate — the composable\n * `heldOutGate`, plus everything else routed through `heldoutSignificance` —\n * still carried the original defect, because the rule had been written into one\n * gate's method body rather than into a shared function. Two copies of a\n * statistical rule is how a defect survives in one of them, so there is now\n * exactly one copy and both gates call it.\n *\n * Three things the rule does that a bare `pairedBootstrap(...).low > threshold`\n * does not:\n *\n * 1. **Two-point (pass/fail) outcomes decide on Tango's SCORE interval.** On a\n * pass/fail eval the paired delta vector is dominated by ties, so the\n * bootstrap of the mean is a resample of a lattice with three atoms and its\n * percentile interval is not valid at a nonzero margin. The score interval\n * (`pairedRiskDifferenceScore`) re-estimates the nuisance loss rate under\n * each hypothesised margin instead of fixing it at the observed value, which\n * is the only construction that stays a confidence interval as the margin\n * moves off zero — the regime every noninferiority threshold lives in.\n * Measured on the composable gate before this change, at a true risk\n * difference sitting exactly on the production caller's -0.05 margin and a\n * nominal 5 %: 14.60 % false promotion at n = 40 and 10.10 % at n = 76.\n * 2. **McNemar's exact test holds a VETO at every non-negative threshold.**\n * Redundant with the interval by construction and kept anyway, so that\n * swapping the estimator for one without that duality cannot silently\n * reintroduce \"promotes what the exact test refuses\". Witness: n = 6, b = 5,\n * c = 0 — no exact argument reaches alpha = 0.05 with 5 discordant pairs\n * (two-sided floor 2/2^5 = 0.0625), whatever an interval says. A NEGATIVE\n * threshold is a noninferiority question, which McNemar's test of \"no\n * difference\" is not the right test for, so the veto does not apply there.\n * 3. **A ZERO-WIDTH interval is refused, wherever it sits.** At [0, 0] it\n * cannot tell a gain from a regression and clears every negative threshold.\n * Away from zero it fails the opposite way: n identical positive deltas give\n * [g, g], which clears threshold 0 on no spread at all. Both are an absence\n * of evidence. Measured on the composable gate before this change, under a\n * bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %\n * false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.\n *\n * Eligibility follows the requested target. A continuous mean needs the\n * bootstrap minimum; binary outcomes and explicit median targets use the\n * exact-test minimum. A small-sample sign diagnostic cannot certify a mean\n * effect. These floors establish estimator eligibility, not adequate power.\n */\n\nimport { minimumPairsForPairedDeltaTest, pairedDeltaTest } from './paired-delta-test'\nimport {\n BOOTSTRAP_GATE_MIN_N,\n type PairedBootstrapResult,\n pairedBinaryScale,\n pairedDeltaTieFraction,\n pairedRiskDifferenceExact,\n pairedRiskDifferenceScore,\n} from './statistics'\n\n/** Which paired estimator produced the deciding interval. */\nexport type PairedDecisionStatistic =\n | 'paired_risk_difference'\n | 'mean_bootstrap'\n | 'median_bootstrap'\n\n/** Which test carried the decision, given the estimator and the sample size. */\nexport type PairedDecisionMethod = 'score-interval' | 'bootstrap-ci' | 'exact-sign'\n\n/** McNemar's exact paired-binary evidence, on the two-point path only. */\nexport interface PairedMcNemarEvidence {\n /** Discordant pairs the treatment won. */\n b: number\n /** Discordant pairs the control won. */\n c: number\n /** b + c — the only pairs carrying information. */\n nDiscordant: number\n /** Two-sided exact p-value. */\n pValue: number\n}\n\nexport interface PairedPromotionDecisionOptions {\n /** Smallest candidate-minus-baseline delta that counts as improvement, in the\n * caller's native units. May be negative (a noninferiority margin). Default 0. */\n threshold?: number\n /** Confidence level. Default 0.95. */\n confidence?: number\n /** Bootstrap resamples, on the paths where a bootstrap decides. Default 2000. */\n resamples?: number\n /** Deterministic bootstrap seed. Omitted ⇒ derived from the deltas. */\n seed?: number\n /** Caller-required paired observations. The test also imposes its own\n * minimum: bootstrap eligibility for a continuous mean, or the exact-test\n * minimum for binary and explicit median targets. This is not a power guarantee. */\n minPairs?: number\n /**\n * `'mean'` (default) routes by SHAPE: a two-point (pass/fail) outcome on any\n * encoding decides on the score interval, everything else on the mean\n * bootstrap. `'median'` forces the median bootstrap on every input, including\n * shapes where it is structurally blind — kept for callers who want outlier\n * robustness on genuinely continuous outcomes and accept that cost.\n */\n statistic?: 'mean' | 'median'\n /** Declared binary support {0, binaryScale}, including when all observations\n * are zero. Must be finite and positive; every observation must use that\n * support. Uses the risk-difference mean, so cannot accompany 'median'.\n * Omitted: infer binary support from observed positive values. */\n binaryScale?: number\n}\n\nexport interface PairedPromotionDecision {\n /** Paired observations supplied. */\n n: number\n /** Threshold the interval was judged against, native units. */\n threshold: number\n confidence: number\n statistic: PairedDecisionStatistic\n method: PairedDecisionMethod\n /** Declared or inferred positive level ({0,1} ⇒ 1, {0,100} ⇒ 100),\n * or null when the outcome is not two-point. Non-null is exactly the\n * condition for the `paired_risk_difference` path, and it is the factor\n * `delta` / `low` / `high` were rescaled by. */\n binaryScale: number | null\n /** Exact-tie fraction over the paired deltas; null when there are no pairs. */\n tieFraction: number | null\n /** Point estimate of the DECIDING statistic, in the caller's native units. */\n delta: number\n /** Lower bound of the DECIDING interval, native units. */\n low: number\n /** Upper bound of the DECIDING interval, native units. */\n high: number\n /** The bootstrap that decided, or null when the score interval did. Callers\n * that need a bootstrap as a diagnostic on the two-point path compute their\n * own — it is not computed here, so the binary path costs no resamples. */\n bootstrap: PairedBootstrapResult | null\n /** McNemar's exact evidence, or null off the two-point path. */\n mcnemar: PairedMcNemarEvidence | null\n /** Exact one-sided sign-test p-value on the small-sample bootstrap path;\n * null otherwise. */\n pValue: number | null\n /** Effective observation minimum for the target, confidence, and caller floor. */\n minimumPairs: number\n /** n >= minimumPairs. */\n sufficient: boolean\n /** The deciding interval is zero-width at numeric precision or non-finite — no evidence in either\n * direction, so it cannot clear any threshold on evidence. */\n indeterminate: boolean\n /** McNemar's exact test refuses at a non-negative threshold. */\n exactTestVetoes: boolean\n /** The deciding interval clears the threshold, ignoring the other two guards. */\n clearsThreshold: boolean\n /** `sufficient && !indeterminate && clearsThreshold && !exactTestVetoes` —\n * the whole rule. */\n promote: boolean\n /** What `delta` measures, for a reason string. */\n label: 'success-rate' | 'mean' | 'median'\n /** Why a zero-width interval is zero-width; empty when it is not. */\n indeterminateCause: string\n /** Sentence naming the test when the exact sign test decided; else empty. */\n methodDetail: string\n}\n\n/** The shape facts that pick the estimator, without computing an interval. */\nexport interface PairedDecisionShape {\n statistic: PairedDecisionStatistic\n /** Declared or inferred positive binary level; null off the binary path. */\n binaryScale: number | null\n /** Exact-tie fraction over the paired deltas; null when there are no pairs. */\n tieFraction: number | null\n}\n\n/**\n * Which estimator {@link decidePairedPromotion} would use on this data, and the\n * shape facts behind it — for callers that must report the shape on a path\n * where no interval is computed at all (an early rejection, or zero pairs).\n * Cheap: no bootstrap, no interval.\n */\nexport function pairedDecisionShape(\n before: number[],\n after: number[],\n statistic: 'mean' | 'median' = 'mean',\n declaredBinaryScale?: number,\n): PairedDecisionShape {\n if (declaredBinaryScale !== undefined) {\n if (!Number.isFinite(declaredBinaryScale) || declaredBinaryScale <= 0) {\n throw new Error('pairedDecisionShape: binaryScale must be finite and positive')\n }\n if (statistic === 'median') {\n throw new Error('pairedDecisionShape: binaryScale requires the mean statistic, not median')\n }\n for (const [name, arm] of [\n ['before', before],\n ['after', after],\n ] as const) {\n for (let i = 0; i < arm.length; i++) {\n const value = arm[i]!\n if (value !== 0 && value !== declaredBinaryScale) {\n throw new Error(\n `pairedDecisionShape: ${name}[${i}] must be 0 or binaryScale (${declaredBinaryScale}); got ${value}`,\n )\n }\n }\n }\n }\n const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after)\n if (statistic === 'median') {\n return { statistic: 'median_bootstrap', binaryScale: null, tieFraction }\n }\n const binaryScale = declaredBinaryScale ?? pairedBinaryScale(before, after)\n if (binaryScale !== null) {\n return { statistic: 'paired_risk_difference', binaryScale, tieFraction }\n }\n return { statistic: 'mean_bootstrap', binaryScale: null, tieFraction }\n}\n\n/**\n * Decide whether a paired candidate-minus-baseline delta clears a promotion\n * threshold. `before` is the baseline arm, `after` the candidate arm, paired by\n * position. Throws on unequal lengths.\n */\nexport function decidePairedPromotion(\n before: number[],\n after: number[],\n options: PairedPromotionDecisionOptions = {},\n): PairedPromotionDecision {\n if (before.length !== after.length) {\n throw new Error(\n `decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`,\n )\n }\n const threshold = options.threshold ?? 0\n if (!Number.isFinite(threshold)) {\n throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`)\n }\n const confidence = options.confidence ?? 0.95\n const exactMinimum = minimumPairsForPairedDeltaTest(confidence)\n const requestedMinimum = options.minPairs ?? exactMinimum\n if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) {\n throw new Error(\n `decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`,\n )\n }\n const { binaryScale, tieFraction } = pairedDecisionShape(\n before,\n after,\n options.statistic,\n options.binaryScale,\n )\n const estimatorMinimum =\n binaryScale === null && options.statistic !== 'median' ? BOOTSTRAP_GATE_MIN_N : exactMinimum\n const minimumPairs = Math.max(requestedMinimum, exactMinimum, estimatorMinimum)\n const n = before.length\n const sufficient = n >= minimumPairs\n\n let core: {\n statistic: PairedDecisionStatistic\n method: PairedDecisionMethod\n delta: number\n low: number\n high: number\n bootstrap: PairedBootstrapResult | null\n mcnemar: PairedMcNemarEvidence | null\n pValue: number | null\n clearsThreshold: boolean\n label: PairedPromotionDecision['label']\n methodDetail: string\n }\n\n if (binaryScale !== null) {\n // Normalise the two-point encoding to {0,1} so the estimators see the\n // pass/fail structure, then rescale the answer back into the caller's\n // native units — the threshold is read in the units of the scores, so a\n // 0-100 pass/fail dimension must be gated in points, not in a rate.\n const unitControl = before.map((v) => v / binaryScale)\n const unitTreatment = after.map((v) => v / binaryScale)\n // TWO estimators with different jobs, because no single one does both.\n // `exact` is the authority on RD = 0 and supplies the veto; its\n // Clopper-Pearson interval conditions on the discordant count and is\n // therefore NOT a confidence interval at a nonzero margin. `score` is\n // Tango's, which is.\n const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence)\n const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence)\n const low = score.lower * binaryScale\n core = {\n statistic: 'paired_risk_difference',\n method: 'score-interval',\n delta: score.riskDifference * binaryScale,\n low,\n high: score.upper * binaryScale,\n bootstrap: null,\n mcnemar: {\n b: exact.b,\n c: exact.c,\n nDiscordant: exact.nDiscordant,\n pValue: exact.pValue,\n },\n pValue: null,\n clearsThreshold: low > threshold,\n label: 'success-rate',\n methodDetail: '',\n }\n } else {\n const bootstrapStatistic = options.statistic === 'median' ? 'median' : 'mean'\n const test = pairedDeltaTest(before, after, {\n confidence,\n resamples: options.resamples,\n statistic: bootstrapStatistic,\n seed: options.seed,\n threshold,\n minPairs: minimumPairs,\n })\n const ci = test.bootstrap\n core = {\n statistic: bootstrapStatistic === 'mean' ? 'mean_bootstrap' : 'median_bootstrap',\n method: test.method,\n delta: bootstrapStatistic === 'mean' ? ci.mean : ci.median,\n low: ci.low,\n high: ci.high,\n bootstrap: ci,\n mcnemar: null,\n pValue: test.pValue,\n clearsThreshold: test.significant,\n label: bootstrapStatistic,\n methodDetail:\n test.method === 'exact-sign'\n ? bootstrapStatistic === 'mean'\n ? ` The mean requires ${minimumPairs} pairs for bootstrap eligibility;` +\n ` the exact sign-test p=${fmt(test.pValue ?? 1)} does not establish a mean effect.`\n : ` Below ${BOOTSTRAP_GATE_MIN_N} pairs the interval is descriptive only;` +\n ` the median decision uses the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.`\n : '',\n }\n }\n\n // Equivalent deltas can differ by rounding after subtraction and averaging.\n // Use the interval's own scale so tiny score units retain the same decision\n // as a rescaled copy of the evidence.\n const intervalScale = Math.max(Math.abs(core.low), Math.abs(core.high))\n const intervalTolerance = intervalScale * Number.EPSILON * 8\n const indeterminate =\n !Number.isFinite(core.low) ||\n !Number.isFinite(core.high) ||\n core.high - core.low <= intervalTolerance\n const indeterminateCause = !indeterminate\n ? ''\n : tieFraction === 1\n ? 'every paired delta is an exact tie'\n : core.mcnemar !== null && core.mcnemar.nDiscordant === 0\n ? 'every pair is concordant (0 discordant pairs)'\n : `the ${core.label} CI collapsed to a point at ${fmt(core.low)}`\n // Only at a non-negative threshold: a negative threshold asks a\n // noninferiority question, which McNemar's test of \"no difference\" does not\n // answer.\n const exactTestVetoes =\n core.mcnemar !== null && threshold >= 0 && !(core.mcnemar.pValue < 1 - confidence)\n\n return {\n n,\n threshold,\n confidence,\n binaryScale,\n tieFraction,\n minimumPairs,\n sufficient,\n indeterminate,\n indeterminateCause,\n exactTestVetoes,\n promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,\n ...core,\n }\n}\n\nfunction fmt(x: number): string {\n return x.toFixed(4)\n}\n","import type { AgentProfileCell } from '../agent-profile-cell'\nimport type { CostProvenance } from '../cost-ledger'\nimport type {\n JudgeScoresRecord,\n RunOutcome,\n RunRecord,\n RunSplitTag,\n RunTerminalOutcome,\n} from '../run-record'\nimport { validateRunRecord } from '../run-record'\nimport type { CampaignCellResult, JudgeScore } from './types'\n\nexport interface CampaignCellRunRecordOptions {\n runId: string\n experimentId: string\n candidateId: string\n model: string\n promptHash: string\n configHash: string\n commitSha: string\n splitTag: RunSplitTag\n seed?: number\n scenarioId?: string\n defaultCostUsd?: number\n agentProfile?: AgentProfileCell\n raw?: Record<string, number>\n}\n\nexport interface CampaignCellQualityProjection {\n score?: number\n judgeScores?: JudgeScoresRecord\n successfulJudgeScores: Record<string, JudgeScore>\n failedJudges: string[]\n raw: Record<string, number>\n}\n\n/**\n * A campaign cell carried a judge score without a `dimensions` record.\n *\n * Two exported types share the name `JudgeScore`: the campaign verdict\n * (`{ dimensions, composite, notes }` from `@tangle-network/agent-eval/campaign`)\n * and the root export's flat per-dimension row (`{ judgeName, dimension, score }`\n * from `@tangle-network/agent-eval`). Campaign aggregation accepts only the\n * campaign shape; the flat shape previously crashed here with an opaque\n * TypeError deep inside aggregation.\n */\nexport class CampaignJudgeScoreShapeError extends TypeError {\n constructor(judgeName: string) {\n super(\n `campaign cell judge '${judgeName}' carries a score without a 'dimensions' record. ` +\n `Use the campaign JudgeScore ({ dimensions, composite, notes }) from ` +\n `'@tangle-network/agent-eval/campaign'; the root export's JudgeScore ` +\n `({ judgeName, dimension, score }) is a different type with the same name.`,\n )\n this.name = 'CampaignJudgeScoreShapeError'\n }\n}\n\nexport interface CampaignCellExecutionEvidence {\n terminalOutcome: RunTerminalOutcome\n executionErrorCount?: number\n judgeErrorCount?: number\n unclassifiedErrorCount?: number\n terminalFailureReason?: string\n}\n\n/**\n * Project one campaign cell into the canonical run format.\n *\n * A dispatch error establishes terminal execution failure. A judge error only\n * establishes that quality measurement failed after dispatch completed.\n * Failures without a stage remain unknown. No failure becomes a zero-quality\n * label.\n */\nexport function campaignCellToRunRecord<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n options: CampaignCellRunRecordOptions,\n): RunRecord {\n const quality = projectCampaignCellQuality(cell)\n const execution = campaignCellExecutionEvidence(cell)\n const judgeErrorCount = Math.max(\n quality.raw.judge_error_count ?? 0,\n execution.judgeErrorCount ?? 0,\n )\n const cellCostProvenance = campaignCellCostProvenance(cell)\n const costProvenance: CostProvenance =\n cellCostProvenance.kind === 'uncaptured' && options.defaultCostUsd !== undefined\n ? { kind: 'estimated', usd: options.defaultCostUsd }\n : cellCostProvenance\n const costUsd = costProvenance.kind === 'uncaptured' ? null : costProvenance.usd\n const raw: Record<string, number> = {\n ...finiteMetrics(options.raw),\n ...quality.raw,\n rep: cell.rep,\n duration_ms: cell.durationMs,\n ...(costUsd === null ? {} : { cost_usd: costUsd }),\n // Retain the observed subtotal even when the caller supplies an estimated total.\n ...(cellCostProvenance.kind === 'uncaptured' ? { cost_known_subtotal_usd: cell.costUsd } : {}),\n cost_observed: costProvenance.kind === 'observed' ? 1 : 0,\n cost_estimated: costProvenance.kind === 'estimated' ? 1 : 0,\n cost_uncaptured: costProvenance.kind === 'uncaptured' ? 1 : 0,\n tokens_input: cell.tokenUsage.input,\n tokens_output: cell.tokenUsage.output,\n tokens_known: cell.tokenUsage.tokensKnown === false ? 0 : 1,\n latency_ms: cell.durationMs,\n ...(execution.executionErrorCount === undefined\n ? {}\n : { execution_error_count: execution.executionErrorCount }),\n ...(judgeErrorCount > 0 ? { judge_error_count: judgeErrorCount } : {}),\n ...(execution.unclassifiedErrorCount === undefined\n ? {}\n : { unclassified_error_count: execution.unclassifiedErrorCount }),\n }\n if (typeof cell.generation === 'number') raw.generation = cell.generation\n if (cell.tokenUsage.reasoning !== undefined) {\n raw.tokens_reasoning = cell.tokenUsage.reasoning\n }\n if (cell.tokenUsage.cached !== undefined) raw.tokens_cached = cell.tokenUsage.cached\n if (cell.tokenUsage.cacheWrite !== undefined) {\n raw.tokens_cache_write = cell.tokenUsage.cacheWrite\n }\n if (cell.tokenUsage.tokensKnown !== false && costUsd !== null && costUsd > 0) {\n raw.tokens_per_dollar = (cell.tokenUsage.input + cell.tokenUsage.output) / costUsd\n }\n if (costUsd !== null && quality.score !== undefined && quality.score > 0.01) {\n raw.cost_per_quality = costUsd / quality.score\n }\n\n const outcome: RunOutcome = {\n raw,\n ...(quality.judgeScores ? { judgeScores: quality.judgeScores } : {}),\n }\n if (quality.score !== undefined) {\n if (options.splitTag === 'holdout') outcome.holdoutScore = quality.score\n else outcome.searchScore = quality.score\n }\n\n return validateRunRecord({\n runId: options.runId,\n experimentId: options.experimentId,\n candidateId: options.candidateId,\n seed: options.seed ?? cell.seed,\n model: options.model,\n promptHash: options.promptHash,\n configHash: options.configHash,\n commitSha: options.commitSha,\n wallMs: cell.durationMs,\n costUsd,\n costProvenance,\n tokenUsage: { ...cell.tokenUsage },\n terminalOutcome: execution.terminalOutcome,\n ...(execution.terminalFailureReason\n ? { terminalFailureReason: execution.terminalFailureReason }\n : {}),\n outcome,\n splitTag: options.splitTag,\n scenarioId: options.scenarioId ?? cell.scenarioId,\n ...(options.agentProfile ? { agentProfile: options.agentProfile } : {}),\n })\n}\n\n/**\n * Validate the cost fields that cross campaign cache and RunRecord boundaries.\n * `costUsd` is a known subtotal for uncaptured cells, but it must equal the\n * authoritative total whenever that total is observed or estimated.\n */\nexport function campaignCellCostProvenance<TArtifact>(\n cell: Pick<CampaignCellResult<TArtifact>, 'cellId' | 'costUsd' | 'costProvenance'>,\n): CostProvenance {\n if (!Number.isFinite(cell.costUsd) || cell.costUsd < 0) {\n throw new Error(`campaign cell '${cell.cellId}' has invalid costUsd`)\n }\n const provenance = cell.costProvenance\n if (!provenance || typeof provenance !== 'object') {\n throw new Error(`campaign cell '${cell.cellId}' has no costProvenance`)\n }\n if (provenance.kind === 'uncaptured') {\n if (provenance.usd !== null) {\n throw new Error(`campaign cell '${cell.cellId}' has invalid uncaptured costProvenance`)\n }\n return { kind: 'uncaptured', usd: null }\n }\n if (\n (provenance.kind !== 'observed' && provenance.kind !== 'estimated') ||\n !Number.isFinite(provenance.usd) ||\n provenance.usd < 0\n ) {\n throw new Error(`campaign cell '${cell.cellId}' has invalid costProvenance`)\n }\n if (provenance.usd !== cell.costUsd) {\n throw new Error(`campaign cell '${cell.cellId}' has costUsd inconsistent with costProvenance`)\n }\n return { kind: provenance.kind, usd: provenance.usd }\n}\n\nexport function campaignCellExecutionEvidence<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): CampaignCellExecutionEvidence {\n if (cell.errorStage === 'dispatch') {\n return {\n terminalOutcome: 'failed',\n executionErrorCount: 1,\n ...(cell.error ? { terminalFailureReason: cell.error } : {}),\n }\n }\n if (cell.errorStage === 'judge') {\n return {\n terminalOutcome: 'succeeded',\n executionErrorCount: 0,\n judgeErrorCount: 1,\n }\n }\n if (!cell.error) {\n return { terminalOutcome: 'succeeded', executionErrorCount: 0 }\n }\n return {\n terminalOutcome: 'unknown',\n unclassifiedErrorCount: 1,\n }\n}\n\n/**\n * Produce the only task-quality view used by campaign aggregates and exports.\n *\n * Successful judge results remain available for diagnosis after another judge\n * fails, but a task score exists only for an error-free cell whose reported\n * judge values are all finite.\n */\nexport function projectCampaignCellQuality<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): CampaignCellQualityProjection {\n if (cell.errorStage === 'dispatch') {\n return { successfulJudgeScores: {}, failedJudges: [], raw: {} }\n }\n\n const perJudge: Record<string, Record<string, number>> = {}\n const successfulJudgeScores: Record<string, JudgeScore> = {}\n const dimensionValues = new Map<string, number[]>()\n const composites: number[] = []\n const notes: string[] = []\n const failedJudges = new Set<string>(\n cell.errorStage === 'judge' ? [cell.errorJudge ?? 'unknown-judge'] : [],\n )\n const raw: Record<string, number> = {}\n\n for (const [judgeName, score] of Object.entries(cell.judgeScores)) {\n const dimensionsShape = (score as { dimensions?: unknown }).dimensions\n if (\n typeof dimensionsShape !== 'object' ||\n dimensionsShape === null ||\n Array.isArray(dimensionsShape)\n ) {\n throw new CampaignJudgeScoreShapeError(judgeName)\n }\n const finiteDimensions = Object.values(score.dimensions).every(Number.isFinite)\n if (score.failed || !Number.isFinite(score.composite) || !finiteDimensions) {\n failedJudges.add(judgeName)\n continue\n }\n\n composites.push(score.composite)\n successfulJudgeScores[judgeName] = score\n const dimensions = { ...score.dimensions }\n perJudge[judgeName] = dimensions\n for (const [dimension, value] of Object.entries(dimensions)) {\n raw[`${judgeName}.${dimension}`] = value\n const values = dimensionValues.get(dimension) ?? []\n values.push(value)\n dimensionValues.set(dimension, values)\n }\n if (score.notes) notes.push(`${judgeName}: ${score.notes}`)\n for (const failedJudge of score.failedJudges ?? []) {\n failedJudges.add(`${judgeName}/${failedJudge}`)\n }\n }\n\n if (failedJudges.size > 0) raw.judge_error_count = failedJudges.size\n const sortedFailedJudges = [...failedJudges].sort()\n if (composites.length === 0) {\n return {\n successfulJudgeScores,\n failedJudges: sortedFailedJudges,\n raw,\n }\n }\n\n const composite = mean(composites)\n const perDimMean = Object.fromEntries(\n [...dimensionValues.entries()].map(([dimension, values]) => [dimension, mean(values)]),\n )\n const complete =\n cell.error === undefined && cell.errorStage === undefined && failedJudges.size === 0\n if (complete) raw.composite = composite\n\n return {\n ...(complete ? { score: composite } : {}),\n raw,\n successfulJudgeScores,\n failedJudges: sortedFailedJudges,\n judgeScores: {\n perJudge,\n perDimMean,\n composite,\n ...(sortedFailedJudges.length > 0 ? { failedJudges: sortedFailedJudges } : {}),\n ...(notes.length > 0 ? { notes: notes.join(' | ') } : {}),\n },\n }\n}\n\n/** Read the canonical task score without recomputing cell quality. */\nexport function campaignCellTaskScore<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): number | undefined {\n return projectCampaignCellQuality(cell).score\n}\n\n/** Read canonical successful judge dimensions without recomputing cell quality. */\nexport function campaignCellJudgeDimensions<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): Record<string, Record<string, number>> {\n return projectCampaignCellQuality(cell).judgeScores?.perJudge ?? {}\n}\n\nfunction finiteMetrics(metrics: Record<string, number> | undefined): Record<string, number> {\n const finite: Record<string, number> = {}\n for (const [key, value] of Object.entries(metrics ?? {})) {\n if (Number.isFinite(value)) finite[key] = value\n }\n return finite\n}\n\nfunction mean(values: number[]): number {\n return values.reduce((sum, value) => sum + value, 0) / values.length\n}\n"],"mappings":";;;;;AAgCA,SAAgB,+BAA+B,aAAa,KAAc;CACxE,IAAI,CAAC,OAAO,SAAS,UAAU,KAAK,cAAc,KAAK,cAAc,GACnE,MAAM,IAAI,MACR,oEAAoE,YACtE;CAEF,MAAM,iBAAiB,IAAI,cAAc;CACzC,OAAO,KAAK,KAAK,KAAK,KAAK,IAAI,aAAa,CAAC;AAC/C;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCA,SAAgB,gBACd,QACA,OACA,UAAkC,CAAC,GACZ;CACvB,MAAM,YAAY,QAAQ,aAAa;CACvC,IAAI,CAAC,OAAO,SAAS,SAAS,GAC5B,MAAM,IAAI,MAAM,kDAAkD,WAAW;CAG/E,MAAM,eAAe,+BADF,QAAQ,cAAc,GACqB;CAC9D,MAAM,mBAAmB,QAAQ,YAAY;CAC7C,IAAI,CAAC,OAAO,UAAU,gBAAgB,KAAK,mBAAmB,GAC5D,MAAM,IAAI,MAAM,6DAA6D,kBAAkB;CAEjG,MAAM,eAAe,KAAK,IAAI,kBAAkB,YAAY;CAC5D,MAAM,YAAY,gBAAgB,QAAQ,OAAO,OAAO;CACxD,MAAM,aAAa,UAAU,KAAK;CAIlC,MAAM,gBACJ,UAAU,IAAI,MACb,CAAC,OAAO,SAAS,UAAU,GAAG,KAC7B,CAAC,OAAO,SAAS,UAAU,IAAI,KAC/B,UAAU,QAAQ,UAAU;CAEhC,IAAI,UAAU,cACZ,OAAO;EACL;EACA,QAAQ;EACR,QAAQ;EACR;EACA;EACA;EACA,aAAa,cAAc,CAAC,iBAAiB,UAAU,MAAM;CAC/D;CAIF,MAAM,QAAQ,eADM,OAAO,KAAK,OAAO,UAAU,MAAM,SAAU,QAAQ,SAClC,GAAG,SAAS;CACnD,MAAM,WAAW,QAAQ,cAAc,SAAS,UAAU,OAAO,UAAU;CAC3E,OAAO;EACL;EACA,QAAQ;EACR,QAAQ,MAAM;EACd;EACA;EACA;EACA,aACE,cACA,CAAC,iBACD,WAAW,aACX,MAAM,WAAW,IAAI,UAAU,cAAc;CACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AC2CA,SAAgB,oBACd,QACA,OACA,YAA+B,QAC/B,qBACqB;CACrB,IAAI,wBAAwB,KAAA,GAAW;EACrC,IAAI,CAAC,OAAO,SAAS,mBAAmB,KAAK,uBAAuB,GAClE,MAAM,IAAI,MAAM,8DAA8D;EAEhF,IAAI,cAAc,UAChB,MAAM,IAAI,MAAM,0EAA0E;EAE5F,KAAK,MAAM,CAAC,MAAM,QAAQ,CACxB,CAAC,UAAU,MAAM,GACjB,CAAC,SAAS,KAAK,CACjB,GACE,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAAK;GACnC,MAAM,QAAQ,IAAI;GAClB,IAAI,UAAU,KAAK,UAAU,qBAC3B,MAAM,IAAI,MACR,wBAAwB,KAAK,GAAG,EAAE,8BAA8B,oBAAoB,SAAS,OAC/F;EAEJ;CAEJ;CACA,MAAM,cAAc,OAAO,WAAW,IAAI,OAAO,uBAAuB,QAAQ,KAAK;CACrF,IAAI,cAAc,UAChB,OAAO;EAAE,WAAW;EAAoB,aAAa;EAAM;CAAY;CAEzE,MAAM,cAAc,uBAAuB,kBAAkB,QAAQ,KAAK;CAC1E,IAAI,gBAAgB,MAClB,OAAO;EAAE,WAAW;EAA0B;EAAa;CAAY;CAEzE,OAAO;EAAE,WAAW;EAAkB,aAAa;EAAM;CAAY;AACvE;;;;;;AAOA,SAAgB,sBACd,QACA,OACA,UAA0C,CAAC,GAClB;CACzB,IAAI,OAAO,WAAW,MAAM,QAC1B,MAAM,IAAI,MACR,gDAAgD,OAAO,OAAO,MAAM,MAAM,OAAO,EACnF;CAEF,MAAM,YAAY,QAAQ,aAAa;CACvC,IAAI,CAAC,OAAO,SAAS,SAAS,GAC5B,MAAM,IAAI,MAAM,wDAAwD,WAAW;CAErF,MAAM,aAAa,QAAQ,cAAc;CACzC,MAAM,eAAe,+BAA+B,UAAU;CAC9D,MAAM,mBAAmB,QAAQ,YAAY;CAC7C,IAAI,CAAC,OAAO,UAAU,gBAAgB,KAAK,mBAAmB,GAC5D,MAAM,IAAI,MACR,mEAAmE,kBACrE;CAEF,MAAM,EAAE,aAAa,gBAAgB,oBACnC,QACA,OACA,QAAQ,WACR,QAAQ,WACV;CACA,MAAM,mBACJ,gBAAgB,QAAQ,QAAQ,cAAc,WAAA,KAAkC;CAClF,MAAM,eAAe,KAAK,IAAI,kBAAkB,cAAc,gBAAgB;CAC9E,MAAM,IAAI,OAAO;CACjB,MAAM,aAAa,KAAK;CAExB,IAAI;CAcJ,IAAI,gBAAgB,MAAM;EAKxB,MAAM,cAAc,OAAO,KAAK,MAAM,IAAI,WAAW;EACrD,MAAM,gBAAgB,MAAM,KAAK,MAAM,IAAI,WAAW;EAMtD,MAAM,QAAQ,0BAA0B,aAAa,eAAe,UAAU;EAC9E,MAAM,QAAQ,0BAA0B,aAAa,eAAe,UAAU;EAC9E,MAAM,MAAM,MAAM,QAAQ;EAC1B,OAAO;GACL,WAAW;GACX,QAAQ;GACR,OAAO,MAAM,iBAAiB;GAC9B;GACA,MAAM,MAAM,QAAQ;GACpB,WAAW;GACX,SAAS;IACP,GAAG,MAAM;IACT,GAAG,MAAM;IACT,aAAa,MAAM;IACnB,QAAQ,MAAM;GAChB;GACA,QAAQ;GACR,iBAAiB,MAAM;GACvB,OAAO;GACP,cAAc;EAChB;CACF,OAAO;EACL,MAAM,qBAAqB,QAAQ,cAAc,WAAW,WAAW;EACvE,MAAM,OAAO,gBAAgB,QAAQ,OAAO;GAC1C;GACA,WAAW,QAAQ;GACnB,WAAW;GACX,MAAM,QAAQ;GACd;GACA,UAAU;EACZ,CAAC;EACD,MAAM,KAAK,KAAK;EAChB,OAAO;GACL,WAAW,uBAAuB,SAAS,mBAAmB;GAC9D,QAAQ,KAAK;GACb,OAAO,uBAAuB,SAAS,GAAG,OAAO,GAAG;GACpD,KAAK,GAAG;GACR,MAAM,GAAG;GACT,WAAW;GACX,SAAS;GACT,QAAQ,KAAK;GACb,iBAAiB,KAAK;GACtB,OAAO;GACP,cACE,KAAK,WAAW,eACZ,uBAAuB,SACrB,sBAAsB,aAAa,0DACT,IAAI,KAAK,UAAU,CAAC,EAAE,sCAChD,+GAC8D,IAAI,KAAK,UAAU,CAAC,EAAE,KACtF;EACR;CACF;CAMA,MAAM,oBADgB,KAAK,IAAI,KAAK,IAAI,KAAK,GAAG,GAAG,KAAK,IAAI,KAAK,IAAI,CAC/B,IAAI,OAAO,UAAU;CAC3D,MAAM,gBACJ,CAAC,OAAO,SAAS,KAAK,GAAG,KACzB,CAAC,OAAO,SAAS,KAAK,IAAI,KAC1B,KAAK,OAAO,KAAK,OAAO;CAC1B,MAAM,qBAAqB,CAAC,gBACxB,KACA,gBAAgB,IACd,uCACA,KAAK,YAAY,QAAQ,KAAK,QAAQ,gBAAgB,IACpD,kDACA,OAAO,KAAK,MAAM,8BAA8B,IAAI,KAAK,GAAG;CAIpE,MAAM,kBACJ,KAAK,YAAY,QAAQ,aAAa,KAAK,EAAE,KAAK,QAAQ,SAAS,IAAI;CAEzE,OAAO;EACL;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,SAAS,cAAc,CAAC,iBAAiB,KAAK,mBAAmB,CAAC;EAClE,GAAG;CACL;AACF;AAEA,SAAS,IAAI,GAAmB;CAC9B,OAAO,EAAE,QAAQ,CAAC;AACpB;;;;;;;;;;;;;ACtUA,IAAa,+BAAb,cAAkD,UAAU;CAC1D,YAAY,WAAmB;EAC7B,MACE,wBAAwB,UAAU,mQAIpC;EACA,KAAK,OAAO;CACd;AACF;;;;;;;;;AAkBA,SAAgB,wBACd,MACA,SACW;CACX,MAAM,UAAU,2BAA2B,IAAI;CAC/C,MAAM,YAAY,8BAA8B,IAAI;CACpD,MAAM,kBAAkB,KAAK,IAC3B,QAAQ,IAAI,qBAAqB,GACjC,UAAU,mBAAmB,CAC/B;CACA,MAAM,qBAAqB,2BAA2B,IAAI;CAC1D,MAAM,iBACJ,mBAAmB,SAAS,gBAAgB,QAAQ,mBAAmB,KAAA,IACnE;EAAE,MAAM;EAAa,KAAK,QAAQ;CAAe,IACjD;CACN,MAAM,UAAU,eAAe,SAAS,eAAe,OAAO,eAAe;CAC7E,MAAM,MAA8B;EAClC,GAAG,cAAc,QAAQ,GAAG;EAC5B,GAAG,QAAQ;EACX,KAAK,KAAK;EACV,aAAa,KAAK;EAClB,GAAI,YAAY,OAAO,CAAC,IAAI,EAAE,UAAU,QAAQ;EAEhD,GAAI,mBAAmB,SAAS,eAAe,EAAE,yBAAyB,KAAK,QAAQ,IAAI,CAAC;EAC5F,eAAe,eAAe,SAAS,aAAa,IAAI;EACxD,gBAAgB,eAAe,SAAS,cAAc,IAAI;EAC1D,iBAAiB,eAAe,SAAS,eAAe,IAAI;EAC5D,cAAc,KAAK,WAAW;EAC9B,eAAe,KAAK,WAAW;EAC/B,cAAc,KAAK,WAAW,gBAAgB,QAAQ,IAAI;EAC1D,YAAY,KAAK;EACjB,GAAI,UAAU,wBAAwB,KAAA,IAClC,CAAC,IACD,EAAE,uBAAuB,UAAU,oBAAoB;EAC3D,GAAI,kBAAkB,IAAI,EAAE,mBAAmB,gBAAgB,IAAI,CAAC;EACpE,GAAI,UAAU,2BAA2B,KAAA,IACrC,CAAC,IACD,EAAE,0BAA0B,UAAU,uBAAuB;CACnE;CACA,IAAI,OAAO,KAAK,eAAe,UAAU,IAAI,aAAa,KAAK;CAC/D,IAAI,KAAK,WAAW,cAAc,KAAA,GAChC,IAAI,mBAAmB,KAAK,WAAW;CAEzC,IAAI,KAAK,WAAW,WAAW,KAAA,GAAW,IAAI,gBAAgB,KAAK,WAAW;CAC9E,IAAI,KAAK,WAAW,eAAe,KAAA,GACjC,IAAI,qBAAqB,KAAK,WAAW;CAE3C,IAAI,KAAK,WAAW,gBAAgB,SAAS,YAAY,QAAQ,UAAU,GACzE,IAAI,qBAAqB,KAAK,WAAW,QAAQ,KAAK,WAAW,UAAU;CAE7E,IAAI,YAAY,QAAQ,QAAQ,UAAU,KAAA,KAAa,QAAQ,QAAQ,KACrE,IAAI,mBAAmB,UAAU,QAAQ;CAG3C,MAAM,UAAsB;EAC1B;EACA,GAAI,QAAQ,cAAc,EAAE,aAAa,QAAQ,YAAY,IAAI,CAAC;CACpE;CACA,IAAI,QAAQ,UAAU,KAAA,GACpB,IAAI,QAAQ,aAAa,WAAW,QAAQ,eAAe,QAAQ;MAC9D,QAAQ,cAAc,QAAQ;CAGrC,OAAO,kBAAkB;EACvB,OAAO,QAAQ;EACf,cAAc,QAAQ;EACtB,aAAa,QAAQ;EACrB,MAAM,QAAQ,QAAQ,KAAK;EAC3B,OAAO,QAAQ;EACf,YAAY,QAAQ;EACpB,YAAY,QAAQ;EACpB,WAAW,QAAQ;EACnB,QAAQ,KAAK;EACb;EACA;EACA,YAAY,EAAE,GAAG,KAAK,WAAW;EACjC,iBAAiB,UAAU;EAC3B,GAAI,UAAU,wBACV,EAAE,uBAAuB,UAAU,sBAAsB,IACzD,CAAC;EACL;EACA,UAAU,QAAQ;EAClB,YAAY,QAAQ,cAAc,KAAK;EACvC,GAAI,QAAQ,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;CACvE,CAAC;AACH;;;;;;AAOA,SAAgB,2BACd,MACgB;CAChB,IAAI,CAAC,OAAO,SAAS,KAAK,OAAO,KAAK,KAAK,UAAU,GACnD,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,sBAAsB;CAEtE,MAAM,aAAa,KAAK;CACxB,IAAI,CAAC,cAAc,OAAO,eAAe,UACvC,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,wBAAwB;CAExE,IAAI,WAAW,SAAS,cAAc;EACpC,IAAI,WAAW,QAAQ,MACrB,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,wCAAwC;EAExF,OAAO;GAAE,MAAM;GAAc,KAAK;EAAK;CACzC;CACA,IACG,WAAW,SAAS,cAAc,WAAW,SAAS,eACvD,CAAC,OAAO,SAAS,WAAW,GAAG,KAC/B,WAAW,MAAM,GAEjB,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,6BAA6B;CAE7E,IAAI,WAAW,QAAQ,KAAK,SAC1B,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,+CAA+C;CAE/F,OAAO;EAAE,MAAM,WAAW;EAAM,KAAK,WAAW;CAAI;AACtD;AAEA,SAAgB,8BACd,MAC+B;CAC/B,IAAI,KAAK,eAAe,YACtB,OAAO;EACL,iBAAiB;EACjB,qBAAqB;EACrB,GAAI,KAAK,QAAQ,EAAE,uBAAuB,KAAK,MAAM,IAAI,CAAC;CAC5D;CAEF,IAAI,KAAK,eAAe,SACtB,OAAO;EACL,iBAAiB;EACjB,qBAAqB;EACrB,iBAAiB;CACnB;CAEF,IAAI,CAAC,KAAK,OACR,OAAO;EAAE,iBAAiB;EAAa,qBAAqB;CAAE;CAEhE,OAAO;EACL,iBAAiB;EACjB,wBAAwB;CAC1B;AACF;;;;;;;;AASA,SAAgB,2BACd,MAC+B;CAC/B,IAAI,KAAK,eAAe,YACtB,OAAO;EAAE,uBAAuB,CAAC;EAAG,cAAc,CAAC;EAAG,KAAK,CAAC;CAAE;CAGhE,MAAM,WAAmD,CAAC;CAC1D,MAAM,wBAAoD,CAAC;CAC3D,MAAM,kCAAkB,IAAI,IAAsB;CAClD,MAAM,aAAuB,CAAC;CAC9B,MAAM,QAAkB,CAAC;CACzB,MAAM,eAAe,IAAI,IACvB,KAAK,eAAe,UAAU,CAAC,KAAK,cAAc,eAAe,IAAI,CAAC,CACxE;CACA,MAAM,MAA8B,CAAC;CAErC,KAAK,MAAM,CAAC,WAAW,UAAU,OAAO,QAAQ,KAAK,WAAW,GAAG;EACjE,MAAM,kBAAmB,MAAmC;EAC5D,IACE,OAAO,oBAAoB,YAC3B,oBAAoB,QACpB,MAAM,QAAQ,eAAe,GAE7B,MAAM,IAAI,6BAA6B,SAAS;EAElD,MAAM,mBAAmB,OAAO,OAAO,MAAM,UAAU,CAAC,CAAC,MAAM,OAAO,QAAQ;EAC9E,IAAI,MAAM,UAAU,CAAC,OAAO,SAAS,MAAM,SAAS,KAAK,CAAC,kBAAkB;GAC1E,aAAa,IAAI,SAAS;GAC1B;EACF;EAEA,WAAW,KAAK,MAAM,SAAS;EAC/B,sBAAsB,aAAa;EACnC,MAAM,aAAa,EAAE,GAAG,MAAM,WAAW;EACzC,SAAS,aAAa;EACtB,KAAK,MAAM,CAAC,WAAW,UAAU,OAAO,QAAQ,UAAU,GAAG;GAC3D,IAAI,GAAG,UAAU,GAAG,eAAe;GACnC,MAAM,SAAS,gBAAgB,IAAI,SAAS,KAAK,CAAC;GAClD,OAAO,KAAK,KAAK;GACjB,gBAAgB,IAAI,WAAW,MAAM;EACvC;EACA,IAAI,MAAM,OAAO,MAAM,KAAK,GAAG,UAAU,IAAI,MAAM,OAAO;EAC1D,KAAK,MAAM,eAAe,MAAM,gBAAgB,CAAC,GAC/C,aAAa,IAAI,GAAG,UAAU,GAAG,aAAa;CAElD;CAEA,IAAI,aAAa,OAAO,GAAG,IAAI,oBAAoB,aAAa;CAChE,MAAM,qBAAqB,CAAC,GAAG,YAAY,CAAC,CAAC,KAAK;CAClD,IAAI,WAAW,WAAW,GACxB,OAAO;EACL;EACA,cAAc;EACd;CACF;CAGF,MAAM,YAAY,KAAK,UAAU;CACjC,MAAM,aAAa,OAAO,YACxB,CAAC,GAAG,gBAAgB,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,WAAW,YAAY,CAAC,WAAW,KAAK,MAAM,CAAC,CAAC,CACvF;CACA,MAAM,WACJ,KAAK,UAAU,KAAA,KAAa,KAAK,eAAe,KAAA,KAAa,aAAa,SAAS;CACrF,IAAI,UAAU,IAAI,YAAY;CAE9B,OAAO;EACL,GAAI,WAAW,EAAE,OAAO,UAAU,IAAI,CAAC;EACvC;EACA;EACA,cAAc;EACd,aAAa;GACX;GACA;GACA;GACA,GAAI,mBAAmB,SAAS,IAAI,EAAE,cAAc,mBAAmB,IAAI,CAAC;GAC5E,GAAI,MAAM,SAAS,IAAI,EAAE,OAAO,MAAM,KAAK,KAAK,EAAE,IAAI,CAAC;EACzD;CACF;AACF;;AAGA,SAAgB,sBACd,MACoB;CACpB,OAAO,2BAA2B,IAAI,CAAC,CAAC;AAC1C;;AAGA,SAAgB,4BACd,MACwC;CACxC,OAAO,2BAA2B,IAAI,CAAC,CAAC,aAAa,YAAY,CAAC;AACpE;AAEA,SAAS,cAAc,SAAqE;CAC1F,MAAM,SAAiC,CAAC;CACxC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,WAAW,CAAC,CAAC,GACrD,IAAI,OAAO,SAAS,KAAK,GAAG,OAAO,OAAO;CAE5C,OAAO;AACT;AAEA,SAAS,KAAK,QAA0B;CACtC,OAAO,OAAO,QAAQ,KAAK,UAAU,MAAM,OAAO,CAAC,IAAI,OAAO;AAChE"}
@@ -1,5 +1,5 @@
1
1
  import { s as ValidationError } from "./errors-Dngq5h35.js";
2
- import { d as validateAgentProfileCell } from "./agent-profile-cell-0gSi5ffD.js";
2
+ import { d as validateAgentProfileCell } from "./agent-profile-cell-Cv6UA-W_.js";
3
3
  import { t as FAILURE_CLASSES } from "./schema-CSf6qWgZ.js";
4
4
  import { n as observedScore } from "./reward-nw2xZGZG.js";
5
5
  //#region src/run-record.ts
@@ -255,4 +255,4 @@ function validSnapshotDate(year, month, day) {
255
255
  //#endregion
256
256
  export { parseRunRecordSafe as a, validateRunRecord as c, modelHasSnapshot as i, UNKNOWN_MODEL as n, roundTripRunRecord as o, isRunRecord as r, runTaskScore as s, RunRecordValidationError as t };
257
257
 
258
- //# sourceMappingURL=run-record-DQpSf7t-.js.map
258
+ //# sourceMappingURL=run-record-DualPTn2.js.map