@tangle-network/agent-eval 0.180.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. package/CHANGELOG.md +60 -0
  2. package/README.md +119 -159
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +3 -3
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +4 -7
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +10 -9
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  38. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  39. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  40. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  41. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  42. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  43. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  44. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  45. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  46. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  47. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  48. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  49. package/dist/experiment/index.d.ts +27 -477
  50. package/dist/experiment/index.d.ts.map +1 -1
  51. package/dist/experiment/index.js +95 -559
  52. package/dist/experiment/index.js.map +1 -1
  53. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  54. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  55. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  56. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  57. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  58. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  59. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  60. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  61. package/dist/hosted/index.d.ts +2 -2
  62. package/dist/hosted/index.d.ts.map +1 -1
  63. package/dist/hosted/index.js +1 -1
  64. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  65. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  66. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  67. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  68. package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
  69. package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  70. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  71. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  72. package/dist/index.d.ts +28 -28
  73. package/dist/index.js +24 -15
  74. package/dist/index.js.map +1 -1
  75. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  76. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  77. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  78. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  79. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  80. package/dist/journal-Cs9f7385.js.map +1 -0
  81. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  82. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  83. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  84. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  85. package/dist/ledger-core/index.d.ts +1 -1
  86. package/dist/ledger-core/index.js +1 -1
  87. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  88. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  89. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  90. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  91. package/dist/meta-eval/index.d.ts +138 -7
  92. package/dist/meta-eval/index.d.ts.map +1 -1
  93. package/dist/meta-eval/index.js +245 -97
  94. package/dist/meta-eval/index.js.map +1 -1
  95. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  96. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  97. package/dist/multishot/golden/index.d.ts +1 -1
  98. package/dist/multishot/index.d.ts +2 -2
  99. package/dist/openapi.json +1 -1
  100. package/dist/outcome-store-BXlkwMPR.js +131 -0
  101. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  102. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  103. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  104. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  105. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  106. package/dist/pipelines/index.js +1 -1
  107. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  108. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  109. package/dist/profile-cell.d.ts +1 -1
  110. package/dist/profile-cell.js +1 -1
  111. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  112. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  113. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  114. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  115. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  116. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  117. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  118. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  119. package/dist/reporting.d.ts +4 -4
  120. package/dist/reporting.js +3 -3
  121. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  122. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  123. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  124. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  125. package/dist/rl.d.ts +53 -99
  126. package/dist/rl.d.ts.map +1 -1
  127. package/dist/rl.js +182 -169
  128. package/dist/rl.js.map +1 -1
  129. package/dist/rollout/index.d.ts +1 -1
  130. package/dist/rollout/index.js +2 -2
  131. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  132. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  133. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  134. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  135. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  136. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  137. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  138. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  139. package/dist/run-record-Br-Yzt_k.js +464 -0
  140. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  141. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  142. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  143. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  144. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  145. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  146. package/dist/sequential-DAsyV2T9.js.map +1 -0
  147. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  148. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  149. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  150. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  151. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  152. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  153. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  154. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  155. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  156. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  157. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  158. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  159. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  160. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  161. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  162. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  163. package/dist/trace-repair/index.d.ts +2 -2
  164. package/dist/traces.d.ts +6 -6
  165. package/dist/traces.js +1 -1
  166. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  167. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  168. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  169. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  170. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  171. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  172. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  173. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  174. package/dist/wire/index.d.ts +2 -2
  175. package/docs/adapters-observability.md +14 -0
  176. package/docs/campaign-proposers.md +86 -128
  177. package/docs/charter.md +108 -112
  178. package/docs/concepts.md +157 -69
  179. package/docs/design/mlbenchmarks-book-review.md +440 -0
  180. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  181. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  182. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  183. package/docs/design/self-improvement-evidence-audit.md +263 -0
  184. package/docs/design.md +2 -1
  185. package/docs/eval-surface-map.md +95 -42
  186. package/docs/evaluation-integrity.md +220 -0
  187. package/docs/experiment.md +111 -55
  188. package/docs/feature-guide.md +5 -6
  189. package/docs/hosted-ingest-spec.md +4 -11
  190. package/docs/insight-report.md +187 -455
  191. package/docs/outcome-validity.md +182 -0
  192. package/docs/product-eval-adoption.md +1 -2
  193. package/docs/research-report-methodology.md +7 -7
  194. package/docs/search-history-receipts.md +8 -0
  195. package/docs/statistical-evidence.md +129 -0
  196. package/docs/verdicts.md +76 -49
  197. package/package.json +1 -1
  198. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  199. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  200. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  201. package/dist/client-BlLY6o2w.js.map +0 -1
  202. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  203. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  204. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  205. package/dist/index-BxWvILU8.d.ts.map +0 -1
  206. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  207. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  208. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  209. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  210. package/dist/outcome-store-ChBKlTd_.js +0 -75
  211. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  212. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  213. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  214. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  215. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  216. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  217. package/dist/run-record-CR63CpHK.js +0 -216
  218. package/dist/run-record-CR63CpHK.js.map +0 -1
  219. package/dist/sequential-B5gXgcyp.js.map +0 -1
  220. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  221. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
@@ -1,12 +1,13 @@
1
1
  import { n as CaptureIntegrityError, s as ValidationError } from "../errors-Dngq5h35.js";
2
- import { a as hashCanonical } from "../canonical-DPyQ_rpt.js";
3
- import { i as makeRng } from "../internal-BMFSR8Ns.js";
2
+ import { a as hashCanonical, i as compareCodeUnits } from "../canonical-DPyQ_rpt.js";
4
3
  import { t as mulberry32 } from "../random-Dn5fPWkt.js";
5
- import { a as spearmanR, r as pearsonR } from "../descriptive-1V17A-qa.js";
4
+ import { a as selfPreference, i as positionalBias, n as calibrateJudgeContinuous, o as verbosityBias, r as continuousAgreement, t as calibrateJudge } from "../judge-calibration-DYmaBtJr.js";
6
5
  import { u as runMetricExtractor } from "../query-D1nLIKt7.js";
6
+ import { r as computeInterval } from "../ast-CP9ae9B0.js";
7
7
  import { t as analyzeSeries } from "../series-convergence-CjO2QdRW.js";
8
- import { t as rubricPredictiveValidity } from "../rubric-predictive-validity-2D5Gw9z9.js";
9
- import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "../outcome-store-ChBKlTd_.js";
8
+ import { a as reduceOutcomeMetric, i as hasVariation, n as assertUniqueObservationIds, o as validateObservationOptions, r as correlationSummary, t as rubricPredictiveValidity } from "../rubric-predictive-validity-CCK-1B7w.js";
9
+ import { n as InMemoryOutcomeStore, r as OutcomeStoreError, t as FileSystemOutcomeStore } from "../outcome-store-BXlkwMPR.js";
10
+ import { z } from "zod";
10
11
  //#region src/meta-eval/calibration.ts
11
12
  /**
12
13
  * Calibration curve — binned "if eval says X, what does reality show?"
@@ -17,7 +18,15 @@ import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "../outco
17
18
  * mean outcome per bucket, reports expected-calibration-error (ECE).
18
19
  */
19
20
  async function calibrationCurve(traceStore, outcomeStore, evalMetric, outcomeMetric, options = {}) {
21
+ const settings = {
22
+ ...options,
23
+ range: options.range === void 0 ? void 0 : { ...options.range }
24
+ };
25
+ validateCalibrationRequest(evalMetric.id, outcomeMetric, settings);
26
+ const extract = evalMetric.extract ?? runMetricExtractor(evalMetric.id);
27
+ const metricId = evalMetric.id;
20
28
  const runs = await traceStore.listRuns();
29
+ assertUniqueObservationIds(runs.map((run) => run.runId), "runId");
21
30
  const outcomes = await outcomeStore.list();
22
31
  const byRun = /* @__PURE__ */ new Map();
23
32
  for (const o of outcomes) {
@@ -25,15 +34,14 @@ async function calibrationCurve(traceStore, outcomeStore, evalMetric, outcomeMet
25
34
  arr.push(o);
26
35
  byRun.set(o.runId, arr);
27
36
  }
28
- const extract = evalMetric.extract ?? runMetricExtractor(evalMetric.id);
29
37
  const pairs = [];
30
38
  for (const run of runs) {
31
39
  const os = byRun.get(run.runId);
32
40
  if (!os?.length) continue;
33
41
  const x = await extract(run, traceStore);
34
42
  if (x === null || !Number.isFinite(x)) continue;
35
- const y = [...os].sort((a, b) => b.capturedAt - a.capturedAt)[0].metrics[outcomeMetric];
36
- if (typeof y !== "number" || !Number.isFinite(y)) continue;
43
+ const y = reduceOutcomeMetric(os, outcomeMetric, "latest");
44
+ if (y === null) continue;
37
45
  pairs.push({
38
46
  x,
39
47
  y
@@ -43,35 +51,44 @@ async function calibrationCurve(traceStore, outcomeStore, evalMetric, outcomeMet
43
51
  return calibrationFromPairs(pairs.map((p) => ({
44
52
  evalScore: p.x,
45
53
  outcome: p.y
46
- })), evalMetric.id, outcomeMetric, options);
54
+ })), metricId, outcomeMetric, settings);
47
55
  }
56
+ /** Measure already joined observations without constructing trace and outcome stores. */
48
57
  function calibrationFromPairs(inputPairs, evalMetric, outcomeMetric, options = {}) {
49
- const pairs = inputPairs.filter((pair) => Number.isFinite(pair.evalScore) && Number.isFinite(pair.outcome));
58
+ validateCalibrationRequest(evalMetric, outcomeMetric, options);
59
+ for (const [index, pair] of inputPairs.entries()) if (pair === null || typeof pair !== "object" || !Number.isFinite(pair.evalScore) || !Number.isFinite(pair.outcome)) throw new Error(`calibration pair ${index} must contain finite evalScore and outcome values`);
60
+ const pairs = inputPairs;
50
61
  if (pairs.length < 2) return null;
51
62
  const numBins = options.bins ?? 10;
52
63
  const binning = options.binning ?? "equal-width";
53
64
  const xs = pairs.map((p) => p.evalScore);
54
65
  const lo = options.range?.lo ?? Math.min(...xs);
55
66
  const hi = options.range?.hi ?? Math.max(...xs);
67
+ const span = hi - lo;
68
+ if (!Number.isFinite(span)) throw new Error("calibration range span must be finite");
69
+ const clipped = pairs.map((pair) => ({
70
+ ...pair,
71
+ evalScore: Math.min(hi, Math.max(lo, pair.evalScore))
72
+ }));
56
73
  const bins = [];
57
- if (binning === "equal-frequency") {
58
- const sorted = [...pairs].sort((a, b) => a.evalScore - b.evalScore);
59
- const perBin = Math.max(1, Math.floor(sorted.length / numBins));
60
- for (let i = 0; i < sorted.length; i += perBin) {
61
- const chunk = sorted.slice(i, i + perBin);
62
- if (chunk.length === 0) continue;
63
- bins.push(toBin(chunk));
74
+ if (span === 0 || clipped.every((pair) => pair.evalScore === clipped[0].evalScore)) bins.push(toBin(clipped));
75
+ else if (binning === "equal-frequency") {
76
+ const sorted = [...clipped].sort((a, b) => a.evalScore - b.evalScore);
77
+ const count = Math.min(numBins, sorted.length);
78
+ for (let i = 0; i < count; i++) {
79
+ const start = Math.floor(i * sorted.length / count);
80
+ const end = Math.floor((i + 1) * sorted.length / count);
81
+ bins.push(toBin(sorted.slice(start, end)));
64
82
  }
65
83
  } else {
66
- const width = (hi - lo) / numBins;
67
- if (width === 0) return null;
68
- for (let i = 0; i < numBins; i++) {
69
- const binLo = lo + i * width;
70
- const binHi = i === numBins - 1 ? hi + 1e-9 : lo + (i + 1) * width;
71
- const chunk = pairs.filter((p) => p.evalScore >= binLo && p.evalScore < binHi);
72
- if (chunk.length === 0) continue;
73
- bins.push(toBin(chunk, binLo, binHi));
84
+ const groups = /* @__PURE__ */ new Map();
85
+ for (const pair of clipped) {
86
+ const index = Math.min(numBins - 1, Math.floor((pair.evalScore - lo) / span * numBins));
87
+ const group = groups.get(index) ?? [];
88
+ group.push(pair);
89
+ groups.set(index, group);
74
90
  }
91
+ for (const [index, chunk] of [...groups].sort(([a], [b]) => a - b)) bins.push(toBin(chunk, lo + span * (index / numBins), lo + span * ((index + 1) / numBins)));
75
92
  }
76
93
  const total = bins.reduce((a, b) => a + b.n, 0);
77
94
  const ece = bins.reduce((a, b) => a + b.n / total * b.gap, 0);
@@ -100,22 +117,41 @@ function toBin(chunk, lower, upper) {
100
117
  };
101
118
  }
102
119
  function mean(xs) {
103
- return xs.reduce((a, b) => a + b, 0) / xs.length;
120
+ return xs.reduce((sum, value) => sum + value / xs.length, 0);
121
+ }
122
+ function validateCalibrationRequest(evalMetric, outcomeMetric, options) {
123
+ assertUniqueObservationIds([evalMetric], "eval metric");
124
+ assertUniqueObservationIds([outcomeMetric], "outcome metric");
125
+ if (evalMetric.trim() !== evalMetric || outcomeMetric.trim() !== outcomeMetric) throw new Error("calibration metric identities must not have surrounding whitespace");
126
+ if (options.bins !== void 0 && (!Number.isSafeInteger(options.bins) || options.bins < 1)) throw new Error("calibration bins must be a positive safe integer");
127
+ if (options.binning !== void 0 && !["equal-width", "equal-frequency"].includes(options.binning)) throw new Error("calibration binning must be equal-width or equal-frequency");
128
+ if (options.range !== void 0 && (!Number.isFinite(options.range.lo) || !Number.isFinite(options.range.hi) || !Number.isFinite(options.range.hi - options.range.lo) || options.range.hi < options.range.lo)) throw new Error("calibration range must have finite ordered bounds");
104
129
  }
105
130
  //#endregion
106
131
  //#region src/meta-eval/correlation-study.ts
107
132
  /**
108
133
  * Correlation study — "does our eval score predict real-world outcomes?"
109
134
  *
110
- * This is the load-bearing signal. Takes a TraceStore + OutcomeStore,
111
- * joins on runId, computes Pearson + Spearman + bootstrap CI for every
112
- * (evalMetric, outcomeMetric) pair the caller declares.
113
- *
114
- * Without this number the framework is ornamental. With it and r > 0.6
115
- * the framework is a moat — no other agent-eval tool publishes one.
135
+ * Joins traces and outcomes by runId and reports descriptive correlations.
136
+ * Independent runs are the bootstrap observation unit.
137
+ * Association alone does not establish causation or held-out predictive performance.
116
138
  */
117
139
  async function correlationStudy(traceStore, outcomeStore, evalMetrics, outcomeMetricNames, options = {}) {
140
+ const reduction = options.reduction ?? "latest";
141
+ const iterations = options.bootstrapIterations ?? 500;
142
+ const seed = options.seed;
143
+ validateObservationOptions(reduction, iterations, seed);
144
+ assertUniqueObservationIds(evalMetrics.map((metric) => metric.id), "eval metric");
145
+ assertUniqueObservationIds(outcomeMetricNames, "outcome metric");
146
+ const maxLag = options.maxCaptureLagMs ?? Infinity;
147
+ if (maxLag < 0 || Number.isNaN(maxLag)) throw new Error("maxCaptureLagMs must be nonnegative");
148
+ const extractors = evalMetrics.map((metric) => ({
149
+ ...metric,
150
+ extract: metric.extract ?? runMetricExtractor(metric.id)
151
+ }));
152
+ const metricNames = [...outcomeMetricNames];
118
153
  const runs = await traceStore.listRuns();
154
+ assertUniqueObservationIds(runs.map((run) => run.runId), "runId");
119
155
  const outcomes = await outcomeStore.list(options.outcomeFilter);
120
156
  const outcomesByRun = /* @__PURE__ */ new Map();
121
157
  for (const o of outcomes) {
@@ -123,10 +159,8 @@ async function correlationStudy(traceStore, outcomeStore, evalMetrics, outcomeMe
123
159
  arr.push(o);
124
160
  outcomesByRun.set(o.runId, arr);
125
161
  }
126
- const reduction = options.reduction ?? "latest";
127
- const maxLag = options.maxCaptureLagMs ?? Infinity;
128
162
  const pairs = [];
129
- for (const em of evalMetrics) for (const om of outcomeMetricNames) pairs.push({
163
+ for (const em of extractors) for (const om of metricNames) pairs.push({
130
164
  evalMetric: em.id,
131
165
  outcomeMetric: om,
132
166
  xs: [],
@@ -140,91 +174,205 @@ async function correlationStudy(traceStore, outcomeStore, evalMetrics, outcomeMe
140
174
  skipped++;
141
175
  continue;
142
176
  }
143
- const eligible = os.filter((o) => o.capturedAt - run.startedAt <= maxLag);
177
+ const eligible = os.filter((o) => {
178
+ const lag = o.capturedAt - run.startedAt;
179
+ return lag >= 0 && lag <= maxLag;
180
+ });
144
181
  if (eligible.length === 0) {
145
182
  skipped++;
146
183
  continue;
147
184
  }
148
- for (const em of evalMetrics) {
149
- const x = await (em.extract ?? runMetricExtractor(em.id))(run, traceStore);
185
+ let joinedThisRun = false;
186
+ for (const em of extractors) {
187
+ const x = await em.extract(run, traceStore);
150
188
  if (x === null || !Number.isFinite(x)) continue;
151
- for (const om of outcomeMetricNames) {
152
- const values = eligible.map((o) => o.metrics[om]).filter((v) => typeof v === "number" && Number.isFinite(v));
153
- if (values.length === 0) continue;
154
- const y = reduce(values, reduction, eligible);
189
+ for (const om of metricNames) {
190
+ const y = reduceOutcomeMetric(eligible, om, reduction);
155
191
  if (y === null) continue;
156
192
  const pair = pairs.find((p) => p.evalMetric === em.id && p.outcomeMetric === om);
157
193
  pair.xs.push(x);
158
194
  pair.ys.push(y);
195
+ joinedThisRun = true;
159
196
  }
160
197
  }
161
- joined++;
198
+ if (joinedThisRun) joined++;
199
+ else skipped++;
162
200
  }
163
- return {
164
- pairs: pairs.filter((p) => p.xs.length >= 3).map((p) => {
165
- const pearson = pearsonR(p.xs, p.ys);
166
- const spearman = spearmanR(p.xs, p.ys);
167
- const pearsonCi95 = bootstrapPearsonCi(p.xs, p.ys, options.bootstrapIterations ?? 500, options.seed);
168
- const verdict = Math.abs(pearson) >= .7 ? "strong" : Math.abs(pearson) >= .4 ? "moderate" : "weak";
169
- return {
201
+ const excludedPairs = [];
202
+ const results = [];
203
+ for (const p of pairs) {
204
+ const reason = p.xs.length < 3 ? "insufficient_samples" : !hasVariation(p.xs) ? "constant_eval_metric" : !hasVariation(p.ys) ? "constant_outcome" : null;
205
+ if (reason !== null) {
206
+ excludedPairs.push({
170
207
  evalMetric: p.evalMetric,
171
208
  outcomeMetric: p.outcomeMetric,
172
209
  n: p.xs.length,
173
- pearson,
174
- spearman,
175
- pearsonCi95,
176
- verdict
177
- };
178
- }),
210
+ reason
211
+ });
212
+ continue;
213
+ }
214
+ const summary = correlationSummary(p.xs, p.ys, iterations, seed);
215
+ const verdict = Math.abs(summary.pearson) >= .7 ? "strong" : Math.abs(summary.pearson) >= .4 ? "moderate" : "weak";
216
+ results.push({
217
+ evalMetric: p.evalMetric,
218
+ outcomeMetric: p.outcomeMetric,
219
+ n: p.xs.length,
220
+ ...summary,
221
+ verdict
222
+ });
223
+ }
224
+ return {
225
+ pairs: results,
226
+ excludedPairs,
179
227
  joinedSamples: joined,
180
228
  skippedRuns: skipped
181
229
  };
182
230
  }
183
- function reduce(values, kind, outcomes) {
184
- if (values.length === 0) return null;
185
- if (kind === "mean") return values.reduce((a, b) => a + b, 0) / values.length;
186
- if (kind === "max") return Math.max(...values);
187
- const latest = [...outcomes].sort((a, b) => b.capturedAt - a.capturedAt)[0];
188
- if (!latest) return null;
189
- const latestKey = Object.keys(latest.metrics)[0];
190
- const v = latestKey !== void 0 ? latest.metrics[latestKey] : void 0;
191
- const paired = outcomes.map((o) => {
192
- const k = Object.keys(o.metrics)[0];
193
- return {
194
- at: o.capturedAt,
195
- v: k !== void 0 ? values.find((x) => o.metrics[k] === x) : void 0
231
+ //#endregion
232
+ //#region src/meta-eval/evaluator-admission.ts
233
+ const identity = z.string().min(1).refine((value) => value.trim() === value);
234
+ const policySchema = z.object({
235
+ confidence: z.number().finite().gt(0).lt(1),
236
+ maxFalseAcceptanceRate: z.number().finite().min(0).lt(1),
237
+ maxFalseRejectionRate: z.number().finite().min(0).lt(1)
238
+ }).strict();
239
+ const observationSchema = z.object({
240
+ id: identity,
241
+ independentUnitId: identity,
242
+ evidenceRef: identity,
243
+ expected: z.enum(["accept", "reject"]),
244
+ observed: z.enum([
245
+ "accept",
246
+ "reject",
247
+ "unknown"
248
+ ]),
249
+ exposure: z.enum(["fresh", "development"])
250
+ }).strict();
251
+ const inputSchema = z.object({
252
+ evaluatorDigest: z.string().regex(/^sha256:[a-f0-9]{64}$/),
253
+ population: identity,
254
+ samplingFrame: identity,
255
+ authority: z.object({
256
+ evaluatorAuthorId: identity,
257
+ auditorId: identity,
258
+ independenceEvidenceRef: identity
259
+ }).strict(),
260
+ policy: policySchema,
261
+ observations: z.array(observationSchema)
262
+ }).strict();
263
+ function errorRate(rows, expected, limit, level) {
264
+ const cases = rows.filter((row) => row.expected === expected);
265
+ const units = /* @__PURE__ */ new Map();
266
+ for (const row of cases) {
267
+ const unit = units.get(row.independentUnitId) ?? {
268
+ error: false,
269
+ unknown: false
196
270
  };
197
- }).filter((p) => p.v !== void 0);
198
- if (paired.length === 0) return v ?? null;
199
- return paired.sort((a, b) => b.at - a.at)[0]?.v ?? null;
200
- }
201
- function bootstrapPearsonCi(xs, ys, iterations, seed) {
202
- const n = xs.length;
203
- if (n < 3) return {
204
- lower: NaN,
205
- upper: NaN
271
+ unit.unknown ||= row.observed === "unknown";
272
+ unit.error ||= row.observed !== "unknown" && row.observed !== expected;
273
+ units.set(row.independentUnitId, unit);
274
+ }
275
+ const n = units.size;
276
+ const errorUnits = [...units.values()].filter((unit) => unit.error).length;
277
+ const unresolvedUnits = [...units.values()].filter((unit) => unit.unknown && !unit.error).length;
278
+ const unknownCases = cases.filter((row) => row.observed === "unknown").length;
279
+ const interval = n === 0 ? null : {
280
+ lower: computeInterval({
281
+ kind: "clopper-pearson",
282
+ level
283
+ }, {
284
+ kind: "binomial",
285
+ successes: errorUnits,
286
+ trials: n
287
+ }).lower,
288
+ upper: computeInterval({
289
+ kind: "clopper-pearson",
290
+ level
291
+ }, {
292
+ kind: "binomial",
293
+ successes: errorUnits + unresolvedUnits,
294
+ trials: n
295
+ }).upper,
296
+ level
206
297
  };
207
- const rng = makeRng(seed, xs, ys);
208
- const rs = [];
209
- for (let b = 0; b < iterations; b++) {
210
- const rx = new Array(n);
211
- const ry = new Array(n);
212
- for (let i = 0; i < n; i++) {
213
- const idx = Math.floor(rng() * n);
214
- rx[i] = xs[idx];
215
- ry[i] = ys[idx];
216
- }
217
- const r = pearsonR(rx, ry);
218
- if (Number.isFinite(r)) rs.push(r);
298
+ const verdict = interval && interval.lower > limit ? "fail" : interval && interval.upper <= limit ? "pass" : "inconclusive";
299
+ return {
300
+ cases: cases.length,
301
+ independentUnits: n,
302
+ errorUnits,
303
+ unresolvedUnits,
304
+ unknownCases,
305
+ errorRate: n === 0 || unresolvedUnits > 0 ? null : errorUnits / n,
306
+ interval,
307
+ limit,
308
+ verdict
309
+ };
310
+ }
311
+ /**
312
+ * Audit frozen judgments using independent source units and exact binomial bounds.
313
+ * A unit fails a class when any control in that class is misjudged.
314
+ * The execution owner enforces auditor separation, fresh sampling, and evidence authenticity.
315
+ */
316
+ function auditEvaluator(input) {
317
+ const parsed = inputSchema.safeParse(input);
318
+ if (!parsed.success) throw new ValidationError(`invalid evaluator audit: ${parsed.error.message}`);
319
+ const audit = parsed.data;
320
+ if (audit.authority.evaluatorAuthorId === audit.authority.auditorId) throw new ValidationError("evaluator admission requires a separate declared audit authority");
321
+ const ids = /* @__PURE__ */ new Set();
322
+ for (const row of audit.observations) {
323
+ if (ids.has(row.id)) throw new ValidationError(`duplicate evaluator audit observation '${row.id}'`);
324
+ ids.add(row.id);
325
+ }
326
+ audit.observations.sort((a, b) => compareCodeUnits(a.id, b.id));
327
+ const developmentUnits = new Set(audit.observations.filter((row) => row.exposure === "development").map((row) => row.independentUnitId));
328
+ const eligible = audit.observations.filter((row) => !developmentUnits.has(row.independentUnitId));
329
+ const exclusions = audit.observations.filter((row) => developmentUnits.has(row.independentUnitId)).map((row) => ({
330
+ id: row.id,
331
+ independentUnitId: row.independentUnitId,
332
+ reason: "development-exposure"
333
+ }));
334
+ const intervalConfidence = 1 - (1 - audit.policy.confidence) / 2;
335
+ if (intervalConfidence >= 1) throw new ValidationError("evaluator audit confidence is too close to one for simultaneous intervals");
336
+ const falseAcceptance = errorRate(eligible, "reject", audit.policy.maxFalseAcceptanceRate, intervalConfidence);
337
+ const falseRejection = errorRate(eligible, "accept", audit.policy.maxFalseRejectionRate, intervalConfidence);
338
+ const rates = [falseAcceptance, falseRejection];
339
+ const verdict = rates.some((rate) => rate.verdict === "fail") ? "reject" : rates.every((rate) => rate.verdict === "pass") ? "admit" : "inconclusive";
340
+ const reasons = [];
341
+ for (const [name, rate] of [["false acceptance", falseAcceptance], ["false rejection", falseRejection]]) {
342
+ if (rate.independentUnits === 0) reasons.push(`${name}: no eligible independent units`);
343
+ else if (rate.unknownCases > 0) reasons.push(`${name}: ${rate.unknownCases} judgments are unknown`);
344
+ if (rate.verdict === "fail") reasons.push(`${name}: lower error bound exceeds ${rate.limit}`);
345
+ else if (rate.verdict === "inconclusive" && rate.interval) reasons.push(`${name}: upper error bound does not establish the required limit`);
219
346
  }
220
- rs.sort((a, b) => a - b);
221
- if (rs.length === 0) return {
222
- lower: NaN,
223
- upper: NaN
347
+ if (verdict === "admit") reasons.push("both error bounds meet the registered limits, including the worst case for unknown judgments");
348
+ const body = {
349
+ evaluatorDigest: audit.evaluatorDigest,
350
+ policyDigest: hashCanonical(audit.policy),
351
+ inputDigest: hashCanonical(audit),
352
+ population: audit.population,
353
+ samplingFrame: audit.samplingFrame,
354
+ authority: audit.authority,
355
+ policy: audit.policy,
356
+ confidence: audit.policy.confidence,
357
+ intervalConfidence,
358
+ verdict,
359
+ reasons,
360
+ observations: audit.observations,
361
+ coverage: {
362
+ cases: audit.observations.length,
363
+ independentUnits: new Set(audit.observations.map((row) => row.independentUnitId)).size,
364
+ eligibleCases: eligible.length,
365
+ eligibleIndependentUnits: new Set(eligible.map((row) => row.independentUnitId)).size,
366
+ excludedCases: exclusions.length,
367
+ unknownCases: eligible.filter((row) => row.observed === "unknown").length
368
+ },
369
+ exclusions,
370
+ falseAcceptance,
371
+ falseRejection
224
372
  };
225
373
  return {
226
- lower: rs[Math.floor(.025 * rs.length)],
227
- upper: rs[Math.min(rs.length - 1, Math.floor(.975 * rs.length))]
374
+ ...body,
375
+ reportDigest: hashCanonical(body)
228
376
  };
229
377
  }
230
378
  //#endregion
@@ -982,6 +1130,6 @@ function evalHealthStamp(report) {
982
1130
  };
983
1131
  }
984
1132
  //#endregion
985
- export { FileSystemOutcomeStore, InMemoryOutcomeStore, calibrationCurve, catchRate, correlationStudy, definePlant, evalHealthStamp, fileSentinelStore, inMemorySentinelStore, judgeSentinelReport, perturbEvidence, plantByPerturbation, rubricPredictiveValidity, seedPlants, snapshotFromAgreement, snapshotFromCalibration, snapshotFromSentinelSet, validateSentinelSnapshot };
1133
+ export { FileSystemOutcomeStore, InMemoryOutcomeStore, OutcomeStoreError, auditEvaluator, calibrateJudge, calibrateJudgeContinuous, calibrationCurve, calibrationFromPairs, catchRate, continuousAgreement, correlationStudy, definePlant, evalHealthStamp, fileSentinelStore, inMemorySentinelStore, judgeSentinelReport, perturbEvidence, plantByPerturbation, positionalBias, rubricPredictiveValidity, seedPlants, selfPreference, snapshotFromAgreement, snapshotFromCalibration, snapshotFromSentinelSet, validateSentinelSnapshot, verbosityBias };
986
1134
 
987
1135
  //# sourceMappingURL=index.js.map