@tangle-network/agent-eval 0.180.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. package/CHANGELOG.md +60 -0
  2. package/README.md +119 -159
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +3 -3
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +4 -7
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +10 -9
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  38. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  39. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  40. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  41. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  42. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  43. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  44. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  45. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  46. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  47. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  48. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  49. package/dist/experiment/index.d.ts +27 -477
  50. package/dist/experiment/index.d.ts.map +1 -1
  51. package/dist/experiment/index.js +95 -559
  52. package/dist/experiment/index.js.map +1 -1
  53. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  54. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  55. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  56. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  57. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  58. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  59. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  60. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  61. package/dist/hosted/index.d.ts +2 -2
  62. package/dist/hosted/index.d.ts.map +1 -1
  63. package/dist/hosted/index.js +1 -1
  64. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  65. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  66. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  67. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  68. package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
  69. package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  70. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  71. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  72. package/dist/index.d.ts +28 -28
  73. package/dist/index.js +24 -15
  74. package/dist/index.js.map +1 -1
  75. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  76. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  77. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  78. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  79. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  80. package/dist/journal-Cs9f7385.js.map +1 -0
  81. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  82. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  83. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  84. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  85. package/dist/ledger-core/index.d.ts +1 -1
  86. package/dist/ledger-core/index.js +1 -1
  87. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  88. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  89. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  90. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  91. package/dist/meta-eval/index.d.ts +138 -7
  92. package/dist/meta-eval/index.d.ts.map +1 -1
  93. package/dist/meta-eval/index.js +245 -97
  94. package/dist/meta-eval/index.js.map +1 -1
  95. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  96. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  97. package/dist/multishot/golden/index.d.ts +1 -1
  98. package/dist/multishot/index.d.ts +2 -2
  99. package/dist/openapi.json +1 -1
  100. package/dist/outcome-store-BXlkwMPR.js +131 -0
  101. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  102. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  103. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  104. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  105. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  106. package/dist/pipelines/index.js +1 -1
  107. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  108. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  109. package/dist/profile-cell.d.ts +1 -1
  110. package/dist/profile-cell.js +1 -1
  111. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  112. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  113. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  114. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  115. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  116. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  117. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  118. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  119. package/dist/reporting.d.ts +4 -4
  120. package/dist/reporting.js +3 -3
  121. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  122. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  123. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  124. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  125. package/dist/rl.d.ts +53 -99
  126. package/dist/rl.d.ts.map +1 -1
  127. package/dist/rl.js +182 -169
  128. package/dist/rl.js.map +1 -1
  129. package/dist/rollout/index.d.ts +1 -1
  130. package/dist/rollout/index.js +2 -2
  131. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  132. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  133. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  134. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  135. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  136. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  137. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  138. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  139. package/dist/run-record-Br-Yzt_k.js +464 -0
  140. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  141. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  142. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  143. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  144. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  145. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  146. package/dist/sequential-DAsyV2T9.js.map +1 -0
  147. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  148. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  149. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  150. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  151. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  152. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  153. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  154. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  155. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  156. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  157. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  158. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  159. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  160. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  161. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  162. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  163. package/dist/trace-repair/index.d.ts +2 -2
  164. package/dist/traces.d.ts +6 -6
  165. package/dist/traces.js +1 -1
  166. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  167. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  168. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  169. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  170. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  171. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  172. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  173. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  174. package/dist/wire/index.d.ts +2 -2
  175. package/docs/adapters-observability.md +14 -0
  176. package/docs/campaign-proposers.md +86 -128
  177. package/docs/charter.md +108 -112
  178. package/docs/concepts.md +157 -69
  179. package/docs/design/mlbenchmarks-book-review.md +440 -0
  180. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  181. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  182. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  183. package/docs/design/self-improvement-evidence-audit.md +263 -0
  184. package/docs/design.md +2 -1
  185. package/docs/eval-surface-map.md +95 -42
  186. package/docs/evaluation-integrity.md +220 -0
  187. package/docs/experiment.md +111 -55
  188. package/docs/feature-guide.md +5 -6
  189. package/docs/hosted-ingest-spec.md +4 -11
  190. package/docs/insight-report.md +187 -455
  191. package/docs/outcome-validity.md +182 -0
  192. package/docs/product-eval-adoption.md +1 -2
  193. package/docs/research-report-methodology.md +7 -7
  194. package/docs/search-history-receipts.md +8 -0
  195. package/docs/statistical-evidence.md +129 -0
  196. package/docs/verdicts.md +76 -49
  197. package/package.json +1 -1
  198. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  199. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  200. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  201. package/dist/client-BlLY6o2w.js.map +0 -1
  202. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  203. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  204. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  205. package/dist/index-BxWvILU8.d.ts.map +0 -1
  206. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  207. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  208. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  209. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  210. package/dist/outcome-store-ChBKlTd_.js +0 -75
  211. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  212. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  213. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  214. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  215. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  216. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  217. package/dist/run-record-CR63CpHK.js +0 -216
  218. package/dist/run-record-CR63CpHK.js.map +0 -1
  219. package/dist/sequential-B5gXgcyp.js.map +0 -1
  220. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  221. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
@@ -1 +0,0 @@
1
- {"version":3,"file":"outcome-store-BYHIuO0e.d.ts","names":[],"sources":["../src/meta-eval/outcome-store.ts"],"mappings":";;;;;;;;;;;;;UAaiB;EACf;EACA;;EAEA,SAAS;;EAET,SAAS;;EAET;;UAGe;EACf;EACA;EACA;EACA;IAAU;IAAa;;EACvB;;UAGe;EACf,OAAO,SAAS,oBAAoB;;;EAGpC,OAAO,gBAAgB,QAAQ;EAC/B,KAAK,SAAS,gBAAgB,QAAQ;;cAG3B,gCAAgC;UACnC;EAEF,OAAO,SAAS,oBAAoB;EAIpC,OAAO,gBAAgB,QAAQ;EAI/B,KAAK,SAAQ,gBAAqB,QAAQ;;UAKjC;EACf;EACA;;cAGW,kCAAkC;UACrC;UACA;UACA;UACA;EAER,YAAY,SAAS;UAKP;EAKR,OAAO,SAAS,oBAAoB;UAiB5B;EAuBR,OAAO,gBAAgB,QAAQ;EAI/B,KAAK,SAAS,gBAAgB,QAAQ"}
@@ -1,75 +0,0 @@
1
- //#region src/meta-eval/outcome-store.ts
2
- var InMemoryOutcomeStore = class {
3
- items = [];
4
- async append(outcome) {
5
- this.items.push({ ...outcome });
6
- }
7
- async forRun(runId) {
8
- return this.items.filter((o) => o.runId === runId).map((o) => ({ ...o }));
9
- }
10
- async list(filter = {}) {
11
- return this.items.filter((o) => matches(o, filter)).map((o) => ({ ...o }));
12
- }
13
- };
14
- var FileSystemOutcomeStore = class {
15
- dir;
16
- maxBytes;
17
- memo;
18
- loaded = false;
19
- constructor(options) {
20
- this.dir = options.dir;
21
- this.maxBytes = options.maxBytes ?? 32 * 1024 * 1024;
22
- }
23
- async ensureDir() {
24
- await (await import("node:fs/promises")).mkdir(this.dir, { recursive: true });
25
- }
26
- async append(outcome) {
27
- await this.ensureDir();
28
- const fs = await import("node:fs/promises");
29
- const path = await import("node:path");
30
- const active = path.join(this.dir, "outcomes.ndjson");
31
- try {
32
- if ((await fs.stat(active)).size >= this.maxBytes) await fs.rename(active, path.join(this.dir, `outcomes.${Date.now()}.ndjson`));
33
- } catch {}
34
- await fs.appendFile(active, `${JSON.stringify(outcome)}\n`, "utf8");
35
- if (this.memo) await this.memo.append(outcome);
36
- }
37
- async load() {
38
- if (this.loaded && this.memo) return this.memo;
39
- const fs = await import("node:fs/promises");
40
- const path = await import("node:path");
41
- const memo = new InMemoryOutcomeStore();
42
- try {
43
- const entries = await fs.readdir(this.dir);
44
- for (const file of entries) {
45
- if (!file.endsWith(".ndjson")) continue;
46
- const content = await fs.readFile(path.join(this.dir, file), "utf8");
47
- for (const line of content.split("\n")) {
48
- if (!line.trim()) continue;
49
- await memo.append(JSON.parse(line));
50
- }
51
- }
52
- } catch {}
53
- this.memo = memo;
54
- this.loaded = true;
55
- return memo;
56
- }
57
- async forRun(runId) {
58
- return (await this.load()).forRun(runId);
59
- }
60
- async list(filter) {
61
- return (await this.load()).list(filter);
62
- }
63
- };
64
- function matches(o, f) {
65
- if (f.runIds && !f.runIds.includes(o.runId)) return false;
66
- if (f.since !== void 0 && o.capturedAt < f.since) return false;
67
- if (f.until !== void 0 && o.capturedAt > f.until) return false;
68
- if (f.source && o.source !== f.source) return false;
69
- if (f.label && o.labels?.[f.label.key] !== f.label.value) return false;
70
- return true;
71
- }
72
- //#endregion
73
- export { InMemoryOutcomeStore as n, FileSystemOutcomeStore as t };
74
-
75
- //# sourceMappingURL=outcome-store-ChBKlTd_.js.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"outcome-store-ChBKlTd_.js","names":[],"sources":["../src/meta-eval/outcome-store.ts"],"sourcesContent":["/**\n * OutcomeStore — deployment outcomes attached to Run IDs.\n *\n * Outcomes arrive asynchronously from production telemetry after the\n * eval run completed: user ratings, retention flags, conversion events,\n * revenue, support-ticket rate, anything a product team can measure.\n * The store is a peer to TraceStore — separate lifecycle, same runId\n * foreign key.\n *\n * The whole point of this module is to make the meta-eval correlation\n * question computable: `correlate(evalMetric, outcomeMetric) → r, ρ, n, CI`.\n */\n\nexport interface DeploymentOutcome {\n runId: string\n capturedAt: number\n /** Numeric outcomes keyed by name — retention_7d, csat, revenue_usd, etc. */\n metrics: Record<string, number>\n /** Dimensions for stratified analysis — cohort, region, user_segment. */\n labels?: Record<string, string>\n /** Free-form provenance (source system, pipeline version). */\n source?: string\n}\n\nexport interface OutcomeFilter {\n runIds?: string[]\n since?: number\n until?: number\n label?: { key: string; value: string }\n source?: string\n}\n\nexport interface OutcomeStore {\n append(outcome: DeploymentOutcome): Promise<void>\n /** All outcomes attached to this run (a single run can have many — multiple\n * capture windows over deployment time). */\n forRun(runId: string): Promise<DeploymentOutcome[]>\n list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>\n}\n\nexport class InMemoryOutcomeStore implements OutcomeStore {\n private items: DeploymentOutcome[] = []\n\n async append(outcome: DeploymentOutcome): Promise<void> {\n this.items.push({ ...outcome })\n }\n\n async forRun(runId: string): Promise<DeploymentOutcome[]> {\n return this.items.filter((o) => o.runId === runId).map((o) => ({ ...o }))\n }\n\n async list(filter: OutcomeFilter = {}): Promise<DeploymentOutcome[]> {\n return this.items.filter((o) => matches(o, filter)).map((o) => ({ ...o }))\n }\n}\n\nexport interface FileSystemOutcomeStoreOptions {\n dir: string\n maxBytes?: number\n}\n\nexport class FileSystemOutcomeStore implements OutcomeStore {\n private dir: string\n private maxBytes: number\n private memo?: InMemoryOutcomeStore\n private loaded = false\n\n constructor(options: FileSystemOutcomeStoreOptions) {\n this.dir = options.dir\n this.maxBytes = options.maxBytes ?? 32 * 1024 * 1024\n }\n\n private async ensureDir(): Promise<void> {\n const fs = await import('node:fs/promises')\n await fs.mkdir(this.dir, { recursive: true })\n }\n\n async append(outcome: DeploymentOutcome): Promise<void> {\n await this.ensureDir()\n const fs = await import('node:fs/promises')\n const path = await import('node:path')\n const active = path.join(this.dir, 'outcomes.ndjson')\n try {\n const stat = await fs.stat(active)\n if (stat.size >= this.maxBytes) {\n await fs.rename(active, path.join(this.dir, `outcomes.${Date.now()}.ndjson`))\n }\n } catch {\n /* first write */\n }\n await fs.appendFile(active, `${JSON.stringify(outcome)}\\n`, 'utf8')\n if (this.memo) await this.memo.append(outcome)\n }\n\n private async load(): Promise<InMemoryOutcomeStore> {\n if (this.loaded && this.memo) return this.memo\n const fs = await import('node:fs/promises')\n const path = await import('node:path')\n const memo = new InMemoryOutcomeStore()\n try {\n const entries = await fs.readdir(this.dir)\n for (const file of entries) {\n if (!file.endsWith('.ndjson')) continue\n const content = await fs.readFile(path.join(this.dir, file), 'utf8')\n for (const line of content.split('\\n')) {\n if (!line.trim()) continue\n await memo.append(JSON.parse(line))\n }\n }\n } catch {\n /* empty */\n }\n this.memo = memo\n this.loaded = true\n return memo\n }\n\n async forRun(runId: string): Promise<DeploymentOutcome[]> {\n return (await this.load()).forRun(runId)\n }\n\n async list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]> {\n return (await this.load()).list(filter)\n }\n}\n\nfunction matches(o: DeploymentOutcome, f: OutcomeFilter): boolean {\n if (f.runIds && !f.runIds.includes(o.runId)) return false\n if (f.since !== undefined && o.capturedAt < f.since) return false\n if (f.until !== undefined && o.capturedAt > f.until) return false\n if (f.source && o.source !== f.source) return false\n if (f.label && o.labels?.[f.label.key] !== f.label.value) return false\n return true\n}\n"],"mappings":";AAwCA,IAAa,uBAAb,MAA0D;CACxD,QAAqC,CAAC;CAEtC,MAAM,OAAO,SAA2C;EACtD,KAAK,MAAM,KAAK,EAAE,GAAG,QAAQ,CAAC;CAChC;CAEA,MAAM,OAAO,OAA6C;EACxD,OAAO,KAAK,MAAM,QAAQ,MAAM,EAAE,UAAU,KAAK,CAAC,CAAC,KAAK,OAAO,EAAE,GAAG,EAAE,EAAE;CAC1E;CAEA,MAAM,KAAK,SAAwB,CAAC,GAAiC;EACnE,OAAO,KAAK,MAAM,QAAQ,MAAM,QAAQ,GAAG,MAAM,CAAC,CAAC,CAAC,KAAK,OAAO,EAAE,GAAG,EAAE,EAAE;CAC3E;AACF;AAOA,IAAa,yBAAb,MAA4D;CAC1D;CACA;CACA;CACA,SAAiB;CAEjB,YAAY,SAAwC;EAClD,KAAK,MAAM,QAAQ;EACnB,KAAK,WAAW,QAAQ,YAAY,KAAK,OAAO;CAClD;CAEA,MAAc,YAA2B;EAEvC,OAAM,MADW,OAAO,oBAAA,CACf,MAAM,KAAK,KAAK,EAAE,WAAW,KAAK,CAAC;CAC9C;CAEA,MAAM,OAAO,SAA2C;EACtD,MAAM,KAAK,UAAU;EACrB,MAAM,KAAK,MAAM,OAAO;EACxB,MAAM,OAAO,MAAM,OAAO;EAC1B,MAAM,SAAS,KAAK,KAAK,KAAK,KAAK,iBAAiB;EACpD,IAAI;GAEF,KAAI,MADe,GAAG,KAAK,MAAM,EAAA,CACxB,QAAQ,KAAK,UACpB,MAAM,GAAG,OAAO,QAAQ,KAAK,KAAK,KAAK,KAAK,YAAY,KAAK,IAAI,EAAE,QAAQ,CAAC;EAEhF,QAAQ,CAER;EACA,MAAM,GAAG,WAAW,QAAQ,GAAG,KAAK,UAAU,OAAO,EAAE,KAAK,MAAM;EAClE,IAAI,KAAK,MAAM,MAAM,KAAK,KAAK,OAAO,OAAO;CAC/C;CAEA,MAAc,OAAsC;EAClD,IAAI,KAAK,UAAU,KAAK,MAAM,OAAO,KAAK;EAC1C,MAAM,KAAK,MAAM,OAAO;EACxB,MAAM,OAAO,MAAM,OAAO;EAC1B,MAAM,OAAO,IAAI,qBAAqB;EACtC,IAAI;GACF,MAAM,UAAU,MAAM,GAAG,QAAQ,KAAK,GAAG;GACzC,KAAK,MAAM,QAAQ,SAAS;IAC1B,IAAI,CAAC,KAAK,SAAS,SAAS,GAAG;IAC/B,MAAM,UAAU,MAAM,GAAG,SAAS,KAAK,KAAK,KAAK,KAAK,IAAI,GAAG,MAAM;IACnE,KAAK,MAAM,QAAQ,QAAQ,MAAM,IAAI,GAAG;KACtC,IAAI,CAAC,KAAK,KAAK,GAAG;KAClB,MAAM,KAAK,OAAO,KAAK,MAAM,IAAI,CAAC;IACpC;GACF;EACF,QAAQ,CAER;EACA,KAAK,OAAO;EACZ,KAAK,SAAS;EACd,OAAO;CACT;CAEA,MAAM,OAAO,OAA6C;EACxD,QAAQ,MAAM,KAAK,KAAK,EAAA,CAAG,OAAO,KAAK;CACzC;CAEA,MAAM,KAAK,QAAsD;EAC/D,QAAQ,MAAM,KAAK,KAAK,EAAA,CAAG,KAAK,MAAM;CACxC;AACF;AAEA,SAAS,QAAQ,GAAsB,GAA2B;CAChE,IAAI,EAAE,UAAU,CAAC,EAAE,OAAO,SAAS,EAAE,KAAK,GAAG,OAAO;CACpD,IAAI,EAAE,UAAU,KAAA,KAAa,EAAE,aAAa,EAAE,OAAO,OAAO;CAC5D,IAAI,EAAE,UAAU,KAAA,KAAa,EAAE,aAAa,EAAE,OAAO,OAAO;CAC5D,IAAI,EAAE,UAAU,EAAE,WAAW,EAAE,QAAQ,OAAO;CAC9C,IAAI,EAAE,SAAS,EAAE,SAAS,EAAE,MAAM,SAAS,EAAE,MAAM,OAAO,OAAO;CACjE,OAAO;AACT"}
@@ -1 +0,0 @@
1
- {"version":3,"file":"promotion-policy-DWOm70gx.js","names":[],"sources":["../src/campaign/gates/promotion-policy.ts"],"sourcesContent":["/**\n * Promotion policy over the evidence VECTOR — the substrate's answer to \"never\n * collapse the multi-objective promotion decision into one scalar.\" A\n * `defaultProductionGate` is one opinionated composition; this module factors\n * the decision into two reusable pieces so MANY policies can compete over the\n * SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):\n *\n * buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus\n * PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy\n * paretoPolicy(ev) // the default strategy\n * paretoSignificanceGate(options): Gate // bus + policy as a Gate\n *\n * The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a\n * potential gain source AND a safety floor (unlike `defaultProductionGate`,\n * where only `composite` can win and `criticalDimensions` are pure floors). A\n * candidate ships iff it weakly DOMINATES the baseline at the confidence level —\n * no objective credibly worse (CI floor breach) AND at least one objective\n * credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work\n * (NOT folded into hold: \"gather more reps\" and \"reject\" are different actions).\n *\n * Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate\n * per-side cost, no per-cell observation vector to bootstrap. Treat them as hard\n * constraints (compose with a budget gate via `composeGate`), not faked CIs.\n */\n\nimport {\n decidePairedPromotion,\n type PairedDecisionMethod,\n type PairedDecisionStatistic,\n type PairedMcNemarEvidence,\n} from '../../paired-promotion-decision'\nimport type { Direction } from '../../pareto'\nimport {\n DECISION_PAIRED_DELTA_STATISTIC,\n type PairedBootstrapResult,\n pairedBootstrap,\n} from '../../statistics'\nimport type { Gate, GateContext, GateDecision, GateResult, JudgeScore, Scenario } from '../types'\nimport { detectScale, pairHoldout } from './statistical-heldout'\n\n/** Where an objective's per-cell scalar comes from. `composite` reads the\n * judge's composite; `dimension` reads a named per-dimension score. */\nexport type ObjectiveSource = { kind: 'composite' } | { kind: 'dimension'; dimension: string }\n\nexport interface PromotionObjective {\n /** Stable label used in reports + `contributingGates`. */\n name: string\n source: ObjectiveSource\n /** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients\n * the paired delta so a positive bootstrap always means \"candidate better\". */\n direction: Direction\n /** The good-direction paired-delta CI lower bound must EXCEED this to count\n * as a significant gain on this axis. Interpreted in the judge's native\n * scale. Default 0 (⇒ \"confidently better\"). */\n gainThreshold?: number\n /** A floor breach (regression) is declared when the good-direction CI lower\n * bound is below −floorTolerance, or when the exact small-sample test proves\n * a drop past it. When omitted it auto-scales off observed magnitudes\n * (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */\n floorTolerance?: number\n}\n\n/** Per-axis verdict from the good-direction paired bootstrap. */\nexport type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs'\n\nexport interface AxisEvidence {\n name: string\n source: ObjectiveSource\n direction: Direction\n /** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):\n * a positive value means the candidate is better on this axis.\n *\n * DIAGNOSTIC on a pass/fail axis: there the verdict is decided on Tango's\n * score interval instead, because a percentile bootstrap over a three-atom\n * delta lattice is not a valid interval at the nonzero margin `floorTolerance`\n * and `gainThreshold` create. `ci` carries the interval that decided. */\n bootstrap: PairedBootstrapResult\n /** Which paired statistic `bootstrap.low`/`.high` bracket. `'mean'` unless the\n * caller asked for the median — on a pass/fail axis the median and its whole\n * CI are pinned at 0 by tie domination and can see neither a gain nor a\n * regression. `bootstrap.median` still carries the median point estimate. */\n bootstrapStatistic: 'median' | 'mean'\n /** The interval the axis verdict was actually decided on, good-direction and\n * in the axis's native units. */\n ci: { low: number; high: number }\n /** Which estimator produced `ci`. */\n decisionStatistic: PairedDecisionStatistic\n /** McNemar's exact evidence on a pass/fail axis; null otherwise. */\n mcnemar: PairedMcNemarEvidence | null\n /** `ci` has zero width — no evidence in either direction, so the axis is\n * neither improved nor regressed however the point estimate sits. */\n indeterminate: boolean\n /** Paired observations contributing to this axis. */\n n: number\n minimumRequired: number\n decisionMethod: PairedDecisionMethod\n gainThreshold: number\n floorTolerance: number\n verdict: AxisVerdict\n}\n\nexport interface EvidenceVector {\n /** One entry per objective — NOTHING averaged across axes. */\n axes: AxisEvidence[]\n /** Smallest paired n across axes that produced observations — the binding\n * evidence-sufficiency constraint. 0 when no axis produced observations. */\n minN: number\n /** Aggregate per-side cost from the gate context (a constraint input, not a\n * CI axis — see the module header). */\n cost: { candidate: number; baseline: number }\n}\n\n/** A promotion strategy: a pure function from the evidence vector to a verdict.\n * Many policies can run over the same `EvidenceVector` and disagree — that's\n * the point (competing strategies, shared evidence). */\nexport type PromotionPolicy = (ev: EvidenceVector) => GateResult\n\nexport interface BuildEvidenceVectorOptions {\n /** Minimum paired observations before an axis can claim significance; below\n * it the axis is `few_runs`. The exact small-sample test may require more\n * observations at the selected confidence. */\n minProductiveRuns?: number\n /** Confidence level for every axis bootstrap. Default 0.95. */\n confidence?: number\n /** Bootstrap resamples. Default 2000. */\n resamples?: number\n /** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */\n seed?: number\n /** Paired statistic every axis CI is computed on. Default `'mean'` — see\n * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */\n statistic?: 'mean' | 'median'\n}\n\n/**\n * The Evidence Bus. For each objective, pair candidate vs baseline by full\n * cellId and bootstrap a CI on the good-direction paired delta. Reuses the\n * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so\n * a single source of truth governs pairing granularity + scale handling.\n */\nexport function buildEvidenceVector<TArtifact, TScenario extends Scenario>(\n ctx: GateContext<TArtifact, TScenario>,\n objectives: PromotionObjective[],\n opts: BuildEvidenceVectorOptions = {},\n): EvidenceVector {\n if (objectives.length === 0) {\n throw new Error('buildEvidenceVector: at least 1 objective required')\n }\n const confidence = opts.confidence ?? 0.95\n const resamples = opts.resamples ?? 2000\n const seed = opts.seed ?? 1337\n const baseline = ctx.baselineJudgeScores ?? ctx.judgeScores\n const scenarioIds = new Set(ctx.scenarios.map((s) => s.id))\n\n const axes: AxisEvidence[] = []\n for (const obj of objectives) {\n let select: (s: JudgeScore) => number | undefined\n if (obj.source.kind === 'composite') {\n select = (s) => s.composite\n } else {\n const dim = obj.source.dimension\n select = (s) => s.dimensions[dim]\n }\n const paired = pairHoldout(ctx.judgeScores, baseline, scenarioIds, select)\n // Orient to the good direction: maximize ⇒ bootstrap (candidate − baseline);\n // minimize ⇒ bootstrap (baseline − candidate) by swapping args, so a\n // positive bootstrap always reads as \"candidate better on this axis\".\n const before = obj.direction === 'maximize' ? paired.before : paired.after\n const after = obj.direction === 'maximize' ? paired.after : paired.before\n const n = paired.before.length\n const floorTolerance =\n obj.floorTolerance ?? 0.05 * detectScale([...paired.before, ...paired.after])\n const gainThreshold = obj.gainThreshold ?? 0\n // Axes are decided on the MEAN paired delta — which for a pass/fail axis is\n // exactly the change in success rate. The median is structurally blind on\n // the shapes eval data lands in: with most pairs tied its bootstrap CI\n // collapses to [0,0] and the axis reads 'flat', hiding real gains AND real\n // regressions — pass/fail axes on ANY encoding ({0,1} and the 0-100 one\n // `detectScale` exists for), and low-cardinality axes even below half ties.\n // Orthogonal to `pairedDeltaTest`'s own small-sample switch: that picks the\n // TEST (bootstrap CI at n ≥ 20, exact sign test below), this picks the\n // ESTIMATOR the test is applied to. Both are needed — an exact sign test on\n // a tie-pinned median is still blind.\n const bootstrapStatistic = opts.statistic ?? DECISION_PAIRED_DELTA_STATISTIC\n // Both burdens of proof route through the ONE shared rule\n // (`decidePairedPromotion`), so a pass/fail axis is judged on Tango's score\n // interval — the only paired-binary construction valid at the nonzero\n // margins `gainThreshold` / `floorTolerance` create — and a zero-width\n // interval cannot buy a verdict in either direction.\n const improvement = decidePairedPromotion(before, after, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n threshold: gainThreshold,\n minPairs: opts.minProductiveRuns,\n })\n const regression = decidePairedPromotion(after, before, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n threshold: floorTolerance,\n minPairs: opts.minProductiveRuns,\n })\n const bootstrap =\n improvement.bootstrap ??\n pairedBootstrap(before, after, {\n confidence,\n resamples,\n statistic: bootstrapStatistic,\n seed,\n })\n // A floor breach fires on EITHER burden of proof, because they cover\n // different failures and the floor is the anti-Goodhart guard:\n // - `improvement.low < -floorTolerance` — the CREDIBLE WORST CASE exceeds\n // the tolerance. This is the contract `AxisEvidence.floorTolerance` and\n // `paretoPolicy`'s own reason string state, and it is the conservative\n // posture a safety axis needs: block unless the data can rule the\n // breach out, rather than waiting for the breach to be proven. Read off\n // the DECIDING interval, so a pass/fail axis is not screened by a\n // bootstrap that is pinned wherever ties dominate.\n // - `regression.promote` — a PROVEN drop past the tolerance. Adds the\n // small-sample path, where the decision is an exact sign test because\n // the bootstrap interval is descriptive only.\n // The credible-worst-case arm stays on the BOOTSTRAP, deliberately. Reading\n // it off the score interval instead would change what the floor MEANS on a\n // pass/fail axis: with every pair concordant the score interval is\n // ±z²/(n+z²) — ±0.39 at n=6, ±0.16 at n=20 — so a completely unchanged\n // safety axis would breach a 0.05 floor at any realistic n, and the gate\n // would refuse everything. That the bootstrap arm is instead fail-OPEN on a\n // tied pass/fail axis is a real and separate weakness: the honest fix is a\n // minimum-power requirement on the floor, not a wider interval, because the\n // data genuinely cannot rule a 5pp drop out at n=20 and a gate that says so\n // by blocking every candidate is not usable. `regression.promote` — the\n // PROVEN-drop arm — does route through the shared rule, so a real pass/fail\n // regression is now caught on an interval valid at the nonzero tolerance.\n const floorBreached = bootstrap.low < -floorTolerance || regression.promote\n // Floor check precedes the gain check: a credible regression must never be\n // masked as \"improved\". With the defaults (gainThreshold 0, positive floor)\n // the regions are disjoint and order is moot, but a consumer who sets a\n // negative gainThreshold (\"accept small dips\") could otherwise have a real\n // floor breach classified as a gain — anti-Goodhart wins the tie.\n const verdict: AxisVerdict = !improvement.sufficient\n ? 'few_runs'\n : floorBreached\n ? 'regressed'\n : improvement.promote\n ? 'improved'\n : 'flat'\n axes.push({\n name: obj.name,\n source: obj.source,\n direction: obj.direction,\n bootstrap,\n bootstrapStatistic,\n ci: { low: improvement.low, high: improvement.high },\n decisionStatistic: improvement.statistic,\n mcnemar: improvement.mcnemar,\n indeterminate: improvement.indeterminate,\n n,\n minimumRequired: improvement.minimumPairs,\n decisionMethod: improvement.method,\n gainThreshold,\n floorTolerance,\n verdict,\n })\n }\n const ns = axes.map((a) => a.n).filter((n) => n > 0)\n const minN = ns.length > 0 ? Math.min(...ns) : 0\n return { axes, minN, cost: { candidate: ctx.cost.candidate, baseline: ctx.cost.baseline } }\n}\n\n/**\n * The default strategy: symmetric multi-objective Pareto significance. Ship iff\n * the candidate weakly dominates the baseline at the confidence level — no axis\n * credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold\n * (anti-Goodhart, dominates everything). Insufficient evidence on any axis →\n * need_more_work. Statistically equivalent → hold (never ship noise).\n */\nexport const paretoPolicy: PromotionPolicy = (ev) => {\n const contributingGates = ev.axes.map((ax) => ({\n name: `objective:${ax.name}`,\n status:\n ax.verdict === 'regressed'\n ? ('fail' as const)\n : ax.verdict === 'few_runs'\n ? ('not_evaluated' as const)\n : ('pass' as const),\n detail: {\n direction: ax.direction,\n source: ax.source,\n verdict: ax.verdict,\n n: ax.n,\n deltaMedian: ax.bootstrap.median,\n ciLow: ax.ci.low,\n ciHigh: ax.ci.high,\n decisionStatistic: ax.decisionStatistic,\n decisionMethod: ax.decisionMethod,\n mcnemar: ax.mcnemar,\n indeterminate: ax.indeterminate,\n bootstrapCiLow: ax.bootstrap.low,\n bootstrapCiHigh: ax.bootstrap.high,\n confidence: ax.bootstrap.confidence,\n gainThreshold: ax.gainThreshold,\n floorTolerance: ax.floorTolerance,\n },\n }))\n\n const regressed = ev.axes.filter((a) => a.verdict === 'regressed')\n const fewRuns = ev.axes.filter((a) => a.verdict === 'few_runs')\n const improved = ev.axes.filter((a) => a.verdict === 'improved')\n\n let decision: GateDecision\n const reasons: string[] = []\n if (regressed.length > 0) {\n // Floor breach dominates: a credible regression on ANY axis blocks ship even\n // if another axis improved. This makes the +gain/−safety false positive\n // structurally impossible whenever the safety dim is an objective.\n decision = 'hold'\n for (const a of regressed) {\n reasons.push(\n `objective '${a.name}' regressed: good-direction CI.low ${a.ci.low.toFixed(3)} < -${a.floorTolerance} (n=${a.n})`,\n )\n }\n } else if (fewRuns.length > 0) {\n // No credible regression on the scored axes, but ≥1 axis lacks the evidence\n // to claim a gain ⇒ gather more reps, do NOT reject.\n decision = 'need_more_work'\n for (const a of fewRuns) {\n reasons.push(\n `objective '${a.name}' has only n=${a.n} paired runs — insufficient evidence to claim significance`,\n )\n }\n } else if (improved.length > 0) {\n // Weakly dominates (no axis worse) AND strictly better on ≥1 axis ⇒ a Pareto\n // improvement at the confidence level.\n decision = 'ship'\n reasons.push(\n `Pareto improvement at the confidence level: ${improved\n .map(\n (a) =>\n `'${a.name}' +${a.ci.low > 0 ? a.ci.low.toFixed(3) : a.bootstrap.mean.toFixed(3)} (CI.low ${a.ci.low.toFixed(3)})`,\n )\n .join(', ')}; no objective regressed`,\n )\n } else {\n // Enough evidence, nothing credibly better or worse ⇒ statistically\n // equivalent. Do NOT ship a no-op.\n decision = 'hold'\n reasons.push(\n 'no Pareto improvement: candidate statistically equivalent to baseline on every objective',\n )\n }\n\n // `delta` surfaces the composite axis if present, else the first axis — a\n // single convenience scalar; the vector lives in `contributingGates`.\n const composite = ev.axes.find((a) => a.source.kind === 'composite') ?? ev.axes[0]\n return { decision, reasons, contributingGates, delta: composite?.bootstrap.median }\n}\n\nexport interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {\n /** The objective vector. Every axis is both a gain source and a safety floor. */\n objectives: PromotionObjective[]\n /** Strategy applied to the evidence vector. Default `paretoPolicy`. Override\n * to run a stricter/looser strategy over the SAME bus (competing policies). */\n policy?: PromotionPolicy\n /** Override the gate name in reports. */\n name?: string\n}\n\n/**\n * Wrap the bus + a policy as a `Gate`. Plugs into the existing\n * `runImprovementLoop({ gate })` slot and composes via `composeGate`; default\n * loop behavior is unchanged because consumers opt in by passing this gate.\n */\nexport function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(\n options: ParetoSignificanceGateOptions,\n): Gate<TArtifact, TScenario> {\n if (options.objectives.length === 0) {\n throw new Error('paretoSignificanceGate: at least 1 objective required')\n }\n const policy = options.policy ?? paretoPolicy\n return {\n name: options.name ?? 'paretoSignificanceGate',\n async decide(ctx: GateContext<TArtifact, TScenario>): Promise<GateResult> {\n const ev = buildEvidenceVector(ctx, options.objectives, options)\n return policy(ev)\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA2IA,SAAgB,oBACd,KACA,YACA,OAAmC,CAAC,GACpB;CAChB,IAAI,WAAW,WAAW,GACxB,MAAM,IAAI,MAAM,oDAAoD;CAEtE,MAAM,aAAa,KAAK,cAAc;CACtC,MAAM,YAAY,KAAK,aAAa;CACpC,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,WAAW,IAAI,uBAAuB,IAAI;CAChD,MAAM,cAAc,IAAI,IAAI,IAAI,UAAU,KAAK,MAAM,EAAE,EAAE,CAAC;CAE1D,MAAM,OAAuB,CAAC;CAC9B,KAAK,MAAM,OAAO,YAAY;EAC5B,IAAI;EACJ,IAAI,IAAI,OAAO,SAAS,aACtB,UAAU,MAAM,EAAE;OACb;GACL,MAAM,MAAM,IAAI,OAAO;GACvB,UAAU,MAAM,EAAE,WAAW;EAC/B;EACA,MAAM,SAAS,YAAY,IAAI,aAAa,UAAU,aAAa,MAAM;EAIzE,MAAM,SAAS,IAAI,cAAc,aAAa,OAAO,SAAS,OAAO;EACrE,MAAM,QAAQ,IAAI,cAAc,aAAa,OAAO,QAAQ,OAAO;EACnE,MAAM,IAAI,OAAO,OAAO;EACxB,MAAM,iBACJ,IAAI,kBAAkB,MAAO,YAAY,CAAC,GAAG,OAAO,QAAQ,GAAG,OAAO,KAAK,CAAC;EAC9E,MAAM,gBAAgB,IAAI,iBAAiB;EAW3C,MAAM,qBAAqB,KAAK,aAAA;EAMhC,MAAM,cAAc,sBAAsB,QAAQ,OAAO;GACvD;GACA;GACA,WAAW;GACX;GACA,WAAW;GACX,UAAU,KAAK;EACjB,CAAC;EACD,MAAM,aAAa,sBAAsB,OAAO,QAAQ;GACtD;GACA;GACA,WAAW;GACX;GACA,WAAW;GACX,UAAU,KAAK;EACjB,CAAC;EACD,MAAM,YACJ,YAAY,aACZ,gBAAgB,QAAQ,OAAO;GAC7B;GACA;GACA,WAAW;GACX;EACF,CAAC;EAyBH,MAAM,gBAAgB,UAAU,MAAM,CAAC,kBAAkB,WAAW;EAMpE,MAAM,UAAuB,CAAC,YAAY,aACtC,aACA,gBACE,cACA,YAAY,UACV,aACA;EACR,KAAK,KAAK;GACR,MAAM,IAAI;GACV,QAAQ,IAAI;GACZ,WAAW,IAAI;GACf;GACA;GACA,IAAI;IAAE,KAAK,YAAY;IAAK,MAAM,YAAY;GAAK;GACnD,mBAAmB,YAAY;GAC/B,SAAS,YAAY;GACrB,eAAe,YAAY;GAC3B;GACA,iBAAiB,YAAY;GAC7B,gBAAgB,YAAY;GAC5B;GACA;GACA;EACF,CAAC;CACH;CACA,MAAM,KAAK,KAAK,KAAK,MAAM,EAAE,CAAC,CAAC,CAAC,QAAQ,MAAM,IAAI,CAAC;CAEnD,OAAO;EAAE;EAAM,MADF,GAAG,SAAS,IAAI,KAAK,IAAI,GAAG,EAAE,IAAI;EAC1B,MAAM;GAAE,WAAW,IAAI,KAAK;GAAW,UAAU,IAAI,KAAK;EAAS;CAAE;AAC5F;;;;;;;;AASA,MAAa,gBAAiC,OAAO;CACnD,MAAM,oBAAoB,GAAG,KAAK,KAAK,QAAQ;EAC7C,MAAM,aAAa,GAAG;EACtB,QACE,GAAG,YAAY,cACV,SACD,GAAG,YAAY,aACZ,kBACA;EACT,QAAQ;GACN,WAAW,GAAG;GACd,QAAQ,GAAG;GACX,SAAS,GAAG;GACZ,GAAG,GAAG;GACN,aAAa,GAAG,UAAU;GAC1B,OAAO,GAAG,GAAG;GACb,QAAQ,GAAG,GAAG;GACd,mBAAmB,GAAG;GACtB,gBAAgB,GAAG;GACnB,SAAS,GAAG;GACZ,eAAe,GAAG;GAClB,gBAAgB,GAAG,UAAU;GAC7B,iBAAiB,GAAG,UAAU;GAC9B,YAAY,GAAG,UAAU;GACzB,eAAe,GAAG;GAClB,gBAAgB,GAAG;EACrB;CACF,EAAE;CAEF,MAAM,YAAY,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,WAAW;CACjE,MAAM,UAAU,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,UAAU;CAC9D,MAAM,WAAW,GAAG,KAAK,QAAQ,MAAM,EAAE,YAAY,UAAU;CAE/D,IAAI;CACJ,MAAM,UAAoB,CAAC;CAC3B,IAAI,UAAU,SAAS,GAAG;EAIxB,WAAW;EACX,KAAK,MAAM,KAAK,WACd,QAAQ,KACN,cAAc,EAAE,KAAK,qCAAqC,EAAE,GAAG,IAAI,QAAQ,CAAC,EAAE,MAAM,EAAE,eAAe,MAAM,EAAE,EAAE,EACjH;CAEJ,OAAO,IAAI,QAAQ,SAAS,GAAG;EAG7B,WAAW;EACX,KAAK,MAAM,KAAK,SACd,QAAQ,KACN,cAAc,EAAE,KAAK,eAAe,EAAE,EAAE,2DAC1C;CAEJ,OAAO,IAAI,SAAS,SAAS,GAAG;EAG9B,WAAW;EACX,QAAQ,KACN,+CAA+C,SAC5C,KACE,MACC,IAAI,EAAE,KAAK,KAAK,EAAE,GAAG,MAAM,IAAI,EAAE,GAAG,IAAI,QAAQ,CAAC,IAAI,EAAE,UAAU,KAAK,QAAQ,CAAC,EAAE,WAAW,EAAE,GAAG,IAAI,QAAQ,CAAC,EAAE,EACpH,CAAC,CACA,KAAK,IAAI,EAAE,yBAChB;CACF,OAAO;EAGL,WAAW;EACX,QAAQ,KACN,0FACF;CACF;CAIA,MAAM,YAAY,GAAG,KAAK,MAAM,MAAM,EAAE,OAAO,SAAS,WAAW,KAAK,GAAG,KAAK;CAChF,OAAO;EAAE;EAAU;EAAS;EAAmB,OAAO,WAAW,UAAU;CAAO;AACpF;;;;;;AAiBA,SAAgB,uBACd,SAC4B;CAC5B,IAAI,QAAQ,WAAW,WAAW,GAChC,MAAM,IAAI,MAAM,uDAAuD;CAEzE,MAAM,SAAS,QAAQ,UAAU;CACjC,OAAO;EACL,MAAM,QAAQ,QAAQ;EACtB,MAAM,OAAO,KAA6D;GACxE,MAAM,KAAK,oBAAoB,KAAK,QAAQ,YAAY,OAAO;GAC/D,OAAO,OAAO,EAAE;EAClB;CACF;AACF"}
@@ -1,131 +0,0 @@
1
- import { i as makeRng } from "./internal-BMFSR8Ns.js";
2
- import { a as spearmanR, r as pearsonR } from "./descriptive-1V17A-qa.js";
3
- //#region src/meta-eval/rubric-predictive-validity.ts
4
- async function rubricPredictiveValidity(input) {
5
- const minSamples = input.minSamples ?? 8;
6
- const reduction = input.reduction ?? "latest";
7
- const resamples = input.bootstrapResamples ?? 500;
8
- const outcomes = await input.outcomes.list();
9
- const outcomesByRun = /* @__PURE__ */ new Map();
10
- for (const o of outcomes) {
11
- const arr = outcomesByRun.get(o.runId) ?? [];
12
- arr.push(o);
13
- outcomesByRun.set(o.runId, arr);
14
- }
15
- const observedRubrics = /* @__PURE__ */ new Set();
16
- for (const r of input.runs) for (const k of Object.keys(r.outcome.raw)) observedRubrics.add(k);
17
- const rubrics = input.rubrics ?? [...observedRubrics];
18
- const buckets = [];
19
- for (const r of rubrics) for (const o of input.outcomeMetrics) buckets.push({
20
- rubric: r,
21
- outcome: o,
22
- xs: [],
23
- ys: []
24
- });
25
- let joined = 0;
26
- let skipped = 0;
27
- for (const run of input.runs) {
28
- const os = outcomesByRun.get(run.runId);
29
- if (!os || os.length === 0) {
30
- skipped++;
31
- continue;
32
- }
33
- let joinedThisRun = false;
34
- for (const r of rubrics) {
35
- const x = run.outcome.raw[r];
36
- if (typeof x !== "number" || !Number.isFinite(x)) continue;
37
- for (const o of input.outcomeMetrics) {
38
- const values = os.map((row) => row.metrics[o]).filter((v) => typeof v === "number" && Number.isFinite(v));
39
- if (values.length === 0) continue;
40
- const y = reduce(values, os, o, reduction);
41
- if (y === null) continue;
42
- const bucket = buckets.find((b) => b.rubric === r && b.outcome === o);
43
- bucket.xs.push(x);
44
- bucket.ys.push(y);
45
- joinedThisRun = true;
46
- }
47
- }
48
- if (joinedThisRun) joined++;
49
- }
50
- const pairs = [];
51
- for (const b of buckets) {
52
- if (b.xs.length < minSamples) continue;
53
- const pearson = pearsonR(b.xs, b.ys);
54
- const spearman = spearmanR(b.xs, b.ys);
55
- const ci = bootstrapCi(b.xs, b.ys, resamples, input.seed);
56
- const verdict = Math.abs(spearman) >= .7 ? "load_bearing" : Math.abs(spearman) >= .4 ? "informative" : "decorative";
57
- pairs.push({
58
- rubric: b.rubric,
59
- outcome: b.outcome,
60
- n: b.xs.length,
61
- pearson,
62
- spearman,
63
- ci95: ci,
64
- verdict
65
- });
66
- }
67
- const byRubric = /* @__PURE__ */ new Map();
68
- for (const p of pairs) {
69
- const arr = byRubric.get(p.rubric) ?? [];
70
- arr.push(p);
71
- byRubric.set(p.rubric, arr);
72
- }
73
- const ranked = [...byRubric.entries()].map(([rubric, ps]) => {
74
- const best = ps.reduce((a, b) => Math.abs(b.spearman) > Math.abs(a.spearman) ? b : a);
75
- return {
76
- rubric,
77
- bestOutcome: best.outcome,
78
- spearman: best.spearman,
79
- pearson: best.pearson,
80
- n: best.n,
81
- verdict: best.verdict
82
- };
83
- }).sort((a, b) => Math.abs(b.spearman) - Math.abs(a.spearman));
84
- const rubricsWithoutData = rubrics.filter((r) => !byRubric.has(r));
85
- return {
86
- pairs,
87
- ranked,
88
- joinedSamples: joined,
89
- skippedRuns: skipped,
90
- rubricsWithoutData
91
- };
92
- }
93
- function reduce(values, outcomes, metric, kind) {
94
- if (values.length === 0) return null;
95
- if (kind === "mean") return values.reduce((s, v) => s + v, 0) / values.length;
96
- if (kind === "max") return Math.max(...values);
97
- return [...outcomes].filter((o) => typeof o.metrics[metric] === "number").sort((a, b) => b.capturedAt - a.capturedAt)[0]?.metrics[metric] ?? null;
98
- }
99
- function bootstrapCi(xs, ys, iterations, seed) {
100
- const n = xs.length;
101
- if (n < 3) return {
102
- low: NaN,
103
- high: NaN
104
- };
105
- const rng = makeRng(seed, xs, ys);
106
- const samples = [];
107
- for (let b = 0; b < iterations; b++) {
108
- const rx = new Array(n);
109
- const ry = new Array(n);
110
- for (let i = 0; i < n; i++) {
111
- const idx = Math.floor(rng() * n);
112
- rx[i] = xs[idx];
113
- ry[i] = ys[idx];
114
- }
115
- const r = pearsonR(rx, ry);
116
- if (Number.isFinite(r)) samples.push(r);
117
- }
118
- samples.sort((a, b) => a - b);
119
- if (samples.length === 0) return {
120
- low: NaN,
121
- high: NaN
122
- };
123
- return {
124
- low: samples[Math.floor(.025 * samples.length)],
125
- high: samples[Math.min(samples.length - 1, Math.floor(.975 * samples.length))]
126
- };
127
- }
128
- //#endregion
129
- export { rubricPredictiveValidity as t };
130
-
131
- //# sourceMappingURL=rubric-predictive-validity-2D5Gw9z9.js.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"rubric-predictive-validity-2D5Gw9z9.js","names":[],"sources":["../src/meta-eval/rubric-predictive-validity.ts"],"sourcesContent":["/**\n * Rubric predictive validity — does our eval rubric predict deployment\n * outcomes?\n *\n * `correlationStudy` (already in this package) joins a `TraceStore` to an\n * `OutcomeStore` and computes Pearson + Spearman + bootstrap CI for each\n * (eval-metric, outcome-metric) pair. That answers \"does X correlate with\n * Y at all.\" `rubricPredictiveValidity` is the campaign-shaped wrapper\n * around it: take a sequence of `RunRecord`s (the canonical campaign\n * artifact) and a `DeploymentOutcomeStore`, join on `runId`, return a\n * ranked verdict on every rubric whose dimension scores were captured in\n * `outcome.raw`.\n *\n * The point — quoting the methodology doc — is that **without this loop\n * every rubric is faith-based**. Once it's wired, you know which rubrics\n * have earned their promotion power and which ones are decoration.\n *\n * const validity = await rubricPredictiveValidity({\n * runs: lastQuarter,\n * outcomes: shipFlagOutcomeStore,\n * outcomeMetrics: ['revenue_lift', 'retention_30d', 'csat'],\n * rubrics: ['anti_slop', 'semantic_concept', 'tool_recovery'],\n * })\n * for (const r of validity.ranked) {\n * console.log(`${r.rubric} → ${r.bestOutcome}: ρ=${r.spearman.toFixed(2)}`)\n * }\n *\n * The function is intentionally read-only. Use the verdict to deprecate\n * decorative rubrics, re-weight composite scores, or trigger a\n * recalibration sweep when predictive validity drops below a threshold.\n */\n\nimport type { RunRecord } from '../run-record'\nimport { pearsonR, spearmanR } from '../statistics'\nimport { makeRng } from '../statistics/internal'\nimport type { DeploymentOutcome, OutcomeStore } from './outcome-store'\n\nexport interface RubricPredictiveValidityInput {\n /**\n * Canonical campaign output. Each record's `outcome.raw[<rubricId>]`\n * provides the eval score; missing keys are silently skipped per pair.\n */\n runs: RunRecord[]\n outcomes: OutcomeStore\n /**\n * Outcome metric names to evaluate against. Each must appear in at\n * least one `DeploymentOutcome.metrics` keyspace; pairs with too few\n * joined samples are excluded from the result.\n */\n outcomeMetrics: string[]\n /**\n * Rubric ids to evaluate. Must appear as keys in `RunRecord.outcome.raw`.\n * If omitted, every numeric key in `outcome.raw` across the run set is\n * treated as a rubric.\n */\n rubrics?: string[]\n /** Minimum joined-sample count before a pair is reported. Default 8. */\n minSamples?: number\n /** Bootstrap resamples for CI. Default 500. */\n bootstrapResamples?: number\n /** Seed for the bootstrap. Absent, the seed is derived from the paired\n * observations, so the same input reproduces the same interval. */\n seed?: number\n /**\n * Reduction when multiple outcomes attach to one runId. Default `'latest'`\n * (most recently captured).\n */\n reduction?: 'latest' | 'mean' | 'max'\n}\n\nexport interface RubricOutcomePair {\n rubric: string\n outcome: string\n n: number\n pearson: number\n spearman: number\n ci95: { low: number; high: number }\n /**\n * Verdict bucket. `load_bearing` ≥ 0.7, `informative` ≥ 0.4,\n * `decorative` < 0.4 in absolute correlation. A negative correlation\n * with a desired outcome is also `decorative` — actively misleading\n * is worse than uninformative.\n */\n verdict: 'load_bearing' | 'informative' | 'decorative'\n}\n\nexport interface RubricRanking {\n rubric: string\n /** Outcome metric this rubric correlated best with. */\n bestOutcome: string\n spearman: number\n pearson: number\n n: number\n verdict: RubricOutcomePair['verdict']\n}\n\nexport interface RubricPredictiveValidityReport {\n pairs: RubricOutcomePair[]\n /** Per-rubric best pair, sorted descending by |spearman|. */\n ranked: RubricRanking[]\n joinedSamples: number\n skippedRuns: number\n /** Rubrics that were declared but never produced a usable score. */\n rubricsWithoutData: string[]\n}\n\nexport async function rubricPredictiveValidity(\n input: RubricPredictiveValidityInput,\n): Promise<RubricPredictiveValidityReport> {\n const minSamples = input.minSamples ?? 8\n const reduction = input.reduction ?? 'latest'\n const resamples = input.bootstrapResamples ?? 500\n\n const outcomes = await input.outcomes.list()\n const outcomesByRun = new Map<string, DeploymentOutcome[]>()\n for (const o of outcomes) {\n const arr = outcomesByRun.get(o.runId) ?? []\n arr.push(o)\n outcomesByRun.set(o.runId, arr)\n }\n\n // Discover rubrics: caller-declared OR every numeric key in outcome.raw\n // observed across runs.\n const observedRubrics = new Set<string>()\n for (const r of input.runs) {\n for (const k of Object.keys(r.outcome.raw)) observedRubrics.add(k)\n }\n const rubrics = input.rubrics ?? [...observedRubrics]\n\n // Collect aligned (x, y) pairs per (rubric, outcome).\n type Bucket = { rubric: string; outcome: string; xs: number[]; ys: number[] }\n const buckets: Bucket[] = []\n for (const r of rubrics) {\n for (const o of input.outcomeMetrics) {\n buckets.push({ rubric: r, outcome: o, xs: [], ys: [] })\n }\n }\n\n let joined = 0\n let skipped = 0\n for (const run of input.runs) {\n const os = outcomesByRun.get(run.runId)\n if (!os || os.length === 0) {\n skipped++\n continue\n }\n let joinedThisRun = false\n for (const r of rubrics) {\n const x = run.outcome.raw[r]\n if (typeof x !== 'number' || !Number.isFinite(x)) continue\n for (const o of input.outcomeMetrics) {\n const values = os\n .map((row) => row.metrics[o])\n .filter((v): v is number => typeof v === 'number' && Number.isFinite(v))\n if (values.length === 0) continue\n const y = reduce(values, os, o, reduction)\n if (y === null) continue\n const bucket = buckets.find((b) => b.rubric === r && b.outcome === o)!\n bucket.xs.push(x)\n bucket.ys.push(y)\n joinedThisRun = true\n }\n }\n if (joinedThisRun) joined++\n }\n\n const pairs: RubricOutcomePair[] = []\n for (const b of buckets) {\n if (b.xs.length < minSamples) continue\n const pearson = pearsonR(b.xs, b.ys)\n const spearman = spearmanR(b.xs, b.ys)\n const ci = bootstrapCi(b.xs, b.ys, resamples, input.seed)\n const verdict: RubricOutcomePair['verdict'] =\n Math.abs(spearman) >= 0.7\n ? 'load_bearing'\n : Math.abs(spearman) >= 0.4\n ? 'informative'\n : 'decorative'\n pairs.push({\n rubric: b.rubric,\n outcome: b.outcome,\n n: b.xs.length,\n pearson,\n spearman,\n ci95: ci,\n verdict,\n })\n }\n\n const byRubric = new Map<string, RubricOutcomePair[]>()\n for (const p of pairs) {\n const arr = byRubric.get(p.rubric) ?? []\n arr.push(p)\n byRubric.set(p.rubric, arr)\n }\n const ranked: RubricRanking[] = [...byRubric.entries()]\n .map(([rubric, ps]) => {\n const best = ps.reduce((a, b) => (Math.abs(b.spearman) > Math.abs(a.spearman) ? b : a))\n return {\n rubric,\n bestOutcome: best.outcome,\n spearman: best.spearman,\n pearson: best.pearson,\n n: best.n,\n verdict: best.verdict,\n }\n })\n .sort((a, b) => Math.abs(b.spearman) - Math.abs(a.spearman))\n\n const rubricsWithoutData = rubrics.filter((r) => !byRubric.has(r))\n\n return { pairs, ranked, joinedSamples: joined, skippedRuns: skipped, rubricsWithoutData }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction reduce(\n values: number[],\n outcomes: DeploymentOutcome[],\n metric: string,\n kind: 'latest' | 'mean' | 'max',\n): number | null {\n if (values.length === 0) return null\n if (kind === 'mean') return values.reduce((s, v) => s + v, 0) / values.length\n if (kind === 'max') return Math.max(...values)\n // 'latest'\n const sorted = [...outcomes]\n .filter((o) => typeof o.metrics[metric] === 'number')\n .sort((a, b) => b.capturedAt - a.capturedAt)\n return sorted[0]?.metrics[metric] ?? null\n}\n\nfunction bootstrapCi(\n xs: number[],\n ys: number[],\n iterations: number,\n seed: number | undefined,\n): { low: number; high: number } {\n const n = xs.length\n if (n < 3) return { low: Number.NaN, high: Number.NaN }\n const rng = makeRng(seed, xs, ys)\n const samples: number[] = []\n for (let b = 0; b < iterations; b++) {\n const rx = new Array<number>(n)\n const ry = new Array<number>(n)\n for (let i = 0; i < n; i++) {\n const idx = Math.floor(rng() * n)\n rx[i] = xs[idx]!\n ry[i] = ys[idx]!\n }\n const r = pearsonR(rx, ry)\n if (Number.isFinite(r)) samples.push(r)\n }\n samples.sort((a, b) => a - b)\n if (samples.length === 0) return { low: Number.NaN, high: Number.NaN }\n return {\n low: samples[Math.floor(0.025 * samples.length)]!,\n high: samples[Math.min(samples.length - 1, Math.floor(0.975 * samples.length))]!,\n }\n}\n"],"mappings":";;;AA0GA,eAAsB,yBACpB,OACyC;CACzC,MAAM,aAAa,MAAM,cAAc;CACvC,MAAM,YAAY,MAAM,aAAa;CACrC,MAAM,YAAY,MAAM,sBAAsB;CAE9C,MAAM,WAAW,MAAM,MAAM,SAAS,KAAK;CAC3C,MAAM,gCAAgB,IAAI,IAAiC;CAC3D,KAAK,MAAM,KAAK,UAAU;EACxB,MAAM,MAAM,cAAc,IAAI,EAAE,KAAK,KAAK,CAAC;EAC3C,IAAI,KAAK,CAAC;EACV,cAAc,IAAI,EAAE,OAAO,GAAG;CAChC;CAIA,MAAM,kCAAkB,IAAI,IAAY;CACxC,KAAK,MAAM,KAAK,MAAM,MACpB,KAAK,MAAM,KAAK,OAAO,KAAK,EAAE,QAAQ,GAAG,GAAG,gBAAgB,IAAI,CAAC;CAEnE,MAAM,UAAU,MAAM,WAAW,CAAC,GAAG,eAAe;CAIpD,MAAM,UAAoB,CAAC;CAC3B,KAAK,MAAM,KAAK,SACd,KAAK,MAAM,KAAK,MAAM,gBACpB,QAAQ,KAAK;EAAE,QAAQ;EAAG,SAAS;EAAG,IAAI,CAAC;EAAG,IAAI,CAAC;CAAE,CAAC;CAI1D,IAAI,SAAS;CACb,IAAI,UAAU;CACd,KAAK,MAAM,OAAO,MAAM,MAAM;EAC5B,MAAM,KAAK,cAAc,IAAI,IAAI,KAAK;EACtC,IAAI,CAAC,MAAM,GAAG,WAAW,GAAG;GAC1B;GACA;EACF;EACA,IAAI,gBAAgB;EACpB,KAAK,MAAM,KAAK,SAAS;GACvB,MAAM,IAAI,IAAI,QAAQ,IAAI;GAC1B,IAAI,OAAO,MAAM,YAAY,CAAC,OAAO,SAAS,CAAC,GAAG;GAClD,KAAK,MAAM,KAAK,MAAM,gBAAgB;IACpC,MAAM,SAAS,GACZ,KAAK,QAAQ,IAAI,QAAQ,EAAE,CAAC,CAC5B,QAAQ,MAAmB,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,CAAC;IACzE,IAAI,OAAO,WAAW,GAAG;IACzB,MAAM,IAAI,OAAO,QAAQ,IAAI,GAAG,SAAS;IACzC,IAAI,MAAM,MAAM;IAChB,MAAM,SAAS,QAAQ,MAAM,MAAM,EAAE,WAAW,KAAK,EAAE,YAAY,CAAC;IACpE,OAAO,GAAG,KAAK,CAAC;IAChB,OAAO,GAAG,KAAK,CAAC;IAChB,gBAAgB;GAClB;EACF;EACA,IAAI,eAAe;CACrB;CAEA,MAAM,QAA6B,CAAC;CACpC,KAAK,MAAM,KAAK,SAAS;EACvB,IAAI,EAAE,GAAG,SAAS,YAAY;EAC9B,MAAM,UAAU,SAAS,EAAE,IAAI,EAAE,EAAE;EACnC,MAAM,WAAW,UAAU,EAAE,IAAI,EAAE,EAAE;EACrC,MAAM,KAAK,YAAY,EAAE,IAAI,EAAE,IAAI,WAAW,MAAM,IAAI;EACxD,MAAM,UACJ,KAAK,IAAI,QAAQ,KAAK,KAClB,iBACA,KAAK,IAAI,QAAQ,KAAK,KACpB,gBACA;EACR,MAAM,KAAK;GACT,QAAQ,EAAE;GACV,SAAS,EAAE;GACX,GAAG,EAAE,GAAG;GACR;GACA;GACA,MAAM;GACN;EACF,CAAC;CACH;CAEA,MAAM,2BAAW,IAAI,IAAiC;CACtD,KAAK,MAAM,KAAK,OAAO;EACrB,MAAM,MAAM,SAAS,IAAI,EAAE,MAAM,KAAK,CAAC;EACvC,IAAI,KAAK,CAAC;EACV,SAAS,IAAI,EAAE,QAAQ,GAAG;CAC5B;CACA,MAAM,SAA0B,CAAC,GAAG,SAAS,QAAQ,CAAC,CAAC,CACpD,KAAK,CAAC,QAAQ,QAAQ;EACrB,MAAM,OAAO,GAAG,QAAQ,GAAG,MAAO,KAAK,IAAI,EAAE,QAAQ,IAAI,KAAK,IAAI,EAAE,QAAQ,IAAI,IAAI,CAAE;EACtF,OAAO;GACL;GACA,aAAa,KAAK;GAClB,UAAU,KAAK;GACf,SAAS,KAAK;GACd,GAAG,KAAK;GACR,SAAS,KAAK;EAChB;CACF,CAAC,CAAC,CACD,MAAM,GAAG,MAAM,KAAK,IAAI,EAAE,QAAQ,IAAI,KAAK,IAAI,EAAE,QAAQ,CAAC;CAE7D,MAAM,qBAAqB,QAAQ,QAAQ,MAAM,CAAC,SAAS,IAAI,CAAC,CAAC;CAEjE,OAAO;EAAE;EAAO;EAAQ,eAAe;EAAQ,aAAa;EAAS;CAAmB;AAC1F;AAIA,SAAS,OACP,QACA,UACA,QACA,MACe;CACf,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,IAAI,SAAS,QAAQ,OAAO,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;CACvE,IAAI,SAAS,OAAO,OAAO,KAAK,IAAI,GAAG,MAAM;CAK7C,OAHe,CAAC,GAAG,QAAQ,CAAC,CACzB,QAAQ,MAAM,OAAO,EAAE,QAAQ,YAAY,QAAQ,CAAC,CACpD,MAAM,GAAG,MAAM,EAAE,aAAa,EAAE,UACvB,CAAC,CAAC,EAAE,EAAE,QAAQ,WAAW;AACvC;AAEA,SAAS,YACP,IACA,IACA,YACA,MAC+B;CAC/B,MAAM,IAAI,GAAG;CACb,IAAI,IAAI,GAAG,OAAO;EAAE,KAAK;EAAY,MAAM;CAAW;CACtD,MAAM,MAAM,QAAQ,MAAM,IAAI,EAAE;CAChC,MAAM,UAAoB,CAAC;CAC3B,KAAK,IAAI,IAAI,GAAG,IAAI,YAAY,KAAK;EACnC,MAAM,KAAK,IAAI,MAAc,CAAC;EAC9B,MAAM,KAAK,IAAI,MAAc,CAAC;EAC9B,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,KAAK;GAC1B,MAAM,MAAM,KAAK,MAAM,IAAI,IAAI,CAAC;GAChC,GAAG,KAAK,GAAG;GACX,GAAG,KAAK,GAAG;EACb;EACA,MAAM,IAAI,SAAS,IAAI,EAAE;EACzB,IAAI,OAAO,SAAS,CAAC,GAAG,QAAQ,KAAK,CAAC;CACxC;CACA,QAAQ,MAAM,GAAG,MAAM,IAAI,CAAC;CAC5B,IAAI,QAAQ,WAAW,GAAG,OAAO;EAAE,KAAK;EAAY,MAAM;CAAW;CACrE,OAAO;EACL,KAAK,QAAQ,KAAK,MAAM,OAAQ,QAAQ,MAAM;EAC9C,MAAM,QAAQ,KAAK,IAAI,QAAQ,SAAS,GAAG,KAAK,MAAM,OAAQ,QAAQ,MAAM,CAAC;CAC/E;AACF"}
@@ -1,75 +0,0 @@
1
- import { a as RunRecord } from "./run-record-DTv1MdjK.js";
2
- import { o as OutcomeStore } from "./outcome-store-BYHIuO0e.js";
3
- //#region src/meta-eval/rubric-predictive-validity.d.ts
4
- interface RubricPredictiveValidityInput {
5
- /**
6
- * Canonical campaign output. Each record's `outcome.raw[<rubricId>]`
7
- * provides the eval score; missing keys are silently skipped per pair.
8
- */
9
- runs: RunRecord[];
10
- outcomes: OutcomeStore;
11
- /**
12
- * Outcome metric names to evaluate against. Each must appear in at
13
- * least one `DeploymentOutcome.metrics` keyspace; pairs with too few
14
- * joined samples are excluded from the result.
15
- */
16
- outcomeMetrics: string[];
17
- /**
18
- * Rubric ids to evaluate. Must appear as keys in `RunRecord.outcome.raw`.
19
- * If omitted, every numeric key in `outcome.raw` across the run set is
20
- * treated as a rubric.
21
- */
22
- rubrics?: string[];
23
- /** Minimum joined-sample count before a pair is reported. Default 8. */
24
- minSamples?: number;
25
- /** Bootstrap resamples for CI. Default 500. */
26
- bootstrapResamples?: number;
27
- /** Seed for the bootstrap. Absent, the seed is derived from the paired
28
- * observations, so the same input reproduces the same interval. */
29
- seed?: number;
30
- /**
31
- * Reduction when multiple outcomes attach to one runId. Default `'latest'`
32
- * (most recently captured).
33
- */
34
- reduction?: 'latest' | 'mean' | 'max';
35
- }
36
- interface RubricOutcomePair {
37
- rubric: string;
38
- outcome: string;
39
- n: number;
40
- pearson: number;
41
- spearman: number;
42
- ci95: {
43
- low: number;
44
- high: number;
45
- };
46
- /**
47
- * Verdict bucket. `load_bearing` ≥ 0.7, `informative` ≥ 0.4,
48
- * `decorative` < 0.4 in absolute correlation. A negative correlation
49
- * with a desired outcome is also `decorative` — actively misleading
50
- * is worse than uninformative.
51
- */
52
- verdict: 'load_bearing' | 'informative' | 'decorative';
53
- }
54
- interface RubricRanking {
55
- rubric: string;
56
- /** Outcome metric this rubric correlated best with. */
57
- bestOutcome: string;
58
- spearman: number;
59
- pearson: number;
60
- n: number;
61
- verdict: RubricOutcomePair['verdict'];
62
- }
63
- interface RubricPredictiveValidityReport {
64
- pairs: RubricOutcomePair[];
65
- /** Per-rubric best pair, sorted descending by |spearman|. */
66
- ranked: RubricRanking[];
67
- joinedSamples: number;
68
- skippedRuns: number;
69
- /** Rubrics that were declared but never produced a usable score. */
70
- rubricsWithoutData: string[];
71
- }
72
- declare function rubricPredictiveValidity(input: RubricPredictiveValidityInput): Promise<RubricPredictiveValidityReport>;
73
- //#endregion
74
- export { rubricPredictiveValidity as a, RubricRanking as i, RubricPredictiveValidityInput as n, RubricPredictiveValidityReport as r, RubricOutcomePair as t };
75
- //# sourceMappingURL=rubric-predictive-validity-Dl1dvKCv.d.ts.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"rubric-predictive-validity-Dl1dvKCv.d.ts","names":[],"sources":["../src/meta-eval/rubric-predictive-validity.ts"],"mappings":";;;UAqCiB;;;;;EAKf,MAAM;EACN,UAAU;;;;;;EAMV;;;;;;EAMA;;EAEA;;EAEA;;;EAGA;;;;;EAKA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;IAAQ;IAAa;;;;;;;;EAOrB;;UAGe;EACf;;EAEA;EACA;EACA;EACA;EACA,SAAS;;UAGM;EACf,OAAO;;EAEP,QAAQ;EACR;EACA;;EAEA;;iBAGoB,yBACpB,OAAO,gCACN,QAAQ"}
@@ -1,216 +0,0 @@
1
- import { c as validateRunRecord } from "./run-record-DQpSf7t-.js";
2
- //#region src/campaign/run-record.ts
3
- /**
4
- * A campaign cell carried a judge score without a `dimensions` record.
5
- *
6
- * Two exported types share the name `JudgeScore`: the campaign verdict
7
- * (`{ dimensions, composite, notes }` from `@tangle-network/agent-eval/campaign`)
8
- * and the root export's flat per-dimension row (`{ judgeName, dimension, score }`
9
- * from `@tangle-network/agent-eval`). Campaign aggregation accepts only the
10
- * campaign shape; the flat shape previously crashed here with an opaque
11
- * TypeError deep inside aggregation.
12
- */
13
- var CampaignJudgeScoreShapeError = class extends TypeError {
14
- constructor(judgeName) {
15
- super(`campaign cell judge '${judgeName}' carries a score without a 'dimensions' record. Use the campaign JudgeScore ({ dimensions, composite, notes }) from '@tangle-network/agent-eval/campaign'; the root export's JudgeScore ({ judgeName, dimension, score }) is a different type with the same name.`);
16
- this.name = "CampaignJudgeScoreShapeError";
17
- }
18
- };
19
- /**
20
- * Project one campaign cell into the canonical run format.
21
- *
22
- * A dispatch error establishes terminal execution failure. A judge error only
23
- * establishes that quality measurement failed after dispatch completed.
24
- * Failures without a stage remain unknown. No failure becomes a zero-quality
25
- * label.
26
- */
27
- function campaignCellToRunRecord(cell, options) {
28
- const quality = projectCampaignCellQuality(cell);
29
- const execution = campaignCellExecutionEvidence(cell);
30
- const judgeErrorCount = Math.max(quality.raw.judge_error_count ?? 0, execution.judgeErrorCount ?? 0);
31
- const cellCostProvenance = campaignCellCostProvenance(cell);
32
- const costProvenance = cellCostProvenance.kind === "uncaptured" && options.defaultCostUsd !== void 0 ? {
33
- kind: "estimated",
34
- usd: options.defaultCostUsd
35
- } : cellCostProvenance;
36
- const costUsd = costProvenance.kind === "uncaptured" ? null : costProvenance.usd;
37
- const raw = {
38
- ...finiteMetrics(options.raw),
39
- ...quality.raw,
40
- rep: cell.rep,
41
- duration_ms: cell.durationMs,
42
- ...costUsd === null ? {} : { cost_usd: costUsd },
43
- ...cellCostProvenance.kind === "uncaptured" ? { cost_known_subtotal_usd: cell.costUsd } : {},
44
- cost_observed: costProvenance.kind === "observed" ? 1 : 0,
45
- cost_estimated: costProvenance.kind === "estimated" ? 1 : 0,
46
- cost_uncaptured: costProvenance.kind === "uncaptured" ? 1 : 0,
47
- tokens_input: cell.tokenUsage.input,
48
- tokens_output: cell.tokenUsage.output,
49
- tokens_known: cell.tokenUsage.tokensKnown === false ? 0 : 1,
50
- latency_ms: cell.durationMs,
51
- ...execution.executionErrorCount === void 0 ? {} : { execution_error_count: execution.executionErrorCount },
52
- ...judgeErrorCount > 0 ? { judge_error_count: judgeErrorCount } : {},
53
- ...execution.unclassifiedErrorCount === void 0 ? {} : { unclassified_error_count: execution.unclassifiedErrorCount }
54
- };
55
- if (typeof cell.generation === "number") raw.generation = cell.generation;
56
- if (cell.tokenUsage.reasoning !== void 0) raw.tokens_reasoning = cell.tokenUsage.reasoning;
57
- if (cell.tokenUsage.cached !== void 0) raw.tokens_cached = cell.tokenUsage.cached;
58
- if (cell.tokenUsage.cacheWrite !== void 0) raw.tokens_cache_write = cell.tokenUsage.cacheWrite;
59
- if (cell.tokenUsage.tokensKnown !== false && costUsd !== null && costUsd > 0) raw.tokens_per_dollar = (cell.tokenUsage.input + cell.tokenUsage.output) / costUsd;
60
- if (costUsd !== null && quality.score !== void 0 && quality.score > .01) raw.cost_per_quality = costUsd / quality.score;
61
- const outcome = {
62
- raw,
63
- ...quality.judgeScores ? { judgeScores: quality.judgeScores } : {}
64
- };
65
- if (quality.score !== void 0) if (options.splitTag === "holdout") outcome.holdoutScore = quality.score;
66
- else outcome.searchScore = quality.score;
67
- return validateRunRecord({
68
- runId: options.runId,
69
- experimentId: options.experimentId,
70
- candidateId: options.candidateId,
71
- seed: options.seed ?? cell.seed,
72
- model: options.model,
73
- promptHash: options.promptHash,
74
- configHash: options.configHash,
75
- commitSha: options.commitSha,
76
- wallMs: cell.durationMs,
77
- costUsd,
78
- costProvenance,
79
- tokenUsage: { ...cell.tokenUsage },
80
- terminalOutcome: execution.terminalOutcome,
81
- ...execution.terminalFailureReason ? { terminalFailureReason: execution.terminalFailureReason } : {},
82
- outcome,
83
- splitTag: options.splitTag,
84
- scenarioId: options.scenarioId ?? cell.scenarioId,
85
- ...options.agentProfile ? { agentProfile: options.agentProfile } : {}
86
- });
87
- }
88
- /**
89
- * Validate the cost fields that cross campaign cache and RunRecord boundaries.
90
- * `costUsd` is a known subtotal for uncaptured cells, but it must equal the
91
- * authoritative total whenever that total is observed or estimated.
92
- */
93
- function campaignCellCostProvenance(cell) {
94
- if (!Number.isFinite(cell.costUsd) || cell.costUsd < 0) throw new Error(`campaign cell '${cell.cellId}' has invalid costUsd`);
95
- const provenance = cell.costProvenance;
96
- if (!provenance || typeof provenance !== "object") throw new Error(`campaign cell '${cell.cellId}' has no costProvenance`);
97
- if (provenance.kind === "uncaptured") {
98
- if (provenance.usd !== null) throw new Error(`campaign cell '${cell.cellId}' has invalid uncaptured costProvenance`);
99
- return {
100
- kind: "uncaptured",
101
- usd: null
102
- };
103
- }
104
- if (provenance.kind !== "observed" && provenance.kind !== "estimated" || !Number.isFinite(provenance.usd) || provenance.usd < 0) throw new Error(`campaign cell '${cell.cellId}' has invalid costProvenance`);
105
- if (provenance.usd !== cell.costUsd) throw new Error(`campaign cell '${cell.cellId}' has costUsd inconsistent with costProvenance`);
106
- return {
107
- kind: provenance.kind,
108
- usd: provenance.usd
109
- };
110
- }
111
- function campaignCellExecutionEvidence(cell) {
112
- if (cell.errorStage === "dispatch") return {
113
- terminalOutcome: "failed",
114
- executionErrorCount: 1,
115
- ...cell.error ? { terminalFailureReason: cell.error } : {}
116
- };
117
- if (cell.errorStage === "judge") return {
118
- terminalOutcome: "succeeded",
119
- executionErrorCount: 0,
120
- judgeErrorCount: 1
121
- };
122
- if (!cell.error) return {
123
- terminalOutcome: "succeeded",
124
- executionErrorCount: 0
125
- };
126
- return {
127
- terminalOutcome: "unknown",
128
- unclassifiedErrorCount: 1
129
- };
130
- }
131
- /**
132
- * Produce the only task-quality view used by campaign aggregates and exports.
133
- *
134
- * Successful judge results remain available for diagnosis after another judge
135
- * fails, but a task score exists only for an error-free cell whose reported
136
- * judge values are all finite.
137
- */
138
- function projectCampaignCellQuality(cell) {
139
- if (cell.errorStage === "dispatch") return {
140
- successfulJudgeScores: {},
141
- failedJudges: [],
142
- raw: {}
143
- };
144
- const perJudge = {};
145
- const successfulJudgeScores = {};
146
- const dimensionValues = /* @__PURE__ */ new Map();
147
- const composites = [];
148
- const notes = [];
149
- const failedJudges = new Set(cell.errorStage === "judge" ? [cell.errorJudge ?? "unknown-judge"] : []);
150
- const raw = {};
151
- for (const [judgeName, score] of Object.entries(cell.judgeScores)) {
152
- const dimensionsShape = score.dimensions;
153
- if (typeof dimensionsShape !== "object" || dimensionsShape === null || Array.isArray(dimensionsShape)) throw new CampaignJudgeScoreShapeError(judgeName);
154
- const finiteDimensions = Object.values(score.dimensions).every(Number.isFinite);
155
- if (score.failed || !Number.isFinite(score.composite) || !finiteDimensions) {
156
- failedJudges.add(judgeName);
157
- continue;
158
- }
159
- composites.push(score.composite);
160
- successfulJudgeScores[judgeName] = score;
161
- const dimensions = { ...score.dimensions };
162
- perJudge[judgeName] = dimensions;
163
- for (const [dimension, value] of Object.entries(dimensions)) {
164
- raw[`${judgeName}.${dimension}`] = value;
165
- const values = dimensionValues.get(dimension) ?? [];
166
- values.push(value);
167
- dimensionValues.set(dimension, values);
168
- }
169
- if (score.notes) notes.push(`${judgeName}: ${score.notes}`);
170
- for (const failedJudge of score.failedJudges ?? []) failedJudges.add(`${judgeName}/${failedJudge}`);
171
- }
172
- if (failedJudges.size > 0) raw.judge_error_count = failedJudges.size;
173
- const sortedFailedJudges = [...failedJudges].sort();
174
- if (composites.length === 0) return {
175
- successfulJudgeScores,
176
- failedJudges: sortedFailedJudges,
177
- raw
178
- };
179
- const composite = mean(composites);
180
- const perDimMean = Object.fromEntries([...dimensionValues.entries()].map(([dimension, values]) => [dimension, mean(values)]));
181
- const complete = cell.error === void 0 && cell.errorStage === void 0 && failedJudges.size === 0;
182
- if (complete) raw.composite = composite;
183
- return {
184
- ...complete ? { score: composite } : {},
185
- raw,
186
- successfulJudgeScores,
187
- failedJudges: sortedFailedJudges,
188
- judgeScores: {
189
- perJudge,
190
- perDimMean,
191
- composite,
192
- ...sortedFailedJudges.length > 0 ? { failedJudges: sortedFailedJudges } : {},
193
- ...notes.length > 0 ? { notes: notes.join(" | ") } : {}
194
- }
195
- };
196
- }
197
- /** Read the canonical task score without recomputing cell quality. */
198
- function campaignCellTaskScore(cell) {
199
- return projectCampaignCellQuality(cell).score;
200
- }
201
- /** Read canonical successful judge dimensions without recomputing cell quality. */
202
- function campaignCellJudgeDimensions(cell) {
203
- return projectCampaignCellQuality(cell).judgeScores?.perJudge ?? {};
204
- }
205
- function finiteMetrics(metrics) {
206
- const finite = {};
207
- for (const [key, value] of Object.entries(metrics ?? {})) if (Number.isFinite(value)) finite[key] = value;
208
- return finite;
209
- }
210
- function mean(values) {
211
- return values.reduce((sum, value) => sum + value, 0) / values.length;
212
- }
213
- //#endregion
214
- export { campaignCellToRunRecord as a, campaignCellTaskScore as i, campaignCellExecutionEvidence as n, projectCampaignCellQuality as o, campaignCellJudgeDimensions as r, campaignCellCostProvenance as t };
215
-
216
- //# sourceMappingURL=run-record-CR63CpHK.js.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"run-record-CR63CpHK.js","names":[],"sources":["../src/campaign/run-record.ts"],"sourcesContent":["import type { AgentProfileCell } from '../agent-profile-cell'\nimport type { CostProvenance } from '../cost-ledger'\nimport type {\n JudgeScoresRecord,\n RunOutcome,\n RunRecord,\n RunSplitTag,\n RunTerminalOutcome,\n} from '../run-record'\nimport { validateRunRecord } from '../run-record'\nimport type { CampaignCellResult, JudgeScore } from './types'\n\nexport interface CampaignCellRunRecordOptions {\n runId: string\n experimentId: string\n candidateId: string\n model: string\n promptHash: string\n configHash: string\n commitSha: string\n splitTag: RunSplitTag\n seed?: number\n scenarioId?: string\n defaultCostUsd?: number\n agentProfile?: AgentProfileCell\n raw?: Record<string, number>\n}\n\nexport interface CampaignCellQualityProjection {\n score?: number\n judgeScores?: JudgeScoresRecord\n successfulJudgeScores: Record<string, JudgeScore>\n failedJudges: string[]\n raw: Record<string, number>\n}\n\n/**\n * A campaign cell carried a judge score without a `dimensions` record.\n *\n * Two exported types share the name `JudgeScore`: the campaign verdict\n * (`{ dimensions, composite, notes }` from `@tangle-network/agent-eval/campaign`)\n * and the root export's flat per-dimension row (`{ judgeName, dimension, score }`\n * from `@tangle-network/agent-eval`). Campaign aggregation accepts only the\n * campaign shape; the flat shape previously crashed here with an opaque\n * TypeError deep inside aggregation.\n */\nexport class CampaignJudgeScoreShapeError extends TypeError {\n constructor(judgeName: string) {\n super(\n `campaign cell judge '${judgeName}' carries a score without a 'dimensions' record. ` +\n `Use the campaign JudgeScore ({ dimensions, composite, notes }) from ` +\n `'@tangle-network/agent-eval/campaign'; the root export's JudgeScore ` +\n `({ judgeName, dimension, score }) is a different type with the same name.`,\n )\n this.name = 'CampaignJudgeScoreShapeError'\n }\n}\n\nexport interface CampaignCellExecutionEvidence {\n terminalOutcome: RunTerminalOutcome\n executionErrorCount?: number\n judgeErrorCount?: number\n unclassifiedErrorCount?: number\n terminalFailureReason?: string\n}\n\n/**\n * Project one campaign cell into the canonical run format.\n *\n * A dispatch error establishes terminal execution failure. A judge error only\n * establishes that quality measurement failed after dispatch completed.\n * Failures without a stage remain unknown. No failure becomes a zero-quality\n * label.\n */\nexport function campaignCellToRunRecord<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n options: CampaignCellRunRecordOptions,\n): RunRecord {\n const quality = projectCampaignCellQuality(cell)\n const execution = campaignCellExecutionEvidence(cell)\n const judgeErrorCount = Math.max(\n quality.raw.judge_error_count ?? 0,\n execution.judgeErrorCount ?? 0,\n )\n const cellCostProvenance = campaignCellCostProvenance(cell)\n const costProvenance: CostProvenance =\n cellCostProvenance.kind === 'uncaptured' && options.defaultCostUsd !== undefined\n ? { kind: 'estimated', usd: options.defaultCostUsd }\n : cellCostProvenance\n const costUsd = costProvenance.kind === 'uncaptured' ? null : costProvenance.usd\n const raw: Record<string, number> = {\n ...finiteMetrics(options.raw),\n ...quality.raw,\n rep: cell.rep,\n duration_ms: cell.durationMs,\n ...(costUsd === null ? {} : { cost_usd: costUsd }),\n // Retain the observed subtotal even when the caller supplies an estimated total.\n ...(cellCostProvenance.kind === 'uncaptured' ? { cost_known_subtotal_usd: cell.costUsd } : {}),\n cost_observed: costProvenance.kind === 'observed' ? 1 : 0,\n cost_estimated: costProvenance.kind === 'estimated' ? 1 : 0,\n cost_uncaptured: costProvenance.kind === 'uncaptured' ? 1 : 0,\n tokens_input: cell.tokenUsage.input,\n tokens_output: cell.tokenUsage.output,\n tokens_known: cell.tokenUsage.tokensKnown === false ? 0 : 1,\n latency_ms: cell.durationMs,\n ...(execution.executionErrorCount === undefined\n ? {}\n : { execution_error_count: execution.executionErrorCount }),\n ...(judgeErrorCount > 0 ? { judge_error_count: judgeErrorCount } : {}),\n ...(execution.unclassifiedErrorCount === undefined\n ? {}\n : { unclassified_error_count: execution.unclassifiedErrorCount }),\n }\n if (typeof cell.generation === 'number') raw.generation = cell.generation\n if (cell.tokenUsage.reasoning !== undefined) {\n raw.tokens_reasoning = cell.tokenUsage.reasoning\n }\n if (cell.tokenUsage.cached !== undefined) raw.tokens_cached = cell.tokenUsage.cached\n if (cell.tokenUsage.cacheWrite !== undefined) {\n raw.tokens_cache_write = cell.tokenUsage.cacheWrite\n }\n if (cell.tokenUsage.tokensKnown !== false && costUsd !== null && costUsd > 0) {\n raw.tokens_per_dollar = (cell.tokenUsage.input + cell.tokenUsage.output) / costUsd\n }\n if (costUsd !== null && quality.score !== undefined && quality.score > 0.01) {\n raw.cost_per_quality = costUsd / quality.score\n }\n\n const outcome: RunOutcome = {\n raw,\n ...(quality.judgeScores ? { judgeScores: quality.judgeScores } : {}),\n }\n if (quality.score !== undefined) {\n if (options.splitTag === 'holdout') outcome.holdoutScore = quality.score\n else outcome.searchScore = quality.score\n }\n\n return validateRunRecord({\n runId: options.runId,\n experimentId: options.experimentId,\n candidateId: options.candidateId,\n seed: options.seed ?? cell.seed,\n model: options.model,\n promptHash: options.promptHash,\n configHash: options.configHash,\n commitSha: options.commitSha,\n wallMs: cell.durationMs,\n costUsd,\n costProvenance,\n tokenUsage: { ...cell.tokenUsage },\n terminalOutcome: execution.terminalOutcome,\n ...(execution.terminalFailureReason\n ? { terminalFailureReason: execution.terminalFailureReason }\n : {}),\n outcome,\n splitTag: options.splitTag,\n scenarioId: options.scenarioId ?? cell.scenarioId,\n ...(options.agentProfile ? { agentProfile: options.agentProfile } : {}),\n })\n}\n\n/**\n * Validate the cost fields that cross campaign cache and RunRecord boundaries.\n * `costUsd` is a known subtotal for uncaptured cells, but it must equal the\n * authoritative total whenever that total is observed or estimated.\n */\nexport function campaignCellCostProvenance<TArtifact>(\n cell: Pick<CampaignCellResult<TArtifact>, 'cellId' | 'costUsd' | 'costProvenance'>,\n): CostProvenance {\n if (!Number.isFinite(cell.costUsd) || cell.costUsd < 0) {\n throw new Error(`campaign cell '${cell.cellId}' has invalid costUsd`)\n }\n const provenance = cell.costProvenance\n if (!provenance || typeof provenance !== 'object') {\n throw new Error(`campaign cell '${cell.cellId}' has no costProvenance`)\n }\n if (provenance.kind === 'uncaptured') {\n if (provenance.usd !== null) {\n throw new Error(`campaign cell '${cell.cellId}' has invalid uncaptured costProvenance`)\n }\n return { kind: 'uncaptured', usd: null }\n }\n if (\n (provenance.kind !== 'observed' && provenance.kind !== 'estimated') ||\n !Number.isFinite(provenance.usd) ||\n provenance.usd < 0\n ) {\n throw new Error(`campaign cell '${cell.cellId}' has invalid costProvenance`)\n }\n if (provenance.usd !== cell.costUsd) {\n throw new Error(`campaign cell '${cell.cellId}' has costUsd inconsistent with costProvenance`)\n }\n return { kind: provenance.kind, usd: provenance.usd }\n}\n\nexport function campaignCellExecutionEvidence<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): CampaignCellExecutionEvidence {\n if (cell.errorStage === 'dispatch') {\n return {\n terminalOutcome: 'failed',\n executionErrorCount: 1,\n ...(cell.error ? { terminalFailureReason: cell.error } : {}),\n }\n }\n if (cell.errorStage === 'judge') {\n return {\n terminalOutcome: 'succeeded',\n executionErrorCount: 0,\n judgeErrorCount: 1,\n }\n }\n if (!cell.error) {\n return { terminalOutcome: 'succeeded', executionErrorCount: 0 }\n }\n return {\n terminalOutcome: 'unknown',\n unclassifiedErrorCount: 1,\n }\n}\n\n/**\n * Produce the only task-quality view used by campaign aggregates and exports.\n *\n * Successful judge results remain available for diagnosis after another judge\n * fails, but a task score exists only for an error-free cell whose reported\n * judge values are all finite.\n */\nexport function projectCampaignCellQuality<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): CampaignCellQualityProjection {\n if (cell.errorStage === 'dispatch') {\n return { successfulJudgeScores: {}, failedJudges: [], raw: {} }\n }\n\n const perJudge: Record<string, Record<string, number>> = {}\n const successfulJudgeScores: Record<string, JudgeScore> = {}\n const dimensionValues = new Map<string, number[]>()\n const composites: number[] = []\n const notes: string[] = []\n const failedJudges = new Set<string>(\n cell.errorStage === 'judge' ? [cell.errorJudge ?? 'unknown-judge'] : [],\n )\n const raw: Record<string, number> = {}\n\n for (const [judgeName, score] of Object.entries(cell.judgeScores)) {\n const dimensionsShape = (score as { dimensions?: unknown }).dimensions\n if (\n typeof dimensionsShape !== 'object' ||\n dimensionsShape === null ||\n Array.isArray(dimensionsShape)\n ) {\n throw new CampaignJudgeScoreShapeError(judgeName)\n }\n const finiteDimensions = Object.values(score.dimensions).every(Number.isFinite)\n if (score.failed || !Number.isFinite(score.composite) || !finiteDimensions) {\n failedJudges.add(judgeName)\n continue\n }\n\n composites.push(score.composite)\n successfulJudgeScores[judgeName] = score\n const dimensions = { ...score.dimensions }\n perJudge[judgeName] = dimensions\n for (const [dimension, value] of Object.entries(dimensions)) {\n raw[`${judgeName}.${dimension}`] = value\n const values = dimensionValues.get(dimension) ?? []\n values.push(value)\n dimensionValues.set(dimension, values)\n }\n if (score.notes) notes.push(`${judgeName}: ${score.notes}`)\n for (const failedJudge of score.failedJudges ?? []) {\n failedJudges.add(`${judgeName}/${failedJudge}`)\n }\n }\n\n if (failedJudges.size > 0) raw.judge_error_count = failedJudges.size\n const sortedFailedJudges = [...failedJudges].sort()\n if (composites.length === 0) {\n return {\n successfulJudgeScores,\n failedJudges: sortedFailedJudges,\n raw,\n }\n }\n\n const composite = mean(composites)\n const perDimMean = Object.fromEntries(\n [...dimensionValues.entries()].map(([dimension, values]) => [dimension, mean(values)]),\n )\n const complete =\n cell.error === undefined && cell.errorStage === undefined && failedJudges.size === 0\n if (complete) raw.composite = composite\n\n return {\n ...(complete ? { score: composite } : {}),\n raw,\n successfulJudgeScores,\n failedJudges: sortedFailedJudges,\n judgeScores: {\n perJudge,\n perDimMean,\n composite,\n ...(sortedFailedJudges.length > 0 ? { failedJudges: sortedFailedJudges } : {}),\n ...(notes.length > 0 ? { notes: notes.join(' | ') } : {}),\n },\n }\n}\n\n/** Read the canonical task score without recomputing cell quality. */\nexport function campaignCellTaskScore<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): number | undefined {\n return projectCampaignCellQuality(cell).score\n}\n\n/** Read canonical successful judge dimensions without recomputing cell quality. */\nexport function campaignCellJudgeDimensions<TArtifact>(\n cell: CampaignCellResult<TArtifact>,\n): Record<string, Record<string, number>> {\n return projectCampaignCellQuality(cell).judgeScores?.perJudge ?? {}\n}\n\nfunction finiteMetrics(metrics: Record<string, number> | undefined): Record<string, number> {\n const finite: Record<string, number> = {}\n for (const [key, value] of Object.entries(metrics ?? {})) {\n if (Number.isFinite(value)) finite[key] = value\n }\n return finite\n}\n\nfunction mean(values: number[]): number {\n return values.reduce((sum, value) => sum + value, 0) / values.length\n}\n"],"mappings":";;;;;;;;;;;;AA8CA,IAAa,+BAAb,cAAkD,UAAU;CAC1D,YAAY,WAAmB;EAC7B,MACE,wBAAwB,UAAU,mQAIpC;EACA,KAAK,OAAO;CACd;AACF;;;;;;;;;AAkBA,SAAgB,wBACd,MACA,SACW;CACX,MAAM,UAAU,2BAA2B,IAAI;CAC/C,MAAM,YAAY,8BAA8B,IAAI;CACpD,MAAM,kBAAkB,KAAK,IAC3B,QAAQ,IAAI,qBAAqB,GACjC,UAAU,mBAAmB,CAC/B;CACA,MAAM,qBAAqB,2BAA2B,IAAI;CAC1D,MAAM,iBACJ,mBAAmB,SAAS,gBAAgB,QAAQ,mBAAmB,KAAA,IACnE;EAAE,MAAM;EAAa,KAAK,QAAQ;CAAe,IACjD;CACN,MAAM,UAAU,eAAe,SAAS,eAAe,OAAO,eAAe;CAC7E,MAAM,MAA8B;EAClC,GAAG,cAAc,QAAQ,GAAG;EAC5B,GAAG,QAAQ;EACX,KAAK,KAAK;EACV,aAAa,KAAK;EAClB,GAAI,YAAY,OAAO,CAAC,IAAI,EAAE,UAAU,QAAQ;EAEhD,GAAI,mBAAmB,SAAS,eAAe,EAAE,yBAAyB,KAAK,QAAQ,IAAI,CAAC;EAC5F,eAAe,eAAe,SAAS,aAAa,IAAI;EACxD,gBAAgB,eAAe,SAAS,cAAc,IAAI;EAC1D,iBAAiB,eAAe,SAAS,eAAe,IAAI;EAC5D,cAAc,KAAK,WAAW;EAC9B,eAAe,KAAK,WAAW;EAC/B,cAAc,KAAK,WAAW,gBAAgB,QAAQ,IAAI;EAC1D,YAAY,KAAK;EACjB,GAAI,UAAU,wBAAwB,KAAA,IAClC,CAAC,IACD,EAAE,uBAAuB,UAAU,oBAAoB;EAC3D,GAAI,kBAAkB,IAAI,EAAE,mBAAmB,gBAAgB,IAAI,CAAC;EACpE,GAAI,UAAU,2BAA2B,KAAA,IACrC,CAAC,IACD,EAAE,0BAA0B,UAAU,uBAAuB;CACnE;CACA,IAAI,OAAO,KAAK,eAAe,UAAU,IAAI,aAAa,KAAK;CAC/D,IAAI,KAAK,WAAW,cAAc,KAAA,GAChC,IAAI,mBAAmB,KAAK,WAAW;CAEzC,IAAI,KAAK,WAAW,WAAW,KAAA,GAAW,IAAI,gBAAgB,KAAK,WAAW;CAC9E,IAAI,KAAK,WAAW,eAAe,KAAA,GACjC,IAAI,qBAAqB,KAAK,WAAW;CAE3C,IAAI,KAAK,WAAW,gBAAgB,SAAS,YAAY,QAAQ,UAAU,GACzE,IAAI,qBAAqB,KAAK,WAAW,QAAQ,KAAK,WAAW,UAAU;CAE7E,IAAI,YAAY,QAAQ,QAAQ,UAAU,KAAA,KAAa,QAAQ,QAAQ,KACrE,IAAI,mBAAmB,UAAU,QAAQ;CAG3C,MAAM,UAAsB;EAC1B;EACA,GAAI,QAAQ,cAAc,EAAE,aAAa,QAAQ,YAAY,IAAI,CAAC;CACpE;CACA,IAAI,QAAQ,UAAU,KAAA,GACpB,IAAI,QAAQ,aAAa,WAAW,QAAQ,eAAe,QAAQ;MAC9D,QAAQ,cAAc,QAAQ;CAGrC,OAAO,kBAAkB;EACvB,OAAO,QAAQ;EACf,cAAc,QAAQ;EACtB,aAAa,QAAQ;EACrB,MAAM,QAAQ,QAAQ,KAAK;EAC3B,OAAO,QAAQ;EACf,YAAY,QAAQ;EACpB,YAAY,QAAQ;EACpB,WAAW,QAAQ;EACnB,QAAQ,KAAK;EACb;EACA;EACA,YAAY,EAAE,GAAG,KAAK,WAAW;EACjC,iBAAiB,UAAU;EAC3B,GAAI,UAAU,wBACV,EAAE,uBAAuB,UAAU,sBAAsB,IACzD,CAAC;EACL;EACA,UAAU,QAAQ;EAClB,YAAY,QAAQ,cAAc,KAAK;EACvC,GAAI,QAAQ,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;CACvE,CAAC;AACH;;;;;;AAOA,SAAgB,2BACd,MACgB;CAChB,IAAI,CAAC,OAAO,SAAS,KAAK,OAAO,KAAK,KAAK,UAAU,GACnD,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,sBAAsB;CAEtE,MAAM,aAAa,KAAK;CACxB,IAAI,CAAC,cAAc,OAAO,eAAe,UACvC,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,wBAAwB;CAExE,IAAI,WAAW,SAAS,cAAc;EACpC,IAAI,WAAW,QAAQ,MACrB,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,wCAAwC;EAExF,OAAO;GAAE,MAAM;GAAc,KAAK;EAAK;CACzC;CACA,IACG,WAAW,SAAS,cAAc,WAAW,SAAS,eACvD,CAAC,OAAO,SAAS,WAAW,GAAG,KAC/B,WAAW,MAAM,GAEjB,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,6BAA6B;CAE7E,IAAI,WAAW,QAAQ,KAAK,SAC1B,MAAM,IAAI,MAAM,kBAAkB,KAAK,OAAO,+CAA+C;CAE/F,OAAO;EAAE,MAAM,WAAW;EAAM,KAAK,WAAW;CAAI;AACtD;AAEA,SAAgB,8BACd,MAC+B;CAC/B,IAAI,KAAK,eAAe,YACtB,OAAO;EACL,iBAAiB;EACjB,qBAAqB;EACrB,GAAI,KAAK,QAAQ,EAAE,uBAAuB,KAAK,MAAM,IAAI,CAAC;CAC5D;CAEF,IAAI,KAAK,eAAe,SACtB,OAAO;EACL,iBAAiB;EACjB,qBAAqB;EACrB,iBAAiB;CACnB;CAEF,IAAI,CAAC,KAAK,OACR,OAAO;EAAE,iBAAiB;EAAa,qBAAqB;CAAE;CAEhE,OAAO;EACL,iBAAiB;EACjB,wBAAwB;CAC1B;AACF;;;;;;;;AASA,SAAgB,2BACd,MAC+B;CAC/B,IAAI,KAAK,eAAe,YACtB,OAAO;EAAE,uBAAuB,CAAC;EAAG,cAAc,CAAC;EAAG,KAAK,CAAC;CAAE;CAGhE,MAAM,WAAmD,CAAC;CAC1D,MAAM,wBAAoD,CAAC;CAC3D,MAAM,kCAAkB,IAAI,IAAsB;CAClD,MAAM,aAAuB,CAAC;CAC9B,MAAM,QAAkB,CAAC;CACzB,MAAM,eAAe,IAAI,IACvB,KAAK,eAAe,UAAU,CAAC,KAAK,cAAc,eAAe,IAAI,CAAC,CACxE;CACA,MAAM,MAA8B,CAAC;CAErC,KAAK,MAAM,CAAC,WAAW,UAAU,OAAO,QAAQ,KAAK,WAAW,GAAG;EACjE,MAAM,kBAAmB,MAAmC;EAC5D,IACE,OAAO,oBAAoB,YAC3B,oBAAoB,QACpB,MAAM,QAAQ,eAAe,GAE7B,MAAM,IAAI,6BAA6B,SAAS;EAElD,MAAM,mBAAmB,OAAO,OAAO,MAAM,UAAU,CAAC,CAAC,MAAM,OAAO,QAAQ;EAC9E,IAAI,MAAM,UAAU,CAAC,OAAO,SAAS,MAAM,SAAS,KAAK,CAAC,kBAAkB;GAC1E,aAAa,IAAI,SAAS;GAC1B;EACF;EAEA,WAAW,KAAK,MAAM,SAAS;EAC/B,sBAAsB,aAAa;EACnC,MAAM,aAAa,EAAE,GAAG,MAAM,WAAW;EACzC,SAAS,aAAa;EACtB,KAAK,MAAM,CAAC,WAAW,UAAU,OAAO,QAAQ,UAAU,GAAG;GAC3D,IAAI,GAAG,UAAU,GAAG,eAAe;GACnC,MAAM,SAAS,gBAAgB,IAAI,SAAS,KAAK,CAAC;GAClD,OAAO,KAAK,KAAK;GACjB,gBAAgB,IAAI,WAAW,MAAM;EACvC;EACA,IAAI,MAAM,OAAO,MAAM,KAAK,GAAG,UAAU,IAAI,MAAM,OAAO;EAC1D,KAAK,MAAM,eAAe,MAAM,gBAAgB,CAAC,GAC/C,aAAa,IAAI,GAAG,UAAU,GAAG,aAAa;CAElD;CAEA,IAAI,aAAa,OAAO,GAAG,IAAI,oBAAoB,aAAa;CAChE,MAAM,qBAAqB,CAAC,GAAG,YAAY,CAAC,CAAC,KAAK;CAClD,IAAI,WAAW,WAAW,GACxB,OAAO;EACL;EACA,cAAc;EACd;CACF;CAGF,MAAM,YAAY,KAAK,UAAU;CACjC,MAAM,aAAa,OAAO,YACxB,CAAC,GAAG,gBAAgB,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,WAAW,YAAY,CAAC,WAAW,KAAK,MAAM,CAAC,CAAC,CACvF;CACA,MAAM,WACJ,KAAK,UAAU,KAAA,KAAa,KAAK,eAAe,KAAA,KAAa,aAAa,SAAS;CACrF,IAAI,UAAU,IAAI,YAAY;CAE9B,OAAO;EACL,GAAI,WAAW,EAAE,OAAO,UAAU,IAAI,CAAC;EACvC;EACA;EACA,cAAc;EACd,aAAa;GACX;GACA;GACA;GACA,GAAI,mBAAmB,SAAS,IAAI,EAAE,cAAc,mBAAmB,IAAI,CAAC;GAC5E,GAAI,MAAM,SAAS,IAAI,EAAE,OAAO,MAAM,KAAK,KAAK,EAAE,IAAI,CAAC;EACzD;CACF;AACF;;AAGA,SAAgB,sBACd,MACoB;CACpB,OAAO,2BAA2B,IAAI,CAAC,CAAC;AAC1C;;AAGA,SAAgB,4BACd,MACwC;CACxC,OAAO,2BAA2B,IAAI,CAAC,CAAC,aAAa,YAAY,CAAC;AACpE;AAEA,SAAS,cAAc,SAAqE;CAC1F,MAAM,SAAiC,CAAC;CACxC,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,WAAW,CAAC,CAAC,GACrD,IAAI,OAAO,SAAS,KAAK,GAAG,OAAO,OAAO;CAE5C,OAAO;AACT;AAEA,SAAS,KAAK,QAA0B;CACtC,OAAO,OAAO,QAAQ,KAAK,UAAU,MAAM,OAAO,CAAC,IAAI,OAAO;AAChE"}