@tangle-network/agent-eval 0.179.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (236) hide show
  1. package/CHANGELOG.md +66 -0
  2. package/README.md +119 -146
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +4 -4
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-CY6Dg5t5.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-CY6Dg5t5.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +5 -8
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +11 -10
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{default-registry-BryMEmr8.js → default-registry-aL7xUrUz.js} +2 -2
  38. package/dist/{default-registry-BryMEmr8.js.map → default-registry-aL7xUrUz.js.map} +1 -1
  39. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  40. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  41. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  42. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  43. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  44. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  45. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  46. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  47. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  48. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  49. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  50. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  51. package/dist/experiment/index.d.ts +27 -477
  52. package/dist/experiment/index.d.ts.map +1 -1
  53. package/dist/experiment/index.js +95 -559
  54. package/dist/experiment/index.js.map +1 -1
  55. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  56. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  57. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  58. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  59. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  60. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  61. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  62. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  63. package/dist/hosted/index.d.ts +2 -2
  64. package/dist/hosted/index.d.ts.map +1 -1
  65. package/dist/hosted/index.js +1 -1
  66. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  67. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  68. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  69. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  70. package/dist/{index-CbLmrWCa.d.ts → index-DNntP4ch.d.ts} +8 -8
  71. package/dist/{index-CbLmrWCa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  72. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  73. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  74. package/dist/index.d.ts +28 -28
  75. package/dist/index.js +25 -16
  76. package/dist/index.js.map +1 -1
  77. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  78. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  79. package/dist/{integrity-DsHWCebQ.js → integrity-DH5ng72x.js} +2 -2
  80. package/dist/{integrity-DsHWCebQ.js.map → integrity-DH5ng72x.js.map} +1 -1
  81. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  82. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  83. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  84. package/dist/journal-Cs9f7385.js.map +1 -0
  85. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  86. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  87. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  88. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  89. package/dist/ledger-core/index.d.ts +1 -1
  90. package/dist/ledger-core/index.js +1 -1
  91. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  92. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  93. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  94. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  95. package/dist/meta-eval/index.d.ts +138 -7
  96. package/dist/meta-eval/index.d.ts.map +1 -1
  97. package/dist/meta-eval/index.js +245 -97
  98. package/dist/meta-eval/index.js.map +1 -1
  99. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  100. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  101. package/dist/multishot/golden/index.d.ts +1 -1
  102. package/dist/multishot/index.d.ts +2 -2
  103. package/dist/openapi.json +1 -1
  104. package/dist/outcome-store-BXlkwMPR.js +131 -0
  105. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  106. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  107. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  108. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  109. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  110. package/dist/pipelines/index.js +1 -1
  111. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  112. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  113. package/dist/profile-cell.d.ts +1 -1
  114. package/dist/profile-cell.js +1 -1
  115. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  116. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  117. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  118. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  119. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  120. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  121. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  122. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  123. package/dist/{report-command-DKlXfU5r.js → report-command-V1ecVgAv.js} +27 -3
  124. package/dist/report-command-V1ecVgAv.js.map +1 -0
  125. package/dist/reporting.d.ts +4 -4
  126. package/dist/reporting.js +3 -3
  127. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  128. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  130. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  131. package/dist/rl.d.ts +53 -99
  132. package/dist/rl.d.ts.map +1 -1
  133. package/dist/rl.js +182 -169
  134. package/dist/rl.js.map +1 -1
  135. package/dist/rollout/index.d.ts +1 -1
  136. package/dist/rollout/index.js +2 -2
  137. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  138. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  139. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  140. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  141. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  142. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  143. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  144. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  145. package/dist/run-record-Br-Yzt_k.js +464 -0
  146. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  147. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  148. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  149. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  150. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  151. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  152. package/dist/sequential-DAsyV2T9.js.map +1 -0
  153. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  154. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  155. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  156. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  157. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  158. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  159. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  160. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  161. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  162. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  163. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  164. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  165. package/dist/supervisor-run/index.d.ts +4 -2
  166. package/dist/supervisor-run/index.d.ts.map +1 -1
  167. package/dist/supervisor-run/index.js +3 -3
  168. package/dist/{terminal-record-Ce9_UjRz.js → terminal-record-BtPwKTSr.js} +58 -26
  169. package/dist/terminal-record-BtPwKTSr.js.map +1 -0
  170. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  171. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  172. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  173. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  174. package/dist/trace-repair/index.d.ts +2 -2
  175. package/dist/traces.d.ts +6 -6
  176. package/dist/traces.js +1 -1
  177. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  178. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  179. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  180. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  181. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  182. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  183. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  184. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  185. package/dist/{types-vUdAx2Cj.d.ts → types-lPkDQNqJ.d.ts} +20 -2
  186. package/dist/{types-vUdAx2Cj.d.ts.map → types-lPkDQNqJ.d.ts.map} +1 -1
  187. package/dist/wire/index.d.ts +2 -2
  188. package/docs/adapters-observability.md +14 -0
  189. package/docs/campaign-proposers.md +86 -128
  190. package/docs/charter.md +108 -112
  191. package/docs/concepts.md +157 -69
  192. package/docs/design/mlbenchmarks-book-review.md +440 -0
  193. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  194. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  195. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  196. package/docs/design/self-improvement-evidence-audit.md +263 -0
  197. package/docs/design.md +2 -1
  198. package/docs/eval-surface-map.md +95 -42
  199. package/docs/evaluation-integrity.md +220 -0
  200. package/docs/experiment.md +111 -55
  201. package/docs/feature-guide.md +5 -6
  202. package/docs/hosted-ingest-spec.md +4 -11
  203. package/docs/insight-report.md +187 -455
  204. package/docs/outcome-validity.md +182 -0
  205. package/docs/product-eval-adoption.md +1 -2
  206. package/docs/research-report-methodology.md +7 -7
  207. package/docs/search-history-receipts.md +8 -0
  208. package/docs/statistical-evidence.md +129 -0
  209. package/docs/verdicts.md +76 -49
  210. package/package.json +1 -1
  211. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  212. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  213. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  214. package/dist/client-BlLY6o2w.js.map +0 -1
  215. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  216. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  217. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  218. package/dist/index-BxWvILU8.d.ts.map +0 -1
  219. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  220. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  221. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  222. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  223. package/dist/outcome-store-ChBKlTd_.js +0 -75
  224. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  225. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  226. package/dist/report-command-DKlXfU5r.js.map +0 -1
  227. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  228. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  229. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  230. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  231. package/dist/run-record-CR63CpHK.js +0 -216
  232. package/dist/run-record-CR63CpHK.js.map +0 -1
  233. package/dist/sequential-B5gXgcyp.js.map +0 -1
  234. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  235. package/dist/terminal-record-Ce9_UjRz.js.map +0 -1
  236. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
@@ -1,6 +1,8 @@
1
- import { a as RunRecord } from "./run-record-DTv1MdjK.js";
2
- import { a as PairedPromotionDecision, d as PairedBootstrapResult, i as PairedMcNemarEvidence, r as PairedDecisionStatistic, t as PairedDecisionMethod, u as PairedBootstrapOptions } from "./paired-promotion-decision-CGzg0cI_.js";
3
- import { R as Scenario, a as CampaignResult, b as GenerationRecord, h as GateContext, j as MutableSurface, p as Gate, v as GateResult, w as JudgeScore } from "./types-C34V4Vto.js";
1
+ import { a as RunRecord } from "./run-record-BiTWauyO.js";
2
+ import { a as PairedPromotionDecision, d as PairedBootstrapResult, i as PairedMcNemarEvidence, r as PairedDecisionStatistic, t as PairedDecisionMethod, u as PairedBootstrapOptions } from "./paired-promotion-decision-DPsMQm-0.js";
3
+ import { R as Scenario, a as CampaignResult, b as GenerationRecord, h as GateContext, j as MutableSurface, p as Gate, v as GateResult, w as JudgeScore } from "./types-CS0qk_Yp.js";
4
+ import { r as LedgerHash } from "./canonical-CFpojCN5.js";
5
+ import { z } from "zod";
4
6
  //#region src/statistics/paired-binary.d.ts
5
7
  /** A binomial proportion estimate with a confidence interval. */
6
8
  interface ProportionInterval {
@@ -323,6 +325,98 @@ interface EProcess {
323
325
  */
324
326
  declare function eProcess(opts?: EProcessOptions): EProcess;
325
327
  //#endregion
328
+ //#region src/experiment/claim.d.ts
329
+ declare const claimSchema: z.ZodObject<{
330
+ use: z.ZodEnum<{
331
+ certification: "certification";
332
+ comparison: "comparison";
333
+ development: "development";
334
+ }>;
335
+ population: z.ZodObject<{
336
+ id: z.ZodString;
337
+ description: z.ZodString;
338
+ }, z.core.$strict>;
339
+ samplingFrame: z.ZodString;
340
+ independentUnit: z.ZodString;
341
+ generalization: z.ZodEnum<{
342
+ "fixed-roster": "fixed-roster";
343
+ "new-units": "new-units";
344
+ }>;
345
+ minimumEffect: z.ZodOptional<z.ZodNumber>;
346
+ }, z.core.$strict>;
347
+ /** The population and independent observations a measured result can describe. */
348
+ type EvaluationClaim = z.infer<typeof claimSchema>;
349
+ /** Validate a claim before binding it to a sealed experiment or final evidence. */
350
+ declare function defineEvaluationClaim(input: EvaluationClaim): EvaluationClaim;
351
+ interface EvaluationUnitSummary {
352
+ observations: number;
353
+ independentUnits: number;
354
+ units: Array<{
355
+ id: string;
356
+ observations: number;
357
+ }>;
358
+ }
359
+ /** Repetitions retain their denominator without becoming additional independent units. */
360
+ declare function summarizeEvaluationUnits(claim: EvaluationClaim, rows: readonly object[]): EvaluationUnitSummary;
361
+ //#endregion
362
+ //#region src/experiment/final-evidence.d.ts
363
+ declare const reservationSchema: z.ZodObject<{
364
+ requestId: z.ZodString;
365
+ claimDigest: z.ZodString;
366
+ populationId: z.ZodString;
367
+ inputDigest: z.ZodString;
368
+ unitIds: z.ZodArray<z.ZodString>;
369
+ }, z.core.$strict>;
370
+ /** One final dataset reserved for one adaptive decision, across processes and campaigns. */
371
+ type FinalEvidenceReservation = z.infer<typeof reservationSchema>;
372
+ declare const measurementSchema: z.ZodObject<{
373
+ evaluatorDigest: z.ZodString;
374
+ candidateDigests: z.ZodArray<z.ZodString>;
375
+ }, z.core.$strict>;
376
+ type FinalEvidenceMeasurement = z.infer<typeof measurementSchema>;
377
+ interface FinalEvidenceRecord {
378
+ reservation: FinalEvidenceReservation;
379
+ reservationHash: LedgerHash;
380
+ exposure: {
381
+ measurement: FinalEvidenceMeasurement;
382
+ entryHash: LedgerHash;
383
+ } | null;
384
+ }
385
+ type FinalEvidenceOutcome<T> = {
386
+ succeeded: true;
387
+ value: T;
388
+ } | {
389
+ succeeded: false;
390
+ error: {
391
+ kind: 'conflict' | 'invalid' | 'unavailable';
392
+ message: string;
393
+ };
394
+ };
395
+ interface FinalEvidenceLedger {
396
+ reserve(input: FinalEvidenceReservation): Promise<FinalEvidenceOutcome<{
397
+ record: FinalEvidenceRecord;
398
+ replayed: boolean;
399
+ }>>;
400
+ expose(requestId: string, measurement: FinalEvidenceMeasurement): Promise<FinalEvidenceOutcome<{
401
+ record: FinalEvidenceRecord;
402
+ replayed: boolean;
403
+ }>>;
404
+ read(): Promise<FinalEvidenceOutcome<FinalEvidenceRecord[]>>;
405
+ }
406
+ /** Retains the distinction between invalid input, consumed evidence, and unavailable storage. */
407
+ declare class FinalEvidenceError extends Error {
408
+ readonly kind: 'conflict' | 'invalid' | 'unavailable';
409
+ constructor(kind: 'conflict' | 'invalid' | 'unavailable', message: string);
410
+ }
411
+ /** A consumed or conflicting dataset cannot authorize another adaptive decision. */
412
+ declare class FinalEvidenceConflictError extends FinalEvidenceError {
413
+ constructor(message: string);
414
+ }
415
+ /** Uses the shared locked journal and requires its trusted head on every reopen. */
416
+ declare function openFinalEvidenceLedger(options: {
417
+ path: string;
418
+ }): FinalEvidenceLedger;
419
+ //#endregion
326
420
  //#region src/attestation.d.ts
327
421
  /**
328
422
  * Reproducibility attestation for any serializable report object.
@@ -366,19 +460,13 @@ interface AttestedReport {
366
460
  reportHash: string;
367
461
  provenance: AttestationProvenance;
368
462
  algorithm: typeof ATTESTATION_ALGORITHM;
369
- /**
370
- * Hex sha-256 over `{ reportHash, provenance, algorithm }`. New attestations
371
- * always carry it. Optional only so persisted pre-envelope attestations can
372
- * still be read and explicitly recognized as legacy by callers.
373
- */
374
- envelopeHash?: string;
463
+ /** Hex sha-256 over `{ reportHash, provenance, algorithm }`. */
464
+ envelopeHash: string;
375
465
  }
376
466
  interface AttestationVerification {
377
467
  valid: boolean;
378
468
  /** Populated iff `valid` is false — names the exact mismatch. */
379
469
  reason?: string;
380
- /** True only for a valid pre-envelope attestation whose provenance is not cryptographically bound. */
381
- legacyUnboundProvenance?: true;
382
470
  }
383
471
  /**
384
472
  * Content-address a report and bind it to its provenance. Throws (via
@@ -393,9 +481,8 @@ declare function attest(report: unknown, provenance: AttestationProvenance): Att
393
481
  * canonicalizes) is a verification failure with the cause in `reason`, not a
394
482
  * crash — verifiers run in pipelines that must record WHY, not die.
395
483
  *
396
- * Legacy attestations without `envelopeHash` remain readable, but verification
397
- * explicitly marks their provenance as unbound so a promotion path can refuse
398
- * them instead of accidentally treating old metadata as cryptographic proof.
484
+ * Both the report hash and the provenance envelope must verify.
485
+ * Missing envelope hashes cannot establish provenance and fail verification.
399
486
  */
400
487
  declare function verifyAttestation(report: unknown, attested: AttestedReport): AttestationVerification;
401
488
  //#endregion
@@ -451,10 +538,8 @@ interface CreateEvidenceReceiptInput extends Omit<EvidenceBinding, 'schemaVersio
451
538
  */
452
539
  declare function createEvidenceReceipt(input: CreateEvidenceReceiptInput, provenance: AttestationProvenance): EvidenceReceipt;
453
540
  /**
454
- * Verify promotion-grade evidence. Generic report attestation keeps a legacy read path,
455
- * but an EvidenceReceipt never accepts unbound provenance: changing the evaluator code,
456
- * model versions, input commitment provenance, or creation record must invalidate the
457
- * evidence rather than merely annotating it as legacy.
541
+ * Verify evidence identity and its provenance envelope. Changing the evaluator,
542
+ * model versions, input commitment, or creation record invalidates the evidence.
458
543
  */
459
544
  declare function verifyEvidenceReceipt(receipt: EvidenceReceipt): EvidenceReceiptVerification;
460
545
  /**
@@ -533,18 +618,23 @@ interface PromotionObjective {
533
618
  /** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients
534
619
  * the paired delta so a positive bootstrap always means "candidate better". */
535
620
  direction: Direction;
621
+ /** Declared binary support {0, binaryScale}, including zero-only observations.
622
+ * Must be finite and positive; paired cell scores must be 0 or this scale.
623
+ * Uses the risk-difference mean and rejects the 'median' statistic. */
624
+ binaryScale?: number;
536
625
  /** The good-direction paired-delta CI lower bound must EXCEED this to count
537
626
  * as a significant gain on this axis. Interpreted in the judge's native
538
627
  * scale. Default 0 (⇒ "confidently better"). */
539
628
  gainThreshold?: number;
540
629
  /** A floor breach (regression) is declared when the good-direction CI lower
541
630
  * bound is below −floorTolerance, or when the exact small-sample test proves
542
- * a drop past it. When omitted it auto-scales off observed magnitudes
543
- * (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */
631
+ * a drop past it. Defaults to 0.05 times the declared binary scale, or
632
+ * auto-scales off observed magnitudes (0.05 on [0,1], 5 on 0-100). */
544
633
  floorTolerance?: number;
545
634
  }
546
- /** Per-axis verdict from the good-direction paired bootstrap. */
547
- type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs';
635
+ /** Per-axis verdict from the shared paired decision rule.
636
+ * 'regressed' includes uncertainty that prevents clearing the regression floor. */
637
+ type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs' | 'indeterminate';
548
638
  interface AxisEvidence {
549
639
  name: string;
550
640
  source: ObjectiveSource;
@@ -572,8 +662,8 @@ interface AxisEvidence {
572
662
  decisionStatistic: PairedDecisionStatistic;
573
663
  /** McNemar's exact evidence on a pass/fail axis; null otherwise. */
574
664
  mcnemar: PairedMcNemarEvidence | null;
575
- /** `ci` has zero width no evidence in either direction, so the axis is
576
- * neither improved nor regressed however the point estimate sits. */
665
+ /** `ci` has zero width or non-finite bounds. It cannot establish a gain or
666
+ * clear a regression floor, regardless of the point estimate. */
577
667
  indeterminate: boolean;
578
668
  /** Paired observations contributing to this axis. */
579
669
  n: number;
@@ -623,11 +713,9 @@ interface BuildEvidenceVectorOptions {
623
713
  */
624
714
  declare function buildEvidenceVector<TArtifact, TScenario extends Scenario>(ctx: GateContext<TArtifact, TScenario>, objectives: PromotionObjective[], opts?: BuildEvidenceVectorOptions): EvidenceVector;
625
715
  /**
626
- * The default strategy: symmetric multi-objective Pareto significance. Ship iff
627
- * the candidate weakly dominates the baseline at the confidence level no axis
628
- * credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold
629
- * (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
630
- * need_more_work. Statistically equivalent → hold (never ship noise).
716
+ * Require a supported gain and every configured regression floor to clear.
717
+ * A failed floor holds the candidate, including when uncertainty permits a loss.
718
+ * Missing or indeterminate evidence requires more work; no gain holds release.
631
719
  */
632
720
  declare const paretoPolicy: PromotionPolicy;
633
721
  interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {
@@ -648,76 +736,59 @@ declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends S
648
736
  //#endregion
649
737
  //#region src/campaign/gates/power-preflight.d.ts
650
738
  /**
651
- * Power preflight "can this budget detect the effect you are hunting?"
652
- *
653
- * The failure it prevents (measured, twice): a live prompt-improvement campaign ran
654
- * 333 sandbox cells over 5.6 hours and produced a +0.08 holdout lift the ship gate
655
- * (paired bootstrap, CI.low > 0.05) could not distinguish from zero — because at
656
- * that holdout size and worker variance the MINIMUM DETECTABLE lift was larger than
657
- * any effect a prompt change plausibly produces. The budget was spent learning what
658
- * a 30-second calculation on the baseline cells already knew. No eval framework we
659
- * know of surfaces this; every underpowered improvement run everywhere ends in an
660
- * uninformative "hold".
739
+ * Approximate detectable lift from baseline variance and the intended sample size.
661
740
  *
662
741
  * Model: the ship rule is `CI.low(paired Δ) > deltaThreshold`. Approximating the
663
742
  * bootstrap CI as normal, `CI.low ≈ effect − z·sd_Δ/√n`, so the smallest shippable
664
- * true effect is `MDE = deltaThreshold + z·sd_Δ/√n`. The paired-delta SD is unknown
665
- * before the candidate exists; we bound it by the zero-correlation case
666
- * `sd_Δ √2·sd_baseline` a CONSERVATIVE (upper) MDE, which is the correct
667
- * direction for a warning. Pairing is per cell (`scenario:rep`), so reps multiply n.
743
+ * effect is `MDE = deltaThreshold + z·sd_Δ/√n`. This estimates `sd_Δ` as
744
+ * `√2·sd_baseline`, assuming equal arm variances and zero paired correlation.
745
+ * Actual candidate variance and paired correlation can move the threshold in either direction.
746
+ * This diagnostic does not replace the gate or guarantee a detection probability.
668
747
  *
669
- * Standalone by design: feed it any baseline composites (a `gate:'none'` run, a
670
- * live-proof table) BEFORE budgeting the real search; `selfImprove` also attaches
671
- * it to every result and warns when the run was structurally unable to ship.
748
+ * Supply one baseline mean per independent observation used by the comparison.
749
+ * With declared source units, repetitions and variants refine those means without increasing n.
750
+ * `selfImprove` attaches this diagnostic after its final measurement.
672
751
  */
673
752
  interface PowerPreflightOptions {
674
- /** Per-cell baseline composites on the HOLDOUT scenarios (one per scenario:rep cell). */
753
+ /** Baseline composites at the comparison's independent observation unit. */
675
754
  baselineComposites: number[];
676
- /** Paired observations the budgeted comparison will produce
677
- * (holdout scenarios × reps). Defaults to `baselineComposites.length`. */
755
+ /** Independent paired observations planned for the comparison.
756
+ * Defaults to `baselineComposites.length`. */
678
757
  pairedN?: number;
679
758
  /** The ship gate's effect-size threshold. Default 0.05 (defaultProductionGate). */
680
759
  deltaThreshold?: number;
681
760
  /** CI confidence the gate uses. Default 0.95. */
682
761
  confidence?: number;
683
- /** True when the holdout is scored by the SAME judge/scorer family as the gate
684
- * (selfImprove's default composition one judge scores everything). Under a
685
- * shared channel, raising paired n reduces only the IDIOSYNCRATIC noise share;
686
- * systematic judge bias is untouched, so the MDE here is a lower bound and the
687
- * only full debiaser is an independent second scoring channel
688
- * (recursive-self-improvement S1c, closed form in EXP-023 P0). Default false. */
762
+ /** Whether the holdout uses the gate's judge family.
763
+ * More observations cannot establish freedom from systematic scoring bias.
764
+ * Default false. */
689
765
  sharedScorerChannel?: boolean;
690
766
  }
691
767
  interface PowerPreflight {
692
768
  /** Paired observations the comparison will have. */
693
769
  n: number;
694
- /** Baseline per-cell composite standard deviation (the variance the effect must beat). */
770
+ /** Sample standard deviation of baseline observation means. */
695
771
  sd: number;
696
- /** Minimum detectable lift: the smallest TRUE effect the gate could ship at this budget. */
772
+ /** Approximate lift needed to put a normal interval above the gate threshold. */
697
773
  mde: number;
698
774
  /** Baseline holdout composite mean. */
699
775
  baselineMean: number;
700
776
  /** Headroom to a perfect 1.0 composite (the largest achievable lift on a [0,1] judge). */
701
777
  headroom: number;
702
- /** True when even the largest achievable effect (headroom) is below the MDE —
703
- * the run is structurally unable to ship regardless of proposal quality.
704
- * Only asserted for [0,1]-scaled judges (see `scaleAssumed`). */
778
+ /** Whether this approximation exceeds the estimated [0,1] score headroom.
779
+ * This is a planning warning, not a proof that promotion is impossible. */
705
780
  underpowered: boolean;
706
781
  /** True when composites look [0,1]-scaled; headroom/underpowered are only
707
782
  * meaningful under that convention (0-100 judges get mde/sd/n but no verdict). */
708
783
  scaleAssumed: boolean;
709
784
  deltaThreshold: number;
710
785
  confidence: number;
711
- /** Set when the holdout shares the gate's scoring channel: more cells cannot
712
- * buy back systematic judge bias — treat the MDE as a lower bound. */
786
+ /** Notes unmeasured systematic bias when the gate shares its scoring channel. */
713
787
  sharedChannelCaveat?: string;
714
788
  /** One actionable sentence for humans and logs. */
715
789
  recommendation: string;
716
790
  }
717
- /** Estimate the minimum detectable lift a paired-holdout improvement run can
718
- * ship at a given budget, from the baseline holdout composites — call it BEFORE
719
- * spending a search to learn whether the effect you are hunting is even
720
- * observable at this holdout size and worker variance. */
791
+ /** Estimate detectable lift from baseline independent observations before budgeting a comparison. */
721
792
  declare function powerPreflight(opts: PowerPreflightOptions): PowerPreflight;
722
793
  //#endregion
723
794
  //#region src/paired-arms.d.ts
@@ -877,10 +948,8 @@ declare function pairRunRecords(baselineRuns: readonly RunRecord[], treatmentRun
877
948
  * evaluate the manifest against observed results — the library refuses
878
949
  * to let you re-interpret a different metric as the declared one.
879
950
  *
880
- * A signed manifest is a portable record: it is written once and verified
881
- * later, possibly by a different release. `algo` names the digest scheme it
882
- * was signed under, and verification selects the encoder by that field, so a
883
- * manifest signed by an earlier release still verifies.
951
+ * A signed manifest carries its required digest scheme. Verification accepts
952
+ * only RFC 8785 canonical JSON, using the same encoder as every new identity.
884
953
  */
885
954
  interface HypothesisManifest {
886
955
  id: string;
@@ -904,31 +973,13 @@ interface HypothesisManifest {
904
973
  baselineLabel?: string;
905
974
  candidateLabel?: string;
906
975
  }
907
- /**
908
- * Identifier for the hashing scheme used to produce `contentHash`.
909
- *
910
- * Both schemes are sha256 hex over the manifest with `contentHash` and `algo`
911
- * stripped, and differ only in how that manifest is serialized:
912
- *
913
- * - `'sha256-rfc8785'` — RFC 8785 canonical JSON. What {@link signManifest}
914
- * emits.
915
- * - `'sha256-content'` — key-sorted `JSON.stringify`. Read-only: manifests
916
- * signed by an earlier release carry it, or carry no `algo` at all, and
917
- * {@link verifyManifest} still verifies them.
918
- */
919
- type SignedManifestAlgo = 'sha256-content' | 'sha256-rfc8785';
976
+ /** SHA-256 over RFC 8785 canonical JSON, excluding `contentHash` and `algo`. */
977
+ type SignedManifestAlgo = 'sha256-rfc8785';
920
978
  interface SignedManifest extends HypothesisManifest {
921
979
  /** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
922
980
  contentHash: string;
923
- /**
924
- * Algorithm string describing how `contentHash` was produced.
925
- *
926
- * Optional on the type so serialized manifests without it still parse,
927
- * but ALWAYS populated by {@link signManifest}. Consumers that want to
928
- * enforce a known algorithm should reject manifests where this field
929
- * is missing or unrecognized.
930
- */
931
- algo?: SignedManifestAlgo;
981
+ /** Required digest scheme. Missing or unsupported schemes fail verification. */
982
+ algo: SignedManifestAlgo;
932
983
  }
933
984
  interface HypothesisResult {
934
985
  manifest: SignedManifest;
@@ -959,10 +1010,8 @@ interface HypothesisResult {
959
1010
  */
960
1011
  declare function hashJson<T>(obj: T): Promise<string>;
961
1012
  /**
962
- * Digest of a manifest under its own declared scheme, with `contentHash` and
963
- * `algo` stripped. Synchronous, so a caller that must fail before consuming an
964
- * observation does not have to await. Throws on an `algo` this release does
965
- * not know — an unverifiable manifest must not read as a valid one.
1013
+ * Digest a manifest after validating its scheme, excluding `contentHash` and
1014
+ * `algo`. This synchronous check can refuse a manifest before consuming data.
966
1015
  */
967
1016
  declare function manifestContentDigest(manifest: SignedManifest): string;
968
1017
  /**
@@ -1002,11 +1051,11 @@ interface SequentialPairedGateOptions {
1002
1051
  /** Type-I budget. With `preRegistration` bound this MUST match
1003
1052
  * `manifest.alpha` (conflict throws). Default 0.05. */
1004
1053
  alpha?: number;
1005
- /** Minimum paired deltas before a promote may fire. The stopping rule is
1054
+ /** Minimum independent paired units before a promote may fire. The stopping rule is
1006
1055
  * "first n ≥ minN with e-value ≥ 1/alpha" — still a valid stopping time.
1007
1056
  * Default 5. */
1008
1057
  minN?: number;
1009
- /** Pre-registered observation budget. Required unless `preRegistration`
1058
+ /** Pre-registered independent-unit budget (scenarios by default in decide()). Required unless `preRegistration`
1010
1059
  * supplies it via `preRegisteredN` (conflict throws). */
1011
1060
  maxN?: number;
1012
1061
  /** Bet truncation forwarded to `eProcess`. Default 0.5. */
@@ -1015,9 +1064,13 @@ interface SequentialPairedGateOptions {
1015
1064
  * x = (d/scale + 1)/2 ∈ [0,1]. A delta outside ±scale throws (use
1016
1065
  * `detectScale` to pick 1 vs 100 BEFORE streaming). Default 1. */
1017
1066
  scale?: number;
1018
- /** Seed for the data-independent shuffle of paired deltas in `decide(ctx)`
1019
- * (exchangeability guard). Default 1337. */
1067
+ /** Seed for reproducible ordering of scenario deltas in decide().
1068
+ * Shuffling does not establish the conditional-mean null. Default 1337. */
1020
1069
  shuffleSeed?: number;
1070
+ /** Independent unit for every scenario passed to decide(). Full cell pairs
1071
+ * are averaged within each unit before testing. Defaults to scenario IDs.
1072
+ * The mapping is copied at construction. Direct observe() is unchanged. */
1073
+ independentUnitByScenarioId?: ReadonlyMap<string, string>;
1021
1074
  /** Bind the pre-registered hypothesis. Verified (content hash) at
1022
1075
  * construction; alpha/maxN/direction/minEffect come FROM the manifest. */
1023
1076
  preRegistration?: SignedManifest;
@@ -1039,8 +1092,10 @@ type SequentialStreamState = EProcessState & {
1039
1092
  decision: SequentialDecision;
1040
1093
  };
1041
1094
  interface SequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario> extends Gate<TArtifact, TScenario> {
1042
- /** Streaming entry point: feed one paired per-scenario delta
1043
- * (candidate − baseline, native scale). Each gate instance carries ONE
1095
+ /** Streaming entry point: feed one paired independent-unit delta
1096
+ * (candidate − baseline, native scale). Aggregate correlated repetitions
1097
+ * before calling; the caller must justify E[delta_t | past] under H0.
1098
+ * Each gate instance carries ONE
1044
1099
  * observe-stream; `decide(ctx)` runs on its own fresh stream and never
1045
1100
  * consumes or advances this one. 'promote' is sticky; observing past the
1046
1101
  * pre-registered maxN throws (extending a finished stream after seeing
@@ -1053,15 +1108,16 @@ interface SequentialPairedGate<TArtifact = unknown, TScenario extends Scenario =
1053
1108
  }
1054
1109
  /**
1055
1110
  * Anytime-valid sequential paired gate. Conforms to the existing `Gate`
1056
- * contract (`decide(ctx)` consumes candidate vs baseline judge scores via
1057
- * `pairHoldout` same pairing granularity as the fixed-n gates: full cellId,
1058
- * never scenarioId) and adds a streaming `observe(delta)` entry for campaigns
1059
- * that score cells incrementally and want to stop mid-stream.
1111
+ * contract: decide(ctx) pairs candidate and baseline by full cellId, then
1112
+ * averages deltas within each configured unit (scenario by default). The
1113
+ * e-process consumes one observation per unit, with equal unit weights.
1114
+ * The streaming observe(delta) entry consumes caller-aggregated independent
1115
+ * units. Repetitions improve a unit's precision and do not increase n.
1060
1116
  *
1061
1117
  * Decision mapping onto the substrate's five-valued `GateDecision`:
1062
1118
  * - 'promote' → 'ship'
1063
1119
  * - 'continue' → 'need_more_work' (stream ended before maxN with
1064
- * the e-value undecided — more reps could decide)
1120
+ * the e-value undecided — more independent units needed)
1065
1121
  * - 'undecided-at-maxN' → 'hold', with the reason stating it is NOT
1066
1122
  * evidence of no effect (never a silent default)
1067
1123
  */
@@ -1087,9 +1143,8 @@ interface SequentialDecideFn {
1087
1143
  state(): EProcessState;
1088
1144
  }
1089
1145
  /**
1090
- * `SurfaceProposer.decide` adapter stops the optimization loop the moment
1091
- * the e-process decides the loop has produced a real improvement, instead of
1092
- * always running `maxGenerations`.
1146
+ * SurfaceProposer.decide adapter that stops exploration at an e-value threshold.
1147
+ * The selected candidate still requires an independent held-out decision.
1093
1148
  *
1094
1149
  * Stream: for each generation g ≥ 1, the per-scenario composite deltas of
1095
1150
  * generation g's top candidate vs the generation-0 top candidate (the
@@ -1100,10 +1155,11 @@ interface SequentialDecideFn {
1100
1155
  * gate (which re-scores on HELD-OUT data — this adapter only spends the
1101
1156
  * exploration budget, it never promotes).
1102
1157
  *
1103
- * Honesty caveats: (1) the incumbent's scores are measured once and shared
1104
- * across all generations' deltas, so type-I control is exact only insofar as
1105
- * those scores approximate the incumbent's true per-scenario means (more reps
1106
- * tighter); (2) an UNDECIDED process never stops the loop — absence of a
1158
+ * This is an exploration stopping heuristic. Selection of the best measured
1159
+ * candidate and reuse of measured incumbent scores generally violate the
1160
+ * conditional-mean null, even when the candidates have no true improvement.
1161
+ * More repetitions reduce noise but do not establish exact type-I control.
1162
+ * An UNDECIDED process never stops the loop — absence of a
1107
1163
  * crossing is NOT evidence of no effect, so the loop simply runs its normal
1108
1164
  * course. Calling the adapter repeatedly with a growing history consumes each
1109
1165
  * generation exactly once (re-feeding an already-seen record would double-count
@@ -1125,8 +1181,8 @@ interface PairedHoldout {
1125
1181
  * pulls the scalar from a cell's judge reports (composite, or a named
1126
1182
  * dimension); a cell contributes the mean of `select` across its judges. Cells
1127
1183
  * whose scenario is not in `scenarioIds`, or where `select` is undefined for
1128
- * every judge on either side, are skipped on BOTH sides so the arrays stay
1129
- * paired. Throws when the two maps disagree on which holdout cells exist — a
1184
+ * every judge on both sides, are skipped. The selected judge IDs must agree
1185
+ * within each pair. Throws when the two maps disagree on holdout cell IDs — a
1130
1186
  * load-bearing invariant: the baseline + winner holdout campaigns run the same
1131
1187
  * scenarios with the same seed base, so their cellIds MUST align; a mismatch
1132
1188
  * means a silent pairing bug, not a soft fallback.
@@ -1165,9 +1221,14 @@ interface HeldoutSignificance {
1165
1221
  * high tie fraction is WHY a median-based gate would have missed a real lift;
1166
1222
  * it is the observability the tie fix adds. */
1167
1223
  tieFraction: number;
1168
- /** n paired observations. */
1224
+ /** Number of paired observation units, after configured aggregation. */
1169
1225
  n: number;
1170
- /** Effective minimum after applying the bootstrap's hard statistical floor. */
1226
+ /** Original matched execution cells, before aggregation. */
1227
+ pairedCellN: number;
1228
+ observationUnit: 'registered' | 'cell';
1229
+ /** Registered unit IDs, or cell IDs on the ungrouped path. */
1230
+ unitIds: string[];
1231
+ /** Effective minimum for the requested target and chosen estimator. */
1171
1232
  minimumRequired: number;
1172
1233
  /** Statistical method that carried the decision. */
1173
1234
  decisionMethod: PairedDecisionMethod;
@@ -1188,6 +1249,8 @@ interface HeldoutSignificanceOptions {
1188
1249
  /** Fixed by default for a deterministic, reproducible gate verdict. */
1189
1250
  seed?: number;
1190
1251
  statistic?: 'mean' | 'median';
1252
+ /** Group full cell pairs into equal-weight independent units before inference. */
1253
+ independentUnitByScenarioId?: ReadonlyMap<string, string>;
1191
1254
  }
1192
1255
  /**
1193
1256
  * Significance of the held-out composite lift: ship only when the lower bound
@@ -1212,8 +1275,8 @@ interface HeldoutSignificanceOptions {
1212
1275
  * and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
1213
1276
  * delta is exactly 0.
1214
1277
  *
1215
- * At small n, where the percentile bootstrap is descriptive only, a
1216
- * pre-registered exact sign test still carries the bootstrap path.
1278
+ * Continuous mean targets require bootstrap eligibility. Explicit median
1279
+ * targets can use the exact sign test at its confidence-dependent minimum.
1217
1280
  */
1218
1281
  declare function heldoutSignificance(paired: PairedHoldout, opts?: HeldoutSignificanceOptions): HeldoutSignificance;
1219
1282
  interface DimensionRegression {
@@ -1236,36 +1299,34 @@ interface DimensionRegression {
1236
1299
  mcnemar: PairedMcNemarEvidence | null;
1237
1300
  /** `ci` has zero width — no evidence in either direction. */
1238
1301
  indeterminate: boolean;
1239
- /** True iff the candidate may have regressed this dimension by more than
1240
- * tolerance: the lower bound of the DECIDING interval on (candidate −
1241
- * baseline) is below −tolerance, OR the exact small-sample test proves a drop
1242
- * past tolerance. */
1302
+ /** The bootstrap lower bound is below negative tolerance, or the shared
1303
+ * paired test supports a drop exceeding tolerance. Missing coverage and
1304
+ * insufficient observations are reported separately. */
1243
1305
  regressed: boolean;
1244
1306
  tolerance: number;
1245
1307
  n: number;
1308
+ pairedCellN: number;
1309
+ observationUnit: 'registered' | 'cell';
1310
+ /** Statistical minimum for the configured independent observation unit. */
1311
+ minimumRequired: number;
1312
+ fewRuns: boolean;
1313
+ /** Both arms lack this dimension on these otherwise matched execution cells. */
1314
+ missingCellIds: string[];
1315
+ /** Configured scenarios with no paired dimension measurement at all. */
1316
+ missingScenarioIds: string[];
1246
1317
  }
1247
1318
  /** Detect the native scale of a set of scores: 0-100 when any magnitude clears
1248
1319
  * 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
1249
1320
  * expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
1250
1321
  declare function detectScale(values: number[]): 1 | 100;
1251
- /** Per-critical-dimension regression guard. For each dimension, pair the
1252
- * candidate vs baseline values by full cellId and bootstrap the paired delta;
1253
- * a dimension is "regressed" when the CI lower bound < −tolerance (conservative
1254
- * blocks if the credible worst case exceeds tolerance, which is the right
1255
- * posture for safety dimensions like `hallucination_free`). When `tolerance`
1256
- * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
1257
- *
1258
- * The interval comes from {@link decidePairedPromotion}, so a pass/fail
1259
- * dimension is judged on Tango's score interval rather than a percentile
1260
- * bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
1261
- * is not a valid interval at one. That matters most here because this guard
1262
- * fails OPEN by construction: `tolerance` is positive, so an interval pinned at
1263
- * [0,0] never satisfies `low < −tolerance` and a real regression on a safety
1264
- * dimension would be reported as `regressed: false`. On the median it fails the
1265
- * same way for the same reason — when most pairs tie, which is automatic for a
1266
- * pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
1267
- * to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
1268
- * restore the pre-0.134 behaviour. */
1322
+ /**
1323
+ * Report required-dimension evidence after full pairing and optional unit means.
1324
+ * A bootstrap floor breach or a shared paired test supporting a drop marks
1325
+ * regression. These two criteria are distinct; `ci` records the shared
1326
+ * estimator and `bootstrap` records the floor interval. Missing observations
1327
+ * and insufficient n remain explicit for the caller's evidence policy.
1328
+ * The default tolerance is 0.05 on [0,1] and 5 on a detected 0-100 scale.
1329
+ */
1269
1330
  declare function dimensionRegressions(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, criticalDimensions: string[], opts?: {
1270
1331
  tolerance?: number;
1271
1332
  confidence?: number;
@@ -1274,7 +1335,9 @@ declare function dimensionRegressions(candidate: Map<string, Record<string, Judg
1274
1335
  /** Paired statistic the CI is computed on. Default `'mean'` — see
1275
1336
  * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */
1276
1337
  statistic?: 'mean' | 'median';
1338
+ independentUnitByScenarioId?: ReadonlyMap<string, string>;
1339
+ minProductiveRuns?: number;
1277
1340
  }): DimensionRegression[];
1278
1341
  //#endregion
1279
- export { paretoSignificanceGate as $, PairArmsOptions as A, McNemarResult as At, PowerPreflight as B, passAtK as Bt, hashJson as C, verifyAttestation as Ct, ComparePairedArmsOptions as D, EProcessStep as Dt, verifyManifest as E, EProcessState as Et, PairedCorrectness as F, mcnemar as Ft, BuildEvidenceVectorOptions as G, powerPreflight as H, PairedMetricDelta as I, pairedBinaryScale as It, ParetoSignificanceGateOptions as J, EvidenceVector as K, comparePairedArms as L, pairedRiskDifference as Lt, PairRunRecordsResult as M, RiskDifferenceResult as Mt, PairedArmRow as N, ScoreRiskDifferenceResult as Nt, MatchedPair as O, eProcess as Ot, PairedArmsComparison as P, isBinaryOutcomeVector as Pt, paretoPolicy as Q, pairArms as R, pairedRiskDifferenceExact as Rt, evaluateHypothesis as S, attest as St, signManifest as T, EProcessOptions as Tt, AxisEvidence as U, PowerPreflightOptions as V, wilson as Vt, AxisVerdict as W, PromotionPolicy as X, PromotionObjective as Y, buildEvidenceVector as Z, sequentialPairedGate as _, verifyEvidenceReceipt as _t, detectScale as a, createCampaignEvidenceReceipt as at, SignedManifest as b, AttestationVerification as bt, pairHoldout as c, EVIDENCE_RECEIPT_VERSION as ct, SequentialDecision as d, EvidenceBinding as dt, Objective as et, SequentialObservation as f, EvidenceReceipt as ft, sequentialDecide as g, isIndependentEvidence as gt, SequentialStreamState as h, createEvidenceReceipt as ht, PairedHoldout as i, CampaignEvidenceContext as it, PairArmsResult as j, ProportionInterval as jt, MatchedRunRecordPair as k, ExactRiskDifferenceResult as kt, SequentialDecideFn as l, EvidenceAuthority as lt, SequentialPairedGateOptions as m, INDEPENDENT_EVIDENCE_AUTHORITY_KINDS as mt, HeldoutSignificance as n, dominates as nt, dimensionRegressions as o, CreateEvidenceReceiptInput as ot, SequentialPairedGate as p, EvidenceReceiptVerification as pt, ObjectiveSource as q, HeldoutSignificanceOptions as r, paretoFrontier as rt, heldoutSignificance as s, EVIDENCE_AUTHORITY_KINDS as st, DimensionRegression as t, ParetoResult as tt, SequentialDecideOptions as u, EvidenceAuthorityKind as ut, HypothesisManifest as v, ATTESTATION_ALGORITHM as vt, manifestContentDigest as w, EProcess as wt, SignedManifestAlgo as x, AttestedReport as xt, HypothesisResult as y, AttestationProvenance as yt, pairRunRecords as z, pairedRiskDifferenceScore as zt };
1280
- //# sourceMappingURL=statistical-heldout-0La5ZTlv.d.ts.map
1342
+ export { paretoSignificanceGate as $, wilson as $t, PairArmsOptions as A, FinalEvidenceReservation as At, PowerPreflight as B, eProcess as Bt, hashJson as C, verifyAttestation as Ct, ComparePairedArmsOptions as D, FinalEvidenceMeasurement as Dt, verifyManifest as E, FinalEvidenceLedger as Et, PairedCorrectness as F, summarizeEvaluationUnits as Ft, BuildEvidenceVectorOptions as G, ScoreRiskDifferenceResult as Gt, powerPreflight as H, McNemarResult as Ht, PairedMetricDelta as I, EProcess as It, ParetoSignificanceGateOptions as J, pairedBinaryScale as Jt, EvidenceVector as K, isBinaryOutcomeVector as Kt, comparePairedArms as L, EProcessOptions as Lt, PairRunRecordsResult as M, EvaluationClaim as Mt, PairedArmRow as N, EvaluationUnitSummary as Nt, MatchedPair as O, FinalEvidenceOutcome as Ot, PairedArmsComparison as P, defineEvaluationClaim as Pt, paretoPolicy as Q, passAtK as Qt, pairArms as R, EProcessState as Rt, evaluateHypothesis as S, attest as St, signManifest as T, FinalEvidenceError as Tt, AxisEvidence as U, ProportionInterval as Ut, PowerPreflightOptions as V, ExactRiskDifferenceResult as Vt, AxisVerdict as W, RiskDifferenceResult as Wt, PromotionPolicy as X, pairedRiskDifferenceExact as Xt, PromotionObjective as Y, pairedRiskDifference as Yt, buildEvidenceVector as Z, pairedRiskDifferenceScore as Zt, sequentialPairedGate as _, verifyEvidenceReceipt as _t, detectScale as a, createCampaignEvidenceReceipt as at, SignedManifest as b, AttestationVerification as bt, pairHoldout as c, EVIDENCE_RECEIPT_VERSION as ct, SequentialDecision as d, EvidenceBinding as dt, Objective as et, SequentialObservation as f, EvidenceReceipt as ft, sequentialDecide as g, isIndependentEvidence as gt, SequentialStreamState as h, createEvidenceReceipt as ht, PairedHoldout as i, CampaignEvidenceContext as it, PairArmsResult as j, openFinalEvidenceLedger as jt, MatchedRunRecordPair as k, FinalEvidenceRecord as kt, SequentialDecideFn as l, EvidenceAuthority as lt, SequentialPairedGateOptions as m, INDEPENDENT_EVIDENCE_AUTHORITY_KINDS as mt, HeldoutSignificance as n, dominates as nt, dimensionRegressions as o, CreateEvidenceReceiptInput as ot, SequentialPairedGate as p, EvidenceReceiptVerification as pt, ObjectiveSource as q, mcnemar as qt, HeldoutSignificanceOptions as r, paretoFrontier as rt, heldoutSignificance as s, EVIDENCE_AUTHORITY_KINDS as st, DimensionRegression as t, ParetoResult as tt, SequentialDecideOptions as u, EvidenceAuthorityKind as ut, HypothesisManifest as v, ATTESTATION_ALGORITHM as vt, manifestContentDigest as w, FinalEvidenceConflictError as wt, SignedManifestAlgo as x, AttestedReport as xt, HypothesisResult as y, AttestationProvenance as yt, pairRunRecords as z, EProcessStep as zt };
1343
+ //# sourceMappingURL=statistical-heldout-CpVd6FmY.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"statistical-heldout-CpVd6FmY.d.ts","names":[],"sources":["../src/statistics/paired-binary.ts","../src/statistics/sequential-eprocess.ts","../src/experiment/claim.ts","../src/experiment/final-evidence.ts","../src/attestation.ts","../src/experiment/evidence-receipt.ts","../src/experiment/campaign-evidence.ts","../src/pareto.ts","../src/campaign/gates/promotion-policy.ts","../src/campaign/gates/power-preflight.ts","../src/paired-arms.ts","../src/pre-registration.ts","../src/campaign/gates/sequential.ts","../src/campaign/gates/statistical-heldout.ts"],"mappings":";;;;;;;UAgBiB;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;;;;;;;;;;;;;;;;;;;;;;;;;iBA2CzD,sBAAsB,QAAQ;;UAU7B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;iBAkBc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;UAkCc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA+Bc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;UAyEc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8Dc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;;;;;;;;;;iBA6Ea,kBACd,QAAQ,mBACR,OAAO;;;;;;;;;;;iBA0BO,QAAQ,WAAW,WAAW;;;UCjgB7B;;;EAGf;;;;EAIA;;;;EAIA;;;;;;;;EAQA,SAAS;;UAGM;;EAEf;;EAEA;;EAEA;;;;;;;;;;UAWe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;EAEA;;;EAGA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;cClFhD,aAAW,EAAA;;;;;;;;;;;;;;;;;GASN,EAAA,KAAA;;KAGC,kBAAkB,EAAE,aAAa;;iBAG7B,sBAAsB,OAAO,kBAAkB;UAO9C;EACf;EACA;EACA,OAAO;IAAQ;IAAY;;;;iBAIb,yBACd,OAAO,iBACP,0BACC;;;cCjCG,mBAAiB,EAAA;;;;;;GAQZ,EAAA,KAAA;;KAGC,2BAA2B,EAAE,aAAa;cAEhD,mBAAiB,EAAA;;;GAKZ,EAAA,KAAA;KAEC,2BAA2B,EAAE,aAAa;UAgCrC;EACf,aAAa;EACb,iBAAiB;EACjB;IAAY,aAAa;IAA0B,WAAW;;;KAGpD,qBAAqB;EAC3B;EAAiB,OAAO;;EACxB;EAAkB;IAAS;IAA8C;;;UAE9D;EACf,QACE,OAAO,2BACN,QAAQ;IAAuB,QAAQ;IAAqB;;EAC/D,OACE,mBACA,aAAa,2BACZ,QAAQ;IAAuB,QAAQ;IAAqB;;EAC/D,QAAQ,QAAQ,qBAAqB;;;cAI1B,2BAA2B;WAE3B;EADX,YACW,8CACT;;;cAQS,mCAAmC;EAC9C,YAAY;;;iBA6HE,wBAAwB;EAAW;IAAiB;;;;;;;;;;;;;;;;;;;;;;;cCrMvD;UAEI;;EAEf,eAAe;;EAEf;;;EAGA;;EAEA;;EAEA;;;EAGA;;UAGe;;EAEf;EACA,YAAY;EACZ,kBAAkB;;EAElB;;UAGe;EACf;;EAEA;;;;;;;;iBAiBc,OAAO,iBAAiB,YAAY,wBAAwB;;;;;;;;;;iBAoB5D,kBACd,iBACA,UAAU,iBACT;;;cC1EU;;cAGA;KAQD,gCAAgC;cAE/B;UAOI;WACN,MAAM;;WAEN;;UAGM;WACN,sBAAsB;;WAEtB;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;WACA,WAAW;;WAEX;;WAEA;;UAGM;WACN,SAAS;WACT,aAAa;;UAGP;WACN;WACA;;UAGM,mCAAmC,KAAK;;;;;;;iBAQzC,sBACd,OAAO,4BACP,YAAY,wBACX;;;;;iBA4Ba,sBAAsB,SAAS,kBAAkB;;;;;;iBAkCjD,sBAAsB,SAAS;;;;KCjJnC,0BAA0B,KACpC;EAOI,YAAY,KAAK;;;iBAGP,8BAA8B,UAAU,UAAU,GAAG;EACnE,UAAU,eAAe,GAAG;EAC5B,SAAS;EACT,SAAS;IAAA;;;;;;;;;;;;;;;;;;KCPC;UAEK,UAAU;;EAEzB;EACA,WAAW;EACX,QAAQ,WAAW;;UAGJ,aAAa;EAC5B,UAAU;EACV,WAAW;;EAEX,cAAc;IAAQ,WAAW;IAAG,WAAW;;;;iBAIjC,UAAU,GAAG,GAAG,GAAG,GAAG,GAAG,YAAY,UAAU;;;;;;iBAmB/C,eAAe,GAAG,YAAY,KAAK,YAAY,UAAU,OAAO,aAAa;;;;;KCxBjF;EAAoB;;EAAwB;EAAmB;;UAE1D;;EAEf;EACA,QAAQ;;;EAGR,WAAW;;;;EAIX;;;;EAIA;;;;;EAKA;;;;KAKU;UAEK;EACf;EACA,QAAQ;EACR,WAAW;;;;;;;;EAQX,WAAW;;;;;EAKX;;;EAGA;IAAM;IAAa;;;EAEnB,mBAAmB;;EAEnB,SAAS;;;EAGT;;EAEA;EACA;EACA,gBAAgB;EAChB;EACA;EACA,SAAS;;UAGM;;EAEf,MAAM;;;EAGN;;;EAGA;IAAQ;IAAmB;;;;;;KAMjB,mBAAmB,IAAI,mBAAmB;UAErC;;;;EAIf;;EAEA;;EAEA;;EAEA;;;EAGA;;;;;;;;iBASc,oBAAoB,WAAW,kBAAkB,UAC/D,KAAK,YAAY,WAAW,YAC5B,YAAY,sBACZ,OAAM,6BACL;;;;;;cAkIU,cAAc;UA+EV,sCAAsC;;EAErD,YAAY;;;EAGZ,SAAS;;EAET;;;;;;;iBAQc,uBAAuB,qBAAqB,kBAAkB,WAAW,UACvF,SAAS,gCACR,KAAK,WAAW;;;;;;;;;;;;;;;;;UCzVF;;EAEf;;;EAGA;;EAEA;;EAEA;;;;EAIA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;;EAGA;EACA;EACA;;EAEA;;EAEA;;;iBAYc,eAAe,MAAM,wBAAwB;;;;;UCzB5C;;;EAGf;;;;;EAKA;;EAEA;;EAEA;;EAEA,UAAU;;UAGK;;EAEf;;EAEA;;;UAIe;EACf;;;;EAIA;EACA,UAAU;EACV,WAAW;;UAGI;;EAEf,OAAO;;;EAGP,kBAAkB;;EAElB,mBAAmB;;;;;;;;;;;;;;;;;;;;iBAqBL,SAAS,eAAe,gBAAgB,MAAM,kBAAkB;;UA8G/D;;EAEf;;EAEA;;EAEA,SAAS;;EAET,gBAAgB;;;UAID;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,aAAa;;EAEb;IAAY;IAAW;;;UAGR,iCAAiC;;;;EAIhD;;EAEA,YAAY;;UAGG;EACf;EACA;EACA;;;EAGA,aAAa;EACb,cAAc;;;;;;;;;;;;;;;iBAgBA,kBACd,eAAe,gBACf,MAAM,2BACL;UAmEc;EACf;EACA;EACA,UAAU;EACV,WAAW;;UAGI;EACf,OAAO;EACP,kBAAkB;EAClB,mBAAmB;;;;;;;;;iBAeL,eACd,uBAAuB,aACvB,wBAAwB,cACvB;;;;;;;;;;;;;;;;UCrWc;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;;;KAIU;UAEK,uBAAuB;;EAEtC;;EAEA,MAAM;;UAGS;EACf,UAAU;EACV;EACA;EACA;;;EAGA;;EAEA,kBAAkB;EAGlB;;;;;;;;;;;;;;;;;iBAkBoB,SAAS,GAAG,KAAK,IAAI;;;;;iBAQ3B,sBAAsB,UAAU;;;;;;iBAc1B,aAAa,GAAG,qBAAqB,QAAQ;;;;;iBAS7C,eAAe,GAAG,iBAAiB;;;;;iBAYnC,mBACpB,UAAU,gBACV;EAAY;EAAW;EAAgB;IACtC,QAAQ;;;KCrFC;UAEK;EACf,UAAU;;EAEV;;EAEA;;;EAGA;;UAGe;;;EAGf;;;;EAIA;;;EAGA;;EAEA;;;;EAIA;;;EAGA;;;;EAIA,8BAA8B;;;EAG9B,kBAAkB;;EAElB;;;;;;;EAOA,SAAS;;;;;;KAOC,wBAAwB;EAAkB,UAAU;;UAE/C,qBAAqB,qBAAqB,kBAAkB,WAAW,kBAC9E,KAAK,WAAW;;;;;;;;;;EAUxB,QAAQ,gBAAgB;;;EAGxB,SAAS;;;;;;;;;;;;;;;;;iBAuPK,qBAAqB,qBAAqB,kBAAkB,WAAW,UACrF,SAAS,8BACR,qBAAqB,WAAW;UA6GlB;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe;GACd;IAAQ,SAAS;;IAAyB;IAAe;;;EAE1D,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;iBA0BK,iBAAiB,UAAS,0BAA+B;;;UC9dxD;;EAEf;;EAEA;;EAEA;;;;;;;;;;;;;iBA4Ec,YACd,WAAW,YAAY,eAAe,cACtC,UAAU,YAAY,eAAe,cACrC,aAAa,aACb,SAAS,GAAG,oCACX;UAkEc;EACf,QAAQ;;;;;;;;;;;;EAYR,WAAW;;;;EAIX,iBAAiB;;;;;;;EAOjB,UAAU;;EAEV,mBAAmB;;EAEnB,SAAS;;;;EAIT;;EAEA;;EAEA;EACA;;EAEA;;EAEA;;EAEA,gBAAgB;;EAEhB;;;;EAIA;;EAEA;;UAGe;EACf;EACA;EACA;EACA;;EAEA;EACA;;EAEA,8BAA8B;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA6BhB,oBACd,QAAQ,eACR,OAAM,6BACL;UA2Ec;EACf;;;EAGA,WAAW;;;;EAIX;;EAEA;IAAM;IAAa;;;EAEnB,mBAAmB;;EAEnB,SAAS;;EAET;;;;EAIA;EACA;EACA;EACA;EACA;;EAEA;EACA;;EAEA;;EAEA;;;;;iBAMc,YAAY;;;;;;;;;iBAYZ,qBACd,WAAW,YAAY,eAAe,cACtC,UAAU,YAAY,eAAe,cACrC,aAAa,aACb,8BACA;EACE;EACA;EACA;EACA;;;EAGA;EACA,8BAA8B;EAC9B;IAED"}