@tangle-network/agent-eval 0.180.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. package/CHANGELOG.md +60 -0
  2. package/README.md +119 -159
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +3 -3
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +4 -7
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +10 -9
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  38. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  39. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  40. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  41. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  42. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  43. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  44. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  45. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  46. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  47. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  48. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  49. package/dist/experiment/index.d.ts +27 -477
  50. package/dist/experiment/index.d.ts.map +1 -1
  51. package/dist/experiment/index.js +95 -559
  52. package/dist/experiment/index.js.map +1 -1
  53. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  54. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  55. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  56. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  57. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  58. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  59. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  60. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  61. package/dist/hosted/index.d.ts +2 -2
  62. package/dist/hosted/index.d.ts.map +1 -1
  63. package/dist/hosted/index.js +1 -1
  64. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  65. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  66. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  67. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  68. package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
  69. package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  70. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  71. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  72. package/dist/index.d.ts +28 -28
  73. package/dist/index.js +24 -15
  74. package/dist/index.js.map +1 -1
  75. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  76. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  77. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  78. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  79. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  80. package/dist/journal-Cs9f7385.js.map +1 -0
  81. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  82. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  83. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  84. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  85. package/dist/ledger-core/index.d.ts +1 -1
  86. package/dist/ledger-core/index.js +1 -1
  87. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  88. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  89. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  90. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  91. package/dist/meta-eval/index.d.ts +138 -7
  92. package/dist/meta-eval/index.d.ts.map +1 -1
  93. package/dist/meta-eval/index.js +245 -97
  94. package/dist/meta-eval/index.js.map +1 -1
  95. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  96. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  97. package/dist/multishot/golden/index.d.ts +1 -1
  98. package/dist/multishot/index.d.ts +2 -2
  99. package/dist/openapi.json +1 -1
  100. package/dist/outcome-store-BXlkwMPR.js +131 -0
  101. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  102. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  103. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  104. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  105. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  106. package/dist/pipelines/index.js +1 -1
  107. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  108. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  109. package/dist/profile-cell.d.ts +1 -1
  110. package/dist/profile-cell.js +1 -1
  111. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  112. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  113. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  114. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  115. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  116. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  117. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  118. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  119. package/dist/reporting.d.ts +4 -4
  120. package/dist/reporting.js +3 -3
  121. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  122. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  123. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  124. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  125. package/dist/rl.d.ts +53 -99
  126. package/dist/rl.d.ts.map +1 -1
  127. package/dist/rl.js +182 -169
  128. package/dist/rl.js.map +1 -1
  129. package/dist/rollout/index.d.ts +1 -1
  130. package/dist/rollout/index.js +2 -2
  131. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  132. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  133. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  134. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  135. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  136. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  137. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  138. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  139. package/dist/run-record-Br-Yzt_k.js +464 -0
  140. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  141. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  142. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  143. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  144. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  145. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  146. package/dist/sequential-DAsyV2T9.js.map +1 -0
  147. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  148. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  149. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  150. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  151. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  152. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  153. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  154. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  155. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  156. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  157. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  158. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  159. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  160. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  161. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  162. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  163. package/dist/trace-repair/index.d.ts +2 -2
  164. package/dist/traces.d.ts +6 -6
  165. package/dist/traces.js +1 -1
  166. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  167. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  168. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  169. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  170. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  171. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  172. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  173. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  174. package/dist/wire/index.d.ts +2 -2
  175. package/docs/adapters-observability.md +14 -0
  176. package/docs/campaign-proposers.md +86 -128
  177. package/docs/charter.md +108 -112
  178. package/docs/concepts.md +157 -69
  179. package/docs/design/mlbenchmarks-book-review.md +440 -0
  180. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  181. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  182. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  183. package/docs/design/self-improvement-evidence-audit.md +263 -0
  184. package/docs/design.md +2 -1
  185. package/docs/eval-surface-map.md +95 -42
  186. package/docs/evaluation-integrity.md +220 -0
  187. package/docs/experiment.md +111 -55
  188. package/docs/feature-guide.md +5 -6
  189. package/docs/hosted-ingest-spec.md +4 -11
  190. package/docs/insight-report.md +187 -455
  191. package/docs/outcome-validity.md +182 -0
  192. package/docs/product-eval-adoption.md +1 -2
  193. package/docs/research-report-methodology.md +7 -7
  194. package/docs/search-history-receipts.md +8 -0
  195. package/docs/statistical-evidence.md +129 -0
  196. package/docs/verdicts.md +76 -49
  197. package/package.json +1 -1
  198. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  199. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  200. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  201. package/dist/client-BlLY6o2w.js.map +0 -1
  202. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  203. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  204. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  205. package/dist/index-BxWvILU8.d.ts.map +0 -1
  206. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  207. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  208. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  209. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  210. package/dist/outcome-store-ChBKlTd_.js +0 -75
  211. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  212. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  213. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  214. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  215. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  216. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  217. package/dist/run-record-CR63CpHK.js +0 -216
  218. package/dist/run-record-CR63CpHK.js.map +0 -1
  219. package/dist/sequential-B5gXgcyp.js.map +0 -1
  220. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  221. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
@@ -1,14 +1,14 @@
1
1
  import { i as JudgeError, s as ValidationError } from "./errors-Dngq5h35.js";
2
- import { a as hashCanonical, i as compareCodeUnits, r as canonicalString } from "./canonical-DPyQ_rpt.js";
2
+ import { a as hashCanonical, i as compareCodeUnits, r as canonicalString, t as LEDGER_HASH_PATTERN } from "./canonical-DPyQ_rpt.js";
3
3
  import { o as summarizeNumberSeries, s as weightedComposite, t as confidenceInterval } from "./descriptive-1V17A-qa.js";
4
- import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
5
- import { i as modelHasSnapshot } from "./run-record-DQpSf7t-.js";
4
+ import { o as projectCampaignCellQuality, s as decidePairedPromotion, t as campaignCellCostProvenance } from "./run-record-Br-Yzt_k.js";
5
+ import { $ as formatCoverageFailures, A as FinalEvidenceConflictError, B as surfaceDispatchRef, C as campaignMeanCompositeOrNull, H as surfaceHashMatches, J as assertCompleteCampaign, K as assertCampaignDesign, N as defineEvaluationClaim, P as summarizeEvaluationUnits, R as renderSurfaceDiff, S as campaignMeanComposite, U as BackendIntegrityError, V as surfaceHash, X as campaignScenarioIdentity, Y as campaignCoverage, Z as campaignSplitDigest, at as pairHoldout, b as assertFiniteRankKey, it as heldoutSignificance, j as FinalEvidenceError, q as assertCampaignSplitIdentity, rt as dimensionRegressions, t as createCampaignEvidenceReceipt, w as compareRankKeys, x as campaignBreakdown, z as surfaceContentHash } from "./campaign-evidence-B8oF9xQ6.js";
6
+ import { i as modelHasSnapshot } from "./run-record-DualPTn2.js";
6
7
  import { n as contentHash } from "./verdict-cache-B3eCVQtY.js";
7
- import { o as projectCampaignCellQuality, t as campaignCellCostProvenance } from "./run-record-CR63CpHK.js";
8
8
  import { i as CostLedger, t as CostAccountingIncompleteError } from "./cost-ledger-B1qx30B4.js";
9
- import { $ as campaignScenarioIdentity, C as campaignMeanCompositeOrNull, G as surfaceHashMatches, H as surfaceContentHash, K as BackendIntegrityError, M as heldoutSignificance, N as pairHoldout, Q as campaignCoverage, S as campaignMeanComposite, U as surfaceDispatchRef, V as renderSurfaceDiff, W as surfaceHash, X as assertCampaignSplitIdentity, Y as assertCampaignDesign, Z as assertCompleteCampaign, b as assertFiniteRankKey, et as campaignSplitDigest, j as dimensionRegressions, nt as formatCoverageFailures, t as createCampaignEvidenceReceipt, w as compareRankKeys, x as campaignBreakdown } from "./campaign-evidence-D8DBLqLI.js";
10
- import { d as mapConcurrent, n as replayLedgerText, t as FileLedgerJournal } from "./ledger-core-Cs9f7385.js";
11
- import { D as SearchLedgerIntegrityError, E as SearchLedgerError, S as fsCampaignStorage, T as SearchLedgerConflictError, g as isRecord, h as isExternalTextCandidate, w as SEARCH_LEDGER_FILE_CONTEXT, x as createRunCostLedger } from "./external-optimizer-subprocess-q3VzlGAO.js";
9
+ import { d as mapConcurrent, n as replayLedgerText, t as FileLedgerJournal } from "./journal-Cs9f7385.js";
10
+ import "./ledger-core/index.js";
11
+ import { D as SearchLedgerIntegrityError, E as SearchLedgerError, S as fsCampaignStorage, T as SearchLedgerConflictError, g as isRecord, h as isExternalTextCandidate, w as SEARCH_LEDGER_FILE_CONTEXT, x as createRunCostLedger } from "./external-optimizer-subprocess-D4dzUBZI.js";
12
12
  import { t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
13
13
  import { t as detectRewardHacking } from "./reward-hacking-D0XwhVWE.js";
14
14
  import { n as assertProposalFindings, t as combineAbortSignals } from "./abort-signal-CtzAM_sJ.js";
@@ -975,6 +975,75 @@ async function runEval(opts) {
975
975
  return runCampaign(opts);
976
976
  }
977
977
  //#endregion
978
+ //#region src/campaign/final-evidence.ts
979
+ /** Capture identities before an asynchronous search can mutate its caller's options. */
980
+ function captureFinalEvidencePolicy(policy) {
981
+ return Object.freeze({
982
+ ledger: policy.ledger,
983
+ requestId: policy.requestId,
984
+ evaluatorDigest: policy.evaluatorDigest
985
+ });
986
+ }
987
+ function evaluationUnitMap(claim, scenarios) {
988
+ const map = /* @__PURE__ */ new Map();
989
+ for (const scenario of scenarios) {
990
+ const unit = summarizeEvaluationUnits(claim, [scenario]).units[0];
991
+ if (!unit) throw new ValidationError("final evidence has a scenario without an independent unit");
992
+ if (map.has(scenario.id)) throw new ValidationError(`duplicate final scenario '${scenario.id}'`);
993
+ map.set(scenario.id, unit.id);
994
+ }
995
+ return map;
996
+ }
997
+ /** Shared source variants cannot serve as unseen-unit evidence after development. */
998
+ function assertIndependentEvaluationSplit(claim, finalScenarios, developmentScenarios) {
999
+ const finalUnitIds = new Set(evaluationUnitMap(claim, finalScenarios).values());
1000
+ const overlap = summarizeEvaluationUnits(claim, developmentScenarios).units.filter((unit) => finalUnitIds.has(unit.id));
1001
+ if (overlap.length) throw new ValidationError(`development and final evidence share independent units: ${overlap.map((unit) => unit.id).join(", ")}`);
1002
+ }
1003
+ /** Reserve before search. An identical retry may resume until final data is exposed. */
1004
+ async function reserveFinalEvidence(policy, claimInput, scenarios, developmentScenarios = []) {
1005
+ policy = captureFinalEvidencePolicy(policy);
1006
+ if (claimInput === void 0) throw new ValidationError("fresh final evidence requires an evaluation claim");
1007
+ const claim = defineEvaluationClaim(claimInput);
1008
+ if (claim.use === "development") throw new ValidationError("final evidence requires a comparison or certification claim");
1009
+ if (!LEDGER_HASH_PATTERN.test(policy.evaluatorDigest)) throw new ValidationError("final evidence requires the evaluator content digest");
1010
+ const units = evaluationUnitMap(claim, scenarios);
1011
+ const finalUnitIds = new Set(units.values());
1012
+ assertIndependentEvaluationSplit(claim, scenarios, developmentScenarios);
1013
+ const result = await policy.ledger.reserve({
1014
+ requestId: policy.requestId,
1015
+ claimDigest: hashCanonical({
1016
+ claim,
1017
+ evaluatorDigest: policy.evaluatorDigest
1018
+ }),
1019
+ populationId: claim.population.id,
1020
+ inputDigest: hashCanonical([...scenarios].sort((a, b) => compareCodeUnits(a.id, b.id))),
1021
+ unitIds: [...finalUnitIds].sort(compareCodeUnits)
1022
+ });
1023
+ if (!result.succeeded) throw new FinalEvidenceError(result.error.kind, result.error.message);
1024
+ if (result.value.record.exposure !== null) throw new FinalEvidenceConflictError(`final evidence for '${policy.requestId}' was already exposed; read its recorded result or use fresh evidence`);
1025
+ return {
1026
+ claim,
1027
+ record: result.value.record
1028
+ };
1029
+ }
1030
+ /** Append exposure before dispatch, so a failed or interrupted measurement still consumes evidence. */
1031
+ async function exposeFinalEvidence(policy, claim, scenarios, surfaces) {
1032
+ policy = captureFinalEvidencePolicy(policy);
1033
+ const measurement = {
1034
+ evaluatorDigest: policy.evaluatorDigest,
1035
+ candidateDigests: surfaces.map((surface) => hashCanonical(surface))
1036
+ };
1037
+ const reserved = await reserveFinalEvidence(policy, claim, scenarios);
1038
+ const result = await policy.ledger.expose(policy.requestId, measurement);
1039
+ if (!result.succeeded) throw new FinalEvidenceError(result.error.kind, result.error.message);
1040
+ if (result.value.replayed) throw new FinalEvidenceConflictError(`final evidence for '${policy.requestId}' is already being measured`);
1041
+ return {
1042
+ claim: reserved.claim,
1043
+ record: result.value.record
1044
+ };
1045
+ }
1046
+ //#endregion
978
1047
  //#region src/canary.ts
979
1048
  /**
980
1049
  * Run all configured canaries against a chronological run list.
@@ -1790,6 +1859,12 @@ function defaultProductionGate(options) {
1790
1859
  const minProductiveRuns = options.minProductiveRuns ?? 3;
1791
1860
  const heldoutStatistic = options.heldoutStatistic ?? "mean";
1792
1861
  const explicitlyRequired = new Set(options.requiredChecks ?? []);
1862
+ const scenarioIds = new Set(options.holdoutScenarios.map((scenario) => scenario.id));
1863
+ const independentUnitByScenarioId = options.independentUnitByScenarioId === void 0 ? void 0 : new Map(options.independentUnitByScenarioId);
1864
+ if (independentUnitByScenarioId !== void 0) for (const id of scenarioIds) {
1865
+ const unit = independentUnitByScenarioId.get(id);
1866
+ if (typeof unit !== "string" || unit.length === 0 || unit.trim() !== unit) throw new Error(`defaultProductionGate: missing independent unit for holdout scenario '${id}'`);
1867
+ }
1793
1868
  return {
1794
1869
  name: "defaultProductionGate",
1795
1870
  async decide(ctx) {
@@ -1811,7 +1886,6 @@ function defaultProductionGate(options) {
1811
1886
  reasons.push(`${name}: ${reason}`);
1812
1887
  }
1813
1888
  };
1814
- const scenarioIds = new Set(options.holdoutScenarios.map((s) => s.id));
1815
1889
  let delta;
1816
1890
  if (!ctx.baselineJudgeScores) unavailable("heldout-significance", true, "baselineJudgeScores is required for a held-out comparison");
1817
1891
  else {
@@ -1821,7 +1895,8 @@ function defaultProductionGate(options) {
1821
1895
  confidence,
1822
1896
  resamples,
1823
1897
  seed,
1824
- statistic: heldoutStatistic
1898
+ statistic: heldoutStatistic,
1899
+ independentUnitByScenarioId
1825
1900
  });
1826
1901
  const dec = sig.decision;
1827
1902
  delta = dec.delta;
@@ -1831,6 +1906,9 @@ function defaultProductionGate(options) {
1831
1906
  status: sig.fewRuns ? "not_evaluated" : heldoutPass ? "pass" : "fail",
1832
1907
  detail: {
1833
1908
  n: sig.n,
1909
+ pairedCellN: sig.pairedCellN,
1910
+ observationUnit: sig.observationUnit,
1911
+ unitIds: sig.unitIds,
1834
1912
  delta,
1835
1913
  decisionStatistic: sig.decisionStatistic,
1836
1914
  decisionMethod: sig.decisionMethod,
@@ -1852,9 +1930,9 @@ function defaultProductionGate(options) {
1852
1930
  });
1853
1931
  if (sig.fewRuns) requiredUnavailable.add("heldout-significance");
1854
1932
  if (!heldoutPass) {
1855
- const tieNote = sig.tieFraction >= .4 ? `; ${(sig.tieFraction * 100).toFixed(0)}% tied scenarios` : "";
1933
+ const tieNote = sig.tieFraction >= .4 ? `; ${(sig.tieFraction * 100).toFixed(0)}% tied observation units` : "";
1856
1934
  const ci = `${(dec.confidence * 100).toFixed(0)}% CI [${dec.low.toFixed(3)}, ${dec.high.toFixed(3)}]`;
1857
- reasons.push(sig.fewRuns ? `held-out: only ${sig.n} paired runs (< ${sig.minimumRequired}) — too few to claim significance` : dec.indeterminate ? `held-out: ${dec.indeterminateCause}, so the paired CI is ${ci} and carries no direction — it cannot clear threshold ${deltaThreshold} on evidence${tieNote}` : dec.exactTestVetoes ? `held-out: McNemar exact p=${dec.mcnemar?.pValue.toExponential(2)} does not reject at α=${(1 - dec.confidence).toFixed(4)} (${dec.label} Δ ${delta.toFixed(3)}, ${ci}${tieNote})` : `held-out CI.low ${dec.low.toFixed(3)} ≤ threshold ${deltaThreshold} (${dec.label} Δ ${delta.toFixed(3)}, ${ci}${tieNote})`);
1935
+ reasons.push(sig.fewRuns ? `held-out: only ${sig.n} paired observation units (< ${sig.minimumRequired}) — too few to claim significance` : dec.indeterminate ? `held-out: ${dec.indeterminateCause}, so the paired CI is ${ci} and carries no direction — it cannot clear threshold ${deltaThreshold} on evidence${tieNote}` : dec.exactTestVetoes ? `held-out: McNemar exact p=${dec.mcnemar?.pValue.toExponential(2)} does not reject at α=${(1 - dec.confidence).toFixed(4)} (${dec.label} Δ ${delta.toFixed(3)}, ${ci}${tieNote})` : dec.method === "exact-sign" ? `held-out: exact one-sided sign test p=${dec.pValue} does not reject at α=${((1 - dec.confidence) / 2).toFixed(4)} (${dec.label} Δ ${delta.toFixed(3)}, diagnostic ${ci}${tieNote})` : `held-out CI.low ${dec.low.toFixed(3)} ≤ threshold ${deltaThreshold} (${dec.label} Δ ${delta.toFixed(3)}, ${ci}${tieNote})`);
1858
1936
  }
1859
1937
  }
1860
1938
  const dimensionsProvided = options.criticalDimensions !== void 0;
@@ -1868,18 +1946,22 @@ function defaultProductionGate(options) {
1868
1946
  tolerance: options.regressionTolerance,
1869
1947
  confidence,
1870
1948
  resamples,
1871
- seed
1949
+ seed,
1950
+ independentUnitByScenarioId,
1951
+ minProductiveRuns
1872
1952
  });
1873
1953
  const measured = new Set(dimRegs.map((result) => result.dimension));
1874
1954
  const missingDimensions = criticalDimensions.filter((dimension) => !measured.has(dimension));
1875
1955
  const regressed = dimRegs.filter((result) => result.regressed);
1876
- const dimensionStatus = regressed.length > 0 ? "fail" : missingDimensions.length > 0 ? "not_evaluated" : "pass";
1956
+ const incompleteDimensions = dimRegs.filter((result) => result.fewRuns || result.missingCellIds.length > 0 || result.missingScenarioIds.length > 0);
1957
+ const dimensionStatus = regressed.length > 0 ? "fail" : missingDimensions.length > 0 || incompleteDimensions.length > 0 ? "not_evaluated" : "pass";
1877
1958
  contributing.push({
1878
1959
  name: "dimension-regression",
1879
1960
  status: dimensionStatus,
1880
1961
  detail: {
1881
1962
  guarded: criticalDimensions,
1882
1963
  missingDimensions,
1964
+ incompleteDimensions: incompleteDimensions.map((result) => result.dimension),
1883
1965
  regressions: dimRegs.map((result) => ({
1884
1966
  dimension: result.dimension,
1885
1967
  ciLow: result.ci.low,
@@ -1890,6 +1972,12 @@ function defaultProductionGate(options) {
1890
1972
  median: result.bootstrap.median,
1891
1973
  tolerance: result.tolerance,
1892
1974
  n: result.n,
1975
+ pairedCellN: result.pairedCellN,
1976
+ observationUnit: result.observationUnit,
1977
+ minimumRequired: result.minimumRequired,
1978
+ fewRuns: result.fewRuns,
1979
+ missingCellIds: result.missingCellIds,
1980
+ missingScenarioIds: result.missingScenarioIds,
1893
1981
  regressed: result.regressed
1894
1982
  }))
1895
1983
  }
@@ -1898,7 +1986,11 @@ function defaultProductionGate(options) {
1898
1986
  requiredUnavailable.add("dimension-regression");
1899
1987
  reasons.push(`critical dimension(s) were not scored: ${missingDimensions.join(", ")}`);
1900
1988
  }
1901
- if (regressed.length > 0) reasons.push(`critical dimension(s) regressed: ${regressed.map((result) => `${result.dimension} CI.low ${result.ci.low.toFixed(3)} < -${result.tolerance}`).join("; ")}`);
1989
+ if (incompleteDimensions.length > 0) {
1990
+ requiredUnavailable.add("dimension-regression");
1991
+ reasons.push(`critical dimension evidence is incomplete: ${incompleteDimensions.map((result) => `${result.dimension} has ${result.n} paired observation units (minimum ${result.minimumRequired}), ${result.missingCellIds.length} unscored cells, ${result.missingScenarioIds.length} unscored scenarios`).join("; ")}`);
1992
+ }
1993
+ if (regressed.length > 0) reasons.push(`critical dimension(s) regressed: ${regressed.map((result) => result.bootstrap.low < -result.tolerance ? `${result.dimension} bootstrap CI.low ${result.bootstrap.low.toFixed(3)} < -${result.tolerance}` : `${result.dimension} paired test supports a drop greater than ${result.tolerance}`).join("; ")}`);
1902
1994
  }
1903
1995
  const budgetUsd = options.budgetUsd;
1904
1996
  const budgetConfigured = budgetUsd !== void 0;
@@ -2050,6 +2142,29 @@ function extractText(artifact) {
2050
2142
  }
2051
2143
  }
2052
2144
  //#endregion
2145
+ //#region src/campaign/judge-snapshot.ts
2146
+ /** Capture configuration and callbacks; receiver and closed-over state remain caller-owned. */
2147
+ function captureJudge(judge) {
2148
+ const dimensions = judge.dimensions.map((dimension) => Object.freeze({
2149
+ key: dimension.key,
2150
+ description: dimension.description
2151
+ }));
2152
+ Object.freeze(dimensions);
2153
+ const captured = {
2154
+ name: judge.name,
2155
+ judgeVersion: judge.judgeVersion,
2156
+ dimensions,
2157
+ score: judge.score,
2158
+ appliesTo: judge.appliesTo
2159
+ };
2160
+ return Object.freeze({
2161
+ ...captured,
2162
+ judgeVersion: judgeVersionFor(captured),
2163
+ score: captured.score.bind(judge),
2164
+ appliesTo: captured.appliesTo?.bind(judge)
2165
+ });
2166
+ }
2167
+ //#endregion
2053
2168
  //#region src/campaign/auto-pr.ts
2054
2169
  /**
2055
2170
  * `openAutoPr` — thin shell-out helper for the `runImprovementLoop` preset's
@@ -2180,6 +2295,12 @@ function quoteArg(arg) {
2180
2295
  //#region src/campaign/presets/run-final-comparison.ts
2181
2296
  /** Compare a search-selected surface without changing the search's selection. */
2182
2297
  async function runFinalComparison(opts) {
2298
+ opts = {
2299
+ ...opts,
2300
+ judges: opts.judges?.map(captureJudge),
2301
+ claim: opts.claim && defineEvaluationClaim(opts.claim),
2302
+ finalEvidence: opts.finalEvidence && captureFinalEvidencePolicy(opts.finalEvidence)
2303
+ };
2183
2304
  const storage = opts.storage ?? fsCampaignStorage();
2184
2305
  const costLedger = opts.costLedger ?? createRunCostLedger({
2185
2306
  storage,
@@ -2193,6 +2314,13 @@ async function runFinalComparison(opts) {
2193
2314
  const finalScenarios = structuredClone(opts.scenarios);
2194
2315
  const winnerIsBaseline = surfaceHash(winnerSurface) === surfaceHash(baselineSurface);
2195
2316
  const holdoutDeferred = (opts.holdout ?? "measured") === "deferred";
2317
+ if (opts.finalEvidence && holdoutDeferred) throw new Error("final evidence requires a measured comparison");
2318
+ const controlSurface = opts.neutralize && !winnerIsBaseline && !holdoutDeferred ? structuredClone(opts.neutralize(structuredClone(winnerSurface), structuredClone(baselineSurface))) : void 0;
2319
+ const finalEvidence = opts.finalEvidence ? await exposeFinalEvidence(opts.finalEvidence, opts.claim, finalScenarios, [
2320
+ baselineSurface,
2321
+ winnerSurface,
2322
+ ...controlSurface === void 0 ? [] : [controlSurface]
2323
+ ]) : void 0;
2196
2324
  const baselineOnHoldout = holdoutDeferred ? await runCampaign({
2197
2325
  ...opts,
2198
2326
  labeledStore: "off",
@@ -2251,8 +2379,8 @@ async function runFinalComparison(opts) {
2251
2379
  let neutralizedJudgeScores;
2252
2380
  let neutralizedOnHoldout;
2253
2381
  let neutralizedSurface;
2254
- if (opts.neutralize && !winnerIsBaseline && !holdoutDeferred) {
2255
- const surface = opts.neutralize(structuredClone(winnerSurface), structuredClone(baselineSurface));
2382
+ if (controlSurface !== void 0) {
2383
+ const surface = controlSurface;
2256
2384
  neutralizedSurface = surface;
2257
2385
  neutralizedOnHoldout = await runCampaign({
2258
2386
  ...opts,
@@ -2308,6 +2436,8 @@ async function runFinalComparison(opts) {
2308
2436
  });
2309
2437
  const promotedDiff = surfaceHash(winnerSurface) === surfaceHash(baselineSurface) ? "" : renderSurfaceDiff(winnerSurface, baselineSurface);
2310
2438
  return {
2439
+ ...opts.claim ? { claim: opts.claim } : {},
2440
+ ...finalEvidence ? { finalEvidence } : {},
2311
2441
  baselineOnHoldout,
2312
2442
  winnerOnHoldout,
2313
2443
  ...neutralizedOnHoldout && neutralizedSurface ? {
@@ -2399,391 +2529,15 @@ function paretoFrontierWithCrowding(candidates, objectives) {
2399
2529
  return crowdingDistance(frontier, objectives).sort((a, b) => b.distance - a.distance);
2400
2530
  }
2401
2531
  //#endregion
2402
- //#region src/campaign/search-ledger.ts
2403
- /**
2404
- * Durable append-only audit log for improvement searches.
2405
- *
2406
- * Existing campaign artifacts keep their own rich records: `RunRecord` owns a
2407
- * measured run and `CostLedger` owns per-call accounting. This ledger does not
2408
- * copy those structures. It binds their immutable ids and receipts into one replayable event stream so a
2409
- * search can answer, after a crash, exactly which candidates and task attempts
2410
- * existed, which surfaces actually fired, what they cost, and why they were
2411
- * selected or rejected.
2412
- *
2413
- * The file format is canonical JSONL with a SHA-256 hash chain. Every append is
2414
- * serialized across processes, fsynced before acknowledgement, and idempotent
2415
- * by `eventId`. A malformed, non-canonical, truncated, reordered, or conflicting
2416
- * log fails loudly; the implementation never skips a bad row.
2417
- *
2418
- * The journal machinery itself (hash chain, locking, fsync, idempotent append)
2419
- * is the generic `ledger-core` journal; this module supplies the campaign
2420
- * codec: event schemas, canonical event ordering, and the search state machine.
2421
- */
2422
- const SEARCH_LEDGER_SCHEMA = "tangle.search-ledger.v1";
2423
- const NON_EMPTY = z.string().min(1).refine((value) => value.trim() === value, "must not contain surrounding whitespace");
2424
- const HASH = z.string().regex(/^sha256:[a-f0-9]{64}$/);
2425
- const LINEAGE_NODE_ID = z.string().regex(/^[a-f0-9]{16}$/);
2426
- const IMMUTABLE_REVISION = z.string().regex(/^(?:[a-f0-9]{40}|[a-f0-9]{64}|sha256:[a-f0-9]{64}|sha512:[A-Za-z0-9+/=]+)$/);
2427
- const ISO_TIMESTAMP = z.string().regex(/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{3})?Z$/).refine((value) => Number.isFinite(Date.parse(value)), "invalid timestamp");
2428
- const NON_NEGATIVE_INT = z.number().int().nonnegative().safe();
2429
- const FINITE_NUMBER = z.number().finite();
2430
- const ArtifactRefSchema = z.object({
2431
- role: NON_EMPTY,
2432
- uri: NON_EMPTY,
2433
- sha256: HASH,
2434
- byteLength: NON_NEGATIVE_INT
2435
- }).strict();
2436
- const SourceRefSchema = z.object({
2437
- uri: NON_EMPTY,
2438
- revision: IMMUTABLE_REVISION
2439
- }).strict();
2440
- const FailureReasonSchema = z.object({
2441
- code: NON_EMPTY,
2442
- message: NON_EMPTY
2443
- }).strict();
2444
- const EventBaseShape = {
2445
- eventId: NON_EMPTY,
2446
- occurredAt: ISO_TIMESTAMP,
2447
- artifacts: z.array(ArtifactRefSchema).min(1)
2448
- };
2449
- const OperationKindSchema = z.enum([
2450
- "candidate-generation",
2451
- "analysis",
2452
- "selection",
2453
- "judge",
2454
- "other"
2455
- ]);
2456
- const CandidateSlotSchema = z.object({
2457
- slotId: NON_EMPTY,
2458
- generationOperationId: NON_EMPTY
2459
- }).strict();
2460
- const PlannedOperationSchema = z.object({
2461
- operationId: NON_EMPTY,
2462
- kind: OperationKindSchema
2463
- }).strict();
2464
- const SearchPlanExtendedSchema = z.object({
2465
- ...EventBaseShape,
2466
- kind: z.literal("search-plan-extended"),
2467
- extension: z.object({
2468
- candidateSlots: z.array(CandidateSlotSchema),
2469
- operations: z.array(PlannedOperationSchema)
2470
- }).strict().superRefine((extension, ctx) => {
2471
- if (extension.candidateSlots.length === 0 && extension.operations.length === 0) ctx.addIssue({
2472
- code: "custom",
2473
- message: "a plan extension must add slots or operations"
2474
- });
2475
- })
2476
- }).strict();
2477
- const SearchPlannedSchema = z.object({
2478
- ...EventBaseShape,
2479
- kind: z.literal("search-planned"),
2480
- plan: z.object({
2481
- candidateSlots: z.array(CandidateSlotSchema).min(1),
2482
- tasks: z.array(z.object({
2483
- taskId: NON_EMPTY,
2484
- source: SourceRefSchema,
2485
- benchmark: SourceRefSchema,
2486
- maxAttempts: z.number().int().positive().safe()
2487
- }).strict()).min(1),
2488
- operations: z.array(PlannedOperationSchema).min(1)
2489
- }).strict()
2490
- }).strict();
2491
- const CandidateRegisteredSchema = z.object({
2492
- ...EventBaseShape,
2493
- kind: z.literal("candidate-registered"),
2494
- slotId: NON_EMPTY,
2495
- generationOperationId: NON_EMPTY,
2496
- candidateId: NON_EMPTY,
2497
- lineage: z.object({
2498
- lineageNodeId: LINEAGE_NODE_ID,
2499
- parentCandidateIds: z.array(NON_EMPTY),
2500
- generation: NON_NEGATIVE_INT,
2501
- proposer: NON_EMPTY,
2502
- proposerSource: SourceRefSchema
2503
- }).strict(),
2504
- surfaces: z.array(z.object({
2505
- surfaceId: NON_EMPTY,
2506
- kind: z.enum([
2507
- "prompt",
2508
- "tool-contract",
2509
- "runtime-config",
2510
- "memory",
2511
- "knowledge",
2512
- "agent-profile",
2513
- "code",
2514
- "deployment"
2515
- ]),
2516
- artifact: ArtifactRefSchema
2517
- }).strict()).min(1)
2518
- }).strict();
2519
- const CandidateSlotClosedSchema = z.object({
2520
- ...EventBaseShape,
2521
- kind: z.literal("candidate-slot-closed"),
2522
- slotId: NON_EMPTY,
2523
- generationOperationId: NON_EMPTY,
2524
- reason: FailureReasonSchema
2525
- }).strict();
2526
- const KnownTokensSchema = z.object({
2527
- status: z.literal("known"),
2528
- inputTokens: NON_NEGATIVE_INT,
2529
- outputTokens: NON_NEGATIVE_INT,
2530
- cachedTokens: NON_NEGATIVE_INT
2531
- }).strict();
2532
- const UnknownSchema = z.object({
2533
- status: z.literal("unknown"),
2534
- reason: NON_EMPTY
2535
- }).strict();
2536
- const KnownCostSchema = z.object({
2537
- status: z.literal("known"),
2538
- usd: z.number().finite().nonnegative(),
2539
- source: z.enum([
2540
- "provider",
2541
- "pricing-table",
2542
- "free"
2543
- ])
2544
- }).strict().superRefine((cost, ctx) => {
2545
- if (cost.source === "free" && cost.usd !== 0) ctx.addIssue({
2546
- code: "custom",
2547
- message: "free cost source must have usd 0"
2548
- });
2549
- });
2550
- const UnknownCostSchema = z.object({
2551
- status: z.literal("unknown"),
2552
- knownLowerBoundUsd: z.number().finite().nonnegative(),
2553
- reason: NON_EMPTY
2554
- }).strict();
2555
- const AccountingSchema = z.object({
2556
- tokens: z.discriminatedUnion("status", [KnownTokensSchema, UnknownSchema]),
2557
- cost: z.discriminatedUnion("status", [KnownCostSchema, UnknownCostSchema])
2558
- }).strict();
2559
- const MetricsSchema = z.record(NON_EMPTY, FINITE_NUMBER).superRefine((metrics, ctx) => {
2560
- for (const key of Object.keys(metrics)) if (key === "__proto__" || key === "constructor" || key === "prototype") ctx.addIssue({
2561
- code: "custom",
2562
- message: `unsafe metric key ${key}`
2563
- });
2564
- });
2565
- const OutcomeSchema = z.discriminatedUnion("status", [
2566
- z.object({
2567
- status: z.literal("passed"),
2568
- score: FINITE_NUMBER,
2569
- metrics: MetricsSchema
2570
- }).strict(),
2571
- z.object({
2572
- status: z.literal("failed"),
2573
- score: FINITE_NUMBER,
2574
- metrics: MetricsSchema,
2575
- failure: FailureReasonSchema
2576
- }).strict(),
2577
- z.object({
2578
- status: z.literal("errored"),
2579
- metrics: MetricsSchema,
2580
- error: FailureReasonSchema.extend({ retryable: z.boolean() }).strict()
2581
- }).strict()
2582
- ]);
2583
- const EffectSchema = z.discriminatedUnion("status", [z.object({
2584
- status: z.literal("measured"),
2585
- metric: NON_EMPTY,
2586
- baselineValue: FINITE_NUMBER,
2587
- candidateValue: FINITE_NUMBER,
2588
- delta: FINITE_NUMBER
2589
- }).strict().superRefine((effect, ctx) => {
2590
- const expected = effect.candidateValue - effect.baselineValue;
2591
- const tolerance = Number.EPSILON * Math.max(1, Math.abs(expected), Math.abs(effect.delta)) * 8;
2592
- if (Math.abs(effect.delta - expected) > tolerance) ctx.addIssue({
2593
- code: "custom",
2594
- message: "delta must equal candidateValue - baselineValue"
2595
- });
2596
- }), z.object({
2597
- status: z.literal("not-measured"),
2598
- reason: NON_EMPTY
2599
- }).strict()]);
2600
- const SurfaceEvidenceSchema = z.object({
2601
- surfaceId: NON_EMPTY,
2602
- fired: z.boolean(),
2603
- firingCount: NON_NEGATIVE_INT,
2604
- effect: EffectSchema,
2605
- evidence: z.array(ArtifactRefSchema).min(1)
2606
- }).strict().superRefine((evidence, ctx) => {
2607
- if (evidence.fired && evidence.firingCount === 0) ctx.addIssue({
2608
- code: "custom",
2609
- message: "a fired surface must have firingCount >= 1"
2610
- });
2611
- if (!evidence.fired && evidence.firingCount !== 0) ctx.addIssue({
2612
- code: "custom",
2613
- message: "a surface that did not fire must have firingCount 0"
2614
- });
2615
- if (!evidence.fired && evidence.effect.status === "measured" && evidence.effect.delta !== 0) ctx.addIssue({
2616
- code: "custom",
2617
- message: "a surface that did not fire cannot claim non-zero effect"
2618
- });
2619
- });
2620
- const TaskAttemptedSchema = z.object({
2621
- ...EventBaseShape,
2622
- kind: z.literal("task-attempted"),
2623
- candidateId: NON_EMPTY,
2624
- runId: NON_EMPTY,
2625
- attemptIndex: NON_NEGATIVE_INT,
2626
- task: z.object({
2627
- taskId: NON_EMPTY,
2628
- source: SourceRefSchema
2629
- }).strict(),
2630
- identity: z.object({
2631
- model: z.object({
2632
- provider: NON_EMPTY,
2633
- snapshot: NON_EMPTY.refine(modelHasSnapshot, "model must include an immutable snapshot")
2634
- }).strict(),
2635
- agent: SourceRefSchema,
2636
- benchmark: SourceRefSchema
2637
- }).strict(),
2638
- outcome: OutcomeSchema,
2639
- accounting: AccountingSchema,
2640
- surfaceEvidence: z.array(SurfaceEvidenceSchema).min(1)
2641
- }).strict();
2642
- const SearchOperationRecordedSchema = z.object({
2643
- ...EventBaseShape,
2644
- kind: z.literal("search-operation-recorded"),
2645
- operationId: NON_EMPTY,
2646
- operationKind: OperationKindSchema,
2647
- execution: z.discriminatedUnion("kind", [z.object({
2648
- kind: z.literal("model"),
2649
- model: z.object({
2650
- provider: NON_EMPTY,
2651
- snapshot: NON_EMPTY.refine(modelHasSnapshot, "model must include an immutable snapshot")
2652
- }).strict(),
2653
- source: SourceRefSchema
2654
- }).strict(), z.object({
2655
- kind: z.literal("deterministic"),
2656
- source: SourceRefSchema
2657
- }).strict()]),
2658
- outcome: z.discriminatedUnion("status", [
2659
- z.object({ status: z.literal("completed") }).strict(),
2660
- z.object({
2661
- status: z.literal("partial"),
2662
- failure: FailureReasonSchema
2663
- }).strict(),
2664
- z.object({
2665
- status: z.literal("failed"),
2666
- failure: FailureReasonSchema
2667
- }).strict()
2668
- ]),
2669
- accounting: AccountingSchema
2670
- }).strict();
2671
- const CandidateDecidedSchema = z.object({
2672
- ...EventBaseShape,
2673
- kind: z.literal("candidate-decided"),
2674
- candidateId: NON_EMPTY,
2675
- decision: z.discriminatedUnion("status", [z.object({ status: z.literal("selected") }).strict(), z.object({
2676
- status: z.literal("rejected"),
2677
- reason: FailureReasonSchema
2678
- }).strict()])
2679
- }).strict();
2680
- const SearchCompletedSchema = z.object({
2681
- ...EventBaseShape,
2682
- kind: z.literal("search-completed"),
2683
- result: z.discriminatedUnion("status", [z.object({
2684
- status: z.literal("selected"),
2685
- candidateId: NON_EMPTY
2686
- }).strict(), z.object({
2687
- status: z.literal("all-rejected"),
2688
- reason: FailureReasonSchema
2689
- }).strict()])
2690
- }).strict();
2691
- const EventSchema = z.discriminatedUnion("kind", [
2692
- SearchPlannedSchema,
2693
- SearchPlanExtendedSchema,
2694
- CandidateRegisteredSchema,
2695
- CandidateSlotClosedSchema,
2696
- TaskAttemptedSchema,
2697
- SearchOperationRecordedSchema,
2698
- CandidateDecidedSchema,
2699
- SearchCompletedSchema
2700
- ]);
2701
- const EntrySchema = z.object({
2702
- schema: z.literal(SEARCH_LEDGER_SCHEMA),
2703
- campaignId: NON_EMPTY,
2704
- sequence: NON_NEGATIVE_INT,
2705
- previousHash: z.union([HASH, z.null()]),
2706
- event: EventSchema,
2707
- entryHash: HASH
2708
- }).strict();
2709
- /** Validate and return a canonical copy. Arrays whose order is not semantic are
2710
- * sorted so retries from different processes produce byte-identical events. */
2711
- function validateSearchLedgerEvent(input) {
2712
- const parsed = EventSchema.safeParse(input);
2713
- if (!parsed.success) throw new SearchLedgerError(`invalid search ledger event: ${formatZodError(parsed.error)}`);
2714
- return normalizeEvent(parsed.data);
2715
- }
2716
- /** Open a durable filesystem search ledger. Construction performs no I/O; the
2717
- * first `append` or `replay` validates the complete existing file. */
2718
- function openSearchLedger(options) {
2719
- if (options.path.trim().length === 0) throw new SearchLedgerError("ledger path is empty");
2720
- return new FileSearchLedger(options.path, options.campaignId, options.trustedHead);
2721
- }
2722
- /** Replay immutable search-ledger JSONL through the same codec as FileSearchLedger. */
2723
- function replaySearchLedgerText(text, campaignId, source) {
2724
- return replayLedgerText(text, source, searchLedgerCodec(campaignId)).projection;
2725
- }
2726
- function searchLedgerCodec(campaignId) {
2727
- return {
2728
- ...SEARCH_LEDGER_FILE_CONTEXT,
2729
- header: {
2730
- schema: SEARCH_LEDGER_SCHEMA,
2731
- campaignId
2732
- },
2733
- conflictError: (message) => new SearchLedgerConflictError(message),
2734
- parseEntry: parseSearchLedgerEntry,
2735
- checkEntryHeader: (entry, index) => {
2736
- if (entry.campaignId !== campaignId) throw new SearchLedgerIntegrityError(`entry ${index} belongs to campaign ${entry.campaignId}, expected ${campaignId}`);
2737
- },
2738
- createProjector: () => createSearchLedgerProjector(campaignId)
2739
- };
2532
+ //#region src/campaign/search-ledger-ordering.ts
2533
+ function artifactKey(artifact) {
2534
+ return canonicalString(artifact);
2740
2535
  }
2741
- /** Append-only file-backed search ledger with idempotent writes and replay. */
2742
- var FileSearchLedger = class {
2743
- path;
2744
- campaignId;
2745
- trustedHeadPath;
2746
- trustedHeadMode;
2747
- journal;
2748
- constructor(path, campaignId, trustedHead = "pin") {
2749
- if (path.trim().length === 0) throw new SearchLedgerError("ledger path is empty");
2750
- if (campaignId.length === 0) throw new SearchLedgerError("campaignId is empty");
2751
- if (campaignId.trim() !== campaignId) throw new SearchLedgerError("campaignId must not contain surrounding whitespace");
2752
- this.campaignId = campaignId;
2753
- this.trustedHeadMode = trustedHead;
2754
- this.journal = new FileLedgerJournal(path, searchLedgerCodec(campaignId), { requireTrustedHead: trustedHead === "require" });
2755
- this.path = this.journal.path;
2756
- this.trustedHeadPath = this.journal.trustedHeadPath;
2757
- }
2758
- async replay() {
2759
- return (await this.journal.replay()).projection;
2760
- }
2761
- async append(input) {
2762
- const event = validateSearchLedgerEvent(input);
2763
- const { entry, appended, projection } = await this.journal.append(event, { pinHead: this.trustedHeadMode !== "off" });
2764
- return {
2765
- entry,
2766
- appended,
2767
- replay: projection
2768
- };
2769
- }
2770
- async trustedHead() {
2771
- return this.journal.trustedHead();
2772
- }
2773
- async pinTrustedHead() {
2774
- return this.journal.pinTrustedHead();
2775
- }
2776
- async clearTrustedHead() {
2777
- return this.journal.clearTrustedHead();
2778
- }
2779
- };
2780
- function parseSearchLedgerEntry(raw, context) {
2781
- const parsed = EntrySchema.safeParse(raw);
2782
- if (!parsed.success) throw new SearchLedgerIntegrityError(`search ledger ${context.path} has a malformed entry at line ${context.line}: ${formatZodError(parsed.error)}`);
2783
- const entry = parsed.data;
2784
- if (canonicalString(validateSearchLedgerEvent(entry.event)) !== canonicalString(entry.event)) throw new SearchLedgerIntegrityError(`search ledger ${context.path} has non-canonical event ordering at line ${context.line}`);
2785
- return entry;
2536
+ function compareStrings(a, b) {
2537
+ return a < b ? -1 : a > b ? 1 : 0;
2786
2538
  }
2539
+ //#endregion
2540
+ //#region src/campaign/search-ledger-projector.ts
2787
2541
  /** Replay the campaign search state machine over chain-verified entries. The
2788
2542
  * generic journal owns sequence, hash, and eventId-uniqueness checks; this
2789
2543
  * projector owns every campaign invariant and builds the replay projection. */
@@ -3062,23 +2816,414 @@ function createSearchLedgerProjector(campaignId) {
3062
2816
  };
3063
2817
  };
3064
2818
  return {
3065
- apply,
3066
- finish
2819
+ apply,
2820
+ finish
2821
+ };
2822
+ }
2823
+ /** Every candidate slot must name a planned candidate-generation operation,
2824
+ * whether it arrives with the plan or with a later extension. */
2825
+ function assertSlotGenerationOperation(slot, plannedOperations) {
2826
+ if (plannedOperations.get(slot.generationOperationId)?.kind !== "candidate-generation") throw new SearchLedgerIntegrityError(`candidate slot ${slot.slotId} references unplanned candidate-generation operation ${slot.generationOperationId}`);
2827
+ }
2828
+ function plannedTaskOutcomeKeys(planEvent, candidates) {
2829
+ const missing = [];
2830
+ const registeredCandidates = [...candidates.values()].sort((a, b) => compareStrings(a.registered.slotId, b.registered.slotId));
2831
+ for (const candidate of registeredCandidates) {
2832
+ const slotId = candidate.registered.slotId;
2833
+ for (const task of planEvent.plan.tasks) if (!candidate.attempts.some((attempt) => attempt.task.taskId === task.taskId && attempt.outcome.status !== "errored")) missing.push(`${slotId}/${task.taskId}`);
2834
+ }
2835
+ return missing;
2836
+ }
2837
+ function assertUnique(values, label, eventId) {
2838
+ if (new Set(values).size !== values.length) throw new SearchLedgerIntegrityError(`event ${eventId} contains duplicate ${label} values`);
2839
+ }
2840
+ //#endregion
2841
+ //#region src/campaign/search-ledger-types.ts
2842
+ const SEARCH_LEDGER_SCHEMA = "tangle.search-ledger.v1";
2843
+ //#endregion
2844
+ //#region src/campaign/search-ledger.ts
2845
+ /**
2846
+ * Durable append-only audit log for improvement searches.
2847
+ *
2848
+ * Existing campaign artifacts keep their own rich records: `RunRecord` owns a
2849
+ * measured run and `CostLedger` owns per-call accounting. This ledger does not
2850
+ * copy those structures. It binds their immutable ids and receipts into one replayable event stream so a
2851
+ * search can answer, after a crash, exactly which candidates and task attempts
2852
+ * existed, which surfaces actually fired, what they cost, and why they were
2853
+ * selected or rejected.
2854
+ *
2855
+ * The file format is canonical JSONL with a SHA-256 hash chain. Every append is
2856
+ * serialized across processes, fsynced before acknowledgement, and idempotent
2857
+ * by `eventId`. A malformed, non-canonical, truncated, reordered, or conflicting
2858
+ * log fails loudly; the implementation never skips a bad row.
2859
+ *
2860
+ * The journal machinery itself (hash chain, locking, fsync, idempotent append)
2861
+ * is the generic `ledger-core` journal; this module supplies the campaign
2862
+ * codec: event schemas, canonical event ordering, and the search state machine.
2863
+ */
2864
+ const NON_EMPTY = z.string().min(1).refine((value) => value.trim() === value, "must not contain surrounding whitespace");
2865
+ const HASH = z.string().regex(/^sha256:[a-f0-9]{64}$/);
2866
+ const LINEAGE_NODE_ID = z.string().regex(/^[a-f0-9]{16}$/);
2867
+ const IMMUTABLE_REVISION = z.string().regex(/^(?:[a-f0-9]{40}|[a-f0-9]{64}|sha256:[a-f0-9]{64}|sha512:[A-Za-z0-9+/=]+)$/);
2868
+ const ISO_TIMESTAMP = z.string().regex(/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d{3})?Z$/).refine((value) => Number.isFinite(Date.parse(value)), "invalid timestamp");
2869
+ const NON_NEGATIVE_INT = z.number().int().nonnegative().safe();
2870
+ const FINITE_NUMBER = z.number().finite();
2871
+ const ArtifactRefSchema = z.object({
2872
+ role: NON_EMPTY,
2873
+ uri: NON_EMPTY,
2874
+ sha256: HASH,
2875
+ byteLength: NON_NEGATIVE_INT
2876
+ }).strict();
2877
+ const SourceRefSchema = z.object({
2878
+ uri: NON_EMPTY,
2879
+ revision: IMMUTABLE_REVISION
2880
+ }).strict();
2881
+ const FailureReasonSchema = z.object({
2882
+ code: NON_EMPTY,
2883
+ message: NON_EMPTY
2884
+ }).strict();
2885
+ const EventBaseShape = {
2886
+ eventId: NON_EMPTY,
2887
+ occurredAt: ISO_TIMESTAMP,
2888
+ artifacts: z.array(ArtifactRefSchema).min(1)
2889
+ };
2890
+ const OperationKindSchema = z.enum([
2891
+ "candidate-generation",
2892
+ "analysis",
2893
+ "selection",
2894
+ "judge",
2895
+ "other"
2896
+ ]);
2897
+ const CandidateSlotSchema = z.object({
2898
+ slotId: NON_EMPTY,
2899
+ generationOperationId: NON_EMPTY
2900
+ }).strict();
2901
+ const PlannedOperationSchema = z.object({
2902
+ operationId: NON_EMPTY,
2903
+ kind: OperationKindSchema
2904
+ }).strict();
2905
+ const SearchPlanExtendedSchema = z.object({
2906
+ ...EventBaseShape,
2907
+ kind: z.literal("search-plan-extended"),
2908
+ extension: z.object({
2909
+ candidateSlots: z.array(CandidateSlotSchema),
2910
+ operations: z.array(PlannedOperationSchema)
2911
+ }).strict().superRefine((extension, ctx) => {
2912
+ if (extension.candidateSlots.length === 0 && extension.operations.length === 0) ctx.addIssue({
2913
+ code: "custom",
2914
+ message: "a plan extension must add slots or operations"
2915
+ });
2916
+ })
2917
+ }).strict();
2918
+ const SearchPlannedSchema = z.object({
2919
+ ...EventBaseShape,
2920
+ kind: z.literal("search-planned"),
2921
+ plan: z.object({
2922
+ candidateSlots: z.array(CandidateSlotSchema).min(1),
2923
+ tasks: z.array(z.object({
2924
+ taskId: NON_EMPTY,
2925
+ source: SourceRefSchema,
2926
+ benchmark: SourceRefSchema,
2927
+ maxAttempts: z.number().int().positive().safe()
2928
+ }).strict()).min(1),
2929
+ operations: z.array(PlannedOperationSchema).min(1)
2930
+ }).strict()
2931
+ }).strict();
2932
+ const CandidateRegisteredSchema = z.object({
2933
+ ...EventBaseShape,
2934
+ kind: z.literal("candidate-registered"),
2935
+ slotId: NON_EMPTY,
2936
+ generationOperationId: NON_EMPTY,
2937
+ candidateId: NON_EMPTY,
2938
+ lineage: z.object({
2939
+ lineageNodeId: LINEAGE_NODE_ID,
2940
+ parentCandidateIds: z.array(NON_EMPTY),
2941
+ generation: NON_NEGATIVE_INT,
2942
+ proposer: NON_EMPTY,
2943
+ proposerSource: SourceRefSchema
2944
+ }).strict(),
2945
+ surfaces: z.array(z.object({
2946
+ surfaceId: NON_EMPTY,
2947
+ kind: z.enum([
2948
+ "prompt",
2949
+ "tool-contract",
2950
+ "runtime-config",
2951
+ "memory",
2952
+ "knowledge",
2953
+ "agent-profile",
2954
+ "code",
2955
+ "deployment"
2956
+ ]),
2957
+ artifact: ArtifactRefSchema
2958
+ }).strict()).min(1)
2959
+ }).strict();
2960
+ const CandidateSlotClosedSchema = z.object({
2961
+ ...EventBaseShape,
2962
+ kind: z.literal("candidate-slot-closed"),
2963
+ slotId: NON_EMPTY,
2964
+ generationOperationId: NON_EMPTY,
2965
+ reason: FailureReasonSchema
2966
+ }).strict();
2967
+ const KnownTokensSchema = z.object({
2968
+ status: z.literal("known"),
2969
+ inputTokens: NON_NEGATIVE_INT,
2970
+ outputTokens: NON_NEGATIVE_INT,
2971
+ cachedTokens: NON_NEGATIVE_INT
2972
+ }).strict();
2973
+ const UnknownSchema = z.object({
2974
+ status: z.literal("unknown"),
2975
+ reason: NON_EMPTY
2976
+ }).strict();
2977
+ const KnownCostSchema = z.object({
2978
+ status: z.literal("known"),
2979
+ usd: z.number().finite().nonnegative(),
2980
+ source: z.enum([
2981
+ "provider",
2982
+ "pricing-table",
2983
+ "free"
2984
+ ])
2985
+ }).strict().superRefine((cost, ctx) => {
2986
+ if (cost.source === "free" && cost.usd !== 0) ctx.addIssue({
2987
+ code: "custom",
2988
+ message: "free cost source must have usd 0"
2989
+ });
2990
+ });
2991
+ const UnknownCostSchema = z.object({
2992
+ status: z.literal("unknown"),
2993
+ knownLowerBoundUsd: z.number().finite().nonnegative(),
2994
+ reason: NON_EMPTY
2995
+ }).strict();
2996
+ const AccountingSchema = z.object({
2997
+ tokens: z.discriminatedUnion("status", [KnownTokensSchema, UnknownSchema]),
2998
+ cost: z.discriminatedUnion("status", [KnownCostSchema, UnknownCostSchema])
2999
+ }).strict();
3000
+ const MetricsSchema = z.record(NON_EMPTY, FINITE_NUMBER).superRefine((metrics, ctx) => {
3001
+ for (const key of Object.keys(metrics)) if (key === "__proto__" || key === "constructor" || key === "prototype") ctx.addIssue({
3002
+ code: "custom",
3003
+ message: `unsafe metric key ${key}`
3004
+ });
3005
+ });
3006
+ const OutcomeSchema = z.discriminatedUnion("status", [
3007
+ z.object({
3008
+ status: z.literal("passed"),
3009
+ score: FINITE_NUMBER,
3010
+ metrics: MetricsSchema
3011
+ }).strict(),
3012
+ z.object({
3013
+ status: z.literal("failed"),
3014
+ score: FINITE_NUMBER,
3015
+ metrics: MetricsSchema,
3016
+ failure: FailureReasonSchema
3017
+ }).strict(),
3018
+ z.object({
3019
+ status: z.literal("errored"),
3020
+ metrics: MetricsSchema,
3021
+ error: FailureReasonSchema.extend({ retryable: z.boolean() }).strict()
3022
+ }).strict()
3023
+ ]);
3024
+ const EffectSchema = z.discriminatedUnion("status", [z.object({
3025
+ status: z.literal("measured"),
3026
+ metric: NON_EMPTY,
3027
+ baselineValue: FINITE_NUMBER,
3028
+ candidateValue: FINITE_NUMBER,
3029
+ delta: FINITE_NUMBER
3030
+ }).strict().superRefine((effect, ctx) => {
3031
+ const expected = effect.candidateValue - effect.baselineValue;
3032
+ const tolerance = Number.EPSILON * Math.max(1, Math.abs(expected), Math.abs(effect.delta)) * 8;
3033
+ if (Math.abs(effect.delta - expected) > tolerance) ctx.addIssue({
3034
+ code: "custom",
3035
+ message: "delta must equal candidateValue - baselineValue"
3036
+ });
3037
+ }), z.object({
3038
+ status: z.literal("not-measured"),
3039
+ reason: NON_EMPTY
3040
+ }).strict()]);
3041
+ const SurfaceEvidenceSchema = z.object({
3042
+ surfaceId: NON_EMPTY,
3043
+ fired: z.boolean(),
3044
+ firingCount: NON_NEGATIVE_INT,
3045
+ effect: EffectSchema,
3046
+ evidence: z.array(ArtifactRefSchema).min(1)
3047
+ }).strict().superRefine((evidence, ctx) => {
3048
+ if (evidence.fired && evidence.firingCount === 0) ctx.addIssue({
3049
+ code: "custom",
3050
+ message: "a fired surface must have firingCount >= 1"
3051
+ });
3052
+ if (!evidence.fired && evidence.firingCount !== 0) ctx.addIssue({
3053
+ code: "custom",
3054
+ message: "a surface that did not fire must have firingCount 0"
3055
+ });
3056
+ if (!evidence.fired && evidence.effect.status === "measured" && evidence.effect.delta !== 0) ctx.addIssue({
3057
+ code: "custom",
3058
+ message: "a surface that did not fire cannot claim non-zero effect"
3059
+ });
3060
+ });
3061
+ const TaskAttemptedSchema = z.object({
3062
+ ...EventBaseShape,
3063
+ kind: z.literal("task-attempted"),
3064
+ candidateId: NON_EMPTY,
3065
+ runId: NON_EMPTY,
3066
+ attemptIndex: NON_NEGATIVE_INT,
3067
+ task: z.object({
3068
+ taskId: NON_EMPTY,
3069
+ source: SourceRefSchema
3070
+ }).strict(),
3071
+ identity: z.object({
3072
+ model: z.object({
3073
+ provider: NON_EMPTY,
3074
+ snapshot: NON_EMPTY.refine(modelHasSnapshot, "model must include an immutable snapshot")
3075
+ }).strict(),
3076
+ agent: SourceRefSchema,
3077
+ benchmark: SourceRefSchema
3078
+ }).strict(),
3079
+ outcome: OutcomeSchema,
3080
+ accounting: AccountingSchema,
3081
+ surfaceEvidence: z.array(SurfaceEvidenceSchema).min(1)
3082
+ }).strict();
3083
+ const SearchOperationRecordedSchema = z.object({
3084
+ ...EventBaseShape,
3085
+ kind: z.literal("search-operation-recorded"),
3086
+ operationId: NON_EMPTY,
3087
+ operationKind: OperationKindSchema,
3088
+ execution: z.discriminatedUnion("kind", [z.object({
3089
+ kind: z.literal("model"),
3090
+ model: z.object({
3091
+ provider: NON_EMPTY,
3092
+ snapshot: NON_EMPTY.refine(modelHasSnapshot, "model must include an immutable snapshot")
3093
+ }).strict(),
3094
+ source: SourceRefSchema
3095
+ }).strict(), z.object({
3096
+ kind: z.literal("deterministic"),
3097
+ source: SourceRefSchema
3098
+ }).strict()]),
3099
+ outcome: z.discriminatedUnion("status", [
3100
+ z.object({ status: z.literal("completed") }).strict(),
3101
+ z.object({
3102
+ status: z.literal("partial"),
3103
+ failure: FailureReasonSchema
3104
+ }).strict(),
3105
+ z.object({
3106
+ status: z.literal("failed"),
3107
+ failure: FailureReasonSchema
3108
+ }).strict()
3109
+ ]),
3110
+ accounting: AccountingSchema
3111
+ }).strict();
3112
+ const CandidateDecidedSchema = z.object({
3113
+ ...EventBaseShape,
3114
+ kind: z.literal("candidate-decided"),
3115
+ candidateId: NON_EMPTY,
3116
+ decision: z.discriminatedUnion("status", [z.object({ status: z.literal("selected") }).strict(), z.object({
3117
+ status: z.literal("rejected"),
3118
+ reason: FailureReasonSchema
3119
+ }).strict()])
3120
+ }).strict();
3121
+ const SearchCompletedSchema = z.object({
3122
+ ...EventBaseShape,
3123
+ kind: z.literal("search-completed"),
3124
+ result: z.discriminatedUnion("status", [z.object({
3125
+ status: z.literal("selected"),
3126
+ candidateId: NON_EMPTY
3127
+ }).strict(), z.object({
3128
+ status: z.literal("all-rejected"),
3129
+ reason: FailureReasonSchema
3130
+ }).strict()])
3131
+ }).strict();
3132
+ const EventSchema = z.discriminatedUnion("kind", [
3133
+ SearchPlannedSchema,
3134
+ SearchPlanExtendedSchema,
3135
+ CandidateRegisteredSchema,
3136
+ CandidateSlotClosedSchema,
3137
+ TaskAttemptedSchema,
3138
+ SearchOperationRecordedSchema,
3139
+ CandidateDecidedSchema,
3140
+ SearchCompletedSchema
3141
+ ]);
3142
+ const EntrySchema = z.object({
3143
+ schema: z.literal(SEARCH_LEDGER_SCHEMA),
3144
+ campaignId: NON_EMPTY,
3145
+ sequence: NON_NEGATIVE_INT,
3146
+ previousHash: z.union([HASH, z.null()]),
3147
+ event: EventSchema,
3148
+ entryHash: HASH
3149
+ }).strict();
3150
+ /** Validate and return a canonical copy. Arrays whose order is not semantic are
3151
+ * sorted so retries from different processes produce byte-identical events. */
3152
+ function validateSearchLedgerEvent(input) {
3153
+ const parsed = EventSchema.safeParse(input);
3154
+ if (!parsed.success) throw new SearchLedgerError(`invalid search ledger event: ${formatZodError(parsed.error)}`);
3155
+ return normalizeEvent(parsed.data);
3156
+ }
3157
+ /** Open a durable filesystem search ledger. Construction performs no I/O; the
3158
+ * first `append` or `replay` validates the complete existing file. */
3159
+ function openSearchLedger(options) {
3160
+ if (options.path.trim().length === 0) throw new SearchLedgerError("ledger path is empty");
3161
+ return new FileSearchLedger(options.path, options.campaignId, options.trustedHead);
3162
+ }
3163
+ /** Replay immutable search-ledger JSONL through the same codec as FileSearchLedger. */
3164
+ function replaySearchLedgerText(text, campaignId, source) {
3165
+ return replayLedgerText(text, source, searchLedgerCodec(campaignId)).projection;
3166
+ }
3167
+ function searchLedgerCodec(campaignId) {
3168
+ return {
3169
+ ...SEARCH_LEDGER_FILE_CONTEXT,
3170
+ header: {
3171
+ schema: SEARCH_LEDGER_SCHEMA,
3172
+ campaignId
3173
+ },
3174
+ conflictError: (message) => new SearchLedgerConflictError(message),
3175
+ parseEntry: parseSearchLedgerEntry,
3176
+ checkEntryHeader: (entry, index) => {
3177
+ if (entry.campaignId !== campaignId) throw new SearchLedgerIntegrityError(`entry ${index} belongs to campaign ${entry.campaignId}, expected ${campaignId}`);
3178
+ },
3179
+ createProjector: () => createSearchLedgerProjector(campaignId)
3067
3180
  };
3068
3181
  }
3069
- /** Every candidate slot must name a planned candidate-generation operation,
3070
- * whether it arrives with the plan or with a later extension. */
3071
- function assertSlotGenerationOperation(slot, plannedOperations) {
3072
- if (plannedOperations.get(slot.generationOperationId)?.kind !== "candidate-generation") throw new SearchLedgerIntegrityError(`candidate slot ${slot.slotId} references unplanned candidate-generation operation ${slot.generationOperationId}`);
3073
- }
3074
- function plannedTaskOutcomeKeys(planEvent, candidates) {
3075
- const missing = [];
3076
- const registeredCandidates = [...candidates.values()].sort((a, b) => compareStrings(a.registered.slotId, b.registered.slotId));
3077
- for (const candidate of registeredCandidates) {
3078
- const slotId = candidate.registered.slotId;
3079
- for (const task of planEvent.plan.tasks) if (!candidate.attempts.some((attempt) => attempt.task.taskId === task.taskId && attempt.outcome.status !== "errored")) missing.push(`${slotId}/${task.taskId}`);
3182
+ /** Append-only file-backed search ledger with idempotent writes and replay. */
3183
+ var FileSearchLedger = class {
3184
+ path;
3185
+ campaignId;
3186
+ trustedHeadPath;
3187
+ trustedHeadMode;
3188
+ journal;
3189
+ constructor(path, campaignId, trustedHead = "pin") {
3190
+ if (path.trim().length === 0) throw new SearchLedgerError("ledger path is empty");
3191
+ if (campaignId.length === 0) throw new SearchLedgerError("campaignId is empty");
3192
+ if (campaignId.trim() !== campaignId) throw new SearchLedgerError("campaignId must not contain surrounding whitespace");
3193
+ this.campaignId = campaignId;
3194
+ this.trustedHeadMode = trustedHead;
3195
+ this.journal = new FileLedgerJournal(path, searchLedgerCodec(campaignId), { requireTrustedHead: trustedHead === "require" });
3196
+ this.path = this.journal.path;
3197
+ this.trustedHeadPath = this.journal.trustedHeadPath;
3080
3198
  }
3081
- return missing;
3199
+ async replay() {
3200
+ return (await this.journal.replay()).projection;
3201
+ }
3202
+ async append(input) {
3203
+ const event = validateSearchLedgerEvent(input);
3204
+ const { entry, appended, projection } = await this.journal.append(event, { pinHead: this.trustedHeadMode !== "off" });
3205
+ return {
3206
+ entry,
3207
+ appended,
3208
+ replay: projection
3209
+ };
3210
+ }
3211
+ async trustedHead() {
3212
+ return this.journal.trustedHead();
3213
+ }
3214
+ async pinTrustedHead() {
3215
+ return this.journal.pinTrustedHead();
3216
+ }
3217
+ async clearTrustedHead() {
3218
+ return this.journal.clearTrustedHead();
3219
+ }
3220
+ };
3221
+ function parseSearchLedgerEntry(raw, context) {
3222
+ const parsed = EntrySchema.safeParse(raw);
3223
+ if (!parsed.success) throw new SearchLedgerIntegrityError(`search ledger ${context.path} has a malformed entry at line ${context.line}: ${formatZodError(parsed.error)}`);
3224
+ const entry = parsed.data;
3225
+ if (canonicalString(validateSearchLedgerEvent(entry.event)) !== canonicalString(entry.event)) throw new SearchLedgerIntegrityError(`search ledger ${context.path} has non-canonical event ordering at line ${context.line}`);
3226
+ return entry;
3082
3227
  }
3083
3228
  function normalizeEvent(event) {
3084
3229
  const artifacts = sortArtifacts(event.artifacts);
@@ -3127,18 +3272,9 @@ function normalizeEvent(event) {
3127
3272
  function sortArtifacts(artifacts) {
3128
3273
  return [...artifacts].map((artifact) => ({ ...artifact })).sort((a, b) => compareStrings(artifactKey(a), artifactKey(b)));
3129
3274
  }
3130
- function artifactKey(artifact) {
3131
- return canonicalString(artifact);
3132
- }
3133
- function compareStrings(a, b) {
3134
- return a < b ? -1 : a > b ? 1 : 0;
3135
- }
3136
3275
  function sortedStrings(values) {
3137
3276
  return [...values].sort();
3138
3277
  }
3139
- function assertUnique(values, label, eventId) {
3140
- if (new Set(values).size !== values.length) throw new SearchLedgerIntegrityError(`event ${eventId} contains duplicate ${label} values`);
3141
- }
3142
3278
  function formatZodError(error) {
3143
3279
  return error.issues.map((issue) => `${issue.path.length > 0 ? issue.path.join(".") : "<root>"}: ${issue.message}`).join("; ");
3144
3280
  }
@@ -4622,6 +4758,16 @@ const DEFAULT_DISPATCH_TIMEOUT_MS = 6e5;
4622
4758
  * Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
4623
4759
  */
4624
4760
  async function runImprovementLoop(opts) {
4761
+ opts = {
4762
+ ...opts,
4763
+ judges: opts.judges?.map(captureJudge),
4764
+ ...opts.claim || opts.finalEvidence ? {
4765
+ claim: opts.claim && defineEvaluationClaim(opts.claim),
4766
+ finalEvidence: opts.finalEvidence && captureFinalEvidencePolicy(opts.finalEvidence),
4767
+ scenarios: structuredClone(opts.scenarios),
4768
+ holdoutScenarios: structuredClone(opts.holdoutScenarios)
4769
+ } : {}
4770
+ };
4625
4771
  if (opts.autoOnPromote === "config") throw new Error("runImprovementLoop: autoOnPromote='config' requires isolated deployment, rollback, and independent validation. Use 'pr' or 'none'.");
4626
4772
  if (opts.tracing === "off" && opts.proposer) throw new Error("runImprovementLoop: tracing='off' is forbidden when a proposer is wired. The improvement loop without traces is unattributable; candidate surfaces cannot be cited back to spans and the optimization dataset goes unfed.");
4627
4773
  if (opts.autoOnPromote === "pr" && (!opts.ghOwner || !opts.ghRepo)) throw new Error("runImprovementLoop: autoOnPromote='pr' requires ghOwner + ghRepo.");
@@ -4637,6 +4783,11 @@ async function runImprovementLoop(opts) {
4637
4783
  costCeilingUsd: opts.costCeiling
4638
4784
  });
4639
4785
  const dispatchTimeoutMs = opts.dispatchTimeoutMs ?? DEFAULT_DISPATCH_TIMEOUT_MS;
4786
+ if (opts.claim?.generalization === "new-units") assertIndependentEvaluationSplit(opts.claim, opts.holdoutScenarios, opts.scenarios);
4787
+ if (opts.finalEvidence) {
4788
+ if (opts.holdout === "deferred") throw new Error("final evidence requires a measured comparison");
4789
+ await reserveFinalEvidence(opts.finalEvidence, opts.claim, opts.holdoutScenarios, opts.scenarios);
4790
+ }
4640
4791
  const optimization = await runOptimization({
4641
4792
  ...opts,
4642
4793
  dispatchTimeoutMs,
@@ -4960,14 +5111,7 @@ async function executeOptimizationMethod(options) {
4960
5111
  baselineSurface: structuredClone(input.baselineSurface),
4961
5112
  trainScenarios: cloneScenarios(input.trainScenarios),
4962
5113
  selectionScenarios: cloneScenarios(input.selectionScenarios),
4963
- judges: Object.freeze(input.judges.map((judge) => {
4964
- const dimensions = judge.dimensions.map((dimension) => Object.freeze({ ...dimension }));
4965
- Object.freeze(dimensions);
4966
- return Object.freeze({
4967
- ...judge,
4968
- dimensions
4969
- });
4970
- })),
5114
+ judges: Object.freeze(input.judges.map(captureJudge)),
4971
5115
  runOptions: Object.freeze({ ...input.runOptions }),
4972
5116
  costLedger: costScope.ledger
4973
5117
  });
@@ -5114,6 +5258,16 @@ function assertOptimizationProvenance(methodName, value) {
5114
5258
  * Compare complete optimization methods on disjoint train, selection, and final test data.
5115
5259
  */
5116
5260
  async function compareOptimizationMethods(opts) {
5261
+ opts = {
5262
+ ...opts,
5263
+ methods: opts.methods.map((method) => ({ ...method })),
5264
+ trainScenarios: structuredClone(opts.trainScenarios),
5265
+ selectionScenarios: structuredClone(opts.selectionScenarios),
5266
+ testScenarios: structuredClone(opts.testScenarios),
5267
+ judges: opts.judges.map(captureJudge),
5268
+ claim: opts.claim && defineEvaluationClaim(opts.claim),
5269
+ finalEvidence: opts.finalEvidence && captureFinalEvidencePolicy(opts.finalEvidence)
5270
+ };
5117
5271
  assertOptimizationMethods(opts.methods);
5118
5272
  assertComparisonPartitions(opts);
5119
5273
  const searchHistoryPolicy = opts.searchHistoryPolicy ?? "allow-missing";
@@ -5160,11 +5314,38 @@ async function compareOptimizationMethods(opts) {
5160
5314
  return byScenario;
5161
5315
  };
5162
5316
  const scenarioIds = opts.testScenarios.map((s) => s.id).sort();
5317
+ const unitByScenario = opts.claim ? evaluationUnitMap(opts.claim, opts.testScenarios) : new Map(scenarioIds.map((id) => [id, id]));
5318
+ const units = opts.claim ? summarizeEvaluationUnits(opts.claim, opts.testScenarios) : {
5319
+ observations: scenarioIds.length,
5320
+ independentUnits: scenarioIds.length,
5321
+ units: scenarioIds.map((id) => ({
5322
+ id,
5323
+ observations: 1
5324
+ }))
5325
+ };
5326
+ const aggregateUnits = (arr) => {
5327
+ const values = /* @__PURE__ */ new Map();
5328
+ scenarioIds.forEach((id, index) => {
5329
+ const unitId = unitByScenario.get(id);
5330
+ const bucket = values.get(unitId) ?? [];
5331
+ bucket.push(arr[index]);
5332
+ values.set(unitId, bucket);
5333
+ });
5334
+ return units.units.map((unit) => mean(values.get(unit.id)));
5335
+ };
5336
+ const decisionOptions = {
5337
+ seed,
5338
+ resamples,
5339
+ confidence: intervalConfidence,
5340
+ threshold: opts.claim?.minimumEffect ?? 0
5341
+ };
5163
5342
  const align = (byScenario, label) => {
5164
5343
  const missing = scenarioIds.filter((id) => !(id in byScenario));
5165
5344
  if (missing.length > 0) throw new Error(`compareOptimizationMethods: ${label} produced no test score for scenario(s) [${missing.join(", ")}]. A cell failed or its judges returned nothing. Fix the dispatch or judge; the comparison will not replace missing scores with zero.`);
5166
5345
  return scenarioIds.map((id) => byScenario[id]);
5167
5346
  };
5347
+ if (opts.claim?.generalization === "new-units") assertIndependentEvaluationSplit(opts.claim, opts.testScenarios, [...opts.trainScenarios, ...opts.selectionScenarios]);
5348
+ if (opts.finalEvidence) await reserveFinalEvidence(opts.finalEvidence, opts.claim, opts.testScenarios, [...opts.trainScenarios, ...opts.selectionScenarios]);
5168
5349
  const optimizationOwner = new AbortController();
5169
5350
  const optimized = await mapConcurrent(opts.methods, optimizationConcurrency, async (method) => {
5170
5351
  try {
@@ -5195,6 +5376,7 @@ async function compareOptimizationMethods(opts) {
5195
5376
  cost: result.cost
5196
5377
  }))).totalCostUsd, opts.costCeiling, "optimization");
5197
5378
  const testCostPhase = finalCostPhase(opts, baselineSurface, optimized, seed);
5379
+ const finalEvidence = opts.finalEvidence ? await exposeFinalEvidence(opts.finalEvidence, opts.claim, opts.testScenarios, [baselineSurface, ...optimized.map((method) => method.winnerSurface)]) : void 0;
5198
5380
  const baselineArr = align(await scoreOnTest(baselineSurface, "test/baseline", testCostPhase), "baseline");
5199
5381
  const testScoresBySurface = /* @__PURE__ */ new Map([[surfaceContentHash(baselineSurface), baselineArr]]);
5200
5382
  const winners = [];
@@ -5211,21 +5393,19 @@ async function compareOptimizationMethods(opts) {
5211
5393
  });
5212
5394
  }
5213
5395
  const scores = winners.map((w) => {
5214
- const boot = pairedBootstrap(baselineArr, w.arr, {
5215
- seed,
5216
- resamples,
5217
- confidence: intervalConfidence,
5218
- statistic: "mean"
5219
- });
5396
+ const baselineUnits = aggregateUnits(baselineArr);
5397
+ const winnerUnits = aggregateUnits(w.arr);
5398
+ const decision = decidePairedPromotion(baselineUnits, winnerUnits, decisionOptions);
5220
5399
  const score = {
5221
5400
  name: w.name,
5222
- baselineComposite: mean(baselineArr),
5223
- winnerComposite: mean(w.arr),
5224
- lift: boot.mean,
5401
+ baselineComposite: mean(baselineUnits),
5402
+ winnerComposite: mean(winnerUnits),
5403
+ lift: decision.delta,
5225
5404
  liftCi: {
5226
- low: boot.low,
5227
- high: boot.high
5405
+ low: decision.low,
5406
+ high: decision.high
5228
5407
  },
5408
+ decision,
5229
5409
  optimizationCost: w.cost,
5230
5410
  scenarioScores: scenarioIds.map((scenarioId, index) => ({
5231
5411
  scenarioId,
@@ -5233,6 +5413,13 @@ async function compareOptimizationMethods(opts) {
5233
5413
  winnerComposite: w.arr[index],
5234
5414
  lift: w.arr[index] - baselineArr[index]
5235
5415
  })),
5416
+ unitScores: units.units.map((unit, index) => ({
5417
+ unitId: unit.id,
5418
+ scenarios: unit.observations,
5419
+ baselineComposite: baselineUnits[index],
5420
+ winnerComposite: winnerUnits[index],
5421
+ lift: winnerUnits[index] - baselineUnits[index]
5422
+ })),
5236
5423
  winnerSurface: structuredClone(w.winnerSurface),
5237
5424
  rank: 0
5238
5425
  };
@@ -5266,23 +5453,20 @@ async function compareOptimizationMethods(opts) {
5266
5453
  });
5267
5454
  const best = scores[0];
5268
5455
  const byName = new Map(winners.map((w) => [w.name, w]));
5269
- const bestArr = byName.get(best.name).arr;
5456
+ const bestArr = aggregateUnits(byName.get(best.name).arr);
5270
5457
  const pairwise = scores.slice(1).map((other) => {
5271
- const otherArr = byName.get(other.name).arr;
5272
- const boot = pairedBootstrap(otherArr, bestArr, {
5273
- seed,
5274
- resamples,
5275
- confidence: intervalConfidence,
5276
- statistic: "mean"
5277
- });
5278
- const favored = !Number.isFinite(boot.low) || !Number.isFinite(boot.high) || boot.low === boot.high ? "tie" : boot.low > 0 ? best.name : boot.high < 0 ? other.name : "tie";
5458
+ const otherArr = aggregateUnits(byName.get(other.name).arr);
5459
+ const decision = decidePairedPromotion(otherArr, bestArr, decisionOptions);
5460
+ const reverse = decidePairedPromotion(bestArr, otherArr, decisionOptions);
5461
+ const favored = decision.promote ? best.name : reverse.promote ? other.name : null;
5279
5462
  return {
5280
5463
  a: best.name,
5281
5464
  b: other.name,
5282
- deltaMean: boot.mean,
5283
- low: boot.low,
5284
- high: boot.high,
5285
- favored
5465
+ deltaMean: decision.delta,
5466
+ low: decision.low,
5467
+ high: decision.high,
5468
+ favored,
5469
+ decision
5286
5470
  };
5287
5471
  });
5288
5472
  const optimizationCost = combineComparisonCosts(scores.map((score) => ({
@@ -5303,6 +5487,11 @@ async function compareOptimizationMethods(opts) {
5303
5487
  best,
5304
5488
  pairwise,
5305
5489
  testScenarioIds: scenarioIds,
5490
+ units,
5491
+ observationUnit: opts.claim ? "registered" : "scenario",
5492
+ pairedCellN: scenarioIds.length * (opts.reps ?? 1),
5493
+ ...opts.claim ? { claim: opts.claim } : {},
5494
+ ...finalEvidence ? { finalEvidence } : {},
5306
5495
  optimizationCost,
5307
5496
  testCost,
5308
5497
  totalCost,
@@ -5963,6 +6152,6 @@ function firstString(value) {
5963
6152
  return typeof value === "string" && value.trim() ? value : void 0;
5964
6153
  }
5965
6154
  //#endregion
5966
- export { openSearchLedger as A, redTeamDataset as B, assertSearchHistoryAdmissionOptions as C, verifySearchHistoryArtifact as D, searchHistoryCoverageRow as E, paretoFrontierWithCrowding as F, runCampaign as G, scoreRedTeamOutput as H, runFinalComparison as I, tangleTracesRoot as J, planCampaignRun as K, openAutoPr as L, validateSearchLedgerEvent as M, dominates as N, verifySearchHistoryReceipt as O, paretoFrontier as P, computeManifestHash as Q, defaultProductionGate as R, assertCompleteSearchHistory as S, createSearchHistoryReceipt as T, runCanaries as U, redTeamReport as V, runEval as W, buildCellSchedule as X, readCachedCell as Y, cellCachePath as Z, isProposedCandidate as _, transientDispatchFailure as a, recordCandidatePopulationSearch as b, compareOptimizationMethods as c, combineComparisonCosts as d, costFromLedgerSummary as f, runOptimization as g, runImprovementLoop as h, quotaExhaustedUntil as i, replaySearchLedgerText as j, FileSearchLedger as k, optimizationTokenUsageFromSummary as l, readGepaCandidatePopulationArtifact as m, JudgeParseError as n, aggregateRunScore as o, assertGepaCandidatePopulationSummary as p, resolveRunDir as q, isTransientTransportFailure as r, clamp01 as s, llmJudge as t, executeOptimizationMethod as u, labelTrustRank as v, assertSearchHistoryMatchesReplay as w, SearchHistoryRequiredError as x, SearchRecorder as y, DEFAULT_RED_TEAM_CORPUS as z };
6155
+ export { tangleTracesRoot as $, openSearchLedger as A, DEFAULT_RED_TEAM_CORPUS as B, assertSearchHistoryAdmissionOptions as C, verifySearchHistoryArtifact as D, searchHistoryCoverageRow as E, paretoFrontierWithCrowding as F, assertIndependentEvaluationSplit as G, redTeamReport as H, runFinalComparison as I, reserveFinalEvidence as J, captureFinalEvidencePolicy as K, openAutoPr as L, validateSearchLedgerEvent as M, dominates as N, verifySearchHistoryReceipt as O, paretoFrontier as P, resolveRunDir as Q, captureJudge as R, assertCompleteSearchHistory as S, createSearchHistoryReceipt as T, scoreRedTeamOutput as U, redTeamDataset as V, runCanaries as W, runCampaign as X, runEval as Y, planCampaignRun as Z, isProposedCandidate as _, transientDispatchFailure as a, recordCandidatePopulationSearch as b, compareOptimizationMethods as c, combineComparisonCosts as d, readCachedCell as et, costFromLedgerSummary as f, runOptimization as g, runImprovementLoop as h, quotaExhaustedUntil as i, replaySearchLedgerText as j, FileSearchLedger as k, optimizationTokenUsageFromSummary as l, readGepaCandidatePopulationArtifact as m, JudgeParseError as n, cellCachePath as nt, aggregateRunScore as o, assertGepaCandidatePopulationSummary as p, evaluationUnitMap as q, isTransientTransportFailure as r, computeManifestHash as rt, clamp01 as s, llmJudge as t, buildCellSchedule as tt, executeOptimizationMethod as u, labelTrustRank as v, assertSearchHistoryMatchesReplay as w, SearchHistoryRequiredError as x, SearchRecorder as y, defaultProductionGate as z };
5967
6156
 
5968
- //# sourceMappingURL=llm-judge-v80Kmu9g.js.map
6157
+ //# sourceMappingURL=llm-judge-DEFZeSiu.js.map