@tangle-network/agent-eval 0.180.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (221) hide show
  1. package/CHANGELOG.md +60 -0
  2. package/README.md +119 -159
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +3 -3
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +4 -7
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +10 -9
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  38. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  39. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  40. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  41. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  42. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  43. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  44. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  45. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  46. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  47. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  48. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  49. package/dist/experiment/index.d.ts +27 -477
  50. package/dist/experiment/index.d.ts.map +1 -1
  51. package/dist/experiment/index.js +95 -559
  52. package/dist/experiment/index.js.map +1 -1
  53. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  54. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  55. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  56. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  57. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  58. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  59. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  60. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  61. package/dist/hosted/index.d.ts +2 -2
  62. package/dist/hosted/index.d.ts.map +1 -1
  63. package/dist/hosted/index.js +1 -1
  64. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  65. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  66. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  67. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  68. package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
  69. package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  70. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  71. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  72. package/dist/index.d.ts +28 -28
  73. package/dist/index.js +24 -15
  74. package/dist/index.js.map +1 -1
  75. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  76. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  77. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  78. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  79. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  80. package/dist/journal-Cs9f7385.js.map +1 -0
  81. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  82. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  83. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  84. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  85. package/dist/ledger-core/index.d.ts +1 -1
  86. package/dist/ledger-core/index.js +1 -1
  87. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  88. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  89. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  90. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  91. package/dist/meta-eval/index.d.ts +138 -7
  92. package/dist/meta-eval/index.d.ts.map +1 -1
  93. package/dist/meta-eval/index.js +245 -97
  94. package/dist/meta-eval/index.js.map +1 -1
  95. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  96. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  97. package/dist/multishot/golden/index.d.ts +1 -1
  98. package/dist/multishot/index.d.ts +2 -2
  99. package/dist/openapi.json +1 -1
  100. package/dist/outcome-store-BXlkwMPR.js +131 -0
  101. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  102. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  103. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  104. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  105. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  106. package/dist/pipelines/index.js +1 -1
  107. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  108. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  109. package/dist/profile-cell.d.ts +1 -1
  110. package/dist/profile-cell.js +1 -1
  111. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  112. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  113. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  114. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  115. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  116. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  117. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  118. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  119. package/dist/reporting.d.ts +4 -4
  120. package/dist/reporting.js +3 -3
  121. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  122. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  123. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  124. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  125. package/dist/rl.d.ts +53 -99
  126. package/dist/rl.d.ts.map +1 -1
  127. package/dist/rl.js +182 -169
  128. package/dist/rl.js.map +1 -1
  129. package/dist/rollout/index.d.ts +1 -1
  130. package/dist/rollout/index.js +2 -2
  131. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  132. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  133. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  134. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  135. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  136. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  137. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  138. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  139. package/dist/run-record-Br-Yzt_k.js +464 -0
  140. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  141. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  142. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  143. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  144. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  145. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  146. package/dist/sequential-DAsyV2T9.js.map +1 -0
  147. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  148. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  149. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  150. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  151. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  152. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  153. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  154. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  155. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  156. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  157. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  158. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  159. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  160. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  161. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  162. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  163. package/dist/trace-repair/index.d.ts +2 -2
  164. package/dist/traces.d.ts +6 -6
  165. package/dist/traces.js +1 -1
  166. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  167. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  168. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  169. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  170. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  171. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  172. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  173. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  174. package/dist/wire/index.d.ts +2 -2
  175. package/docs/adapters-observability.md +14 -0
  176. package/docs/campaign-proposers.md +86 -128
  177. package/docs/charter.md +108 -112
  178. package/docs/concepts.md +157 -69
  179. package/docs/design/mlbenchmarks-book-review.md +440 -0
  180. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  181. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  182. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  183. package/docs/design/self-improvement-evidence-audit.md +263 -0
  184. package/docs/design.md +2 -1
  185. package/docs/eval-surface-map.md +95 -42
  186. package/docs/evaluation-integrity.md +220 -0
  187. package/docs/experiment.md +111 -55
  188. package/docs/feature-guide.md +5 -6
  189. package/docs/hosted-ingest-spec.md +4 -11
  190. package/docs/insight-report.md +187 -455
  191. package/docs/outcome-validity.md +182 -0
  192. package/docs/product-eval-adoption.md +1 -2
  193. package/docs/research-report-methodology.md +7 -7
  194. package/docs/search-history-receipts.md +8 -0
  195. package/docs/statistical-evidence.md +129 -0
  196. package/docs/verdicts.md +76 -49
  197. package/package.json +1 -1
  198. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  199. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  200. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  201. package/dist/client-BlLY6o2w.js.map +0 -1
  202. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  203. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  204. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  205. package/dist/index-BxWvILU8.d.ts.map +0 -1
  206. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  207. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  208. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  209. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  210. package/dist/outcome-store-ChBKlTd_.js +0 -75
  211. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  212. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  213. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  214. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  215. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  216. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  217. package/dist/run-record-CR63CpHK.js +0 -216
  218. package/dist/run-record-CR63CpHK.js.map +0 -1
  219. package/dist/sequential-B5gXgcyp.js.map +0 -1
  220. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  221. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
@@ -0,0 +1,220 @@
1
+ # Evaluation claims and automated improvement
2
+
3
+ Use reusable evaluations to guide development and select candidates.
4
+ Declare stronger evidence requirements when a result must support a broader claim.
5
+ `selfImprove()` returns its selected surface and measured lift even when its release gate remains inconclusive.
6
+
7
+ The package separates three questions:
8
+
9
+ | Question | Evidence | Public entry |
10
+ | --- | --- | --- |
11
+ | Did this change help on these cases? | Paired scores, failures, cost, and case coverage. | `defineAgentEval()` or `selfImprove()` from `/contract`. |
12
+ | Does the improvement extend to new tasks? | Representative independent tasks and an appropriate comparison; a useful effect when making an improvement decision. | Optional `claim` on campaign comparisons; registered rules from `/experiment`. |
13
+ | Can fresh final evidence support this adaptive decision? | A frozen comparison, retained access boundaries, and a durable exposure record. | Optional `finalEvidence` on the same comparison. |
14
+
15
+ These controls reuse the existing execution path, paired estimators, sealed experiments, and locked journal.
16
+ They do not add another optimizer or agent runner.
17
+
18
+ ## Declare the comparison without consuming data
19
+
20
+ ```ts
21
+ import { defineEvaluationClaim } from '@tangle-network/agent-eval/experiment'
22
+
23
+ const claim = defineEvaluationClaim({
24
+ use: 'comparison',
25
+ population: {
26
+ id: 'support-incidents',
27
+ description: 'Support incidents from the deployed product',
28
+ },
29
+ samplingFrame: 'A random sample of incidents from the declared collection window',
30
+ independentUnit: 'source.incidentId',
31
+ generalization: 'new-units',
32
+ minimumEffect: 0.05,
33
+ })
34
+ ```
35
+
36
+ Pass `claim` to `selfImprove()`, `runImprovementLoop()`, or `compareOptimizationMethods()`.
37
+ It remains independent of final-evidence storage.
38
+ `minimumEffect` is optional because development reports and absolute-rate measurements need not test an improvement threshold.
39
+ When omitted, the default `selfImprove()` gate uses a 0.05 gain threshold; method comparison uses 0.
40
+ Custom gates choose their own thresholds.
41
+
42
+ `independentUnit` names a field path in each scenario or evidence row.
43
+ Variants from one incident must carry the same source identity.
44
+ For new-unit claims, automatic `selfImprove()` partitions keep those variants together.
45
+ Fixed-roster claims retain source units in reports without requiring disjoint sources between development partitions.
46
+ Explicit partitions for new-unit claims must not share source units between development and final evaluation.
47
+
48
+ The default self-improvement gate averages paired cells within registered units.
49
+ Method comparison first averages repetitions within each scenario, then averages scenarios within each source unit.
50
+ It weights source units equally.
51
+ Results retain `scenarioScores`, `unitScores`, `units`, and `pairedCellN` so callers can inspect each denominator.
52
+ Custom improvement gates receive raw cell scores and scenarios.
53
+ Pass the claim into your gate's configuration and apply its grouping and decision rules there.
54
+
55
+ `fixed-roster` describes the specified cases.
56
+ Repeated executions can measure execution variability on that roster.
57
+ They do not establish task diversity or performance on unseen users.
58
+ Claim metadata records the intended scope; it does not authenticate sampling or turn an exploratory result into certification.
59
+
60
+ ## Interpret small and inconclusive results
61
+
62
+ There is no universal task count that proves an improvement.
63
+ The effect, outcome type, dependence, confidence level, and decision procedure determine what the evidence supports.
64
+
65
+ Paired binary decisions use the shared score interval.
66
+ At nonnegative gain thresholds, the exact discordance check can also veto promotion.
67
+ Negative thresholds ask whether a regression stays within a tolerance and do not use that veto.
68
+ A sufficiently large binary gain can pass with fewer than 20 independent pairs.
69
+ Continuous mean decisions require the existing bootstrap path's 20-pair eligibility threshold.
70
+ That implementation threshold does not establish adequate power or guarantee interval coverage for every distribution.
71
+ The smaller-sample sign test answers a different question about directional or median change.
72
+
73
+ Method rankings describe observed lift.
74
+ `favored: null` means the evidence does not establish a favored method; it does not establish equivalence.
75
+ Each score and pairwise contrast retains its full `decision`, including the estimator, threshold, minimum, and sufficiency.
76
+ An inconclusive gate leaves the selected candidate available for further development or a narrower evaluation.
77
+
78
+ For a design that matches its outcome model, `clusteredPower()` simulates power at the declared `minimumEffect`.
79
+ A registered `power-floor` gate checks a supplied power curve against its minimum effect and target power.
80
+ These optional design checks do not run automatically when you pass a `claim`.
81
+ High power at a much larger effect cannot substitute for power at the improvement that matters.
82
+
83
+ ## Opt into fresh final evidence
84
+
85
+ ```ts
86
+ import { openFinalEvidenceLedger } from '@tangle-network/agent-eval/experiment'
87
+
88
+ const finalEvidence = {
89
+ ledger: openFinalEvidenceLedger({ path: '.agent-eval/final-evidence.jsonl' }),
90
+ requestId: 'support-comparison-2026-09-13',
91
+ evaluatorDigest, // Content identity of the actual evaluator and its configuration.
92
+ }
93
+
94
+ // Pass both claim and finalEvidence to the existing comparison entrypoint.
95
+ ```
96
+
97
+ Ordinary regression and development evaluations remain reusable.
98
+ Supply this policy when freshness is part of the evidence supporting a particular final comparison.
99
+ It requires a comparison or certification claim and measured final execution.
100
+
101
+ Comparison entrypoints capture judge configuration and callbacks before search or final exposure.
102
+ The host must keep callback receiver and closed-over state stable throughout the comparison.
103
+
104
+ The campaign reserves source units before candidate search.
105
+ It records exposure before dispatching the baseline, selected candidates, or an optional neutralized control.
106
+ Failed or interrupted final execution still consumes that evidence.
107
+ An already exposed request cannot start another final measurement.
108
+
109
+ The ledger permits exact retries of reservation and exposure writes.
110
+ Campaign entrypoints refuse a replayed exposure so competing workers cannot each start a new measurement.
111
+ Read retained campaign artifacts after exposure; a fresh request ID does not restore freshness.
112
+
113
+ Source unit identities are unique across the shared ledger, including across population labels.
114
+ The ledger also rejects the same dataset digest under another request.
115
+ Use one persistent ledger for related decisions and preserve stable source identities.
116
+ Opening another empty ledger or inventing new lineage identities cannot establish independent evidence.
117
+
118
+ `FinalEvidenceLedger` returns typed outcomes.
119
+ Inspect `succeeded` before reading `value`.
120
+ Failures distinguish conflicting use, invalid input, and unavailable or damaged storage.
121
+ Campaign errors preserve these categories through `FinalEvidenceError.kind`.
122
+
123
+ The filesystem implementation uses the existing hash-chained journal, process locks, durable writes, and required trusted head.
124
+ Preserve both the journal and its `.head` file.
125
+ The head detects truncation while it remains trusted.
126
+ An actor who can replace both files can replace the recorded history.
127
+
128
+ The host owns answer-file permissions, model context, credentials, and author/evaluator separation.
129
+ The ledger records exposure; it cannot prove that earlier undisclosed access never occurred.
130
+
131
+ ## Seal the rule and measured field
132
+
133
+ Attach the same `claim` to `defineExperiment()` before calling `sealExperiment()`.
134
+ Cluster intervals register both the source unit and measured field:
135
+
136
+ ```ts
137
+ const interval = {
138
+ kind: 'cluster-bootstrap' as const,
139
+ clusterBy: 'source.incidentId',
140
+ value: 'pairedDelta',
141
+ resamples: 2000,
142
+ seed: 7,
143
+ level: 0.95,
144
+ method: 'percentile' as const,
145
+ }
146
+
147
+ // Register interval in the experiment's intervals map before sealing.
148
+ // Then execute registered.interval('lift', { kind: 'rows', rows }).
149
+ ```
150
+
151
+ For new-unit claims, each cluster interval's `clusterBy` must equal `claim.independentUnit`.
152
+ Registered binomial intervals require one unique `unitId` per trial for new-unit claims.
153
+ Opened seals capture validated rules before asynchronous execution.
154
+ Caller mutation cannot change the opened experiment's rules.
155
+
156
+ Only current canonical digest schemes can execute.
157
+ Retain historical artifacts with their original identities; re-register current work under the supported format.
158
+ See [registered experiments](./experiment.md) for the complete rule language.
159
+
160
+ ## Audit an evaluator's errors
161
+
162
+ Use `auditEvaluator()` from `/meta-eval` when admitting a new checker or model judge.
163
+ Provide actual judgments of independently verified good and bad controls.
164
+ Each observation names its source unit, evidence reference, expected decision, observed decision, and development exposure.
165
+ This audit is separate from `selfImprove()`; the caller decides whether to require admission before search or release.
166
+
167
+ The audit measures false acceptance and false rejection separately.
168
+ A source unit fails a class when any variant in that class is misjudged.
169
+ Repeated variants increase case coverage without increasing the independent-unit count.
170
+ If a source appeared during evaluator development, every supplied variant from that source is excluded from fresh audit evidence.
171
+
172
+ The audit uses exact binomial bounds and adjusts the two intervals for simultaneous confidence.
173
+ Unknown judgments remain visible and contribute their most adverse possible outcomes to each upper bound.
174
+ Admission is possible when both worst-case upper bounds meet policy.
175
+ If unresolved outcomes affect the measured error rate, that rate is `null`.
176
+ An always-accept checker fails false acceptance; an always-reject checker fails false rejection.
177
+
178
+ Reports retain inputs, source coverage, exclusions, unknowns, limits, and content digests.
179
+ The declared audit authority must differ from the evaluator author.
180
+ Different identifiers alone do not prove independence; the host must enforce and record separation.
181
+ Audit cases must represent the stated population and a consistent control-generation procedure.
182
+ Changing the number or kind of variants changes the meaning of an any-variant error rate.
183
+
184
+ `auditEvaluator()` measures supplied judgments.
185
+ It does not execute models or automatically approve a deployment.
186
+ For outcome associations and direct score calibration, use [the outcome-validity tools](./outcome-validity.md).
187
+
188
+ ## Test whether self-improvement is useful
189
+
190
+ These integrity checks establish execution and measurement behavior.
191
+ They do not establish that a particular optimizer improves agents across domains.
192
+ The [historical evidence audit](./design/self-improvement-evidence-audit.md) records prior gains, nulls, regressions, and their limits.
193
+
194
+ For a benefit experiment, define the user behavior and useful effect before search.
195
+ Compare the starting agent, a direct edit or simple search baseline, and the proposed improvement method at equal actual resources.
196
+ Give every method the same allowed preparation, feedback, tools, and candidate surface.
197
+ Retain all attempts, costs, failures, selected candidates, and final comparisons.
198
+
199
+ Before interpreting a null result, verify that candidate generation executed and the evaluator distinguishes plausible improvements from regressions.
200
+ Measure improvement under the conditions where the method claims an advantage.
201
+ Use fresh tasks when the conclusion concerns unseen tasks.
202
+ Treat a result on a fixed product workflow as evidence for that workflow.
203
+ Repeat across distinct domains before making a broad claim.
204
+
205
+ The [offline example](../examples/evaluation-integrity/) exercises these public APIs and exports its report without paid calls.
206
+ It verifies the integration with deterministic fixtures; it is not an optimizer-benefit study.
207
+
208
+ ## Source and design rationale
209
+
210
+ The book motivates the distinctions; the API and policies are project design choices.
211
+
212
+ | Source | Applied idea |
213
+ | --- | --- |
214
+ | [Chapter 4: purposes of holdout](https://mlbenchmarks.org/04-holdout-method.html#whats-the-holdout-method-for) | Development feedback, selection, and capability measurement require different evidence. |
215
+ | [Chapter 3: detecting differences](https://mlbenchmarks.org/03-detecting-differences.html#comparing-similar-models) | Pair comparisons and evaluate precision against the actual effect and independent observations. |
216
+ | [Chapter 5: test-set reuse](https://mlbenchmarks.org/05-test-set-reuse.html) | Preserve development feedback while tracking adaptive final-data exposure. |
217
+ | [Chapter 11: confounded evaluations](https://mlbenchmarks.org/11-evaluating-language-models.html#confounded-evaluations) | Give methods comparable preparation before judging their adaptation potential. |
218
+ | [Chapter 14: judge agreement](https://mlbenchmarks.org/14-evaluation-frontier.html#agreement-alone-is-not-enough) | Measure consequential evaluator errors; agreement alone does not establish correct rankings. |
219
+
220
+ The [complete review](./design/mlbenchmarks-book-review.md) records all available chapters, repository evidence, and remaining research questions.
@@ -1,31 +1,31 @@
1
1
  # The experiment subpath
2
2
 
3
- `@tangle-network/agent-eval/experiment` turns an experiment's registration into the object that runs it.
3
+ `@tangle-network/agent-eval/experiment` registers decision rules as data and provides interpreters bound to a verified seal.
4
4
 
5
- The registry of measured claims those experiments produce lives in [`evidence/`](../evidence/README.md); a sealed experiment's digest is the `experimentDigest` its registry record carries.
5
+ The [`evidence/` registry](../evidence/README.md) stores published measurements and their experiment identities.
6
+ Sealing does not execute an agent or publish an evidence record.
6
7
 
7
- ## The covenant
8
+ ## What a seal enforces
8
9
 
9
- 1. **The registered rule is the executed rule.**
10
- Every rule row admission, subset selection, estimand, interval, decision table, validity gate, halt, budget, matched budget, reissue — is a typed data node, never a closure or prose.
11
- `sealExperiment` canonicalizes and hashes the whole tree into one digest.
12
- Every interpreter takes only a sealed node plus evidence records.
13
- No execution surface has a parameter for alpha, threshold, metric, or stopping rule, so registered-vs-ran drift is unrepresentable rather than checked.
14
- 2. **Refusals live inside artifacts.**
15
- An inadequate cluster count, a mismatched arm budget, a non-monotone funnel stage, a non-total decision table — each produces a typed verdict object (or a typed error), never a warning sentence beside a number.
16
- 3. **A change is a re-seal.**
17
- `amendExperiment` verifies the current seal, validates the new spec, and appends a `{at, reason, blind[], digest}` entry.
18
- The digest history is the audit trail; changing what is decided without a new digest is impossible.
10
+ 1. `sealExperiment()` validates and hashes the specification, including its registered rules and optional claim.
11
+ 2. `openSealedExperiment()` verifies that digest and captures the rules before returning the execution handle.
12
+ Its decision, interval, gate, and budget methods read their rules from that captured specification.
13
+ 3. `amendExperiment()` verifies the current seal, validates the replacement specification, and records its digest, reason, time, and declared blindness.
14
+
15
+ A seal verifies the current specification's identity.
16
+ It does not authenticate registration time, amendment history, or the origin of supplied measurements.
17
+ The host must retain evidence, invoke the required checks, and honor their refusal results.
18
+ Calling `registered.decide()` does not automatically run admission, power, budget, or halt checks.
19
19
 
20
20
  ## The objects
21
21
 
22
- | object | entry point | what it closes |
22
+ | Object | Entry point | Purpose |
23
23
  | --- | --- | --- |
24
- | Registered-rule AST | `src/experiment/ast.ts` (14 node families) | prose rules; every registered condition compiles to data the seal covers |
25
- | Define / seal / execute | `defineExperiment`, `sealExperiment`, `amendExperiment`, `openSealedExperiment` | hand-written PREREG.md files; the runner executes the sealed rule itself |
26
- | Cluster-aware power | `clusteredPower`, `assertDesignAdequate` | "4 clusters cannot certify any effect size, including 1.0" learned by running the experiment, now refused before a dollar is spent |
27
- | Denominator chain | `buildFunnel`, `executeAdmissionRule`, `composeFunnels`, `renderFunnelTable` | hand-assembled `20 15 14 → admitted` chains, formatted differently each run |
28
- | Matched budgets | `verifyMatchedBudgets`, `assertMatchedBudgets` | "realized tokens must agree within 5%" verified by hand |
24
+ | Registered rules | [Rule types](../src/experiment/ast.ts) and [`ExperimentSpec`](../src/experiment/define.ts) | Describe admission, estimation, intervals, and decisions as data. |
25
+ | Define / seal / execute | `defineExperiment`, `sealExperiment`, `amendExperiment`, `openSealedExperiment` | Validate, identify, and execute the registered rules. |
26
+ | Cluster-aware power | `clusteredPower`, `assertDesignAdequate` | Assess a declared effect under a simulated outcome model and cluster-count policy. |
27
+ | Denominator chain | `buildFunnel`, `executeAdmissionRule`, `composeFunnels`, `renderFunnelTable` | Reconcile retained and excluded evidence. |
28
+ | Matched budgets | `verifyMatchedBudgets`, `assertMatchedBudgets` | Check realized tokens against a declared tolerance. |
29
29
 
30
30
  ### Sealing and execution
31
31
 
@@ -35,67 +35,123 @@ import {
35
35
  sealExperiment,
36
36
  } from '@tangle-network/agent-eval/experiment'
37
37
 
38
- const sealed = await sealExperiment(spec) // RFC 8785 + sha256 over the whole tree
39
- const registered = await openSealedExperiment(sealed) // verifies the digest first
38
+ const sealed = await sealExperiment(spec)
39
+ const registered = await openSealedExperiment(sealed)
40
40
 
41
- const admission = registered.admit(rows) // funnel + survivors, from the sealed rule
41
+ const admission = registered.admit(rows)
42
42
  const gate = registered.gate('power-floor', { kind: 'power-floor', curve })
43
- const halt = registered.halt([gate]) // refuse-spend fires before any contrast
44
- const outcome = registered.decide(quantities) // the sealed table; non-total tables throw
43
+ const halt = registered.halt([gate])
44
+ if (halt.fired) throw new Error(`Experiment halted: ${halt.failedGates.join(', ')}`)
45
+ // Compute quantities from the admitted evidence, then call registered.decide(quantities).
45
46
  ```
46
47
 
47
- #### Digest schemes and retention
48
+ This fragment assumes `spec` registers admission, the named gate, and a halt rule.
49
+ Malformed specifications and unusable evidence throw typed errors; decision and validity refusals remain in returned artifacts.
48
50
 
49
- A seal and a signed `HypothesisManifest` are durable records: each is written
50
- once and verified later, possibly by a different release. Both carry an `algo`
51
- field that names the digest scheme, and verification selects the encoder from
52
- that field.
51
+ Cluster intervals register both `clusterBy` and `value` inside the sealed `IntervalSpec`.
52
+ Call `registered.interval('gain95', { kind: 'rows', rows })` to apply those fields.
53
+ The row evidence cannot override the registered value field.
54
+ Changing the measured field requires a new seal.
53
55
 
54
- | `algo` | Serialization | Status |
55
- |---|---|---|
56
- | `sha256-rfc8785` | RFC 8785 canonical JSON, from `ledger-core/canonical` | What `sealExperiment` and `signManifest` write |
57
- | `sha256-content` | key-sorted `JSON.stringify` | Read-only. Records written before the RFC 8785 scheme carry it, or carry no `algo` at all, and still verify |
56
+ For a paired contrast, prepare one difference per pair and register that difference field as `value`.
57
+ A pooled pass rate from both arms measures a different quantity.
58
+ The [runnable sealed experiment](../examples/sealed-experiment/index.ts) demonstrates the paired path.
59
+ Confidence levels, field paths, seeds, and resample counts are validated before sealing and direct computation.
60
+
61
+ Older cluster interval registrations omitted `value` and require their original package version for execution.
62
+ Retain their original bytes and evidence; create a new registration for subsequent measurements.
58
63
 
59
- The `sha256-content` encoder is private to the module that verifies with it and
60
- is unreachable from any path that writes a digest. Retire it once no record
61
- carrying that tag needs to verify; until then, deleting it would make those
62
- records unverifiable rather than invalid.
64
+ #### Canonical identities and migration
63
65
 
64
- `openSealedExperiment` is the only execution surface.
65
- A rule that is not in the sealed spec cannot run; a rule that is cannot run differently.
66
+ Readers and writers use RFC 8785 canonical JSON from `ledger-core/canonical`.
67
+ Verification refuses missing or unsupported digest schemes.
68
+
69
+ | Artifact | Required identity | Refusal |
70
+ |---|---|---|
71
+ | Sealed experiment | `algo: 'sha256-rfc8785'` | `verifySealedExperiment()` returns `false`; `openSealedExperiment()` refuses execution |
72
+ | Signed hypothesis | `algo: 'sha256-rfc8785'` | `verifyManifest()` returns `false`; synchronous digest checks and hypothesis evaluation refuse the record |
73
+ | Agent profile cell | `agent-profile-cell:sha256-rfc8785:<digest>` | Cell validation refuses any other scheme |
74
+ | Report attestation | Report hash and required `envelopeHash` over its provenance | `verifyAttestation()` returns an invalid result with a reason |
75
+
76
+ The package no longer verifies `sha256-content` records, untagged manifests, bare `agent-profile-cell:sha256:` identifiers, or attestations without provenance envelopes.
77
+ Keep those records unchanged as historical artifacts with their original package version.
78
+ Create new registrations with `sealExperiment()` or `signManifest()` before collecting new decision evidence.
79
+ Use `buildAgentProfileCell()` and `attest()` to produce current identities from independently verified source material.
80
+ Never relabel an existing digest or reconstruct a provenance envelope from unverified metadata.
81
+ A new digest cannot establish that a registration existed before its evidence was observed.
82
+
83
+ Use the opened handle when the result must follow a particular registration.
84
+ Direct helpers such as `computeInterval()` and `executeDecisionRule()` also accept unsealed rules for development.
85
+ They do not establish a link to a registered experiment.
66
86
 
67
87
  ### Cluster-aware power refusal
68
88
 
69
- Two floors, one simulation:
89
+ `clusteredPower()` combines a cluster-count policy with a simulated power curve:
90
+
91
+ - The exact whole-cluster sign-flip test has a minimum two-sided p-value of `2^(1-C)` for `C` independent clusters.
92
+ At alpha 0.05, this policy requires at least six clusters; four give 0.125 and three give 0.25.
93
+ - Seeded simulations draw paired contrasts under the configured win/loss model and apply a whole-cluster percentile bootstrap.
94
+ Power is the fraction of simulated intervals that exclude zero.
70
95
 
71
- - **Closed form, zero spend.** With `C` independent clusters, the exact whole-cluster sign-flip test can never produce a two-sided p below `2^(1-C)`.
72
- Four clusters give 0.125 and three give 0.25 — both above alpha 0.05, so those designs are refused at any effect size.
73
- Six clusters is the smallest certifiable count at 0.05.
74
- - **Seeded simulation.** Per-row paired contrasts are drawn under a registered effect model (base win/loss rates, optional noisy clusters), each trial takes a whole-cluster percentile bootstrap, and power is the fraction of trials whose interval excludes zero.
96
+ The six-cluster floor is a policy for this helper, not a universal requirement for every estimator or fixed-roster evaluation.
97
+ The helper computes both results; it does not skip simulation when the cluster-count policy fails.
75
98
 
76
99
  The refusal is a verdict inside the returned artifact (`result.refusal`), with `assertDesignAdequate` as the throwing form.
77
- The `power-floor` validity gate consumes a power curve as evidence and fails when the curve tops out under the registered target.
100
+ Both `clusteredPower` and the registered `power-floor` gate require `minimumEffect`.
101
+ The gate evaluates a supplied curve; it does not run the simulation itself.
102
+ The effect must appear exactly in the supplied grid; the API does not interpolate.
103
+ Adequacy requires target power at that effect.
104
+ `maxPower` describes the grid and cannot establish adequacy at a smaller effect.
105
+
106
+ ```ts
107
+ import { assertDesignAdequate, clusteredPower } from '@tangle-network/agent-eval/experiment'
108
+
109
+ const power = clusteredPower({
110
+ clusterSizes: Array.from({ length: 24 }, () => 3),
111
+ effects: [0.05, 0.1, 0.2],
112
+ minimumEffect: 0.1,
113
+ targetPower: 0.8,
114
+ seed: 17,
115
+ })
116
+
117
+ assertDesignAdequate(power)
118
+ ```
119
+
120
+ Simulation effects are expected paired contrasts within signal clusters.
121
+ The zero-effect model requires equal `baseWinRate` and `baseLossRate`.
122
+ Configured noisy clusters retain zero expected contrast, so they dilute the pooled population effect.
123
+ Power remains conditional on this outcome model, the registered sampling structure, and the simulated test.
124
+ The [statistical evidence guide](./statistical-evidence.md) explains unit counts, adaptation comparisons, and sequential assumptions.
78
125
 
79
126
  ### The funnel
80
127
 
81
128
  `buildFunnel` refuses a stage that gains rows, named exclusions that do not sum, and partitions that overdraw their source stage.
82
- `executeAdmissionRule` runs a sealed admission rule over rows and returns the funnel, the survivors, and the partition rows in one object — the chain and the rows can never disagree.
83
- Partitions carry `pooling: 'never'`: a secondary set is reported beside the primary chain and cannot be pooled into it.
129
+ `registered.admit(rows)` applies the sealed admission rule and returns the funnel, survivors, and partition rows together.
130
+ The standalone `executeAdmissionRule(rule, rows)` also accepts an unsealed rule.
131
+ Partitions carry `pooling: 'never'`: report each secondary set separately from the primary chain.
84
132
  The object is its own JSON render; `renderFunnelTable` prints the text table with the reconciliation line (`input = surviving + excluded`).
85
133
 
86
134
  ### Matched budgets
87
135
 
88
136
  `verifyMatchedBudgets` compares realized per-arm tokens under the registered tolerance and returns a verdict whose `refusal` field carries `onFail: 'refuse-contrast'` when arms diverge.
89
- A contrast between arms that spent differently is not a contrast; the refusal is the artifact that says so.
137
+ Use this check when the claim requires matched token use.
138
+ An unequal-budget comparison answers a different question and must retain the resource difference in its interpretation.
90
139
 
91
140
  ## Acceptance: the three preregistrations
92
141
 
93
- The module's acceptance suite (`tests/experiment/preregistration-acceptance.test.ts`) re-derives the week's three hand-written preregistrations as sealed specs and reproduces each recorded decision by executing the sealed rules against the recorded evidence:
142
+ The [acceptance suite](../tests/experiment/preregistration-acceptance.test.ts) encodes three historical preregistrations as sealed specifications.
143
+ It checks their recorded decisions against fixed evidence:
94
144
 
95
- - **killtest-20260810** all four validity gates fail on the recorded evidence (the rep-4 oracle flip, the 2-row population drift, the zero-call control, the 0.692 power ceiling) and the halt rule refuses the spend, matching the recorded `$0.00, contrast never run`.
145
+ - **killtest-20260810**: all four validity gates fail on the recorded evidence.
146
+ The failures are the rep-4 oracle flip, 2-row population drift, zero-call control, and 0.692 power ceiling.
147
+ The halt rule refuses spend, matching the recorded `$0.00, contrast never run`.
96
148
  The obligation node routes a positive interval without the registered control to `blocked-pending-registered-control`, never to `thesis-survives`.
97
- - **freelunch-20260810** the admission funnel reproduces the recorded `48 > 43 > 35 > 35 > 32` chain with the 3-row secondary partition; the uniform-pass budget reproduces the recorded uniform n=2; the amendment-6 ledger under the same sealed rule refuses pass 2 — the registered-vs-ran divergence the seal makes unrepresentable; the report-only decision reproduces `3/64` and `2/32`.
98
- - **tbench-20260808 milestone 2** — the round-robin selection reproduces the recorded 20-row subset in pick order; the m3 subset filters the SEALED m2 draw (16 rows); the decision table on the recorded interval reproduces `not-certified-at-this-n`.
149
+ - **freelunch-20260810**: the admission funnel reproduces `48 > 43 > 35 > 35 > 32` with the 3-row secondary partition.
150
+ The uniform-pass budget reproduces uniform n=2; the amendment-6 ledger under the same sealed rule refuses pass 2.
151
+ The report-only decision reproduces `3/64` and `2/32`.
152
+ - **tbench-20260808 milestone 2**: round-robin selection reproduces the recorded 20-row subset in pick order.
153
+ The m3 subset filters the sealed m2 draw to 16 rows.
154
+ The decision table on the recorded interval reproduces `not-certified-at-this-n`.
99
155
 
100
156
  ## What is composed, not duplicated
101
157
 
@@ -103,7 +159,7 @@ The statistical machinery underneath is re-exported from its existing homes; thi
103
159
 
104
160
  | family | home |
105
161
  | --- | --- |
106
- | `pairedBootstrap`, `mcnemar`/`mcnemarPower`/`mcnemarRequiredN`, `pairedRiskDifference*`, `holm`, `benjaminiHochberg`, `eProcess`, `wilson`, `mulberry32`, sample-size helpers | `src/statistics.ts` |
162
+ | `pairedBootstrap`, `mcnemar`/`mcnemarPower`/`mcnemarRequiredN`, `pairedRiskDifference*`, `holm`, `benjaminiHochberg`, `eProcess`, `wilson`, `mulberry32`, sample-size helpers | [`src/statistics/index.ts`](../src/statistics/index.ts) |
107
163
  | `pairedEvalueSequence` (anytime-valid) | `src/sequential.ts` |
108
164
  | `powerPreflight` (variance-based MDE refusal) | `src/campaign/gates/power-preflight.ts` |
109
165
  | `sequentialPairedGate`, `sequentialDecide` (manifest-bound) | `src/campaign/gates/sequential.ts` |
@@ -118,5 +174,5 @@ The trace-repair admission machinery (`buildDenominatorChain`, oracle determinis
118
174
 
119
175
  ## Where this sits
120
176
 
121
- This is Wave 2 of the [charter](./charter.md): the experiment subpath, built after the kill test that re-derived the three preregistrations as decision-rule objects (verdict: extended — ten node families beyond the seed AST, no opaque node, no rule dropped).
122
- Wave 3 wires these objects to the live-sandbox seam; the improvement receipt (Wave 4) serializes a sealed experiment's digest, gates, and refusal outcomes into one attested file.
177
+ The [charter](./charter.md) describes current package ownership and host responsibilities.
178
+ Use [evaluation claims and final evidence](./evaluation-integrity.md) when connecting a registration to an automated improvement workflow.
@@ -55,10 +55,10 @@ user intent
55
55
  -> datasets and optimizers replay the same adapter
56
56
  ```
57
57
 
58
- The important part is that production and eval do not use different loops. The
59
- adapter can swap dependencies (real user session, replay fixture, sandbox), but
60
- the state shape, validators, actions, budgets, and stop policies should stay
61
- the same. That is what makes benchmark gains transfer to real usage.
58
+ Keep the production state, validators, actions, budgets, and stop policies in the evaluation path.
59
+ The adapter can supply a real user session, replay fixture, or sandbox.
60
+ This tests the behavior that production executes.
61
+ Transfer to future tasks still requires representative evaluation data and a measured comparison.
62
62
 
63
63
  ### Agent Runtime Integration
64
64
 
@@ -82,8 +82,7 @@ Implementation ownership:
82
82
  assignment, and optimizer row conversion in `agent-eval`.
83
83
  - Put product state readers, action executors, approval policy, credentials,
84
84
  workspace paths, and UI-specific storage in the downstream repo.
85
- - Promote a product adapter into `agent-eval` only after at least two products
86
- need the same adapter shape.
85
+ - Keep product execution adapters in the consuming repository.
87
86
 
88
87
  ### Code Generator
89
88
 
@@ -178,17 +178,10 @@ It is not production storage because process restart clears its in-memory data.
178
178
  TENANT_KEY=dev-token TENANT_ID=acme pnpm tsx examples/hosted-ingest-server/server.ts
179
179
  ```
180
180
 
181
- In another terminal:
182
-
183
- ```sh
184
- HOSTED_ENDPOINT=http://localhost:8080 \
185
- HOSTED_TENANT_KEY=dev-token \
186
- HOSTED_TENANT_ID=acme \
187
- pnpm tsx examples/foreign-agent-quickstart/index.ts
188
- ```
189
-
190
- The quickstart's eval-run gets POSTed to the reference receiver; the
191
- receiver's `GET /v1/runs` lists it back.
181
+ Send events with [`createHostedClient`](../src/hosted/client.ts) from `@tangle-network/agent-eval/hosted`.
182
+ Call `client.ingestEvalRun(event)` with a valid `EvalRunEvent` after configuring the client's endpoint, tenant ID, and API key.
183
+ For `selfImprove`, configure `hostedTenant` as shown in the [receiver example](../examples/hosted-ingest-server/).
184
+ The receiver's authenticated `GET /v1/runs` endpoint lists ingested runs.
192
185
 
193
186
  ---
194
187