@tangle-network/agent-eval 0.179.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (236) hide show
  1. package/CHANGELOG.md +66 -0
  2. package/README.md +119 -146
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +4 -4
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-CY6Dg5t5.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-CY6Dg5t5.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +5 -8
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +11 -10
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{default-registry-BryMEmr8.js → default-registry-aL7xUrUz.js} +2 -2
  38. package/dist/{default-registry-BryMEmr8.js.map → default-registry-aL7xUrUz.js.map} +1 -1
  39. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  40. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  41. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  42. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  43. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  44. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  45. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  46. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  47. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  48. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  49. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  50. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  51. package/dist/experiment/index.d.ts +27 -477
  52. package/dist/experiment/index.d.ts.map +1 -1
  53. package/dist/experiment/index.js +95 -559
  54. package/dist/experiment/index.js.map +1 -1
  55. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  56. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  57. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  58. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  59. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  60. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  61. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  62. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  63. package/dist/hosted/index.d.ts +2 -2
  64. package/dist/hosted/index.d.ts.map +1 -1
  65. package/dist/hosted/index.js +1 -1
  66. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  67. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  68. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  69. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  70. package/dist/{index-CbLmrWCa.d.ts → index-DNntP4ch.d.ts} +8 -8
  71. package/dist/{index-CbLmrWCa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  72. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  73. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  74. package/dist/index.d.ts +28 -28
  75. package/dist/index.js +25 -16
  76. package/dist/index.js.map +1 -1
  77. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  78. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  79. package/dist/{integrity-DsHWCebQ.js → integrity-DH5ng72x.js} +2 -2
  80. package/dist/{integrity-DsHWCebQ.js.map → integrity-DH5ng72x.js.map} +1 -1
  81. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  82. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  83. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  84. package/dist/journal-Cs9f7385.js.map +1 -0
  85. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  86. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  87. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  88. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  89. package/dist/ledger-core/index.d.ts +1 -1
  90. package/dist/ledger-core/index.js +1 -1
  91. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  92. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  93. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  94. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  95. package/dist/meta-eval/index.d.ts +138 -7
  96. package/dist/meta-eval/index.d.ts.map +1 -1
  97. package/dist/meta-eval/index.js +245 -97
  98. package/dist/meta-eval/index.js.map +1 -1
  99. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  100. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  101. package/dist/multishot/golden/index.d.ts +1 -1
  102. package/dist/multishot/index.d.ts +2 -2
  103. package/dist/openapi.json +1 -1
  104. package/dist/outcome-store-BXlkwMPR.js +131 -0
  105. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  106. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  107. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  108. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  109. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  110. package/dist/pipelines/index.js +1 -1
  111. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  112. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  113. package/dist/profile-cell.d.ts +1 -1
  114. package/dist/profile-cell.js +1 -1
  115. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  116. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  117. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  118. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  119. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  120. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  121. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  122. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  123. package/dist/{report-command-DKlXfU5r.js → report-command-V1ecVgAv.js} +27 -3
  124. package/dist/report-command-V1ecVgAv.js.map +1 -0
  125. package/dist/reporting.d.ts +4 -4
  126. package/dist/reporting.js +3 -3
  127. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  128. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  130. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  131. package/dist/rl.d.ts +53 -99
  132. package/dist/rl.d.ts.map +1 -1
  133. package/dist/rl.js +182 -169
  134. package/dist/rl.js.map +1 -1
  135. package/dist/rollout/index.d.ts +1 -1
  136. package/dist/rollout/index.js +2 -2
  137. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  138. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  139. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  140. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  141. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  142. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  143. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  144. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  145. package/dist/run-record-Br-Yzt_k.js +464 -0
  146. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  147. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  148. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  149. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  150. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  151. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  152. package/dist/sequential-DAsyV2T9.js.map +1 -0
  153. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  154. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  155. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  156. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  157. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  158. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  159. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  160. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  161. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  162. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  163. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  164. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  165. package/dist/supervisor-run/index.d.ts +4 -2
  166. package/dist/supervisor-run/index.d.ts.map +1 -1
  167. package/dist/supervisor-run/index.js +3 -3
  168. package/dist/{terminal-record-Ce9_UjRz.js → terminal-record-BtPwKTSr.js} +58 -26
  169. package/dist/terminal-record-BtPwKTSr.js.map +1 -0
  170. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  171. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  172. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  173. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  174. package/dist/trace-repair/index.d.ts +2 -2
  175. package/dist/traces.d.ts +6 -6
  176. package/dist/traces.js +1 -1
  177. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  178. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  179. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  180. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  181. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  182. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  183. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  184. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  185. package/dist/{types-vUdAx2Cj.d.ts → types-lPkDQNqJ.d.ts} +20 -2
  186. package/dist/{types-vUdAx2Cj.d.ts.map → types-lPkDQNqJ.d.ts.map} +1 -1
  187. package/dist/wire/index.d.ts +2 -2
  188. package/docs/adapters-observability.md +14 -0
  189. package/docs/campaign-proposers.md +86 -128
  190. package/docs/charter.md +108 -112
  191. package/docs/concepts.md +157 -69
  192. package/docs/design/mlbenchmarks-book-review.md +440 -0
  193. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  194. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  195. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  196. package/docs/design/self-improvement-evidence-audit.md +263 -0
  197. package/docs/design.md +2 -1
  198. package/docs/eval-surface-map.md +95 -42
  199. package/docs/evaluation-integrity.md +220 -0
  200. package/docs/experiment.md +111 -55
  201. package/docs/feature-guide.md +5 -6
  202. package/docs/hosted-ingest-spec.md +4 -11
  203. package/docs/insight-report.md +187 -455
  204. package/docs/outcome-validity.md +182 -0
  205. package/docs/product-eval-adoption.md +1 -2
  206. package/docs/research-report-methodology.md +7 -7
  207. package/docs/search-history-receipts.md +8 -0
  208. package/docs/statistical-evidence.md +129 -0
  209. package/docs/verdicts.md +76 -49
  210. package/package.json +1 -1
  211. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  212. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  213. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  214. package/dist/client-BlLY6o2w.js.map +0 -1
  215. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  216. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  217. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  218. package/dist/index-BxWvILU8.d.ts.map +0 -1
  219. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  220. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  221. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  222. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  223. package/dist/outcome-store-ChBKlTd_.js +0 -75
  224. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  225. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  226. package/dist/report-command-DKlXfU5r.js.map +0 -1
  227. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  228. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  229. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  230. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  231. package/dist/run-record-CR63CpHK.js +0 -216
  232. package/dist/run-record-CR63CpHK.js.map +0 -1
  233. package/dist/sequential-B5gXgcyp.js.map +0 -1
  234. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  235. package/dist/terminal-record-Ce9_UjRz.js.map +0 -1
  236. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
package/docs/charter.md CHANGED
@@ -1,112 +1,108 @@
1
- # Charter: what agent-eval is for
2
-
3
- This document states what this package is, derived from the four end-states the stack must reach.
4
- It was written on 2026-08-10 from a measured inventory of this repo, agent-runtime, discovery, discovery-lab, braid, traces, and supervisor-lab.
5
- Every claim about current code cites the module that carries it.
6
- When behavior moves, move this document in the same change.
7
-
8
- ## One sentence
9
-
10
- agent-eval is the honesty layer of the agent stack: instruments that make it structurally hard to fool ourselves at machine speed, and that never choose a research method.
11
-
12
- The anchor is discovery's covenant (discovery `docs/01-vision.md`): shared code enforces evidence integrity and immutable observations; it never chooses roles, methods, or winners.
13
- Everything above this package — runtime, discovery, braid, verticals gets its freedom because honesty is enforced below.
14
-
15
- ## The four end-states, and what each demands from this package
16
-
17
- 1. **Build complex software rapidly, end to end, without issues.**
18
- The gate is not generation.
19
- The gate is knowing what is true about the work while it runs.
20
- Measured on our own corpus: 87% of failed long runs end on a clean exit with the agent claiming success.
21
- Demand: executable verification wired into the runtime loop, not beside it.
22
-
23
- 2. **Do novel research: physics, quantum computing, math, unsolved problems.**
24
- An unsolved problem has no held-out test suite by definition.
25
- Every grader this package shipped before 2026-08 assumes an answer key.
26
- Demand: verification strategies that certify without one — proof kernels, invariant checks, independent derivation agreement, replication — plus the statistics of a careful experimentalist.
27
- This demand is measured, not speculative: discovery-lab holds 99 pursuit directories and 71 blind-graded oracle files on frontier problems, all verified today by hand-rolled run tools outside this package.
28
-
29
- 3. **Build self-improving agents easily, and explain them easily.**
30
- Improvement requires an ungameable signal; the grader, not the edit, decides improve-versus-game.
31
- Explanation requires a portable artifact: the receipt that proves B beat A, verifiable by a third party who does not trust us.
32
-
33
- 4. **Build new interfaces (braid-class): detached sessions, forking, analysts on tap.**
34
- An interface can only expose what the layer below makes addressable.
35
- Demand: sessions, traces, experiments, and verdicts as durable, forkable, queryable objects.
36
-
37
- ## What already exists (the 2026-08-10 inventory, corrected)
38
-
39
- The fragmentation story is smaller than it feels.
40
- The audit refuted "built four times": traces imports this package's whole analyst suite; agent-runtime's live detectors import the detection kernel verbatim; supervisor-lab's judges are AgentProfiles dispatched through this package's judge primitives.
41
-
42
- - **Ask any question over any trace: exists.**
43
- `TraceAnalysisEngine` (`src/analyst/engine.ts`) takes a free-text question.
44
- A new custom question costs zero library files: `defineTraceAnalyst` + `runTraceAnalyst` on the `./analyst` subpath, with the DSPy RLM engine, seven byte-budgeted trace tools, and a metered model proxy behind it.
45
- - **Default failure analysts: exist.**
46
- `buildDefaultAnalystRegistry` ships failure-mode, intent-divergence, knowledge-gap, knowledge-poisoning, improvement, control-integrity, and skill-usage kinds, engine-agnostic and versionable.
47
- - **Statistics: most of an A-plus toolkit, publicly exported.**
48
- Paired bootstrap, clustered paired binary, exact and score risk differences, McNemar with power and required-n, MDE, multiplicity (Holm, Benjamini-Hochberg), e-process sequential gates, corpus inter-rater agreement, pre-registration manifests with content hashes.
49
- - **Executable process verification: proven this week.**
50
- The trace-repair grader scores a proposed fix by executing it and running the task's own held-out suite from outside the container.
51
- Oracle-fix separates from inert-probe (+0.353 vs 0.000 on milestone 1) with the floor pinned at zero.
52
- - **Integrity instruments hardened by this week's burns:**
53
- served-model assertions (a gateway can answer one id with another model), oracle determinism certification (a wall-clock grader flipped 8 of 16 units on identical bytes), control-policy declaration (a zero-step control screened two milestones and could never fire), and equal-terms refusal between comparison arms (`repairArmAsymmetries`).
54
-
55
- ## What is missing (the honest, short list)
56
-
57
- 1. **Cluster-aware power with design-time refusal.**
58
- "Four task clusters cannot certify any effect size, including 1.0" was learned by running the experiment.
59
- A `clusteredPower` simulator must refuse the design before a dollar is spent.
60
- 2. **Pre-registration as code, bound everywhere.**
61
- The manifest binds exactly one statistic family today.
62
- The registered decision rule must be the object the runner executes; drift between registered and ran must be unrepresentable.
63
- 3. **A general prime query surface.**
64
- The prime engine is benchmark-bound; there is no `runPrimeAnalyst` symmetric to `runTraceAnalyst`.
65
- 4. **The funnel as a first-class object.**
66
- Denominator chains are assembled by hand every run.
67
- 5. **The unified analyst definition.**
68
- One declarative unit AgentProfile + evidence projection + reply contract + budget declaration — compiled to any engine, guarded by a byte-identity kill test against the bespoke arms.
69
- 6. **Verification without an answer key.**
70
- A strategy family where the held-out suite is one member: proof kernels, invariant and metamorphic checks, independent derivation agreement, replication.
71
- The type layer exists (`docs/verification-strategies.md`) and the certifications are now produced, not just typed: every in-package verifier — the layer pipeline, the completion oracle, trace contracts, declarative oracles, trajectory replay, the repair grader — lands in `DefaultVerdict` with a certification naming its checker, strategy member, and unverified assumptions (`docs/verdicts.md`).
72
- What remains is execution without an answer key: a real kernel checker bound through the port.
73
- 7. **Session forking as one primitive.**
74
- Both halves exist unjoined: agent-runtime's `SandboxLineage.fork` (live checkpoint) and this package's trajectory replay (recorded prefix).
75
- Braid branches are metadata pointers; the provider session is always new.
76
- 8. **The improvement receipt.**
77
- Digests, attestations, and sealed manifests exist; the single portable, third-party-verifiable file does not.
78
- Five gaming attacks on the receipt are named; each refusal must live inside the receipt.
79
-
80
- ## Build order
81
-
82
- Wave 1 — in flight now.
83
- The analyst definition contract with its byte-identity CI kill test; prime as a HarnessType in agent-interface; the three-arm review fixes.
84
-
85
- Wave 2 the experiment subpath. **Shipped: [docs/experiment.md](./experiment.md).**
86
- `./experiment`: compose the exported statistics into `defineExperiment` / `sealExperiment` with cluster-aware power refusal, pre-registration as executable decision rules, the funnel object, and matched-budget verification.
87
- Kill test first: re-derive this week's three hand-written PREREG.md files as decision-rule objects; any rule that needs an opaque escape hatch kills or extends the design.
88
- The kill test extended the design — ten node families beyond the seed AST, no opaque node — and the three preregistrations are the subpath's acceptance suite.
89
-
90
- Wave 3 runtime wiring (blocked until agent-runtime's supervision merge resolves).
91
- The executable checker bound as a validator at the live-sandbox seam; a stop policy that consumes executable verdicts; then the supervisor budget-allocation experiment at equal compute.
92
- The published negative result to beat: StateSeal, −3.0pp, CI [−8.5, 1.1], n=540.
93
- Our measured headroom: blind continuation rescues 4.7% of rollouts; the done-signal has 62.5% precision.
94
-
95
- Wave 4 the joins.
96
- `fork(session, step, modification)` joining lineage-fork and trajectory replay, consumed by braid; the improvement receipt v1 serializing evidence vector, pre-registration hash, grader calibration, and refusal outcomes into one attested file.
97
-
98
- Wave 5 science.
99
- The verification-strategy family, proof kernels first.
100
- Its pilot is live: two independent Lean formalizations of the BCWW (4.6) inequality — one from the paper, one from the campaign's artifacts — with a kernel-checked equivalence verdict and the campaign's counterexample checked against both.
101
- A statement mismatch is a successful outcome; it is the formalization gap made visible.
102
- The contract layer is shipped: the family, checker port, and blind two-arm equivalence protocol (`src/verification-strategy.ts`) and verdict certifications (`src/verdict.ts`) — see `docs/verification-strategies.md` for the family, each member's failure mode, and the pilot as the worked example.
103
-
104
- ## Standing principles (each earned by a measured burn)
105
-
106
- - **Instruments, never methods.** This package refuses to choose roles, prompts, models, or winners; it makes whatever runs honest.
107
- - **No upward dependencies.** Consumers import this package; never the reverse.
108
- - **A check that cannot run must never render as green.** (505 of 5,459 automated reviews published verdicts no evidence supported.)
109
- - **Certification is task-scoped.** A prompt certified on one task carries nothing onto another; re-authoring voids it.
110
- - **Access to the world beats loop sophistication.** Measured twice in one week: the analyst that could execute beat the one that could only read; framing carried more than the agent loop.
111
- - **Controls must be able to fire.** A screening control that cannot in principle produce the outcome it screens for is uncalibrated, not conservative.
112
- - **The refusal lives inside the artifact.** An adequacy check, an equal-terms check, a determinism check that runs beside the result can be skipped; one that the artifact carries cannot.
1
+ # Charter: what agent-eval owns
2
+
3
+ `agent-eval` owns evaluation data, scoring, experiment decisions, and release evidence.
4
+ It lets a host automate candidate generation while preserving the evidence needed to challenge the result.
5
+
6
+ ## Package boundary
7
+
8
+ | Concern | Owner |
9
+ |---|---|
10
+ | Portable agent contracts and canonical encodings | `agent-interface` |
11
+ | Cases, judge scores, run records, statistical comparisons, evidence admission, and release rules | `agent-eval` |
12
+ | Agent sessions, workers, tool access, model execution, and research orchestration | The host, including `agent-runtime` |
13
+ | Product activation, business outcomes, storage authorization, and access to final data | The consuming application |
14
+
15
+ `agent-runtime` and `agent-knowledge` can depend on Eval.
16
+ Eval must not depend on either consumer, including through development or type-only imports.
17
+ Execution enters through caller-supplied functions.
18
+
19
+ ## Decisions the package supports
20
+
21
+ **Did the agent perform the required task?**
22
+ A clean process exit or a fluent answer cannot establish that the required artifact works.
23
+ Campaigns retain results, failures, deterministic checks, semantic judgments, traces, and measured usage.
24
+ [Completion verification](../src/completion-verifier.ts), [layered verification](../src/multi-layer-verifier.ts), and [trace replay](./trajectory-replay.md) support checks on produced work.
25
+
26
+ **Did a change improve the agent?**
27
+ A comparison needs paired evidence, explicit exclusions, and an appropriate independent observation unit.
28
+ An improvement decision also needs a declared meaningful effect.
29
+ [Campaign gates](./eval-surface-map.md) and [registered experiments](./experiment.md) make those decisions inspectable.
30
+ Train and selection data can guide search; final evidence supports the resulting comparison.
31
+
32
+ **Does the evaluator measure the intended outcome?**
33
+ Known good and known bad controls test different errors.
34
+ [Evaluator admission](./evaluation-integrity.md), [judge calibration](./concepts.md#judge-calibration), and [outcome validity](./outcome-validity.md) describe what the measurements establish.
35
+ Outcome association can motivate an experiment; it cannot establish that changing a rubric causes improvement.
36
+
37
+ **Can another reader verify the evidence?**
38
+ [Evidence receipts](./experiment.md) bind reports to declared identities and provenance.
39
+ [Search-history receipts](./search-history-receipts.md) account for planned and attempted search slots.
40
+ [Verdict certifications](./verdicts.md) name the checker and its unverified assumptions.
41
+ The [evidence registry](../evidence/README.md) retains published measurements and their freshness state.
42
+
43
+ ## Implemented foundations
44
+
45
+ The current implementation includes:
46
+
47
+ - Registered decision rules, sealed experiments, admission funnels, matched-budget checks, and cluster-aware power refusal in [`/experiment`](../src/experiment/index.ts).
48
+ - Paired comparisons, exact binary inference, multiplicity corrections, and sequential gates in [statistics](../src/statistics/index.ts) and [campaign gates](../src/campaign/gates/).
49
+ - Declarative analyst definitions and caller-owned engine binding in [`/analyst`](../src/analyst/index.ts).
50
+ - Executed repair grading and replay in [`/trace-repair`](../src/trace-repair/index.ts) and [`/trajectory-replay`](../src/trajectory-replay/index.ts).
51
+ - Verification strategies and blind equivalence checks through a [caller-supplied checker](./verification-strategies.md).
52
+ - Claim metadata, durable final-evidence reservations, and evaluator admission through the [evaluation integrity API](./evaluation-integrity.md).
53
+
54
+ The [benchmark-book review](./design/mlbenchmarks-book-review.md) separates observed defects, existing capabilities, and proposed research.
55
+ Its archived measurements describe the reviewed revision.
56
+ Current source and regression tests define present behavior.
57
+
58
+ ## Automating evaluation engineering
59
+
60
+ The host can generate candidate cases, checks, rubrics, and agent changes.
61
+ Eval checks whether their evidence supports the declared decision.
62
+ The same authoring loop must not silently turn its own generated labels into independent certification.
63
+
64
+ A host can compose this loop:
65
+
66
+ ```mermaid
67
+ flowchart LR
68
+ A[Production failures and task requirements] --> B[Candidate cases and evaluators]
69
+ B --> C[Independent evaluator audit]
70
+ C --> D[Search on train and selection cases]
71
+ D --> E[Reserved final evidence]
72
+ E --> F[Paired comparison and release decision]
73
+ F --> G[Observed deployment outcomes]
74
+ G --> A
75
+ ```
76
+
77
+ Every revision to an evaluator or candidate changes the object being tested.
78
+ Once final evidence influences that revision, the next confirmation needs fresh evidence.
79
+ Reusable comparisons can declare their population and unit without consuming final evidence.
80
+ Opting into a shared final-evidence ledger records fresh-confirmation exposure across campaigns.
81
+ It cannot enforce secrecy outside the host that uses it.
82
+
83
+ ## Remaining boundaries
84
+
85
+ A package cannot establish population coverage from a dataset name.
86
+ Sampling plans still need production context, source lineage, and checks for missing groups.
87
+ More repetitions improve measurements on existing units; they do not add independent tasks.
88
+
89
+ Generated evaluators need an independent source of expected behavior.
90
+ The host must enforce author/auditor separation and prevent access to final evidence.
91
+ Declared identities and digests make these assumptions inspectable without proving them.
92
+
93
+ A checker for an open research problem must run through its actual verification backend.
94
+ The checker port supports proof kernels, invariants, replication, and agreement checks.
95
+ A strategy name alone supplies no evidence that any of those checks executed.
96
+
97
+ Product activation and continuous monitoring stay with the host.
98
+ Eval returns evidence and decisions; it does not grant deployment authority or choose a research agenda.
99
+
100
+ ## Standing rules
101
+
102
+ - Keep missing evidence distinct from measured zero, failed execution, and a successful empty result.
103
+ - Preserve every attempted slot and its cost, including rejected candidates and service failures.
104
+ - Check practical effect, independence, power, and capture completeness before interpreting a positive score.
105
+ - Keep refusals and exclusions inside the result artifact.
106
+ - Require current canonical envelopes for seals and attestations.
107
+ - Preserve historical evidence as recorded, even when current APIs reject its retired format.
108
+ - Treat a negative result as evidence about the measured conditions and mechanism.