@tangle-network/agent-eval 0.179.0 → 0.181.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (236) hide show
  1. package/CHANGELOG.md +66 -0
  2. package/README.md +119 -146
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
  5. package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
  6. package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
  7. package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
  8. package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
  9. package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
  10. package/dist/analyst/index.d.ts +10 -10
  11. package/dist/analyst/index.js +4 -4
  12. package/dist/ast-CP9ae9B0.js +557 -0
  13. package/dist/ast-CP9ae9B0.js.map +1 -0
  14. package/dist/ast-hI-vjW6J.d.ts +457 -0
  15. package/dist/ast-hI-vjW6J.d.ts.map +1 -0
  16. package/dist/{benchmark-command-CY6Dg5t5.js → benchmark-command-B57n9vjz.js} +7 -6
  17. package/dist/{benchmark-command-CY6Dg5t5.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
  18. package/dist/benchmarks/index.d.ts +4 -4
  19. package/dist/benchmarks/index.js +3 -3
  20. package/dist/campaign/index.d.ts +6 -6
  21. package/dist/campaign/index.js +8 -8
  22. package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
  23. package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
  24. package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
  25. package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
  26. package/dist/cli.js +5 -8
  27. package/dist/cli.js.map +1 -1
  28. package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
  29. package/dist/client-CXE-U1SA.js.map +1 -0
  30. package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
  31. package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
  32. package/dist/contract/index.d.ts +13 -13
  33. package/dist/contract/index.js +11 -10
  34. package/dist/contract/index.js.map +1 -1
  35. package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
  36. package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
  37. package/dist/{default-registry-BryMEmr8.js → default-registry-aL7xUrUz.js} +2 -2
  38. package/dist/{default-registry-BryMEmr8.js.map → default-registry-aL7xUrUz.js.map} +1 -1
  39. package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
  40. package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
  41. package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
  42. package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
  43. package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
  44. package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
  45. package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
  46. package/dist/engine-DS1cysJy.d.ts.map +1 -0
  47. package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
  48. package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
  49. package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
  50. package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
  51. package/dist/experiment/index.d.ts +27 -477
  52. package/dist/experiment/index.d.ts.map +1 -1
  53. package/dist/experiment/index.js +95 -559
  54. package/dist/experiment/index.js.map +1 -1
  55. package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
  56. package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
  57. package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
  58. package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
  59. package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
  60. package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
  61. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
  62. package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
  63. package/dist/hosted/index.d.ts +2 -2
  64. package/dist/hosted/index.d.ts.map +1 -1
  65. package/dist/hosted/index.js +1 -1
  66. package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
  67. package/dist/index-Bp_6sj3x.d.ts.map +1 -0
  68. package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
  69. package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
  70. package/dist/{index-CbLmrWCa.d.ts → index-DNntP4ch.d.ts} +8 -8
  71. package/dist/{index-CbLmrWCa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
  72. package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
  73. package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
  74. package/dist/index.d.ts +28 -28
  75. package/dist/index.js +25 -16
  76. package/dist/index.js.map +1 -1
  77. package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
  78. package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
  79. package/dist/{integrity-DsHWCebQ.js → integrity-DH5ng72x.js} +2 -2
  80. package/dist/{integrity-DsHWCebQ.js.map → integrity-DH5ng72x.js.map} +1 -1
  81. package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
  82. package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
  83. package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
  84. package/dist/journal-Cs9f7385.js.map +1 -0
  85. package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
  86. package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
  87. package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
  88. package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
  89. package/dist/ledger-core/index.d.ts +1 -1
  90. package/dist/ledger-core/index.js +1 -1
  91. package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
  92. package/dist/llm-judge-DEFZeSiu.js.map +1 -0
  93. package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
  94. package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
  95. package/dist/meta-eval/index.d.ts +138 -7
  96. package/dist/meta-eval/index.d.ts.map +1 -1
  97. package/dist/meta-eval/index.js +245 -97
  98. package/dist/meta-eval/index.js.map +1 -1
  99. package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
  100. package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
  101. package/dist/multishot/golden/index.d.ts +1 -1
  102. package/dist/multishot/index.d.ts +2 -2
  103. package/dist/openapi.json +1 -1
  104. package/dist/outcome-store-BXlkwMPR.js +131 -0
  105. package/dist/outcome-store-BXlkwMPR.js.map +1 -0
  106. package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
  107. package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
  108. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
  109. package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
  110. package/dist/pipelines/index.js +1 -1
  111. package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
  112. package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
  113. package/dist/profile-cell.d.ts +1 -1
  114. package/dist/profile-cell.js +1 -1
  115. package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
  116. package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
  117. package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
  118. package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
  119. package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
  120. package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
  121. package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
  122. package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
  123. package/dist/{report-command-DKlXfU5r.js → report-command-V1ecVgAv.js} +27 -3
  124. package/dist/report-command-V1ecVgAv.js.map +1 -0
  125. package/dist/reporting.d.ts +4 -4
  126. package/dist/reporting.js +3 -3
  127. package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
  128. package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
  129. package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
  130. package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
  131. package/dist/rl.d.ts +53 -99
  132. package/dist/rl.d.ts.map +1 -1
  133. package/dist/rl.js +182 -169
  134. package/dist/rl.js.map +1 -1
  135. package/dist/rollout/index.d.ts +1 -1
  136. package/dist/rollout/index.js +2 -2
  137. package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
  138. package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
  139. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
  140. package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
  141. package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
  142. package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
  143. package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
  144. package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
  145. package/dist/run-record-Br-Yzt_k.js +464 -0
  146. package/dist/run-record-Br-Yzt_k.js.map +1 -0
  147. package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
  148. package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
  149. package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
  150. package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
  151. package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
  152. package/dist/sequential-DAsyV2T9.js.map +1 -0
  153. package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
  154. package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
  155. package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
  156. package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
  157. package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
  158. package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
  159. package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
  160. package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
  161. package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
  162. package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
  163. package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
  164. package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
  165. package/dist/supervisor-run/index.d.ts +4 -2
  166. package/dist/supervisor-run/index.d.ts.map +1 -1
  167. package/dist/supervisor-run/index.js +3 -3
  168. package/dist/{terminal-record-Ce9_UjRz.js → terminal-record-BtPwKTSr.js} +58 -26
  169. package/dist/terminal-record-BtPwKTSr.js.map +1 -0
  170. package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
  171. package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
  172. package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
  173. package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
  174. package/dist/trace-repair/index.d.ts +2 -2
  175. package/dist/traces.d.ts +6 -6
  176. package/dist/traces.js +1 -1
  177. package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
  178. package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
  179. package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
  180. package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
  181. package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
  182. package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
  183. package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
  184. package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
  185. package/dist/{types-vUdAx2Cj.d.ts → types-lPkDQNqJ.d.ts} +20 -2
  186. package/dist/{types-vUdAx2Cj.d.ts.map → types-lPkDQNqJ.d.ts.map} +1 -1
  187. package/dist/wire/index.d.ts +2 -2
  188. package/docs/adapters-observability.md +14 -0
  189. package/docs/campaign-proposers.md +86 -128
  190. package/docs/charter.md +108 -112
  191. package/docs/concepts.md +157 -69
  192. package/docs/design/mlbenchmarks-book-review.md +440 -0
  193. package/docs/design/mlbenchmarks-review/observations.json +713 -0
  194. package/docs/design/mlbenchmarks-review/probes.mts +476 -0
  195. package/docs/design/mlbenchmarks-review/sources.json +200 -0
  196. package/docs/design/self-improvement-evidence-audit.md +263 -0
  197. package/docs/design.md +2 -1
  198. package/docs/eval-surface-map.md +95 -42
  199. package/docs/evaluation-integrity.md +220 -0
  200. package/docs/experiment.md +111 -55
  201. package/docs/feature-guide.md +5 -6
  202. package/docs/hosted-ingest-spec.md +4 -11
  203. package/docs/insight-report.md +187 -455
  204. package/docs/outcome-validity.md +182 -0
  205. package/docs/product-eval-adoption.md +1 -2
  206. package/docs/research-report-methodology.md +7 -7
  207. package/docs/search-history-receipts.md +8 -0
  208. package/docs/statistical-evidence.md +129 -0
  209. package/docs/verdicts.md +76 -49
  210. package/package.json +1 -1
  211. package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
  212. package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
  213. package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
  214. package/dist/client-BlLY6o2w.js.map +0 -1
  215. package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
  216. package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
  217. package/dist/engine-CX8ReXkn.d.ts.map +0 -1
  218. package/dist/index-BxWvILU8.d.ts.map +0 -1
  219. package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
  220. package/dist/ledger-core-Cs9f7385.js.map +0 -1
  221. package/dist/llm-judge-v80Kmu9g.js.map +0 -1
  222. package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
  223. package/dist/outcome-store-ChBKlTd_.js +0 -75
  224. package/dist/outcome-store-ChBKlTd_.js.map +0 -1
  225. package/dist/promotion-policy-DWOm70gx.js.map +0 -1
  226. package/dist/report-command-DKlXfU5r.js.map +0 -1
  227. package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
  228. package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
  229. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
  230. package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
  231. package/dist/run-record-CR63CpHK.js +0 -216
  232. package/dist/run-record-CR63CpHK.js.map +0 -1
  233. package/dist/sequential-B5gXgcyp.js.map +0 -1
  234. package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
  235. package/dist/terminal-record-Ce9_UjRz.js.map +0 -1
  236. package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
@@ -0,0 +1,476 @@
1
+ import { createHash } from 'node:crypto'
2
+ import { lstat, mkdtemp, readFile, readdir, rm } from 'node:fs/promises'
3
+ import { tmpdir } from 'node:os'
4
+ import { join } from 'node:path'
5
+ import { fileURLToPath } from 'node:url'
6
+ import { campaignSplitDigest } from '../../../src/campaign/coverage.ts'
7
+ import { sequentialPairedGate } from '../../../src/campaign/gates/sequential.ts'
8
+ import type { JudgeScore } from '../../../src/campaign/types.ts'
9
+ import { HoldoutAuditor } from '../../../src/contamination-guard.ts'
10
+ import { selfImprove } from '../../../src/contract/self-improve.ts'
11
+ import { evaluatePowerFloorGate } from '../../../src/experiment/ast.ts'
12
+ import { compareCodeUnits, hashCanonical } from '../../../src/ledger-core/canonical.ts'
13
+ import { correlationStudy } from '../../../src/meta-eval/correlation-study.ts'
14
+ import { InMemoryOutcomeStore } from '../../../src/meta-eval/outcome-store.ts'
15
+ import { rubricPredictiveValidity } from '../../../src/meta-eval/rubric-predictive-validity.ts'
16
+ import { compareAdaptationCurves, runAdaptationCurve } from '../../../src/rl/adaptation-eval.ts'
17
+ import { runContaminationProbe } from '../../../src/rl/contamination.ts'
18
+ import { PredictiveValidityResearcher } from '../../../src/rl/predictive-validity-researcher.ts'
19
+ import type { RunRecord } from '../../../src/run-record.ts'
20
+ import { TraceEmitter } from '../../../src/trace/emitter.ts'
21
+ import { InMemoryTraceStore } from '../../../src/trace/store.ts'
22
+
23
+ // These diagnostics record behavior without asserting that a defect must remain.
24
+ const reviewedBaseRevision = 'fe1cc5111aab5d588bf7db3a3785325635937a91'
25
+ const command = 'pnpm exec tsx docs/design/mlbenchmarks-review/probes.mts'
26
+ const repositoryRoot = fileURLToPath(new URL('../../../', import.meta.url))
27
+ const sourcePaths = ['src', 'package.json', 'pnpm-lock.yaml', 'tsconfig.json']
28
+
29
+ async function fileEntries(path: string): Promise<Array<{ path: string; sha256: string }>> {
30
+ const absolutePath = join(repositoryRoot, path)
31
+ const metadata = await lstat(absolutePath)
32
+ if (metadata.isDirectory()) {
33
+ const children = await readdir(absolutePath)
34
+ return (await Promise.all(children.map(child => fileEntries(`${path}/${child}`)))).flat()
35
+ }
36
+ if (!metadata.isFile()) throw new Error(`Source identity requires a regular file: ${path}`)
37
+ return [{ path, sha256: createHash('sha256').update(await readFile(absolutePath)).digest('hex') }]
38
+ }
39
+
40
+ // Hash working files, including untracked files, so Git index state cannot hide changes.
41
+ const sourceFiles = (await Promise.all(sourcePaths.map(fileEntries)))
42
+ .flat()
43
+ .sort((left, right) => compareCodeUnits(left.path, right.path))
44
+ const sourceIdentity = {
45
+ algorithm: 'sha256-canonical-file-manifest',
46
+ paths: sourcePaths,
47
+ fileCount: sourceFiles.length,
48
+ digest: hashCanonical({ domain: 'agent-eval-mlbenchmarks-review-source-v1', files: sourceFiles }),
49
+ dependencyScope: 'Records the manifest and lockfile; assumes dependencies were installed from that lockfile.',
50
+ }
51
+ const diagnosticPath = 'docs/design/mlbenchmarks-review/probes.mts'
52
+ const diagnosticIdentity = {
53
+ path: diagnosticPath,
54
+ sha256: createHash('sha256').update(await readFile(join(repositoryRoot, diagnosticPath))).digest('hex'),
55
+ }
56
+
57
+ async function holdoutReuse() {
58
+ const runRoot = await mkdtemp(join(tmpdir(), 'agent-eval-mlbenchmarks-review-'))
59
+ const train = Array.from({ length: 6 }, (_, i) => ({
60
+ id: `train-${i}`,
61
+ kind: 'offline-probe',
62
+ }))
63
+ const final = Array.from({ length: 6 }, (_, i) => ({
64
+ id: `final-${i}`,
65
+ kind: 'offline-probe',
66
+ }))
67
+ const rounds = []
68
+ try {
69
+ for (const round of [1, 2]) {
70
+ const finalDispatches: string[] = []
71
+ let agentCallbacks = 0
72
+ let judgeCallbacks = 0
73
+ let proposerCallbacks = 0
74
+ const result = await selfImprove({
75
+ scenarios: train,
76
+ baselineSurface: 'baseline',
77
+ model: 'deterministic-probe@2026-09-12',
78
+ agent: async (surface, scenario) => {
79
+ agentCallbacks++
80
+ if (scenario.id.startsWith('final-')) finalDispatches.push(scenario.id)
81
+ return String(surface)
82
+ },
83
+ judge: {
84
+ name: 'marker',
85
+ dimensions: [{ key: 'pass', description: 'Output has the marker' }],
86
+ score: ({ artifact }) => {
87
+ judgeCallbacks++
88
+ const pass = artifact.includes('marker') ? 1 : 0
89
+ return { dimensions: { pass }, composite: pass, notes: '' }
90
+ },
91
+ },
92
+ proposer: {
93
+ kind: 'offline-probe',
94
+ propose: async () => {
95
+ proposerCallbacks++
96
+ return [{
97
+ surface: 'marker',
98
+ label: 'marker',
99
+ rationale: 'Synthetic repeat-access probe',
100
+ }]
101
+ },
102
+ },
103
+ budget: { generations: 1, populationSize: 1, holdoutScenarios: final },
104
+ runDir: join(runRoot, `round-${round}`),
105
+ expectUsage: 'off',
106
+ })
107
+ rounds.push({
108
+ round,
109
+ gateDecision: result.gateDecision,
110
+ finalDispatches: finalDispatches.length,
111
+ distinctFinalIds: new Set(finalDispatches).size,
112
+ finalSplitDigest: campaignSplitDigest(final, 1),
113
+ agentCallbacks,
114
+ judgeCallbacks,
115
+ proposerCallbacks,
116
+ })
117
+ }
118
+ } finally {
119
+ // Remove only the directory allocated for this invocation.
120
+ await rm(runRoot, { recursive: true, force: true })
121
+ }
122
+ const auditor = new HoldoutAuditor([
123
+ { id: 'heldout', payload: 'hidden', split: 'holdout' },
124
+ ])
125
+ auditor.get('heldout', 'debugging')
126
+ auditor.get('heldout', 'evaluation')
127
+ return {
128
+ inputs: {
129
+ calls: 2,
130
+ trainCases: train.length,
131
+ finalCases: final.length,
132
+ generationsPerCall: 1,
133
+ populationPerGeneration: 1,
134
+ replicatesPerCase: 1,
135
+ sameFinalPayloadsOnBothCalls: true,
136
+ firstResultFedToSecondProposer: false,
137
+ baselineSurface: 'baseline',
138
+ proposedSurface: 'marker',
139
+ scoring: '1 if artifact contains marker, otherwise 0',
140
+ expectUsage: 'off',
141
+ },
142
+ results: {
143
+ rounds,
144
+ accessPurposes: auditor.getAccessLog().map(entry => entry.purpose),
145
+ temporaryRunDirectoriesRemoved: true,
146
+ },
147
+ limitations: [
148
+ 'Measures repeated final-set access and debugging access, not empirical overfitting.',
149
+ 'The calls use deterministic local callbacks and independent temporary run directories.',
150
+ 'Does not estimate false-promotion frequency or test a downstream access-control service.',
151
+ ],
152
+ }
153
+ }
154
+
155
+ async function outcomeKeyOrder() {
156
+ const rows = Array.from({ length: 10 }, (_, i) => {
157
+ const score = (i + 1) / 10
158
+ return { score, retention: 1 - score, csat: score }
159
+ })
160
+ const results = []
161
+ for (const keyOrder of ['retention-first', 'csat-first'] as const) {
162
+ const traces = new InMemoryTraceStore()
163
+ const outcomes = new InMemoryOutcomeStore()
164
+ let tick = 0
165
+ for (const [i, row] of rows.entries()) {
166
+ const emitter = new TraceEmitter(traces, {
167
+ runId: `outcome-fixture-${i}`,
168
+ now: () => ++tick,
169
+ })
170
+ await emitter.startRun({ scenarioId: `scenario-${i}` })
171
+ await emitter.endRun({ pass: true, score: row.score })
172
+ await outcomes.append({
173
+ runId: emitter.runId,
174
+ capturedAt: ++tick,
175
+ metrics: keyOrder === 'retention-first'
176
+ ? { retention: row.retention, csat: row.csat }
177
+ : { csat: row.csat, retention: row.retention },
178
+ })
179
+ }
180
+ results.push({
181
+ keyOrder,
182
+ latest: await correlationStudy(traces, outcomes, [{ id: 'score' }], ['csat'], {
183
+ seed: 1,
184
+ bootstrapIterations: 500,
185
+ }),
186
+ mean: await correlationStudy(traces, outcomes, [{ id: 'score' }], ['csat'], {
187
+ seed: 1,
188
+ bootstrapIterations: 500,
189
+ reduction: 'mean',
190
+ }),
191
+ })
192
+ }
193
+ return {
194
+ inputs: {
195
+ n: rows.length,
196
+ rows,
197
+ outcomeRowsPerRun: 1,
198
+ requestedMetric: 'csat',
199
+ expectedPearsonForRequestedMetric: 1,
200
+ expectedSpearmanForRequestedMetric: 1,
201
+ seed: 1,
202
+ bootstrapIterations: 500,
203
+ },
204
+ results,
205
+ limitations: [
206
+ 'Tests metric selection and JSON key order with constructed data, not deployment validity.',
207
+ 'Each run has one outcome row, so latest and mean refer to the same requested observation.',
208
+ ],
209
+ }
210
+ }
211
+
212
+ async function adaptationPairing() {
213
+ const scenariosA = [{ scenarioId: 'easy-only', score: 0.9 }]
214
+ const scenariosB = [{ scenarioId: 'hard-only', score: 0.1 }]
215
+ const runner = {
216
+ run: async ({ scenario }: { scenario: { score: number } }) => scenario.score,
217
+ }
218
+ const ks = [0, 1]
219
+ const reps = 1
220
+ const a = await runAdaptationCurve<{ scenarioId: string; score: number }>({
221
+ scenarios: scenariosA, ks, reps, runner,
222
+ })
223
+ const b = await runAdaptationCurve<{ scenarioId: string; score: number }>({
224
+ scenarios: scenariosB, ks, reps, runner,
225
+ })
226
+ return {
227
+ inputs: {
228
+ scenariosA,
229
+ scenariosB,
230
+ ks,
231
+ reps,
232
+ observationsPerArm: ks.length * reps,
233
+ commonScenarios: scenariosA.filter(a => scenariosB.some(b => a.scenarioId === b.scenarioId)).length,
234
+ sameRunnerForBothArms: true,
235
+ bootstrapSeed: 1,
236
+ },
237
+ results: compareAdaptationCurves(a, b, { seed: 1 }),
238
+ limitations: [
239
+ 'The two arms differ in task difficulty; zero task identities overlap.',
240
+ 'This probes the adaptation helper, not the separately implemented campaign paired comparison.',
241
+ ],
242
+ }
243
+ }
244
+
245
+ async function contaminationDisplay() {
246
+ const originals = Array.from({ length: 12 }, (_, i) => ({
247
+ id: `case-${i}`,
248
+ score: 1,
249
+ }))
250
+ const perturbed = originals.map(scenario => ({ ...scenario, score: 0.4 }))
251
+ const result = await runContaminationProbe({
252
+ scenarioId: scenario => scenario.id,
253
+ originals,
254
+ perturbed,
255
+ scoreFn: async scenario => scenario.score,
256
+ })
257
+ return {
258
+ inputs: {
259
+ n: originals.length,
260
+ originalScorePerCase: 1,
261
+ perturbedScorePerCase: 0.4,
262
+ observationsPerCasePerCondition: 1,
263
+ modelTrainingExposure: 'No model is used; scores are constructed fixture values.',
264
+ },
265
+ results: result,
266
+ limitations: [
267
+ 'The global Wilcoxon test measures the constructed paired difference; it does not identify contamination as its cause.',
268
+ 'Per-item qValue uses BH on 1 - abs(delta), without a per-item sampling null; it is a display aid in the inspected source.',
269
+ 'The per-item qValues do not drive the global contaminationSuspected result.',
270
+ ],
271
+ }
272
+ }
273
+
274
+ async function negativeOutcomeDirection() {
275
+ const rows = Array.from({ length: 8 }, (_, i) => ({
276
+ quality: i / 7,
277
+ successRate: 1 - i / 7,
278
+ }))
279
+ const runs: RunRecord[] = []
280
+ const outcomes = new InMemoryOutcomeStore()
281
+ for (const [i, row] of rows.entries()) {
282
+ const runId = `direction-fixture-${i}`
283
+ runs.push({
284
+ runId,
285
+ experimentId: 'book-review-fixture',
286
+ candidateId: 'same-candidate',
287
+ scenarioId: `direction-scenario-${i}`,
288
+ seed: 0,
289
+ model: 'fixture@1',
290
+ promptHash: '0'.repeat(64),
291
+ configHash: '1'.repeat(64),
292
+ commitSha: reviewedBaseRevision,
293
+ wallMs: 0,
294
+ costUsd: 0,
295
+ costProvenance: { kind: 'observed', usd: 0 },
296
+ tokenUsage: { input: 0, output: 0 },
297
+ terminalOutcome: 'succeeded',
298
+ splitTag: 'holdout',
299
+ outcome: { holdoutScore: row.quality, raw: { quality: row.quality } },
300
+ })
301
+ await outcomes.append({
302
+ runId,
303
+ capturedAt: i,
304
+ metrics: { success_rate: row.successRate },
305
+ })
306
+ }
307
+ const report = await rubricPredictiveValidity({
308
+ runs,
309
+ outcomes,
310
+ outcomeMetrics: ['success_rate'],
311
+ rubrics: ['quality'],
312
+ seed: 1,
313
+ bootstrapResamples: 100,
314
+ })
315
+ const researcher = new PredictiveValidityResearcher({
316
+ outcomes,
317
+ outcomeMetrics: ['success_rate'],
318
+ rubrics: ['quality'],
319
+ })
320
+ const researcherReport = await researcher.runValidityCheck(runs)
321
+ const failures = await researcher.inspectFailures(runs)
322
+ const changes = await researcher.proposeChange(failures)
323
+ return {
324
+ inputs: {
325
+ n: rows.length,
326
+ rows,
327
+ rubric: 'quality',
328
+ outcome: 'success_rate',
329
+ desiredOutcomeDirection: 'increase',
330
+ seed: 1,
331
+ bootstrapResamples: 100,
332
+ researcherBootstrapResamples: 500,
333
+ researcherSeed: 'Derived deterministically by the validity helper',
334
+ researcherFailureThreshold: 0.5,
335
+ },
336
+ results: {
337
+ report,
338
+ researcher: {
339
+ report: researcherReport,
340
+ failureGroups: failures.length,
341
+ failures: failures.map(failure => ({
342
+ code: failure.code,
343
+ description: failure.description,
344
+ samples: failure.evidence.samples,
345
+ })),
346
+ proposedChanges: changes,
347
+ },
348
+ },
349
+ limitations: [
350
+ 'Magnitude-based bucketing is intentional in existing tests, despite contradictory interface prose.',
351
+ 'A negative association can be desirable for an outcome such as failure rate; direction needs explicit interpretation.',
352
+ 'The researcher recommends increasing rubric weight despite its negative association with desired success rate; it does not execute or deploy that recommendation.',
353
+ 'Constructed perfect correlation establishes neither causal validity nor held-out predictive performance.',
354
+ ],
355
+ }
356
+ }
357
+
358
+ async function sequentialDependence() {
359
+ const options = { alpha: 0.05, minN: 5, maxN: 100, shuffleSeed: 1337 }
360
+ const branches = []
361
+ for (const delta of [-1, 1]) {
362
+ const scores = (composite: number): Record<string, JudgeScore> => ({
363
+ judge: { composite, dimensions: {}, notes: '' },
364
+ })
365
+ const baseline = new Map(Array.from({ length: 100 }, (_, i) => [
366
+ `task:${i}`,
367
+ scores(delta > 0 ? 0 : 1),
368
+ ]))
369
+ const candidate = new Map(Array.from({ length: 100 }, (_, i) => [
370
+ `task:${i}`,
371
+ scores(delta > 0 ? 1 : 0),
372
+ ]))
373
+ const result = await sequentialPairedGate(options).decide({
374
+ scenarios: [{ id: 'task', kind: 'synthetic-common-sign' }],
375
+ judgeScores: candidate,
376
+ baselineJudgeScores: baseline,
377
+ candidateArtifacts: new Map(),
378
+ baselineArtifacts: new Map(),
379
+ cost: { candidate: 0, baseline: 0 },
380
+ signal: new AbortController().signal,
381
+ })
382
+ branches.push({ commonDelta: delta, probability: 0.5, result })
383
+ }
384
+ return {
385
+ inputs: {
386
+ ...options,
387
+ branchesEnumerated: branches.length,
388
+ cellsPerBranch: 100,
389
+ independentRandomSignsPerExperiment: 1,
390
+ dataGeneratingProcess: 'Draw one fair sign Z; set all 100 paired deltas equal to Z.',
391
+ exchangeable: true,
392
+ marginalMeanDelta: 0,
393
+ conditionalMeanAfterFirstObservation: 'Z, not necessarily <= 0',
394
+ },
395
+ results: {
396
+ branches,
397
+ promotionProbabilityUnderMarginalZeroProcess: branches.reduce(
398
+ (sum, branch) => sum + (branch.result.decision === 'ship' ? branch.probability : 0),
399
+ 0,
400
+ ),
401
+ },
402
+ limitations: [
403
+ 'Enumerates both equiprobable branches exactly; this is not a Monte Carlo estimate.',
404
+ 'The process violates the conditional-mean null required by the e-process core.',
405
+ 'This refutes sufficiency of exchangeability and shuffling, not the valid e-process theorem or any measured production dataset.',
406
+ ],
407
+ }
408
+ }
409
+
410
+ function powerFloor() {
411
+ const gate = {
412
+ kind: 'power-floor' as const,
413
+ target: 0.8,
414
+ effectGrid: [0.01, 1],
415
+ sim: { trials: 1, resamples: 1, seed: 1 },
416
+ }
417
+ const curve = [{ effect: 0.01, power: 0.1 }, { effect: 1, power: 1 }]
418
+ return {
419
+ inputs: {
420
+ gate,
421
+ curve,
422
+ practicalEffectForInterpretation: 0.01,
423
+ curveSource: 'Supplied deterministic fixture; no power simulation is run.',
424
+ },
425
+ results: evaluatePowerFloorGate('floor', gate, curve),
426
+ limitations: [
427
+ 'The inspected gate documents a maximum-over-grid structural feasibility check; this output matches that contract.',
428
+ 'Passing this gate does not establish target power at the practical effect of 0.01.',
429
+ 'The fixture powers are inputs, not measured or simulated power estimates.',
430
+ ],
431
+ }
432
+ }
433
+
434
+ const observations = {
435
+ reviewedBaseRevision,
436
+ sourceIdentity,
437
+ diagnosticIdentity,
438
+ command,
439
+ paidModelCalls: 0,
440
+ execution: {
441
+ kind: 'offline deterministic diagnostic',
442
+ modelCalls: 0,
443
+ callbackImplementation: 'Local arithmetic and string checks only; no provider clients are supplied.',
444
+ temporaryRunStorage: 'Allocated under the OS temporary directory and removed in finally.',
445
+ outputPolicy: 'Preserves current returned measurements; excludes temporary paths, run IDs, and wallclock fields.',
446
+ assertionPolicy: 'No assertions require the observed defects or policy boundaries to persist.',
447
+ },
448
+ probes: {
449
+ holdoutReuse: await holdoutReuse(),
450
+ outcomeKeyOrder: await outcomeKeyOrder(),
451
+ adaptationPairing: await adaptationPairing(),
452
+ contaminationDisplay: await contaminationDisplay(),
453
+ negativeOutcomeDirection: await negativeOutcomeDirection(),
454
+ sequentialDependence: await sequentialDependence(),
455
+ powerFloor: powerFloor(),
456
+ },
457
+ separateVerificationAtReviewedBase: {
458
+ provenance: 'Historical checks at reviewedBaseRevision; this diagnostic does not rerun them.',
459
+ sourceRevision: reviewedBaseRevision,
460
+ checks: [
461
+ { command: 'pnpm typecheck', result: 'passed' },
462
+ { command: 'pnpm build', result: 'passed' },
463
+ { command: 'pnpm verify:package', result: 'passed' },
464
+ ],
465
+ tests: {
466
+ command: 'pnpm test -- tests/experiment/preregistration-acceptance.test.ts tests/experiment/power.test.ts tests/contamination-guard.test.ts tests/rl-predictive-validity-researcher.test.ts tests/rubric-predictive-validity.test.ts tests/meta-eval.test.ts',
467
+ observedScope: 'The package command expanded to the full Vitest suite.',
468
+ files: { passed: 399, skipped: 2 },
469
+ tests: { passed: 5876, skipped: 3 },
470
+ result: 'passed',
471
+ },
472
+ limitation: 'Passing repository checks do not establish correctness of the counterexample behaviors recorded above.',
473
+ },
474
+ }
475
+
476
+ console.log(JSON.stringify(observations, null, 2))
@@ -0,0 +1,200 @@
1
+ {
2
+ "accessed_at_utc": "2026-09-13T01:01:32.062338+00:00",
3
+ "index_url": "https://mlbenchmarks.org/",
4
+ "index_sha256": "edd63754d75578741714d1fcbf8766e97782c865a4e65345f00967a26b8b4589",
5
+ "pages": [
6
+ {
7
+ "file": "00-preface.html",
8
+ "url": "https://mlbenchmarks.org/00-preface.html",
9
+ "title": "Preface - The Emerging Science of Machine Learning Benchmarks",
10
+ "html_sha256": "df83e9d6dd7ee845e05a4086db47f53af0d85f807be0f512f22f2323f7c00391",
11
+ "body_words": 2689,
12
+ "review_status": "complete-text"
13
+ },
14
+ {
15
+ "file": "00-prologue.html",
16
+ "url": "https://mlbenchmarks.org/00-prologue.html",
17
+ "title": "Prologue - The Emerging Science of Machine Learning Benchmarks",
18
+ "html_sha256": "1885c0e15e19f55c2f267d96a0ebd06ff694582f0ad8905ccddd6e85b8033553",
19
+ "body_words": 625,
20
+ "review_status": "complete-text"
21
+ },
22
+ {
23
+ "file": "01-introduction.html",
24
+ "url": "https://mlbenchmarks.org/01-introduction.html",
25
+ "title": "1 – Introduction - The Emerging Science of Machine Learning Benchmarks",
26
+ "html_sha256": "507797a0b6c72a0036b8367602ee41a0d0c339a7269a3ea7934128bb3a9185a0",
27
+ "body_words": 3377,
28
+ "review_status": "complete-text"
29
+ },
30
+ {
31
+ "file": "02-populations-predictions.html",
32
+ "url": "https://mlbenchmarks.org/02-populations-predictions.html",
33
+ "title": "2 – Populations and predictions - The Emerging Science of Machine Learning Benchmarks",
34
+ "html_sha256": "e6ff7c7ddd73154d199f04d3a43f92b3a16195456f4f23cc0296e257bbb225bc",
35
+ "body_words": 5563,
36
+ "review_status": "complete-text"
37
+ },
38
+ {
39
+ "file": "03-detecting-differences.html",
40
+ "url": "https://mlbenchmarks.org/03-detecting-differences.html",
41
+ "title": "3 – Detecting differences - The Emerging Science of Machine Learning Benchmarks",
42
+ "html_sha256": "b9b1df7d9339a71a70d825f0d51c27d17293a0d3dd55a6272f84591c78285cde",
43
+ "body_words": 5542,
44
+ "review_status": "complete-text"
45
+ },
46
+ {
47
+ "file": "04-holdout-method.html",
48
+ "url": "https://mlbenchmarks.org/04-holdout-method.html",
49
+ "title": "4 – Holdout method - The Emerging Science of Machine Learning Benchmarks",
50
+ "html_sha256": "c3dd581b0fbc8227d989f0c5863b2060d902cf866c1167dc1fb840a224a04907",
51
+ "body_words": 8441,
52
+ "review_status": "complete-text"
53
+ },
54
+ {
55
+ "file": "05-test-set-reuse.html",
56
+ "url": "https://mlbenchmarks.org/05-test-set-reuse.html",
57
+ "title": "5 – Test set reuse - The Emerging Science of Machine Learning Benchmarks",
58
+ "html_sha256": "09261cb5ddeea95fcb84b042eaa2b721393e1a52c40377d13247085d394d6138",
59
+ "body_words": 5585,
60
+ "review_status": "complete-text"
61
+ },
62
+ {
63
+ "file": "06-scientific-crisis.html",
64
+ "url": "https://mlbenchmarks.org/06-scientific-crisis.html",
65
+ "title": "6 – Scientific crisis - The Emerging Science of Machine Learning Benchmarks",
66
+ "html_sha256": "500d42c5bce9fbd67b4908e5a5a3fd956d954969ab73c0828e22211be952ef15",
67
+ "body_words": 7256,
68
+ "review_status": "complete-text"
69
+ },
70
+ {
71
+ "file": "07-replication-machine-learning.html",
72
+ "url": "https://mlbenchmarks.org/07-replication-machine-learning.html",
73
+ "title": "7 – Replication in machine learning - The Emerging Science of Machine Learning Benchmarks",
74
+ "html_sha256": "c5103477aa1a72b2bbd75fe58e8f1ab2254b247e6529f06c0bebb12950335bf4",
75
+ "body_words": 8899,
76
+ "review_status": "complete-text"
77
+ },
78
+ {
79
+ "file": "08-forces-against-crisis.html",
80
+ "url": "https://mlbenchmarks.org/08-forces-against-crisis.html",
81
+ "title": "8 – Forces against crisis - The Emerging Science of Machine Learning Benchmarks",
82
+ "html_sha256": "874764d22ba2c712877fc86fa2e465297670071ca8c58fd011769c1dfb1be4f8",
83
+ "body_words": 8128,
84
+ "review_status": "complete-text"
85
+ },
86
+ {
87
+ "file": "10-generative-models.html",
88
+ "url": "https://mlbenchmarks.org/10-generative-models.html",
89
+ "title": "10 – Generative models - The Emerging Science of Machine Learning Benchmarks",
90
+ "html_sha256": "2e89354b87cc473e1190f687a4286f80754efc86257ec06b3569ea13d6f23729",
91
+ "body_words": 11490,
92
+ "review_status": "complete-text"
93
+ },
94
+ {
95
+ "file": "11-evaluating-language-models.html",
96
+ "url": "https://mlbenchmarks.org/11-evaluating-language-models.html",
97
+ "title": "11 – Evaluating language models - The Emerging Science of Machine Learning Benchmarks",
98
+ "html_sha256": "01a08570e33f9446bc33f0b0ac6fa60c18ec9d356672944da51b4ff802059bad",
99
+ "body_words": 13399,
100
+ "review_status": "complete-text"
101
+ },
102
+ {
103
+ "file": "12-problem-aggregation.html",
104
+ "url": "https://mlbenchmarks.org/12-problem-aggregation.html",
105
+ "title": "12 – The problem of aggregation - The Emerging Science of Machine Learning Benchmarks",
106
+ "html_sha256": "f09917ed330f01aa573d9609469e82745ebc50a218484c88685ff21065217104",
107
+ "body_words": 8857,
108
+ "review_status": "complete-text"
109
+ },
110
+ {
111
+ "file": "13-model-moves-data.html",
112
+ "url": "https://mlbenchmarks.org/13-model-moves-data.html",
113
+ "title": "13 – When the model moves the data - The Emerging Science of Machine Learning Benchmarks",
114
+ "html_sha256": "673f0e4f300cf4d27b4bdab92593a0e2b72fcd63092a51afbebf614ed4173ae8",
115
+ "body_words": 8832,
116
+ "review_status": "complete-text"
117
+ },
118
+ {
119
+ "file": "14-evaluation-frontier.html",
120
+ "url": "https://mlbenchmarks.org/14-evaluation-frontier.html",
121
+ "title": "14 – Evaluation at the frontier - The Emerging Science of Machine Learning Benchmarks",
122
+ "html_sha256": "71a5d336dc99e9135651218344564ffaea2e9140e67295860735c3b76df15154",
123
+ "body_words": 9464,
124
+ "review_status": "complete-text"
125
+ },
126
+ {
127
+ "file": "15-epilogue.html",
128
+ "url": "https://mlbenchmarks.org/15-epilogue.html",
129
+ "title": "15 – Epilogue - The Emerging Science of Machine Learning Benchmarks",
130
+ "html_sha256": "a0b57ccdb46107e3a82d2a9f3bd16601e950fd3fb13006030062edc61379c651",
131
+ "body_words": 1624,
132
+ "review_status": "complete-text"
133
+ }
134
+ ],
135
+ "review_date_local": "2026-09-12",
136
+ "review_timezone": "America/Los_Angeles",
137
+ "repository": {
138
+ "url": "https://github.com/tangle-network/agent-eval",
139
+ "revision": "fe1cc5111aab5d588bf7db3a3785325635937a91",
140
+ "package_version": "0.180.0",
141
+ "base": "origin/main"
142
+ },
143
+ "coverage": {
144
+ "linked_reading_pages": 16,
145
+ "reviewed_pages": 16,
146
+ "body_words": 109771,
147
+ "word_count_method": "Whitespace split of BeautifulSoup chapter-body get_text with spaces; includes headings, tables, equations, notes, and references.",
148
+ "extraction_audit": "Paragraph extraction plus every omitted non-whitespace text node were read; figures/tables supporting conclusions were checked in HTML or PDF context.",
149
+ "missing_chapter": {
150
+ "number": 9,
151
+ "status": "unavailable from inspected live index",
152
+ "checked_404_paths": [
153
+ "https://mlbenchmarks.org/09-annotation.html",
154
+ "https://mlbenchmarks.org/09-data-annotation.html",
155
+ "https://mlbenchmarks.org/09-annotations.html"
156
+ ]
157
+ },
158
+ "independent_reproduction_of_cited_studies": false,
159
+ "print_edition_reviewed": false
160
+ },
161
+ "review_partitions": [
162
+ {
163
+ "pages": [
164
+ "00-preface.html",
165
+ "00-prologue.html",
166
+ "01-introduction.html",
167
+ "02-populations-predictions.html",
168
+ "03-detecting-differences.html",
169
+ "04-holdout-method.html",
170
+ "05-test-set-reuse.html",
171
+ "06-scientific-crisis.html"
172
+ ],
173
+ "body_words": 39078
174
+ },
175
+ {
176
+ "pages": [
177
+ "07-replication-machine-learning.html",
178
+ "08-forces-against-crisis.html",
179
+ "10-generative-models.html",
180
+ "11-evaluating-language-models.html"
181
+ ],
182
+ "body_words": 41916
183
+ },
184
+ {
185
+ "pages": [
186
+ "12-problem-aggregation.html",
187
+ "13-model-moves-data.html",
188
+ "14-evaluation-frontier.html",
189
+ "15-epilogue.html"
190
+ ],
191
+ "body_words": 28777
192
+ }
193
+ ],
194
+ "visual_inspection": {
195
+ "chapter_7": "Figures 7.2–7.4 checked using linked chapter PDF; HTML used for references.",
196
+ "chapter_11": "Figures 11.3–11.6 checked using linked chapter PDF; HTML used for references.",
197
+ "chapters_12_to_14": "All eight SVG figures checked; all three chapter-12 tables checked in HTML.",
198
+ "chapters_1_to_6": "Captions and surrounding arguments read; chapter-2 confusion matrix and chapter-5 rate table checked in HTML."
199
+ }
200
+ }