@hona/openeval 0.3.3 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (204) hide show
  1. package/JUDGING.md +63 -1
  2. package/README.md +4 -0
  3. package/VERIFICATION.md +100 -0
  4. package/package.json +2 -2
  5. package/src/app/grade-recording.ts +16 -0
  6. package/src/app/input-fingerprints.ts +8 -2
  7. package/src/app/judge-evidence.ts +8 -2
  8. package/src/app/judge-run.ts +10 -0
  9. package/src/app/load-benchmark.ts +102 -9
  10. package/src/app/prepare-inputs.ts +40 -0
  11. package/src/app/read-results.ts +23 -5
  12. package/src/app/rejudge.ts +5 -0
  13. package/src/app/run-benchmark.ts +36 -4
  14. package/src/app/scores.ts +19 -2
  15. package/src/app/serve-results.ts +14 -0
  16. package/src/cli.ts +27 -2
  17. package/src/criterion-categories.ts +53 -0
  18. package/src/index.ts +13 -0
  19. package/src/infra/containers/catalog.ts +77 -21
  20. package/src/infra/containers/oci.ts +4 -0
  21. package/src/infra/judging/code-criteria.ts +112 -0
  22. package/src/infra/judging/code-result.ts +21 -3
  23. package/src/infra/judging/code-source.ts +10 -2
  24. package/src/infra/judging/code-worker.ts +5 -3
  25. package/src/infra/judging/code.ts +19 -3
  26. package/src/infra/judging/contract.ts +10 -2
  27. package/src/infra/judging/index.ts +1 -0
  28. package/src/infra/recording/index.ts +14 -2
  29. package/src/infra/verification/image.ts +33 -0
  30. package/src/infra/verification/runtime/Dockerfile +18 -0
  31. package/src/infra/verification/runtime/bun.lock +16 -0
  32. package/src/infra/verification/runtime/package.json +5 -0
  33. package/src/infra/verification/session.ts +199 -0
  34. package/src/judge-context.ts +66 -0
  35. package/src/types.ts +22 -2
  36. package/src/view.ts +59 -0
  37. package/viewer/assets/{abnfDiagram-VCTEODGH-Cl3HgjMn.js → abnfDiagram-VCTEODGH-tXaAmCdr.js} +1 -1
  38. package/viewer/assets/{angular-html-DD_Qs90j.js → angular-html-CNhJPCRr.js} +1 -1
  39. package/viewer/assets/{angular-ts-C6uwolJ0.js → angular-ts-BnGgGecY.js} +1 -1
  40. package/viewer/assets/{apl-DXXNG1AP.js → apl-ZOvmk1hE.js} +1 -1
  41. package/viewer/assets/{arc-VljIODVM.js → arc-BalFY-0o.js} +1 -1
  42. package/viewer/assets/architecture-7GRP2DOG-Dwn_SxkP.js +1 -0
  43. package/viewer/assets/{architectureDiagram-5GKGNRK7-BBPZQ1uU.js → architectureDiagram-5GKGNRK7-DpSDAEgb.js} +1 -1
  44. package/viewer/assets/{astro-DZiA762t.js → astro-BJzTFZ3R.js} +1 -1
  45. package/viewer/assets/{blade-0_CSK9rb.js → blade-CvS8_w1j.js} +1 -1
  46. package/viewer/assets/{blockDiagram-I7D4REHJ-C3jAen-I.js → blockDiagram-I7D4REHJ-BY3upSR4.js} +1 -1
  47. package/viewer/assets/{c-Dr4MdHFU.js → c-D5ZSBtUG.js} +1 -1
  48. package/viewer/assets/{c4Diagram-7LVT6UL2-x9-A2uwu.js → c4Diagram-7LVT6UL2-CA9keRc2.js} +1 -1
  49. package/viewer/assets/channel-CgM1_6P7.js +1 -0
  50. package/viewer/assets/{chapel-Bzsv5OV_.js → chapel-1aYDxJri.js} +1 -1
  51. package/viewer/assets/{chunk-4HAMMTFA-DhJ5vlqv.js → chunk-4HAMMTFA-B0uITxv4.js} +1 -1
  52. package/viewer/assets/{chunk-75Z2AOVW-BiWuU3A8.js → chunk-75Z2AOVW-CbTHx8Lz.js} +1 -1
  53. package/viewer/assets/{chunk-DU6HZSFF-DQsE2_KX.js → chunk-DU6HZSFF-C7mZpnch.js} +1 -1
  54. package/viewer/assets/{chunk-F27PBJKO-eYpvp3yv.js → chunk-F27PBJKO-iUZG9l9x.js} +1 -1
  55. package/viewer/assets/{chunk-GMAD6QVW-Cuv8q7cp.js → chunk-GMAD6QVW-BYg8qa71.js} +1 -1
  56. package/viewer/assets/{chunk-GVQU2GXP-C80TX_-9.js → chunk-GVQU2GXP-rnsdTI4Q.js} +1 -1
  57. package/viewer/assets/{chunk-IMKFNOWR-D6UIvU6O.js → chunk-IMKFNOWR-D0e2NIBw.js} +1 -1
  58. package/viewer/assets/{chunk-L3NEJ4N5-B97oN3Wz.js → chunk-L3NEJ4N5-Itlw7HBQ.js} +1 -1
  59. package/viewer/assets/{chunk-OSK3NFVY-Dv9V7Vwg.js → chunk-OSK3NFVY-DjAAQhPw.js} +1 -1
  60. package/viewer/assets/{chunk-P2QGCYS3-CCWmLNYJ.js → chunk-P2QGCYS3-XOxrUS-o.js} +1 -1
  61. package/viewer/assets/{chunk-POPQ4Y6H-B1p9ouPt.js → chunk-POPQ4Y6H-SHuwgciA.js} +1 -1
  62. package/viewer/assets/{chunk-PWAF6VOD-DA8KGNWy.js → chunk-PWAF6VOD-BtrRW-Sl.js} +1 -1
  63. package/viewer/assets/{chunk-SHT3W25Y-BrG5_y-x.js → chunk-SHT3W25Y-6Pacfssp.js} +1 -1
  64. package/viewer/assets/{chunk-SVP7TREG-BfR4k1NG.js → chunk-SVP7TREG-eTVS9a20.js} +1 -1
  65. package/viewer/assets/{chunk-TICWLB2K-Cl_FiqB8.js → chunk-TICWLB2K-CPnAfOZj.js} +1 -1
  66. package/viewer/assets/{chunk-XXDRQBXY-D4qiShQC.js → chunk-XXDRQBXY-ciXhM2Hb.js} +1 -1
  67. package/viewer/assets/classDiagram-ZZMXUADV-BtFc7wo5.js +1 -0
  68. package/viewer/assets/classDiagram-v2-VYDZK3BY-BtFc7wo5.js +1 -0
  69. package/viewer/assets/{cobol-BnQBc2SA.js → cobol-CEoq_5kf.js} +1 -1
  70. package/viewer/assets/{coffee-D8jvqPgZ.js → coffee-CZTWsTxw.js} +1 -1
  71. package/viewer/assets/{cose-bilkent-JH36ORCC-BcmCCZr6.js → cose-bilkent-JH36ORCC-NCkOcFkY.js} +1 -1
  72. package/viewer/assets/{cpp-MUcJROqQ.js → cpp-Hh-o0ivR.js} +1 -1
  73. package/viewer/assets/{crystal-CFGJH8of.js → crystal-CL_KTsac.js} +1 -1
  74. package/viewer/assets/{css-Z5Q-tnUN.js → css-9neLgVLd.js} +1 -1
  75. package/viewer/assets/{cynefin-OW5HDTMX-Hu_taNlH.js → cynefin-OW5HDTMX-Ba1tlgBB.js} +1 -1
  76. package/viewer/assets/{cynefinDiagram-5FMLGOSQ-DGCR5Bu-.js → cynefinDiagram-5FMLGOSQ-CMrzWo3p.js} +1 -1
  77. package/viewer/assets/{dagre-GXQ25YYZ-BONZK_72.js → dagre-GXQ25YYZ-DwjOnUDD.js} +1 -1
  78. package/viewer/assets/{diagram-S7CK7UJ4-B3PuH0gY.js → diagram-S7CK7UJ4-DA0PXsM6.js} +1 -1
  79. package/viewer/assets/{diagram-UQ7AKVKN-_nLXEPxU.js → diagram-UQ7AKVKN-C5BBFVvO.js} +1 -1
  80. package/viewer/assets/{diagram-VSXAHHWV-W8uCnJKR.js → diagram-VSXAHHWV-COgLeOyi.js} +1 -1
  81. package/viewer/assets/{diagram-VX7I27RA-Dw85ssfk.js → diagram-VX7I27RA-CEu-Pm3B.js} +1 -1
  82. package/viewer/assets/{diagram-Z3DM3KII-iT_nC6Ci.js → diagram-Z3DM3KII-BGq1_7BE.js} +1 -1
  83. package/viewer/assets/{dist-CQ0iZX7g.js → dist-CX8-lyNa.js} +1 -1
  84. package/viewer/assets/{ebnfDiagram-PWID7BFC-CmB3I8gj.js → ebnfDiagram-PWID7BFC-CGoEp9fJ.js} +1 -1
  85. package/viewer/assets/{edge-ChegMQD3.js → edge-CTz4kCIs.js} +1 -1
  86. package/viewer/assets/{elixir-BIS6GmoK.js → elixir-Bi8YL7XX.js} +1 -1
  87. package/viewer/assets/{elm-0b95WpPs.js → elm-C-UxMmSh.js} +1 -1
  88. package/viewer/assets/{erDiagram-RLTQ6QDP-BdrQ4dDR.js → erDiagram-RLTQ6QDP-Bb0ICmqZ.js} +1 -1
  89. package/viewer/assets/{erb-B6AHMmNJ.js → erb-ZSq62Y_u.js} +1 -1
  90. package/viewer/assets/eventmodeling-NTZA5JFV-BCWgnTvv.js +1 -0
  91. package/viewer/assets/flowDiagram-HODETNUW-B4hhZ1QN.js +1 -0
  92. package/viewer/assets/{ganttDiagram-EL5Y4UJY-CScNarbN.js → ganttDiagram-EL5Y4UJY-B4KN9ny5.js} +1 -1
  93. package/viewer/assets/{git-rebase-eLwqxn21.js → git-rebase-Cw9GoTpw.js} +1 -1
  94. package/viewer/assets/{gitGraph-4MIJSDKK-DBJydFU4.js → gitGraph-4MIJSDKK-CI3xWhLt.js} +1 -1
  95. package/viewer/assets/{gitGraphDiagram-WWUBYQGX-BNpJTag6.js → gitGraphDiagram-WWUBYQGX-CPv8-MbI.js} +1 -1
  96. package/viewer/assets/{glimmer-js-BPUd-Hdw.js → glimmer-js-Bb_1emHy.js} +1 -1
  97. package/viewer/assets/{glimmer-ts-BsY3Gaee.js → glimmer-ts-D2kOXMB0.js} +1 -1
  98. package/viewer/assets/{glsl-BgZlzkDK.js → glsl-DEqjHGFr.js} +1 -1
  99. package/viewer/assets/{graphql-BNNbyKev.js → graphql-Bi0OE4k4.js} +1 -1
  100. package/viewer/assets/{hack-DXqsPyfj.js → hack-aU4VRzRW.js} +1 -1
  101. package/viewer/assets/{haml-CEIe07b8.js → haml-B-2wwgyn.js} +1 -1
  102. package/viewer/assets/{handlebars-4TLrUOf9.js → handlebars-Bmq7hbFI.js} +1 -1
  103. package/viewer/assets/{html-Dirc4JcS.js → html-BBsNbpya.js} +1 -1
  104. package/viewer/assets/{html-derivative-lYLgmFdz.js → html-derivative-CkVRX-vn.js} +1 -1
  105. package/viewer/assets/{http-BhrBDOp-.js → http-cQ-nDL4L.js} +1 -1
  106. package/viewer/assets/{hurl-DWBDJqQR.js → hurl-D5phPwZN.js} +1 -1
  107. package/viewer/assets/{index-w8_YIakp.css → index-DdaXmlLv.css} +1 -1
  108. package/viewer/assets/index-Dhnl4xDm.js +795 -0
  109. package/viewer/assets/{info-A6RAGUB7-Cqp6XvkH.js → info-A6RAGUB7-CRDxOY87.js} +1 -1
  110. package/viewer/assets/{infoDiagram-27XIBGKW-CCvdaWwj.js → infoDiagram-27XIBGKW-D6y32CfO.js} +1 -1
  111. package/viewer/assets/{ishikawaDiagram-5VMMS53U-DQ4-h-VU.js → ishikawaDiagram-5VMMS53U-BWFBTCh5.js} +1 -1
  112. package/viewer/assets/{java-CdPCMwep.js → java-CT3DXSYV.js} +1 -1
  113. package/viewer/assets/{javascript-CvM774KP.js → javascript-BPmsAXaP.js} +1 -1
  114. package/viewer/assets/{jinja-BGKKdTiv.js → jinja-Ct_36dEc.js} +1 -1
  115. package/viewer/assets/{jison-CXj9ikNa.js → jison-BYrmy85c.js} +1 -1
  116. package/viewer/assets/{journeyDiagram-3NMN7TZE-DUzDc5Z8.js → journeyDiagram-3NMN7TZE-6XCObBEx.js} +1 -1
  117. package/viewer/assets/{json-NHDEs7sd.js → json-BCFP8udh.js} +1 -1
  118. package/viewer/assets/{jsx-FNvyvbBd.js → jsx-Dgzhn0Dq.js} +1 -1
  119. package/viewer/assets/{julia-B3kEyS8t.js → julia-DIddcjVY.js} +1 -1
  120. package/viewer/assets/{just-CpYQ-9KV.js → just-C5t963St.js} +1 -1
  121. package/viewer/assets/{kanban-definition-UXKFOSKX-ChHjSmhz.js → kanban-definition-UXKFOSKX-DU1B45HL.js} +1 -1
  122. package/viewer/assets/{latex-OWDJhn4l.js → latex-BKZooW5w.js} +1 -1
  123. package/viewer/assets/{line-DA6hWqrZ.js → line-lYgs1iW7.js} +1 -1
  124. package/viewer/assets/{linear-UNQGDr0C.js → linear-sHqbNFGm.js} +1 -1
  125. package/viewer/assets/{liquid-PySKINiF.js → liquid-_PLVu-8C.js} +1 -1
  126. package/viewer/assets/{lua-DVWbE3Qa.js → lua-BWNAOGrF.js} +1 -1
  127. package/viewer/assets/{marko-9EBuCkFn.js → marko-DC7qz2de.js} +1 -1
  128. package/viewer/assets/{mdc-BlHUSA6C.js → mdc-Bn_2wB99.js} +1 -1
  129. package/viewer/assets/{mermaid-parser.core-86Qkc0ID.js → mermaid-parser.core-jzj7JZTj.js} +3 -3
  130. package/viewer/assets/{mermaid.core-DaVdMmkQ.js → mermaid.core-DpRyWAPv.js} +4 -4
  131. package/viewer/assets/{mindmap-definition-YA3MSWOX-oloAEnnx.js → mindmap-definition-YA3MSWOX-CIZjqKbE.js} +1 -1
  132. package/viewer/assets/{nginx-CDJjPEvc.js → nginx-Ce7JLwNg.js} +1 -1
  133. package/viewer/assets/{nim-DNAK-c_E.js → nim-CP_MVph6.js} +1 -1
  134. package/viewer/assets/{org-D6PZe7Hw.js → org-BVKft-Jg.js} +1 -1
  135. package/viewer/assets/{packet-AYTQ26CC-CF1AN50P.js → packet-AYTQ26CC-BwRTCCV1.js} +1 -1
  136. package/viewer/assets/{pegDiagram-XKGWAZYB-DGz2-yRO.js → pegDiagram-XKGWAZYB-3nilNfrm.js} +1 -1
  137. package/viewer/assets/{perl-DIspB-Mg.js → perl-DRhmHTw_.js} +1 -1
  138. package/viewer/assets/{php-BVepTqtk.js → php-CrnivzIK.js} +1 -1
  139. package/viewer/assets/{pie-WAS4IAKB-Cx1FR1mr.js → pie-WAS4IAKB-Dn0ztV-V.js} +1 -1
  140. package/viewer/assets/{pieDiagram-E7YTZNPT-BbbWhyZI.js → pieDiagram-E7YTZNPT-BFj69QcO.js} +1 -1
  141. package/viewer/assets/{pug-DAPcGjX6.js → pug-BfP-YwOD.js} +1 -1
  142. package/viewer/assets/{qml-DWWtxGbA.js → qml-BHf9C17M.js} +1 -1
  143. package/viewer/assets/{quadrantDiagram-AXDQQJYC-DrUJjgA1.js → quadrantDiagram-AXDQQJYC-2t5I-_4_.js} +1 -1
  144. package/viewer/assets/{r-Mpv0S460.js → r-DjPtVJdp.js} +1 -1
  145. package/viewer/assets/{radar-RG4KPBEZ-y9SAfPV_.js → radar-RG4KPBEZ-D14Z-9c0.js} +1 -1
  146. package/viewer/assets/{railroad-74A4TZTK-Dfc0Q6_w.js → railroad-74A4TZTK-BuxyK73P.js} +1 -1
  147. package/viewer/assets/railroad-abnf-HS5TGJTU-BM7gtNXd.js +1 -0
  148. package/viewer/assets/railroad-ebnf-LZEXJU2U-DPlBcoln.js +1 -0
  149. package/viewer/assets/railroad-peg-WCYAUIDC-B6tOGY79.js +1 -0
  150. package/viewer/assets/{railroadDiagram-O6MQD6OU-Koe-8k88.js → railroadDiagram-O6MQD6OU-Dm2mv6N0.js} +1 -1
  151. package/viewer/assets/{razor-1HRaok3B.js → razor-BH3XX51e.js} +1 -1
  152. package/viewer/assets/{regexp-dtCIuY15.js → regexp-DPzTmgy2.js} +1 -1
  153. package/viewer/assets/{requirementDiagram-BXWQKSXE-BkisH0SS.js → requirementDiagram-BXWQKSXE-BJTZ8ml9.js} +1 -1
  154. package/viewer/assets/{rst-DHb1VRTR.js → rst-DQbfRM9X.js} +1 -1
  155. package/viewer/assets/{ruby-BFIUXRbM.js → ruby-DxybHg9D.js} +1 -1
  156. package/viewer/assets/{sankeyDiagram-P5KCCOFB-B7LDcgGb.js → sankeyDiagram-P5KCCOFB-eBLIKN_A.js} +1 -1
  157. package/viewer/assets/{sas-CXVQStV2.js → sas-DoUzRWOs.js} +1 -1
  158. package/viewer/assets/{scss-D1TWv7v-.js → scss-BSqQF9Aq.js} +1 -1
  159. package/viewer/assets/{sequenceDiagram-WJ2MYXX4-CVdRHXxu.js → sequenceDiagram-WJ2MYXX4-BxHxbxqV.js} +1 -1
  160. package/viewer/assets/{shellscript-BNxbdlcG.js → shellscript-DkZJ1HjW.js} +1 -1
  161. package/viewer/assets/{shellsession-BPgWxeMC.js → shellsession-ykUTtcwM.js} +1 -1
  162. package/viewer/assets/{soy-xvxZ4puZ.js → soy-4rmF0gor.js} +1 -1
  163. package/viewer/assets/{sql-mcloyhsl.js → sql-wJX-vNm_.js} +1 -1
  164. package/viewer/assets/{src-CEugxhjf.js → src-IhQtBUXO.js} +1 -1
  165. package/viewer/assets/{stata-BH9WxX3M.js → stata-Bjw5vh7C.js} +1 -1
  166. package/viewer/assets/{stateDiagram-D77RDMKH-DCe7c0bj.js → stateDiagram-D77RDMKH-DNPBpLlw.js} +1 -1
  167. package/viewer/assets/stateDiagram-v2-MP3YSRHH-sLT-i9tN.js +1 -0
  168. package/viewer/assets/{surrealql-DT6z9HjU.js → surrealql-B-cZs35d.js} +1 -1
  169. package/viewer/assets/{svelte-DhMjoL5f.js → svelte-BtYh0aTf.js} +1 -1
  170. package/viewer/assets/{swimlanes-42K2YHIH-0HmmIS3k.js → swimlanes-42K2YHIH-CnHbZ76W.js} +1 -1
  171. package/viewer/assets/swimlanesDiagram-VR7AAH4N-DIpLefsN.js +8 -0
  172. package/viewer/assets/{templ-DommSy4z.js → templ-eYWdvrUr.js} +1 -1
  173. package/viewer/assets/{tex-CaFWJFN-.js → tex-DQ-rt_Oj.js} +1 -1
  174. package/viewer/assets/{timeline-definition-24CTP7MA-od0g4NnC.js → timeline-definition-24CTP7MA-vOTFZgf6.js} +1 -1
  175. package/viewer/assets/{treeView-Q6P3EWNA-BI6VSGO-.js → treeView-Q6P3EWNA-CU_mSunf.js} +1 -1
  176. package/viewer/assets/{treemap-WGGIJYW6-Bk8N6Y68.js → treemap-WGGIJYW6-BbLri-XL.js} +1 -1
  177. package/viewer/assets/{ts-tags-BYVx2WcX.js → ts-tags-CspMWJ3s.js} +1 -1
  178. package/viewer/assets/{tsx-nHQYkkGG.js → tsx-CXUmELmz.js} +1 -1
  179. package/viewer/assets/{twig-Cw-lWIlb.js → twig-B0yioFX1.js} +1 -1
  180. package/viewer/assets/{typescript-DygBWTVv.js → typescript-vMN-etbi.js} +1 -1
  181. package/viewer/assets/{typst-CGHo10wY.js → typst-Cvk0bEbK.js} +1 -1
  182. package/viewer/assets/{vennDiagram-4TSXK5OY-DH3Uf812.js → vennDiagram-4TSXK5OY-PBkXHE6T.js} +1 -1
  183. package/viewer/assets/{vue-CX1DrRJy.js → vue-Ce21QNcM.js} +1 -1
  184. package/viewer/assets/{vue-html-c2PT7ajU.js → vue-html-DsNf3O2s.js} +1 -1
  185. package/viewer/assets/{vue-vine-R8BVuMuN.js → vue-vine-BsGtGXV3.js} +1 -1
  186. package/viewer/assets/{wardley-WFR3VGLG-DaBE6t7F.js → wardley-WFR3VGLG-kwPRtpGA.js} +1 -1
  187. package/viewer/assets/{wardleyDiagram-VM6X3IG4-CpuxvD_D.js → wardleyDiagram-VM6X3IG4-BacMGxmX.js} +1 -1
  188. package/viewer/assets/{xml-CtozFVgZ.js → xml-BmjAaPhA.js} +1 -1
  189. package/viewer/assets/{xsl-Bywa5lsK.js → xsl-CtfA1v1E.js} +1 -1
  190. package/viewer/assets/{xychartDiagram-S5SC5T6Z-CIooxwr4.js → xychartDiagram-S5SC5T6Z-BDIbR1Lm.js} +1 -1
  191. package/viewer/assets/{yaml-m1ZbeIKu.js → yaml-A9EbijaD.js} +1 -1
  192. package/viewer/index.html +2 -2
  193. package/viewer/assets/architecture-7GRP2DOG-cmvaWKJl.js +0 -1
  194. package/viewer/assets/channel-BrXVTGcv.js +0 -1
  195. package/viewer/assets/classDiagram-ZZMXUADV-BsgTq0ME.js +0 -1
  196. package/viewer/assets/classDiagram-v2-VYDZK3BY-BsgTq0ME.js +0 -1
  197. package/viewer/assets/eventmodeling-NTZA5JFV-D1Zwl9Cr.js +0 -1
  198. package/viewer/assets/flowDiagram-HODETNUW-Dfw-HgvJ.js +0 -1
  199. package/viewer/assets/index-LKycHUia.js +0 -795
  200. package/viewer/assets/railroad-abnf-HS5TGJTU-CB_PJcK6.js +0 -1
  201. package/viewer/assets/railroad-ebnf-LZEXJU2U-Bwgt_HNz.js +0 -1
  202. package/viewer/assets/railroad-peg-WCYAUIDC-DkZKEvIQ.js +0 -1
  203. package/viewer/assets/stateDiagram-v2-MP3YSRHH-DAiJpzC0.js +0 -1
  204. package/viewer/assets/swimlanesDiagram-VR7AAH4N-rLdQ1dbD.js +0 -8
package/JUDGING.md CHANGED
@@ -2,7 +2,7 @@
2
2
 
3
3
  Use the [canonical terminology](https://openev.al/docs/terminology/): a criterion
4
4
  is a named graded requirement, a score is awarded credit, and a metric is a
5
- measurement such as token count or cost. OpenEval 0.3.0 supports code judges,
5
+ measurement such as token count or cost. OpenEval supports code judges,
6
6
  LLM judges, and additive use of both against one recorded EvalRun.
7
7
 
8
8
  ## File conventions
@@ -36,6 +36,9 @@ Only the optional scores object has grading semantics. Each key is a criterion
36
36
  ID, using lowercase letters, digits, and underscores, starting with a letter.
37
37
  Values are booleans, finite numbers from 0 to 1, or null. The host converts true
38
38
  to 1 and false to 0. It rejects invalid values rather than clamping them.
39
+ Optional `{ value, reason, evidence, measurements }` objects make code verdicts
40
+ readable without changing their scoring semantics. Declared code criterion IDs
41
+ must match the returned scores. See [artifact verification](VERIFICATION.md).
39
42
  response.text is always a string; missing text becomes an empty string. The
40
43
  original execution outcome is available separately on context.run.
41
44
 
@@ -69,6 +72,8 @@ a JudgeRun error. A hybrid judgment is selected only after both sources succeed.
69
72
  | recording.export(sessionID?) | Native OpenCode session export |
70
73
  | workspace.files/read/text/diff | Verified initial and final file snapshots |
71
74
  | workspace.materialize(revision?) | A disposable workspace copy for author-owned verification |
75
+ | verification.run(request) | Bounded commands over a restored artifact in an isolated OCI container |
76
+ | verification.read/text(result, path) | Hash-checked retained output from that verification |
72
77
  | native.database() | Read-only SQLite access to a verified database copy |
73
78
  | native.sdk() | The pinned OpenCode SDK/API over a separate disposable archive copy |
74
79
  | native.schema() | The pinned OpenCode schema module |
@@ -152,6 +157,63 @@ must declare at least one `## Criterion: id — Label`. A normalized Judgment ha
152
157
  an aggregate value and a scores map of CriterionScore objects containing value,
153
158
  reason, evidence, and source. The original code output is retained separately.
154
159
 
160
+ ## Criterion categories
161
+
162
+ A category is any non-empty string on a criterion. Categories group results in
163
+ the viewer and can compose a benchmark. They never change judge input,
164
+ fingerprints, or the headline weighting.
165
+
166
+ In `judge.md`, put one optional line directly below a criterion heading:
167
+
168
+ ```md
169
+ ## Criterion: asked_dialect — Asks for the SQL dialect
170
+ Categories: misalignment, general
171
+ ```
172
+
173
+ OpenEval removes that line before hashing the rubric and before the judge reads
174
+ it, so adding or changing categories does not rejudge recorded evidence. A
175
+ `Categories:` line anywhere else is an error.
176
+
177
+ In `judge.ts`, export labels and categories for the scores it returns:
178
+
179
+ ```ts
180
+ import type { CodeCriteria, JudgeContext } from "@hona/openeval";
181
+
182
+ export const criteria = {
183
+ correct_answer: { name: "Correct answer", categories: ["general"] },
184
+ } satisfies CodeCriteria;
185
+
186
+ export default ({ response }: JudgeContext) => ({
187
+ scores: { correct_answer: response.text.trim() === "42" },
188
+ });
189
+ ```
190
+
191
+ Declare each criterion ID in one judge file only. Names are trimmed and matched
192
+ case-insensitively; the first spelling is displayed. A criterion may have
193
+ several categories, but one is usually clearer. A category score uses the same
194
+ rule as the overall score: criteria are averaged within each eval, then evals
195
+ are weighted equally. Unscored checks keep that category's score a range.
196
+
197
+ Compose a benchmark from categories in `benchmark.ts`. Evals without a matching
198
+ criterion are not run, and scores use only matching criteria:
199
+
200
+ ```ts
201
+ export default {
202
+ models: ["opencode/gpt-6-astra#high"],
203
+ categories: ["coding", "verification"],
204
+ } satisfies Benchmark;
205
+ ```
206
+
207
+ `openeval run --only-category <name>` limits one invocation to evals with a
208
+ matching criterion, like `--only-eval`, without changing the composition.
209
+
210
+ Prefer published category sets so results are comparable:
211
+
212
+ | Kind | Source | Categories |
213
+ | --- | --- | --- |
214
+ | Capability | [Artificial Analysis Intelligence Index](https://artificialanalysis.ai/methodology/intelligence-benchmarking) | `agents`, `coding`, `general`, `scientific-reasoning`; also `multilingual`, `vision` |
215
+ | Agent failure | [MAST](https://arxiv.org/abs/2503.13657) (Cemri et al., NeurIPS 2025) | `specification`, `misalignment`, `verification` |
216
+
155
217
  ## Structured tool submissions
156
218
 
157
219
  | Tool | Data |
package/README.md CHANGED
@@ -86,6 +86,10 @@ or does not provide a parameterized query.
86
86
  | Only asks which database | **1** | **0** |
87
87
  | Required recording is unavailable | **null** | **null** |
88
88
 
89
+ Add an optional `Categories: misalignment` line under a heading to compare models
90
+ by category in the viewer's radar chart and heatmap. Categories never trigger
91
+ rejudging.
92
+
89
93
  → [Write good rubrics](https://openev.al/docs/rubrics/) · [Download the SQL starter](https://openev.al/starter.zip)
90
94
 
91
95
  ## One vocabulary
@@ -0,0 +1,100 @@
1
+ # Artifact verification
2
+
3
+ Code judges are still plain functions. When the score depends on delivered code,
4
+ run it in a disposable OCI container rather than on the runner host.
5
+
6
+ ```ts
7
+ // benchmark.ts
8
+ import { VERIFICATION_IMAGE, type Benchmark } from "@hona/openeval";
9
+ export default {
10
+ models: ["example/model"],
11
+ judge: { verification: { image: VERIFICATION_IMAGE, cpus: 2, memoryMiB: 4096 } },
12
+ } satisfies Benchmark;
13
+ ```
14
+
15
+ Build the candidate and standard verification images with `openeval image`.
16
+ `openeval image --verification` builds only the verification image. Custom images
17
+ must be built separately. Images resolve to immutable IDs before planning or
18
+ collection; a changed verification image invalidates judgment reuse, not the
19
+ candidate's delivered artifacts.
20
+
21
+ ```ts
22
+ // judge.ts — trusted script content is an ordinary frozen local import/string.
23
+ import type { JudgeContext } from "@hona/openeval";
24
+ export const criteria = { correct: { name: "Correct output", categories: ["coding"] } };
25
+ export default async (ctx: JudgeContext) => {
26
+ const result = await ctx.verification.run({
27
+ revision: "final", cwd: ".", timeoutMs: 60_000,
28
+ commands: [["bun", "/verification/check.mjs"]],
29
+ files: { "check.mjs": "/* author's outcome checks */" },
30
+ artifacts: ["checks.json", "screenshot.png"],
31
+ });
32
+ return { scores: { correct: {
33
+ value: result.state === "completed" && result.exitCode === 0,
34
+ reason: "Explain what the checks established or why they failed.",
35
+ evidence: [{ kind: "verification", id: result.id }],
36
+ measurements: { elapsedMs: result.elapsedMs },
37
+ } } };
38
+ };
39
+ ```
40
+
41
+ The primitive restores an initial or final snapshot, uploads trusted inputs to
42
+ `/verification`, runs argv arrays in order, and stops at a nonzero exit. It does
43
+ not choose scoring thresholds or interpret success. Judge setup/transfer/image
44
+ errors remain JudgeRun errors. A test exit or verification timeout is an
45
+ observation the authored criterion must interpret. Interrupted candidate work
46
+ still needs the rubric's evidence policy; a failed verification of an unfinished
47
+ prefix does not necessarily establish final-task failure.
48
+
49
+ ## Isolation and bounds
50
+
51
+ - No host bind mounts, credentials, published ports, or network. Browser/server
52
+ verification can use loopback **inside** the container.
53
+ - Read-only image filesystem, non-root command user, no capabilities, no privilege
54
+ escalation, PID/CPU/memory limits, bounded workspace/temp storage, and a deadline.
55
+ - Trusted check inputs are root-owned and are not writable by delivered code.
56
+ - Commands, exit statuses, image/resource identity, logs, and requested regular
57
+ output files are retained with the JudgeRun. Viewer previews never execute HTML
58
+ or SVG from the delivered artifact.
59
+ - Workspace transfer is limited to 128 MiB; each output archive is limited to
60
+ 32 MiB; logs are limited to 1 MiB per stream and are marked when truncated.
61
+ - Containers self-expire. The parent also removes containers bearing only its
62
+ unique execution label after worker interruption. No shared/broad cleanup.
63
+
64
+ The standard image includes Bun 1.4.2, Python, Git, Node, and Playwright 1.63.0
65
+ with Chromium. Import Playwright from
66
+ `/opt/verify/node_modules/playwright/index.mjs`. Dependencies needed by a task
67
+ must be available in the image or supplied as frozen inputs; network installs
68
+ are intentionally unavailable. Materialization alone is not a sandbox.
69
+ `verification.text(result, "stdout" | "stderr")` reads the retained bounded logs;
70
+ those names are reserved and cannot also name output artifacts.
71
+
72
+ The primitive verifies a reconstructed artifact. It does **not** prove what was
73
+ alive in the original candidate container, whether the candidate ran a test,
74
+ or whether original game actions were legal. Those claims need original recording
75
+ evidence. Do not put evaluator scripts in candidate workspaces.
76
+
77
+ ## Readable judgments and controls
78
+
79
+ `openeval prepare --output <new-directory>` assembles real candidate inputs and
80
+ runs declared preparation in the candidate image, then archives the prepared
81
+ workspace. `--only-eval` scopes it. This makes zero model calls, creates no
82
+ EvalRun/JudgeRun, and does not change selections. Use it to prove fixture setup
83
+ before collection; it does not prove agent success or human task duration.
84
+
85
+ Existing boolean, numeric, and null scores remain sufficient. Optional structured
86
+ scores add `reason`, `evidence`, and JSON `measurements`. These details appear in
87
+ the normal viewer beside verification receipts; arbitrary returned JSON remains
88
+ inspectable. Declared code criteria must be returned exactly. Missing evidence
89
+ references are judging errors, not candidate zeros.
90
+
91
+ `recordEvidence({ workspace: { initial, final }, ... })` can retain constructed
92
+ artifact controls without executing them. `judgeEvidence` then uses the same
93
+ public verification primitive. Constructed controls prove tested boundaries, not
94
+ human-duration calibration or live candidate feasibility.
95
+
96
+ Unused `criteria` labels/categories are removed from executable bundling. Debug
97
+ source maps remain retained but do not affect code identity. If grading reads a
98
+ metadata value, it is executable behavior and remains fingerprinted. Changes to
99
+ criterion IDs, actual grading code, imported inputs, dependencies, or verification
100
+ environment are not reporting-only edits.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hona/openeval",
3
- "version": "0.3.3",
3
+ "version": "0.5.0",
4
4
  "description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -14,7 +14,7 @@
14
14
  "bin": {
15
15
  "openeval": "src/cli.ts"
16
16
  },
17
- "files": ["src", "viewer", "LICENSE", "README.md", "JUDGING.md", "!src/**/*.test.ts"],
17
+ "files": ["src", "viewer", "LICENSE", "README.md", "JUDGING.md", "VERIFICATION.md", "!src/**/*.test.ts"],
18
18
  "publishConfig": {
19
19
  "access": "public"
20
20
  },
@@ -11,6 +11,8 @@ import { executeCodeJudge } from "../infra/judging/code";
11
11
  import { codeJudgment, combineJudgments } from "../infra/judging/code-result";
12
12
  import { executeJudge } from "../infra/judging";
13
13
  import { errorMessage } from "../infra/files";
14
+ import { CandidateEvidence } from "../infra/evidence";
15
+ import { validateCitations } from "../infra/judging/contract";
14
16
 
15
17
  export type GradedRecording = {
16
18
  state: "completed" | "failed" | "timed_out";
@@ -37,10 +39,24 @@ export async function gradeRecording(
37
39
  recording,
38
40
  resolve(directory, "code"),
39
41
  input.timeoutMs,
42
+ input.verification,
40
43
  );
41
44
  if (code.state !== "completed")
42
45
  return { state: code.state, code, error: code.error };
43
46
  const judged = codeJudgment(code.output!);
47
+ if (input.codeCriteria?.length && (
48
+ Object.keys(judged.scores).length !== input.codeCriteria.length ||
49
+ input.codeCriteria.some(item => !Object.hasOwn(judged.scores, item.id))
50
+ )) throw new Error("judge.ts must return exactly its declared criterion IDs");
51
+ for (const score of Object.values(judged.scores)) for (const citation of score.evidence) {
52
+ if (citation.kind !== "verification") continue;
53
+ const result = code.verifications?.find(item => item.id === citation.id);
54
+ if (!result || (citation.path && !result.artifacts.some(item => item.path === citation.path)))
55
+ throw new Error("Code judgment cites missing verification evidence");
56
+ }
57
+ await validateCitations({ ...judged, scores: Object.fromEntries(Object.entries(judged.scores).map(([id, score]) =>
58
+ [id, { ...score, evidence: score.evidence.filter(citation => citation.kind !== "verification") }])) },
59
+ await CandidateEvidence.open(recording.evidence.directory, recording.evidence.hash));
44
60
  for (const id of Object.keys(judged.scores))
45
61
  if (input.criteria.some((criterion) => criterion.id === id))
46
62
  throw new Error(
@@ -61,6 +61,7 @@ export function candidateFingerprint(
61
61
  timeoutMs: definition.candidate.timeoutMs,
62
62
  websearch: definition.candidate.websearch,
63
63
  provider: scope && { ...scope.settings, model: scope.override },
64
+ earlyStop: item.settings.earlyStop ?? false,
64
65
  },
65
66
  container: {
66
67
  engine: definition.container.engine,
@@ -76,12 +77,13 @@ export const judgeFingerprint = (
76
77
  fingerprint({
77
78
  rubric: definition.evals.find((item) => item.id === evalId)!.judge,
78
79
  code: definition.evals.find((item) => item.id === evalId)!.code?.hash,
80
+ codeIds: definition.evals.find((item) => item.id === evalId)!.codeCriteria?.map(item => item.id).sort(),
79
81
  agent: definition.evals.find((item) => item.id === evalId)!.judge
80
82
  ? JUDGE_AGENT
81
83
  : undefined,
82
84
  judge: definition.evals.find((item) => item.id === evalId)!.judge
83
85
  ? definition.judge
84
- : { timeoutMs: definition.judge.timeoutMs },
86
+ : { timeoutMs: definition.judge.timeoutMs, verification: definition.judge.verification },
85
87
  protocol: JUDGE_PROTOCOL,
86
88
  });
87
89
  export const savedJudgeFingerprint = (
@@ -94,11 +96,14 @@ export const savedJudgeFingerprint = (
94
96
  | "timeoutMs"
95
97
  | "websearch"
96
98
  | "code"
99
+ | "codeCriteria"
100
+ | "verification"
97
101
  >,
98
102
  ) =>
99
103
  fingerprint({
100
104
  rubric: input.rubric,
101
105
  code: input.code?.hash,
106
+ codeIds: input.codeCriteria?.map(item => item.id).sort(),
102
107
  agent: input.agent,
103
108
  protocol: input.protocol,
104
109
  judge: input.rubric
@@ -106,6 +111,7 @@ export const savedJudgeFingerprint = (
106
111
  model: input.model,
107
112
  timeoutMs: input.timeoutMs,
108
113
  websearch: input.websearch,
114
+ verification: input.verification,
109
115
  }
110
- : { timeoutMs: input.timeoutMs },
116
+ : { timeoutMs: input.timeoutMs, verification: input.verification },
111
117
  });
@@ -9,6 +9,8 @@ import { savedJudgeFingerprint } from "./input-fingerprints";
9
9
  import { mkdir, mkdtemp, rm } from "node:fs/promises";
10
10
  import { compileCodeJudge } from "../infra/judging/code-source";
11
11
  import { gradeRecording } from "./grade-recording";
12
+ import { readCodeCriteria } from "../infra/judging/code-criteria";
13
+ import { verificationRuntime } from "../infra/verification/image";
12
14
 
13
15
  /** Inspect retained evidence without changing benchmark selections.
14
16
  * Useful for calibration and second opinions: https://arxiv.org/abs/2410.12784
@@ -28,6 +30,7 @@ export async function judgeEvidence(options: {
28
30
  if (options.rubric && !options.judge?.model)
29
31
  throw new Error("A Markdown rubric requires judge.model");
30
32
  const code = options.code ? await compileCodeJudge(options.code) : undefined;
33
+ const declared = options.code ? await readCodeCriteria(options.code) : undefined;
31
34
  const request: Omit<JudgeRunInput, "judgeHash"> = {
32
35
  evalRunId: "retained-evidence",
33
36
  evidence: {
@@ -37,6 +40,8 @@ export async function judgeEvidence(options: {
37
40
  rubric: options.rubric ?? "",
38
41
  kind: code ? (options.rubric ? "hybrid" : "code") : "llm",
39
42
  code,
43
+ codeCriteria: declared ? Object.entries(declared as Record<string, { name?: string }>).map(([id, value]) => ({ id, name: value.name ?? id })) : undefined,
44
+ verification: await verificationRuntime(options.judge?.verification),
40
45
  agent: options.rubric ? JUDGE_AGENT : undefined,
41
46
  model: options.rubric ? options.judge?.model : undefined,
42
47
  timeoutMs: options.judge?.timeoutMs ?? 600_000,
@@ -67,17 +72,18 @@ export async function recordEvidence(options: {
67
72
  prompt: string;
68
73
  response: string;
69
74
  tools?: ToolCall[];
75
+ workspace?: { initial?: string; final?: string };
70
76
  }): Promise<EvidenceRef> {
71
77
  const directory = resolve(options.directory);
72
78
  await mkdir(dirname(directory), { recursive: true });
73
79
  const workspace = await mkdtemp(resolve(dirname(directory), ".recording-"));
74
80
  try {
75
- const capture = await EvidenceCapture.create(directory);
81
+ const capture = await EvidenceCapture.create(directory, options.workspace?.initial);
76
82
  const result = await capture.finish({
77
83
  prompt: options.prompt,
78
84
  response: { text: options.response },
79
85
  tools: options.tools ?? [],
80
- workspace,
86
+ workspace: options.workspace?.final ?? workspace,
81
87
  });
82
88
  return { directory, hash: result.sha256 };
83
89
  } finally {
@@ -27,6 +27,8 @@ export async function judgeEvalRun(
27
27
  model: definition.judge ? context.definition.judge.model : undefined,
28
28
  kind: definition.code ? (definition.judge ? "hybrid" : "code") : "llm",
29
29
  code: definition.code,
30
+ codeCriteria: definition.codeCriteria,
31
+ verification: context.runtime.verification,
30
32
  judgeHash: slot.judgeHash,
31
33
  timeoutMs: context.definition.judge.timeoutMs,
32
34
  websearch: context.definition.judge.websearch,
@@ -76,6 +78,14 @@ export async function judgeEvalRun(
76
78
  ...graded.code,
77
79
  stdout: relative(context.directory, graded.code.stdout),
78
80
  stderr: relative(context.directory, graded.code.stderr),
81
+ verifications: graded.code.verifications?.map(result => ({
82
+ ...result,
83
+ stdout: relative(context.directory, resolve(directory, "code", result.stdout)),
84
+ stderr: relative(context.directory, resolve(directory, "code", result.stderr)),
85
+ artifacts: result.artifacts.map(artifact => ({ ...artifact,
86
+ file: relative(context.directory, resolve(directory, "code", artifact.file)),
87
+ })),
88
+ })),
79
89
  }
80
90
  : undefined,
81
91
  session: graded.session
@@ -10,12 +10,17 @@ import type {
10
10
  } from "../types";
11
11
  import { CANDIDATE_TIMEOUT_MS } from "../types";
12
12
  import { rubricCriteria } from "../judgment";
13
+ import {
14
+ categoryKey,
15
+ categoryList,
16
+ rubricCategories,
17
+ } from "../criterion-categories";
13
18
  import { compileCodeJudge } from "../infra/judging/code-source";
19
+ import { readCodeCriteria } from "../infra/judging/code-criteria";
14
20
  import { RUNTIME_IMAGE } from "../infra/opencode/version";
15
21
  import { monitorPolicy } from "./monitor-policy";
16
22
  import {
17
23
  fingerprint,
18
- hash,
19
24
  relativePath,
20
25
  treeHash,
21
26
  contained,
@@ -35,10 +40,13 @@ export const modelRef = (value: unknown): ModelRef => {
35
40
  throw new Error(`Invalid model reference: ${String(value)}`);
36
41
  return value as ModelRef;
37
42
  };
43
+ /** Bun caches modules by path, ignoring URL queries; clear the entry so edits load. */
44
+ async function freshModule(path: string): Promise<Record<string, unknown>> {
45
+ delete require.cache[path];
46
+ return import(pathToFileURL(path).href);
47
+ }
38
48
  async function declaration<T>(path: string): Promise<T> {
39
- const url = pathToFileURL(path);
40
- url.searchParams.set("version", hash(await Bun.file(path).bytes()));
41
- return (await import(url.href)).default;
49
+ return (await freshModule(path)).default as T;
42
50
  }
43
51
  async function loadEval(directory: string): Promise<EvalDefinition> {
44
52
  const id = basename(directory),
@@ -168,10 +176,23 @@ async function loadEval(directory: string): Promise<EvalDefinition> {
168
176
  throw new Error(`${id}: preparation requires argv`);
169
177
  }
170
178
  const promptText = await prompt.text(),
171
- judgeText = hasMarkdown ? await judge.text() : "";
179
+ rubric = rubricCategories(hasMarkdown ? await judge.text() : ""),
180
+ judgeText = rubric.rubric;
172
181
  if (!promptText.trim() || (hasMarkdown && !judgeText.trim()))
173
182
  throw new Error(`${id}: prompt and judge must not be empty`);
174
183
  const code = hasCode ? await compileCodeJudge(codeFile) : undefined;
184
+ const criteria = hasMarkdown ? rubricCriteria(judgeText) : [];
185
+ const declared = hasCode ? await codeCriteria(id, codeFile) : [];
186
+ if (declared.some((item) => criteria.some(({ id }) => id === item.id)))
187
+ throw new Error(
188
+ `${id}: declare each criterion in either judge.md or judge.ts, not both`,
189
+ );
190
+ const categories = Object.fromEntries(
191
+ [
192
+ ...Object.entries(rubric.categories),
193
+ ...declared.map((item) => [item.id, item.categories] as const),
194
+ ].filter(([, names]) => names.length),
195
+ );
175
196
  return {
176
197
  id,
177
198
  directory,
@@ -183,12 +204,49 @@ async function loadEval(directory: string): Promise<EvalDefinition> {
183
204
  source,
184
205
  }),
185
206
  judgeHash: fingerprint({ rubric: judgeText, code: code?.hash }),
186
- criteria: hasMarkdown ? rubricCriteria(judgeText) : [],
207
+ criteria,
208
+ ...(declared.length
209
+ ? { codeCriteria: declared.map(({ id, name }) => ({ id, name })) }
210
+ : {}),
211
+ ...(Object.keys(categories).length ? { categories } : {}),
187
212
  ...(code ? { code } : {}),
188
213
  name: /^# (.+)$/m.exec(judgeText)?.[1] ?? id,
189
214
  };
190
215
  }
191
216
 
217
+ /** Reads judge.ts's optional `criteria` export as data; planning never executes judge code. */
218
+ async function codeCriteria(id: string, path: string) {
219
+ const declared = await readCodeCriteria(path);
220
+ if (declared === undefined) return [];
221
+ if (!declared || typeof declared !== "object" || Array.isArray(declared))
222
+ throw new Error(`${id}/judge.ts criteria must be an object`);
223
+ return Object.entries(declared).map(([criterion, value]) => {
224
+ const where = `${id}/judge.ts criterion ${criterion}`;
225
+ if (!/^[a-z][a-z0-9_]*$/.test(criterion))
226
+ throw new Error(`${where}: use a lowercase snake_case ID`);
227
+ if (
228
+ !value ||
229
+ typeof value !== "object" ||
230
+ Array.isArray(value) ||
231
+ Object.keys(value).some((key) => !["name", "categories"].includes(key))
232
+ )
233
+ throw new Error(`${where}: declare only name and categories`);
234
+ const { name, categories = [] } = value as {
235
+ name?: unknown;
236
+ categories?: unknown;
237
+ };
238
+ if (name !== undefined && (typeof name !== "string" || !name.trim()))
239
+ throw new Error(`${where}: name must be a non-empty string`);
240
+ if (!Array.isArray(categories))
241
+ throw new Error(`${where}: categories must be an array`);
242
+ return {
243
+ id: criterion,
244
+ name: (name as string | undefined)?.trim() ?? criterion.replaceAll("_", " "),
245
+ categories: categoryList(categories, where),
246
+ };
247
+ });
248
+ }
249
+
192
250
  export async function loadBenchmark(
193
251
  path: string,
194
252
  ): Promise<BenchmarkDefinition> {
@@ -214,10 +272,19 @@ export async function loadBenchmark(
214
272
  "concurrency",
215
273
  "candidate",
216
274
  "container",
275
+ "categories",
217
276
  ].includes(key),
218
277
  )
219
278
  )
220
279
  throw new Error("benchmark.ts contains unsupported settings");
280
+ if (
281
+ definition.categories !== undefined &&
282
+ (!Array.isArray(definition.categories) || !definition.categories.length)
283
+ )
284
+ throw new Error("benchmark.ts categories must be a non-empty array");
285
+ const categories = definition.categories
286
+ ? categoryList(definition.categories, "benchmark.ts")
287
+ : undefined;
221
288
  const models = definition.models.map(modelRef);
222
289
  if (
223
290
  definition.judge !== undefined &&
@@ -225,11 +292,11 @@ export async function loadBenchmark(
225
292
  typeof definition.judge !== "object" ||
226
293
  Array.isArray(definition.judge) ||
227
294
  Object.keys(definition.judge).some(
228
- (key) => !["model", "timeoutMs", "websearch"].includes(key),
295
+ (key) => !["model", "timeoutMs", "websearch", "verification"].includes(key),
229
296
  ))
230
297
  )
231
298
  throw new Error(
232
- "judge must be an object with model, timeoutMs, or websearch settings",
299
+ "judge must be an object with model, timeoutMs, websearch, or verification settings",
233
300
  );
234
301
  if (new Set(models).size !== models.length)
235
302
  throw new Error("Benchmark contains duplicate models");
@@ -245,9 +312,21 @@ export async function loadBenchmark(
245
312
  .filter((entry) => entry.isDirectory() && !entry.name.startsWith("."))
246
313
  .sort((a, b) => a.name.localeCompare(b.name));
247
314
  if (!directories.length) throw new Error("Benchmark has no eval folders");
248
- const evals = await Promise.all(
315
+ const declared = await Promise.all(
249
316
  directories.map((entry) => loadEval(resolve(evalRoot, entry.name))),
250
317
  );
318
+ const keys = categories?.map(categoryKey);
319
+ const evals = keys
320
+ ? declared.filter((item) =>
321
+ Object.values(item.categories ?? {}).some((names) =>
322
+ names.some((name) => keys.includes(categoryKey(name))),
323
+ ),
324
+ )
325
+ : declared;
326
+ if (!evals.length)
327
+ throw new Error(
328
+ `No criteria match the benchmark categories: ${categories!.join(", ")}`,
329
+ );
251
330
  if (evals.some((item) => item.judge) && !definition.judge?.model)
252
331
  throw new Error("A benchmark containing judge.md requires judge.model");
253
332
  const concurrency = positive(definition.concurrency, 10, "Concurrency");
@@ -258,6 +337,13 @@ export async function loadBenchmark(
258
337
  const engine = definition.container?.engine ?? "docker";
259
338
  if (engine !== "docker" && engine !== "podman")
260
339
  throw new Error("Container engine must be docker or podman");
340
+ const verification = definition.judge?.verification;
341
+ if (verification && (
342
+ typeof verification !== "object" || Array.isArray(verification) ||
343
+ Object.keys(verification).some(key => !["image", "engine", "cpus", "memoryMiB"].includes(key)) ||
344
+ typeof verification.image !== "string" || !verification.image.trim() || /\s/.test(verification.image) ||
345
+ !["docker", "podman"].includes(verification.engine ?? engine)
346
+ )) throw new Error("judge.verification requires an image and valid container limits");
261
347
  for (const search of [
262
348
  definition.candidate?.websearch,
263
349
  definition.judge?.websearch,
@@ -288,6 +374,12 @@ export async function loadBenchmark(
288
374
  "Judge timeout",
289
375
  ),
290
376
  websearch: definition.judge?.websearch ?? "exa",
377
+ ...(verification ? { verification: {
378
+ image: verification.image,
379
+ engine: verification.engine ?? engine,
380
+ cpus: positive(verification.cpus, 2, "Verification CPU count"),
381
+ memoryMiB: positive(verification.memoryMiB, 4096, "Verification memory"),
382
+ } } : {}),
291
383
  },
292
384
  container: {
293
385
  engine,
@@ -295,5 +387,6 @@ export async function loadBenchmark(
295
387
  cpus: positive(definition.container?.cpus, 2, "CPU count"),
296
388
  memoryMiB: positive(definition.container?.memoryMiB, 4096, "Memory"),
297
389
  },
390
+ ...(categories ? { categories } : {}),
298
391
  };
299
392
  }
@@ -0,0 +1,40 @@
1
+ import { mkdir, mkdtemp, rm } from "node:fs/promises";
2
+ import { resolve } from "node:path";
3
+ import { tmpdir } from "node:os";
4
+ import { loadBenchmark } from "./load-benchmark";
5
+ import { prepareWorkspace } from "../infra/containers/workspace";
6
+ import { CandidateContainer, inspectImage } from "../infra/containers/oci";
7
+ import { createSessionDatabase } from "../infra/opencode/host";
8
+ import { treeHash, writeJson } from "../infra/files";
9
+
10
+ /** Prove preparation without prompting a model, creating EvalRuns, or changing selections. */
11
+ export async function prepareInputs(path: string, options: { directory: string; onlyEvals?: readonly string[] }) {
12
+ const definition = await loadBenchmark(path);
13
+ if (options.onlyEvals?.some(id => !definition.evals.some(item => item.id === id)))
14
+ throw new Error("Select configured evals for preparation");
15
+ const directory = resolve(options.directory);
16
+ if (await Bun.file(resolve(directory, "prepared.json")).exists()) throw new Error("Choose a new prepared-input directory");
17
+ const imageId = await inspectImage(definition.container);
18
+ const scratch = await mkdtemp(resolve(process.platform === "win32" ? "C:/tmp/opencode" : tmpdir(), "openeval-prepare-"));
19
+ const inputs: Array<{ eval: string; directory: string; sourceHash: string; preparedHash: string }> = [];
20
+ try {
21
+ await mkdir(directory, { recursive: true });
22
+ for (const item of definition.evals.filter(item => !options.onlyEvals || options.onlyEvals.includes(item.id))) {
23
+ const staging = resolve(scratch, item.id);
24
+ await mkdir(staging, { recursive: true });
25
+ const workspace = resolve(staging, "workspace");
26
+ await prepareWorkspace(item, workspace);
27
+ const database = resolve(staging, "opencode.db");
28
+ // No credentials or model route are needed to prepare task inputs.
29
+ await createSessionDatabase(database, []);
30
+ await using container = await CandidateContainer.create(definition.container, imageId);
31
+ await container.prepare(workspace, database, definition.candidate.websearch, item.settings.prepare ?? [], staging, definition.candidate.providers);
32
+ const output = resolve(directory, item.id, "workspace");
33
+ await container.snapshot(output, staging);
34
+ inputs.push({ eval: item.id, directory: output, sourceHash: item.sourceHash, preparedHash: await treeHash(output) });
35
+ }
36
+ const result = { imageId, inputs, candidateExecutions: 0, judgeExecutions: 0 };
37
+ await writeJson(resolve(directory, "prepared.json"), result);
38
+ return result;
39
+ } finally { await rm(scratch, { recursive: true, force: true }); }
40
+ }