@hona/openeval 0.4.0 → 0.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (201) hide show
  1. package/JUDGING.md +6 -1
  2. package/README.md +2 -2
  3. package/VERIFICATION.md +105 -0
  4. package/package.json +2 -2
  5. package/src/app/grade-recording.ts +16 -0
  6. package/src/app/input-fingerprints.ts +24 -14
  7. package/src/app/judge-evidence.ts +8 -2
  8. package/src/app/judge-run.ts +10 -0
  9. package/src/app/load-benchmark.ts +15 -2
  10. package/src/app/plan-benchmark.ts +17 -0
  11. package/src/app/prepare-inputs.ts +40 -0
  12. package/src/app/read-results.ts +21 -4
  13. package/src/app/rejudge.ts +5 -0
  14. package/src/app/retry-run.ts +4 -1
  15. package/src/app/run-benchmark.ts +2 -1
  16. package/src/app/serve-results.ts +14 -0
  17. package/src/cli.ts +21 -2
  18. package/src/index.ts +8 -0
  19. package/src/infra/containers/oci.ts +4 -0
  20. package/src/infra/judging/code-result.ts +21 -3
  21. package/src/infra/judging/code-source.ts +10 -2
  22. package/src/infra/judging/code-worker.ts +5 -3
  23. package/src/infra/judging/code.ts +19 -3
  24. package/src/infra/judging/contract.ts +10 -2
  25. package/src/infra/judging/index.ts +1 -0
  26. package/src/infra/recording/index.ts +14 -2
  27. package/src/infra/verification/image.ts +33 -0
  28. package/src/infra/verification/runtime/Dockerfile +18 -0
  29. package/src/infra/verification/runtime/bun.lock +16 -0
  30. package/src/infra/verification/runtime/package.json +5 -0
  31. package/src/infra/verification/session.ts +199 -0
  32. package/src/judge-context.ts +66 -0
  33. package/src/types.ts +10 -2
  34. package/viewer/assets/{abnfDiagram-VCTEODGH-D-idCGaW.js → abnfDiagram-VCTEODGH-4dcAM__t.js} +1 -1
  35. package/viewer/assets/{angular-html-B-7vkhmj.js → angular-html-DCa1K9Q5.js} +1 -1
  36. package/viewer/assets/{angular-ts-CqJWLTIZ.js → angular-ts-BwOaP4mi.js} +1 -1
  37. package/viewer/assets/{apl-DVTjqAhB.js → apl-BDVbqe8d.js} +1 -1
  38. package/viewer/assets/{arc-Bmz8zsvq.js → arc-vf_TPbdA.js} +1 -1
  39. package/viewer/assets/architecture-7GRP2DOG-ClUScBK0.js +1 -0
  40. package/viewer/assets/{architectureDiagram-5GKGNRK7-CxqP2ijS.js → architectureDiagram-5GKGNRK7-D9I5g97H.js} +1 -1
  41. package/viewer/assets/{astro-Ih6QH4I8.js → astro-C4AVO9I1.js} +1 -1
  42. package/viewer/assets/{blade-vzkWgA61.js → blade-1K2d7b5P.js} +1 -1
  43. package/viewer/assets/{blockDiagram-I7D4REHJ-Dm35S95Q.js → blockDiagram-I7D4REHJ-CnebaxAP.js} +1 -1
  44. package/viewer/assets/{c-092Q-y5e.js → c-CrFxx71c.js} +1 -1
  45. package/viewer/assets/{c4Diagram-7LVT6UL2-DArgyJNC.js → c4Diagram-7LVT6UL2-CZv4bp92.js} +1 -1
  46. package/viewer/assets/channel-DOp5Ibzq.js +1 -0
  47. package/viewer/assets/{chapel-DUOt2X3X.js → chapel-CaV_aD0h.js} +1 -1
  48. package/viewer/assets/{chunk-4HAMMTFA-Bm3UCwQ6.js → chunk-4HAMMTFA-1oAN0Hqj.js} +1 -1
  49. package/viewer/assets/{chunk-75Z2AOVW-CpmjsCJX.js → chunk-75Z2AOVW-DG-4lG9j.js} +1 -1
  50. package/viewer/assets/{chunk-DU6HZSFF-DYT2KEwe.js → chunk-DU6HZSFF-K8r4K98Q.js} +1 -1
  51. package/viewer/assets/{chunk-F27PBJKO-BFGd2OPj.js → chunk-F27PBJKO-DOLcDIe-.js} +1 -1
  52. package/viewer/assets/{chunk-GMAD6QVW-D8gzxWqz.js → chunk-GMAD6QVW-fbOr8e3A.js} +1 -1
  53. package/viewer/assets/{chunk-GVQU2GXP-BZJsi-wS.js → chunk-GVQU2GXP-DGskHf86.js} +1 -1
  54. package/viewer/assets/{chunk-IMKFNOWR-BKbF8JvM.js → chunk-IMKFNOWR-BkiL77Of.js} +1 -1
  55. package/viewer/assets/{chunk-L3NEJ4N5-SWKb6tCy.js → chunk-L3NEJ4N5-rHC84FEy.js} +1 -1
  56. package/viewer/assets/{chunk-OSK3NFVY-BrqoFOmR.js → chunk-OSK3NFVY-CDEFeiSo.js} +1 -1
  57. package/viewer/assets/{chunk-P2QGCYS3-B10Cyxbx.js → chunk-P2QGCYS3-BHNuVs3t.js} +1 -1
  58. package/viewer/assets/{chunk-POPQ4Y6H-DXpfBPGQ.js → chunk-POPQ4Y6H-8IU6HL0N.js} +1 -1
  59. package/viewer/assets/{chunk-PWAF6VOD-DZmaXnfr.js → chunk-PWAF6VOD--FR_J-_V.js} +1 -1
  60. package/viewer/assets/{chunk-SHT3W25Y-1U6zFGxP.js → chunk-SHT3W25Y-DHfu0YQd.js} +1 -1
  61. package/viewer/assets/{chunk-SVP7TREG-CaW6KcpM.js → chunk-SVP7TREG-CZxnTp0N.js} +1 -1
  62. package/viewer/assets/{chunk-TICWLB2K-X4pj7G4S.js → chunk-TICWLB2K-DeYv6m7z.js} +1 -1
  63. package/viewer/assets/{chunk-XXDRQBXY-0afAM3OP.js → chunk-XXDRQBXY-B-stp2Jg.js} +1 -1
  64. package/viewer/assets/classDiagram-ZZMXUADV-fi0_kah3.js +1 -0
  65. package/viewer/assets/classDiagram-v2-VYDZK3BY-fi0_kah3.js +1 -0
  66. package/viewer/assets/{cobol-CtspwbZ4.js → cobol-Dfqw6e9L.js} +1 -1
  67. package/viewer/assets/{coffee-qbHB9gj8.js → coffee-B0VyEiZ_.js} +1 -1
  68. package/viewer/assets/{cose-bilkent-JH36ORCC-7s0vcDw1.js → cose-bilkent-JH36ORCC-DwO-FoMO.js} +1 -1
  69. package/viewer/assets/{cpp-Dx37X5A_.js → cpp-DtAnLTpm.js} +1 -1
  70. package/viewer/assets/{crystal-CiWjJvk4.js → crystal-DR276aYT.js} +1 -1
  71. package/viewer/assets/{css-CLeuyJyb.js → css-jcQdyKCB.js} +1 -1
  72. package/viewer/assets/{cynefin-OW5HDTMX-B4XFGNNZ.js → cynefin-OW5HDTMX-CkCcLfFW.js} +1 -1
  73. package/viewer/assets/{cynefinDiagram-5FMLGOSQ-DdNMKbw6.js → cynefinDiagram-5FMLGOSQ-Btlq2-ol.js} +1 -1
  74. package/viewer/assets/{dagre-GXQ25YYZ-BkDo4A91.js → dagre-GXQ25YYZ-BtEJZ3id.js} +1 -1
  75. package/viewer/assets/{diagram-S7CK7UJ4-DU8cnnVy.js → diagram-S7CK7UJ4-BdLily-u.js} +1 -1
  76. package/viewer/assets/{diagram-UQ7AKVKN-Dg2emYKa.js → diagram-UQ7AKVKN-DBfD46yQ.js} +1 -1
  77. package/viewer/assets/{diagram-VSXAHHWV-C_-YSzcn.js → diagram-VSXAHHWV-BJuAmu9T.js} +1 -1
  78. package/viewer/assets/{diagram-VX7I27RA-BLCEnxqe.js → diagram-VX7I27RA-BVEEYjj-.js} +1 -1
  79. package/viewer/assets/{diagram-Z3DM3KII-62CpIiAg.js → diagram-Z3DM3KII-DqYpuagj.js} +1 -1
  80. package/viewer/assets/{dist-CQIgbdem.js → dist-B0PK9u1q.js} +1 -1
  81. package/viewer/assets/{ebnfDiagram-PWID7BFC-FkGGETgC.js → ebnfDiagram-PWID7BFC-DafUAY97.js} +1 -1
  82. package/viewer/assets/{edge-CzmGSXHr.js → edge-CMwzwqao.js} +1 -1
  83. package/viewer/assets/{elixir-DDfD_-oS.js → elixir-D0m0fb5e.js} +1 -1
  84. package/viewer/assets/{elm-l7aRfQIq.js → elm-Dm3Vhwyo.js} +1 -1
  85. package/viewer/assets/{erDiagram-RLTQ6QDP-GlCpsS1v.js → erDiagram-RLTQ6QDP-BjTQFP-n.js} +1 -1
  86. package/viewer/assets/{erb-qVffN5xM.js → erb-BWrVRqIb.js} +1 -1
  87. package/viewer/assets/eventmodeling-NTZA5JFV-QswxRol8.js +1 -0
  88. package/viewer/assets/flowDiagram-HODETNUW-DolIsASs.js +1 -0
  89. package/viewer/assets/{ganttDiagram-EL5Y4UJY-CH_Tk4Ex.js → ganttDiagram-EL5Y4UJY-cvpUkVwQ.js} +1 -1
  90. package/viewer/assets/{git-rebase-jEJ8nKk1.js → git-rebase-Bx9WSw2T.js} +1 -1
  91. package/viewer/assets/{gitGraph-4MIJSDKK-Cc9o1l4Y.js → gitGraph-4MIJSDKK-YF9cEiQc.js} +1 -1
  92. package/viewer/assets/{gitGraphDiagram-WWUBYQGX-UdfHBATc.js → gitGraphDiagram-WWUBYQGX-Cf7LUlAN.js} +1 -1
  93. package/viewer/assets/{glimmer-js-BMk2JCW8.js → glimmer-js-jJ5_T2M9.js} +1 -1
  94. package/viewer/assets/{glimmer-ts-BH7RwQxU.js → glimmer-ts-DqUZ7xnF.js} +1 -1
  95. package/viewer/assets/{glsl-CAWrZoEq.js → glsl-DHzsK5Yf.js} +1 -1
  96. package/viewer/assets/{graphql-Bac_hecs.js → graphql-TivQ3Vbm.js} +1 -1
  97. package/viewer/assets/{hack-1OgpAtm-.js → hack-CXzUKlVo.js} +1 -1
  98. package/viewer/assets/{haml-CvwG3BzW.js → haml-DB9hsLVR.js} +1 -1
  99. package/viewer/assets/{handlebars-Cq4oEEp8.js → handlebars-Dfxd-Vhz.js} +1 -1
  100. package/viewer/assets/{html-D5wZ_JR6.js → html-CwO_HZk6.js} +1 -1
  101. package/viewer/assets/{html-derivative-ZfzPj0aS.js → html-derivative-Da9QhqUr.js} +1 -1
  102. package/viewer/assets/{http-BDE2UDvO.js → http-DaNELgRt.js} +1 -1
  103. package/viewer/assets/{hurl-CQbXNLlk.js → hurl-DCibLrD6.js} +1 -1
  104. package/viewer/assets/index-Cc4pf8mQ.js +795 -0
  105. package/viewer/assets/{index-OL_rrWNy.css → index-DVYWRkK6.css} +1 -1
  106. package/viewer/assets/{info-A6RAGUB7-YD13oMfx.js → info-A6RAGUB7-B7Ru1Oky.js} +1 -1
  107. package/viewer/assets/{infoDiagram-27XIBGKW-jYKjA_l7.js → infoDiagram-27XIBGKW-RlWd2kYA.js} +1 -1
  108. package/viewer/assets/{ishikawaDiagram-5VMMS53U-BBPzHiVd.js → ishikawaDiagram-5VMMS53U-CS2-lnGi.js} +1 -1
  109. package/viewer/assets/{java-Cbpu4oyT.js → java-D2z6ozlO.js} +1 -1
  110. package/viewer/assets/{javascript-UMuq64YD.js → javascript-CxVsIZNi.js} +1 -1
  111. package/viewer/assets/{jinja-DVvZtqgU.js → jinja-BSUcBlS_.js} +1 -1
  112. package/viewer/assets/{jison-CZCXBIV3.js → jison-qcnqXcSy.js} +1 -1
  113. package/viewer/assets/{journeyDiagram-3NMN7TZE-CMqG5ndb.js → journeyDiagram-3NMN7TZE-DqPSRu4-.js} +1 -1
  114. package/viewer/assets/{json-CzTvWngu.js → json-eiyev3x0.js} +1 -1
  115. package/viewer/assets/{jsx-CNJnEGR4.js → jsx-rDHm7tH3.js} +1 -1
  116. package/viewer/assets/{julia-Bj0q4uoH.js → julia-cFfLusXt.js} +1 -1
  117. package/viewer/assets/{just-93-MRGUC.js → just-D6wClmab.js} +1 -1
  118. package/viewer/assets/{kanban-definition-UXKFOSKX-VyFYPq9b.js → kanban-definition-UXKFOSKX-BI_a5cRc.js} +1 -1
  119. package/viewer/assets/{latex-CXd1tMjA.js → latex-BvnFBRMG.js} +1 -1
  120. package/viewer/assets/{line-D5nvVDHp.js → line-ChPI9Rz5.js} +1 -1
  121. package/viewer/assets/{linear-Cq-FJZ_z.js → linear-CBr5H_et.js} +1 -1
  122. package/viewer/assets/{liquid-Cb-ALXSr.js → liquid-CIOCL1uY.js} +1 -1
  123. package/viewer/assets/{lua-BxTQiamn.js → lua-Dq6WuAHP.js} +1 -1
  124. package/viewer/assets/{marko-mFYyU5jn.js → marko-Nl8_Nklj.js} +1 -1
  125. package/viewer/assets/{mdc-Oryqox7_.js → mdc-obW39E44.js} +1 -1
  126. package/viewer/assets/{mermaid-parser.core-CCDsanlH.js → mermaid-parser.core-BFOx64cQ.js} +3 -3
  127. package/viewer/assets/{mermaid.core-Crd1YMgi.js → mermaid.core-BDe6Tfbh.js} +4 -4
  128. package/viewer/assets/{mindmap-definition-YA3MSWOX-BKLrEFWt.js → mindmap-definition-YA3MSWOX-CjzA1VTL.js} +1 -1
  129. package/viewer/assets/{nginx-C1PwWP7d.js → nginx-Czz1mxQD.js} +1 -1
  130. package/viewer/assets/{nim-PszXW-l4.js → nim-Boz13cJ3.js} +1 -1
  131. package/viewer/assets/{org-DO0CuJlO.js → org-BOlhXpYw.js} +1 -1
  132. package/viewer/assets/{packet-AYTQ26CC-BNPwLE6g.js → packet-AYTQ26CC-MMkKPhzF.js} +1 -1
  133. package/viewer/assets/{pegDiagram-XKGWAZYB-DjBg_iPh.js → pegDiagram-XKGWAZYB-q1Rf8_Y9.js} +1 -1
  134. package/viewer/assets/{perl-CHhAgoDL.js → perl-BA7VmyiG.js} +1 -1
  135. package/viewer/assets/{php-CA4H6qnu.js → php-dYlaksWx.js} +1 -1
  136. package/viewer/assets/{pie-WAS4IAKB-BzNmHX-Y.js → pie-WAS4IAKB-CLljEA6l.js} +1 -1
  137. package/viewer/assets/{pieDiagram-E7YTZNPT-cgMB-fBw.js → pieDiagram-E7YTZNPT-DqzGIO9H.js} +1 -1
  138. package/viewer/assets/{pug-DDuKTe7C.js → pug-WDsqiRUK.js} +1 -1
  139. package/viewer/assets/{qml-DSd2VypD.js → qml-B5P0chYn.js} +1 -1
  140. package/viewer/assets/{quadrantDiagram-AXDQQJYC-Crzhs7Np.js → quadrantDiagram-AXDQQJYC-JV3sYNXt.js} +1 -1
  141. package/viewer/assets/{r-qf-5qR5Q.js → r-DmZfBDub.js} +1 -1
  142. package/viewer/assets/{radar-RG4KPBEZ-B3oI70pB.js → radar-RG4KPBEZ-DH4HiGh1.js} +1 -1
  143. package/viewer/assets/{railroad-74A4TZTK-DOM5Od17.js → railroad-74A4TZTK-BwJg41V9.js} +1 -1
  144. package/viewer/assets/railroad-abnf-HS5TGJTU-Byryy4k9.js +1 -0
  145. package/viewer/assets/railroad-ebnf-LZEXJU2U-BABBdJeJ.js +1 -0
  146. package/viewer/assets/railroad-peg-WCYAUIDC-C7YNbOxe.js +1 -0
  147. package/viewer/assets/{railroadDiagram-O6MQD6OU-M3SvSYf-.js → railroadDiagram-O6MQD6OU-CrclSzRj.js} +1 -1
  148. package/viewer/assets/{razor-B9n9DtIL.js → razor-BLQfBppW.js} +1 -1
  149. package/viewer/assets/{regexp-DhGN0EOR.js → regexp-CePe_vV5.js} +1 -1
  150. package/viewer/assets/{requirementDiagram-BXWQKSXE-CFisiZTo.js → requirementDiagram-BXWQKSXE-5QlkAVAN.js} +1 -1
  151. package/viewer/assets/{rst-D1SdhuXd.js → rst-D9pCulha.js} +1 -1
  152. package/viewer/assets/{ruby-BybsgZgf.js → ruby-r3EAH5Rz.js} +1 -1
  153. package/viewer/assets/{sankeyDiagram-P5KCCOFB-BKf2TMWg.js → sankeyDiagram-P5KCCOFB-BJzQ5vkn.js} +1 -1
  154. package/viewer/assets/{sas-Io7QwCtD.js → sas-Dxc2YEvn.js} +1 -1
  155. package/viewer/assets/{scss-D6XY0yo3.js → scss-rZd3d-ZI.js} +1 -1
  156. package/viewer/assets/{sequenceDiagram-WJ2MYXX4-CHawNQZ9.js → sequenceDiagram-WJ2MYXX4--tS5gMY4.js} +1 -1
  157. package/viewer/assets/{shellscript-DSk8kvCh.js → shellscript-B-b5xnkP.js} +1 -1
  158. package/viewer/assets/{shellsession-Be1CN8DO.js → shellsession-DlWXnhiW.js} +1 -1
  159. package/viewer/assets/{soy-CGkkqGyT.js → soy-DXWgq6qD.js} +1 -1
  160. package/viewer/assets/{sql-DIgb716V.js → sql-5vPpIo1-.js} +1 -1
  161. package/viewer/assets/{src-DP6Z6M4G.js → src-eDHFRM9C.js} +1 -1
  162. package/viewer/assets/{stata-CS_2tb9p.js → stata-BPQLlmzg.js} +1 -1
  163. package/viewer/assets/{stateDiagram-D77RDMKH-Cx9Rf6fY.js → stateDiagram-D77RDMKH-xapk0kUZ.js} +1 -1
  164. package/viewer/assets/stateDiagram-v2-MP3YSRHH-Cb8zYuMi.js +1 -0
  165. package/viewer/assets/{surrealql-Do4o8w1B.js → surrealql-CTEIAZfm.js} +1 -1
  166. package/viewer/assets/{svelte-3bKxeV6-.js → svelte-B6EVsDV1.js} +1 -1
  167. package/viewer/assets/{swimlanes-42K2YHIH-yL2oIw_D.js → swimlanes-42K2YHIH-BLyG_o3-.js} +1 -1
  168. package/viewer/assets/swimlanesDiagram-VR7AAH4N-RLMdNA3S.js +8 -0
  169. package/viewer/assets/{templ-BgEYPNGP.js → templ-Bu4quHqE.js} +1 -1
  170. package/viewer/assets/{tex-D_p6whQw.js → tex-Dv_54fPG.js} +1 -1
  171. package/viewer/assets/{timeline-definition-24CTP7MA-BJeLq4a5.js → timeline-definition-24CTP7MA-DBaZM3qq.js} +1 -1
  172. package/viewer/assets/{treeView-Q6P3EWNA-COcnImiE.js → treeView-Q6P3EWNA-C3zH0Rwt.js} +1 -1
  173. package/viewer/assets/{treemap-WGGIJYW6-DpbxWots.js → treemap-WGGIJYW6-BTho8OjG.js} +1 -1
  174. package/viewer/assets/{ts-tags-GBmMk7Oh.js → ts-tags-0DHV6Smm.js} +1 -1
  175. package/viewer/assets/{tsx-BSD-yNhi.js → tsx-CmZ6PB1U.js} +1 -1
  176. package/viewer/assets/{twig-dwAG5SzX.js → twig-u99yWuG6.js} +1 -1
  177. package/viewer/assets/{typescript-B8YTPl9v.js → typescript-WF79ZIBQ.js} +1 -1
  178. package/viewer/assets/{typst-Dpzojbzo.js → typst-DQ7XxvRZ.js} +1 -1
  179. package/viewer/assets/{vennDiagram-4TSXK5OY-nF34Gjtn.js → vennDiagram-4TSXK5OY-B5Cb69qm.js} +1 -1
  180. package/viewer/assets/{vue-Djbmk2DC.js → vue-DXac3mhj.js} +1 -1
  181. package/viewer/assets/{vue-html-BXc5rfco.js → vue-html-FSSTYK94.js} +1 -1
  182. package/viewer/assets/{vue-vine-DHCfQh17.js → vue-vine-BDF8epqI.js} +1 -1
  183. package/viewer/assets/{wardley-WFR3VGLG-C1qhMifc.js → wardley-WFR3VGLG-DswcXSmz.js} +1 -1
  184. package/viewer/assets/{wardleyDiagram-VM6X3IG4-B0BfyYWQ.js → wardleyDiagram-VM6X3IG4-BPgNi0Kp.js} +1 -1
  185. package/viewer/assets/{xml-CdCEskcV.js → xml-HI856iuk.js} +1 -1
  186. package/viewer/assets/{xsl-BbNukwXP.js → xsl-Ck-aKGwi.js} +1 -1
  187. package/viewer/assets/{xychartDiagram-S5SC5T6Z-DJKYce1U.js → xychartDiagram-S5SC5T6Z-B8UAJr6c.js} +1 -1
  188. package/viewer/assets/{yaml-BYLDET4A.js → yaml-DrPyW5uV.js} +1 -1
  189. package/viewer/index.html +2 -2
  190. package/viewer/assets/architecture-7GRP2DOG-D3Kw34KW.js +0 -1
  191. package/viewer/assets/channel-D2lVv6ST.js +0 -1
  192. package/viewer/assets/classDiagram-ZZMXUADV-DqBNvABf.js +0 -1
  193. package/viewer/assets/classDiagram-v2-VYDZK3BY-DqBNvABf.js +0 -1
  194. package/viewer/assets/eventmodeling-NTZA5JFV-07UfGVCm.js +0 -1
  195. package/viewer/assets/flowDiagram-HODETNUW-CyIUf7TW.js +0 -1
  196. package/viewer/assets/index-BezQIu6a.js +0 -795
  197. package/viewer/assets/railroad-abnf-HS5TGJTU-ScZ2h_5Q.js +0 -1
  198. package/viewer/assets/railroad-ebnf-LZEXJU2U-_Cl5zxJ-.js +0 -1
  199. package/viewer/assets/railroad-peg-WCYAUIDC-CRlGg7vV.js +0 -1
  200. package/viewer/assets/stateDiagram-v2-MP3YSRHH-DRb-6t_S.js +0 -1
  201. package/viewer/assets/swimlanesDiagram-VR7AAH4N-cDiCT6xi.js +0 -8
package/JUDGING.md CHANGED
@@ -2,7 +2,7 @@
2
2
 
3
3
  Use the [canonical terminology](https://openev.al/docs/terminology/): a criterion
4
4
  is a named graded requirement, a score is awarded credit, and a metric is a
5
- measurement such as token count or cost. OpenEval 0.3.0 supports code judges,
5
+ measurement such as token count or cost. OpenEval supports code judges,
6
6
  LLM judges, and additive use of both against one recorded EvalRun.
7
7
 
8
8
  ## File conventions
@@ -36,6 +36,9 @@ Only the optional scores object has grading semantics. Each key is a criterion
36
36
  ID, using lowercase letters, digits, and underscores, starting with a letter.
37
37
  Values are booleans, finite numbers from 0 to 1, or null. The host converts true
38
38
  to 1 and false to 0. It rejects invalid values rather than clamping them.
39
+ Optional `{ value, reason, evidence, measurements }` objects make code verdicts
40
+ readable without changing their scoring semantics. Declared code criterion IDs
41
+ must match the returned scores. See [artifact verification](VERIFICATION.md).
39
42
  response.text is always a string; missing text becomes an empty string. The
40
43
  original execution outcome is available separately on context.run.
41
44
 
@@ -69,6 +72,8 @@ a JudgeRun error. A hybrid judgment is selected only after both sources succeed.
69
72
  | recording.export(sessionID?) | Native OpenCode session export |
70
73
  | workspace.files/read/text/diff | Verified initial and final file snapshots |
71
74
  | workspace.materialize(revision?) | A disposable workspace copy for author-owned verification |
75
+ | verification.run(request) | Bounded commands over a restored artifact in an isolated OCI container |
76
+ | verification.read/text(result, path) | Hash-checked retained output from that verification |
72
77
  | native.database() | Read-only SQLite access to a verified database copy |
73
78
  | native.sdk() | The pinned OpenCode SDK/API over a separate disposable archive copy |
74
79
  | native.schema() | The pinned OpenCode schema module |
package/README.md CHANGED
@@ -87,8 +87,8 @@ or does not provide a parameterized query.
87
87
  | Required recording is unavailable | **null** | **null** |
88
88
 
89
89
  Add an optional `Categories: misalignment` line under a heading to compare models
90
- by category in the viewer's radar chart and heatmap. Categories never trigger
91
- rejudging.
90
+ by category in the viewer's radar chart, heatmap, and benchmark scorecard. Export
91
+ the scorecard as a PNG from the results page. Categories never trigger rejudging.
92
92
 
93
93
  → [Write good rubrics](https://openev.al/docs/rubrics/) · [Download the SQL starter](https://openev.al/starter.zip)
94
94
 
@@ -0,0 +1,105 @@
1
+ # Artifact verification
2
+
3
+ Code judges are still plain functions. When the score depends on delivered code,
4
+ run it in a disposable OCI container rather than on the runner host.
5
+
6
+ ```ts
7
+ // benchmark.ts
8
+ import { VERIFICATION_IMAGE, type Benchmark } from "@hona/openeval";
9
+ export default {
10
+ models: ["example/model"],
11
+ judge: { verification: { image: VERIFICATION_IMAGE, cpus: 2, memoryMiB: 4096 } },
12
+ } satisfies Benchmark;
13
+ ```
14
+
15
+ Build the candidate and standard verification images with `openeval image`.
16
+ `openeval image --verification` builds only the verification image. Custom images
17
+ must be built separately. Images resolve to immutable IDs before planning or
18
+ collection; a changed verification image invalidates judgment reuse, not the
19
+ candidate's delivered artifacts.
20
+
21
+ ```ts
22
+ // judge.ts — trusted script content is an ordinary frozen local import/string.
23
+ import type { JudgeContext } from "@hona/openeval";
24
+ export const criteria = { correct: { name: "Correct output", categories: ["coding"] } };
25
+ export default async (ctx: JudgeContext) => {
26
+ const result = await ctx.verification.run({
27
+ revision: "final", cwd: ".", timeoutMs: 60_000,
28
+ commands: [["bun", "/verification/check.mjs"]],
29
+ files: { "check.mjs": "/* author's outcome checks */" },
30
+ artifacts: ["checks.json", "screenshot.png"],
31
+ });
32
+ return { scores: { correct: {
33
+ value: result.state === "completed" && result.exitCode === 0,
34
+ reason: "Explain what the checks established or why they failed.",
35
+ evidence: [{ kind: "verification", id: result.id }],
36
+ measurements: { elapsedMs: result.elapsedMs },
37
+ } } };
38
+ };
39
+ ```
40
+
41
+ The primitive restores an initial or final snapshot, uploads trusted inputs to
42
+ `/verification`, runs argv arrays in order, and stops at a nonzero exit. It does
43
+ not choose scoring thresholds or interpret success. Judge setup/transfer/image
44
+ errors remain JudgeRun errors. A test exit or verification timeout is an
45
+ observation the authored criterion must interpret. Interrupted candidate work
46
+ still needs the rubric's evidence policy; a failed verification of an unfinished
47
+ prefix does not necessarily establish final-task failure.
48
+
49
+ ## Isolation and bounds
50
+
51
+ - No host bind mounts, credentials, published ports, or network. Browser/server
52
+ verification can use loopback **inside** the container.
53
+ - Read-only image filesystem, non-root command user, no capabilities, no privilege
54
+ escalation, PID/CPU/memory limits, bounded workspace/temp storage, and a deadline.
55
+ - Trusted check inputs are root-owned and are not writable by delivered code.
56
+ - Commands, exit statuses, image/resource identity, logs, and requested regular
57
+ output files are retained with the JudgeRun. Viewer previews never execute HTML
58
+ or SVG from the delivered artifact.
59
+ - Workspace transfer is limited to 128 MiB; each output archive is limited to
60
+ 32 MiB; logs are limited to 1 MiB per stream and are marked when truncated.
61
+ - Containers self-expire. The parent also removes containers bearing only its
62
+ unique execution label after worker interruption. No shared/broad cleanup.
63
+
64
+ The standard image includes Bun 1.4.2, Python, Git, Node, and Playwright 1.63.0
65
+ with Chromium. Import Playwright from
66
+ `/opt/verify/node_modules/playwright/index.mjs`. Dependencies needed by a task
67
+ must be available in the image or supplied as frozen inputs; network installs
68
+ are intentionally unavailable. Materialization alone is not a sandbox.
69
+ `verification.text(result, "stdout" | "stderr")` reads the retained bounded logs;
70
+ those names are reserved and cannot also name output artifacts.
71
+
72
+ The primitive verifies a reconstructed artifact. It does **not** prove what was
73
+ alive in the original candidate container, whether the candidate ran a test,
74
+ or whether original game actions were legal. Those claims need original recording
75
+ evidence. Do not put evaluator scripts in candidate workspaces.
76
+
77
+ ## Readable judgments and controls
78
+
79
+ `openeval prepare --output <new-directory>` assembles real candidate inputs and
80
+ runs declared preparation in the candidate image, then archives the prepared
81
+ workspace. `--only-eval` scopes it. This makes zero model calls, creates no
82
+ EvalRun/JudgeRun, and does not change selections. Use it to prove fixture setup
83
+ before collection; it does not prove agent success or human task duration.
84
+
85
+ Existing boolean, numeric, and null scores remain sufficient. Optional structured
86
+ scores add `reason`, `evidence`, and JSON `measurements`. These details appear in
87
+ the normal viewer beside verification receipts; arbitrary returned JSON remains
88
+ inspectable. Declared code criteria must be returned exactly. Missing evidence
89
+ references are judging errors, not candidate zeros.
90
+ Arrays of named check observations render as readable tables, including author
91
+ supplied pass/fail details. They do not introduce additional scores or weights.
92
+ Large tables are explicitly bounded in the display; full returned JSON is retained.
93
+
94
+ `recordEvidence({ workspace: { initial, final }, ... })` can retain constructed
95
+ artifact controls without executing them. `judgeEvidence` then uses the same
96
+ public verification primitive. Constructed controls prove tested boundaries, not
97
+ human-duration calibration or live candidate feasibility.
98
+
99
+ Unused `criteria` labels/categories are removed from executable bundling. Debug
100
+ source maps remain retained but do not affect code identity. If grading reads a
101
+ metadata value, it is executable behavior and remains fingerprinted. Changes to
102
+ criterion IDs, actual grading code, imported inputs, dependencies, or verification
103
+ environment are not reporting-only edits. The verification environment is part
104
+ of judge identity only for evals with `judge.ts`. Rebuilding the image does not
105
+ rejudge Markdown-only evals.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hona/openeval",
3
- "version": "0.4.0",
3
+ "version": "0.5.1",
4
4
  "description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -14,7 +14,7 @@
14
14
  "bin": {
15
15
  "openeval": "src/cli.ts"
16
16
  },
17
- "files": ["src", "viewer", "LICENSE", "README.md", "JUDGING.md", "!src/**/*.test.ts"],
17
+ "files": ["src", "viewer", "LICENSE", "README.md", "JUDGING.md", "VERIFICATION.md", "!src/**/*.test.ts"],
18
18
  "publishConfig": {
19
19
  "access": "public"
20
20
  },
@@ -11,6 +11,8 @@ import { executeCodeJudge } from "../infra/judging/code";
11
11
  import { codeJudgment, combineJudgments } from "../infra/judging/code-result";
12
12
  import { executeJudge } from "../infra/judging";
13
13
  import { errorMessage } from "../infra/files";
14
+ import { CandidateEvidence } from "../infra/evidence";
15
+ import { validateCitations } from "../infra/judging/contract";
14
16
 
15
17
  export type GradedRecording = {
16
18
  state: "completed" | "failed" | "timed_out";
@@ -37,10 +39,24 @@ export async function gradeRecording(
37
39
  recording,
38
40
  resolve(directory, "code"),
39
41
  input.timeoutMs,
42
+ input.verification,
40
43
  );
41
44
  if (code.state !== "completed")
42
45
  return { state: code.state, code, error: code.error };
43
46
  const judged = codeJudgment(code.output!);
47
+ if (input.codeCriteria?.length && (
48
+ Object.keys(judged.scores).length !== input.codeCriteria.length ||
49
+ input.codeCriteria.some(item => !Object.hasOwn(judged.scores, item.id))
50
+ )) throw new Error("judge.ts must return exactly its declared criterion IDs");
51
+ for (const score of Object.values(judged.scores)) for (const citation of score.evidence) {
52
+ if (citation.kind !== "verification") continue;
53
+ const result = code.verifications?.find(item => item.id === citation.id);
54
+ if (!result || (citation.path && !result.artifacts.some(item => item.path === citation.path)))
55
+ throw new Error("Code judgment cites missing verification evidence");
56
+ }
57
+ await validateCitations({ ...judged, scores: Object.fromEntries(Object.entries(judged.scores).map(([id, score]) =>
58
+ [id, { ...score, evidence: score.evidence.filter(citation => citation.kind !== "verification") }])) },
59
+ await CandidateEvidence.open(recording.evidence.directory, recording.evidence.hash));
44
60
  for (const id of Object.keys(judged.scores))
45
61
  if (input.criteria.some((criterion) => criterion.id === id))
46
62
  throw new Error(
@@ -43,7 +43,8 @@ export function candidateProviders(
43
43
  : undefined;
44
44
  }
45
45
 
46
- /** Declared, candidate-visible inputs. Harness implementation hashes are provenance. */
46
+ /** Declared, candidate-visible inputs. Harness implementation hashes are provenance.
47
+ * Early stopping is a judge-side policy; the planner re-collects sessions it cut short. */
47
48
  export function candidateFingerprint(
48
49
  definition: BenchmarkDefinition,
49
50
  evalId: string,
@@ -69,21 +70,24 @@ export function candidateFingerprint(
69
70
  },
70
71
  });
71
72
  }
73
+ /** Only judge.ts can run checks, so the verification runtime never changes an LLM-only judgment. */
72
74
  export const judgeFingerprint = (
73
75
  definition: BenchmarkDefinition,
74
76
  evalId: string,
75
- ) =>
76
- fingerprint({
77
- rubric: definition.evals.find((item) => item.id === evalId)!.judge,
78
- code: definition.evals.find((item) => item.id === evalId)!.code?.hash,
79
- agent: definition.evals.find((item) => item.id === evalId)!.judge
80
- ? JUDGE_AGENT
81
- : undefined,
82
- judge: definition.evals.find((item) => item.id === evalId)!.judge
83
- ? definition.judge
84
- : { timeoutMs: definition.judge.timeoutMs },
77
+ ) => {
78
+ const item = definition.evals.find((item) => item.id === evalId)!;
79
+ const verification = item.code ? definition.judge.verification : undefined;
80
+ return fingerprint({
81
+ rubric: item.judge,
82
+ code: item.code?.hash,
83
+ codeIds: item.codeCriteria?.map((criterion) => criterion.id).sort(),
84
+ agent: item.judge ? JUDGE_AGENT : undefined,
85
+ judge: item.judge
86
+ ? { ...definition.judge, verification }
87
+ : { timeoutMs: definition.judge.timeoutMs, verification },
85
88
  protocol: JUDGE_PROTOCOL,
86
89
  });
90
+ };
87
91
  export const savedJudgeFingerprint = (
88
92
  input: Pick<
89
93
  JudgeRunInput,
@@ -94,11 +98,15 @@ export const savedJudgeFingerprint = (
94
98
  | "timeoutMs"
95
99
  | "websearch"
96
100
  | "code"
101
+ | "codeCriteria"
102
+ | "verification"
97
103
  >,
98
- ) =>
99
- fingerprint({
104
+ ) => {
105
+ const verification = input.code ? input.verification : undefined;
106
+ return fingerprint({
100
107
  rubric: input.rubric,
101
108
  code: input.code?.hash,
109
+ codeIds: input.codeCriteria?.map((criterion) => criterion.id).sort(),
102
110
  agent: input.agent,
103
111
  protocol: input.protocol,
104
112
  judge: input.rubric
@@ -106,6 +114,8 @@ export const savedJudgeFingerprint = (
106
114
  model: input.model,
107
115
  timeoutMs: input.timeoutMs,
108
116
  websearch: input.websearch,
117
+ verification,
109
118
  }
110
- : { timeoutMs: input.timeoutMs },
119
+ : { timeoutMs: input.timeoutMs, verification },
111
120
  });
121
+ };
@@ -9,6 +9,8 @@ import { savedJudgeFingerprint } from "./input-fingerprints";
9
9
  import { mkdir, mkdtemp, rm } from "node:fs/promises";
10
10
  import { compileCodeJudge } from "../infra/judging/code-source";
11
11
  import { gradeRecording } from "./grade-recording";
12
+ import { readCodeCriteria } from "../infra/judging/code-criteria";
13
+ import { verificationRuntime } from "../infra/verification/image";
12
14
 
13
15
  /** Inspect retained evidence without changing benchmark selections.
14
16
  * Useful for calibration and second opinions: https://arxiv.org/abs/2410.12784
@@ -28,6 +30,7 @@ export async function judgeEvidence(options: {
28
30
  if (options.rubric && !options.judge?.model)
29
31
  throw new Error("A Markdown rubric requires judge.model");
30
32
  const code = options.code ? await compileCodeJudge(options.code) : undefined;
33
+ const declared = options.code ? await readCodeCriteria(options.code) : undefined;
31
34
  const request: Omit<JudgeRunInput, "judgeHash"> = {
32
35
  evalRunId: "retained-evidence",
33
36
  evidence: {
@@ -37,6 +40,8 @@ export async function judgeEvidence(options: {
37
40
  rubric: options.rubric ?? "",
38
41
  kind: code ? (options.rubric ? "hybrid" : "code") : "llm",
39
42
  code,
43
+ codeCriteria: declared ? Object.entries(declared as Record<string, { name?: string }>).map(([id, value]) => ({ id, name: value.name ?? id })) : undefined,
44
+ verification: await verificationRuntime(options.judge?.verification),
40
45
  agent: options.rubric ? JUDGE_AGENT : undefined,
41
46
  model: options.rubric ? options.judge?.model : undefined,
42
47
  timeoutMs: options.judge?.timeoutMs ?? 600_000,
@@ -67,17 +72,18 @@ export async function recordEvidence(options: {
67
72
  prompt: string;
68
73
  response: string;
69
74
  tools?: ToolCall[];
75
+ workspace?: { initial?: string; final?: string };
70
76
  }): Promise<EvidenceRef> {
71
77
  const directory = resolve(options.directory);
72
78
  await mkdir(dirname(directory), { recursive: true });
73
79
  const workspace = await mkdtemp(resolve(dirname(directory), ".recording-"));
74
80
  try {
75
- const capture = await EvidenceCapture.create(directory);
81
+ const capture = await EvidenceCapture.create(directory, options.workspace?.initial);
76
82
  const result = await capture.finish({
77
83
  prompt: options.prompt,
78
84
  response: { text: options.response },
79
85
  tools: options.tools ?? [],
80
- workspace,
86
+ workspace: options.workspace?.final ?? workspace,
81
87
  });
82
88
  return { directory, hash: result.sha256 };
83
89
  } finally {
@@ -27,6 +27,8 @@ export async function judgeEvalRun(
27
27
  model: definition.judge ? context.definition.judge.model : undefined,
28
28
  kind: definition.code ? (definition.judge ? "hybrid" : "code") : "llm",
29
29
  code: definition.code,
30
+ codeCriteria: definition.codeCriteria,
31
+ verification: context.runtime.verification,
30
32
  judgeHash: slot.judgeHash,
31
33
  timeoutMs: context.definition.judge.timeoutMs,
32
34
  websearch: context.definition.judge.websearch,
@@ -76,6 +78,14 @@ export async function judgeEvalRun(
76
78
  ...graded.code,
77
79
  stdout: relative(context.directory, graded.code.stdout),
78
80
  stderr: relative(context.directory, graded.code.stderr),
81
+ verifications: graded.code.verifications?.map(result => ({
82
+ ...result,
83
+ stdout: relative(context.directory, resolve(directory, "code", result.stdout)),
84
+ stderr: relative(context.directory, resolve(directory, "code", result.stderr)),
85
+ artifacts: result.artifacts.map(artifact => ({ ...artifact,
86
+ file: relative(context.directory, resolve(directory, "code", artifact.file)),
87
+ })),
88
+ })),
79
89
  }
80
90
  : undefined,
81
91
  session: graded.session
@@ -292,11 +292,11 @@ export async function loadBenchmark(
292
292
  typeof definition.judge !== "object" ||
293
293
  Array.isArray(definition.judge) ||
294
294
  Object.keys(definition.judge).some(
295
- (key) => !["model", "timeoutMs", "websearch"].includes(key),
295
+ (key) => !["model", "timeoutMs", "websearch", "verification"].includes(key),
296
296
  ))
297
297
  )
298
298
  throw new Error(
299
- "judge must be an object with model, timeoutMs, or websearch settings",
299
+ "judge must be an object with model, timeoutMs, websearch, or verification settings",
300
300
  );
301
301
  if (new Set(models).size !== models.length)
302
302
  throw new Error("Benchmark contains duplicate models");
@@ -337,6 +337,13 @@ export async function loadBenchmark(
337
337
  const engine = definition.container?.engine ?? "docker";
338
338
  if (engine !== "docker" && engine !== "podman")
339
339
  throw new Error("Container engine must be docker or podman");
340
+ const verification = definition.judge?.verification;
341
+ if (verification && (
342
+ typeof verification !== "object" || Array.isArray(verification) ||
343
+ Object.keys(verification).some(key => !["image", "engine", "cpus", "memoryMiB"].includes(key)) ||
344
+ typeof verification.image !== "string" || !verification.image.trim() || /\s/.test(verification.image) ||
345
+ !["docker", "podman"].includes(verification.engine ?? engine)
346
+ )) throw new Error("judge.verification requires an image and valid container limits");
340
347
  for (const search of [
341
348
  definition.candidate?.websearch,
342
349
  definition.judge?.websearch,
@@ -367,6 +374,12 @@ export async function loadBenchmark(
367
374
  "Judge timeout",
368
375
  ),
369
376
  websearch: definition.judge?.websearch ?? "exa",
377
+ ...(verification ? { verification: {
378
+ image: verification.image,
379
+ engine: verification.engine ?? engine,
380
+ cpus: positive(verification.cpus, 2, "Verification CPU count"),
381
+ memoryMiB: positive(verification.memoryMiB, 4096, "Verification memory"),
382
+ } } : {}),
370
383
  },
371
384
  container: {
372
385
  engine,
@@ -7,6 +7,7 @@ import type {
7
7
  } from "../types";
8
8
  import type { Results } from "../infra/sqlite";
9
9
  import { canJudgeEval } from "./eval-state";
10
+ import { monitorPolicy } from "./monitor-policy";
10
11
  import {
11
12
  candidateFingerprint,
12
13
  judgeFingerprint,
@@ -75,6 +76,22 @@ export function planBenchmark(
75
76
  action: "failed",
76
77
  reason: `Candidate ${execution.state}; explicit retry required`,
77
78
  };
79
+ // A session cut short is evidence only under a stopping rule that still applies.
80
+ if (
81
+ execution.state === "stopped" &&
82
+ execution.stop &&
83
+ !monitorPolicy(evalDefinition.settings.earlyStop, model)
84
+ )
85
+ return {
86
+ slot: {
87
+ ...slot,
88
+ evalRunId: null,
89
+ judgeRunId: null,
90
+ previousEvalRunId: execution.id,
91
+ },
92
+ action: "candidate",
93
+ reason: "Early stopping is off; the recorded session stopped early",
94
+ };
78
95
  if (slot.judgeRunId)
79
96
  if (
80
97
  results.judgeRun(slot.judgeRunId)?.judgment?.value === null &&
@@ -0,0 +1,40 @@
1
+ import { mkdir, mkdtemp, rm } from "node:fs/promises";
2
+ import { resolve } from "node:path";
3
+ import { tmpdir } from "node:os";
4
+ import { loadBenchmark } from "./load-benchmark";
5
+ import { prepareWorkspace } from "../infra/containers/workspace";
6
+ import { CandidateContainer, inspectImage } from "../infra/containers/oci";
7
+ import { createSessionDatabase } from "../infra/opencode/host";
8
+ import { treeHash, writeJson } from "../infra/files";
9
+
10
+ /** Prove preparation without prompting a model, creating EvalRuns, or changing selections. */
11
+ export async function prepareInputs(path: string, options: { directory: string; onlyEvals?: readonly string[] }) {
12
+ const definition = await loadBenchmark(path);
13
+ if (options.onlyEvals?.some(id => !definition.evals.some(item => item.id === id)))
14
+ throw new Error("Select configured evals for preparation");
15
+ const directory = resolve(options.directory);
16
+ if (await Bun.file(resolve(directory, "prepared.json")).exists()) throw new Error("Choose a new prepared-input directory");
17
+ const imageId = await inspectImage(definition.container);
18
+ const scratch = await mkdtemp(resolve(process.platform === "win32" ? "C:/tmp/opencode" : tmpdir(), "openeval-prepare-"));
19
+ const inputs: Array<{ eval: string; directory: string; sourceHash: string; preparedHash: string }> = [];
20
+ try {
21
+ await mkdir(directory, { recursive: true });
22
+ for (const item of definition.evals.filter(item => !options.onlyEvals || options.onlyEvals.includes(item.id))) {
23
+ const staging = resolve(scratch, item.id);
24
+ await mkdir(staging, { recursive: true });
25
+ const workspace = resolve(staging, "workspace");
26
+ await prepareWorkspace(item, workspace);
27
+ const database = resolve(staging, "opencode.db");
28
+ // No credentials or model route are needed to prepare task inputs.
29
+ await createSessionDatabase(database, []);
30
+ await using container = await CandidateContainer.create(definition.container, imageId);
31
+ await container.prepare(workspace, database, definition.candidate.websearch, item.settings.prepare ?? [], staging, definition.candidate.providers);
32
+ const output = resolve(directory, item.id, "workspace");
33
+ await container.snapshot(output, staging);
34
+ inputs.push({ eval: item.id, directory: output, sourceHash: item.sourceHash, preparedHash: await treeHash(output) });
35
+ }
36
+ const result = { imageId, inputs, candidateExecutions: 0, judgeExecutions: 0 };
37
+ await writeJson(resolve(directory, "prepared.json"), result);
38
+ return result;
39
+ } finally { await rm(scratch, { recursive: true, force: true }); }
40
+ }
@@ -1,5 +1,5 @@
1
- import { readdir } from "node:fs/promises";
2
- import { resolve, dirname } from "node:path";
1
+ import { readdir, realpath } from "node:fs/promises";
2
+ import { resolve, dirname, sep } from "node:path";
3
3
  import { Results } from "../infra/sqlite";
4
4
  import type { BenchmarkRun, EvalRun, JudgeRun, Slot, Cost } from "../types";
5
5
  import type {
@@ -26,7 +26,7 @@ import {
26
26
  } from "../runtime";
27
27
  import { canJudgeEval } from "./eval-state";
28
28
  import { CandidateEvidence, type EvidenceQuery } from "../infra/evidence";
29
- import { contained } from "../infra/files";
29
+ import { contained, hash } from "../infra/files";
30
30
 
31
31
  const scheduled = (slot: Slot, benchmark: BenchmarkRun, now: number) =>
32
32
  benchmark.state === "running" &&
@@ -167,7 +167,7 @@ export class ResultReader {
167
167
  if (!judge) throw new Error("Unknown judge run");
168
168
  const candidate = results.evalRun(judge.input.evalRunId);
169
169
  const criteria = new Map(
170
- judge.input.criteria.map((criterion) => [criterion.id, criterion]),
170
+ [...judge.input.criteria, ...judge.input.codeCriteria ?? []].map((criterion) => [criterion.id, criterion]),
171
171
  );
172
172
  for (const id of Object.keys(judge.judgment?.scores ?? {}))
173
173
  if (!criteria.has(id))
@@ -225,6 +225,23 @@ export class ResultReader {
225
225
  })),
226
226
  };
227
227
  }
228
+ async verificationEvidence(benchmarkId: string, judgeRunId: string, id: string, path?: string) {
229
+ using results = await this.database(benchmarkId);
230
+ const judge = results.judgeRun(judgeRunId);
231
+ const verification = judge?.code?.verifications?.find(item => item.id === id);
232
+ if (!verification) throw new Error("Unknown verification");
233
+ if (!path) return { receipt: verification };
234
+ const artifact = verification.artifacts.find(item => item.path === path);
235
+ const file = path === "stdout" ? verification.stdout : path === "stderr" ? verification.stderr : artifact?.file;
236
+ if (!file) throw new Error("Unknown verification artifact");
237
+ const root = await realpath(dirname(results.path));
238
+ const target = await realpath(contained(root, file));
239
+ if (target !== root && !target.startsWith(root + sep)) throw new Error("Verification artifact leaves its run directory");
240
+ const bytes = await Bun.file(target).bytes();
241
+ if (bytes.length > 32 * 1024 * 1024 || (artifact && hash(bytes) !== artifact.sha256))
242
+ throw new Error("Verification artifact is too large or its hash differs");
243
+ return { bytes, path };
244
+ }
228
245
  async checkEvidence(
229
246
  benchmarkId: string,
230
247
  judgeRunId: string,
@@ -7,6 +7,7 @@ import { judgeEvalRun } from "./judge-run";
7
7
  import { executeQueue, finishBenchmark } from "./run-benchmark";
8
8
  import { judgingFingerprint } from "../infra/judging";
9
9
  import { canJudgeEval } from "./eval-state";
10
+ import { verificationRuntime } from "../infra/verification/image";
10
11
 
11
12
  /** Rejudging changes only the selected judgment. The candidate execution is never repeated. */
12
13
  export async function judgeRun(
@@ -59,6 +60,8 @@ export async function judgeRuns(
59
60
  judgeHash: updated.judgeHash,
60
61
  criteria: updated.criteria,
61
62
  code: updated.code,
63
+ codeCriteria: updated.codeCriteria,
64
+ categories: updated.categories,
62
65
  name: updated.name,
63
66
  }
64
67
  : evalDefinition;
@@ -67,7 +70,9 @@ export async function judgeRuns(
67
70
  const runtime = {
68
71
  ...previous.runtime,
69
72
  judgeHash: await judgingFingerprint(),
73
+ verification: await verificationRuntime(definition.judge.verification),
70
74
  };
75
+ if (runtime.verification) definition.judge.verification = runtime.verification;
71
76
  const selected = new Set(candidates.map((candidate) => candidate.slotId));
72
77
  results.transaction(() => {
73
78
  if (results.benchmark?.state === "running")
@@ -28,7 +28,10 @@ export async function retryEvalRun(
28
28
  const slot = results.slot(previous.slotId)!;
29
29
  if (slot.evalRunId !== previous.id)
30
30
  throw new Error("Select the current eval run for this repetition");
31
- const runtime = await runtimeFingerprint(benchmark.definition.container);
31
+ const runtime = await runtimeFingerprint(
32
+ benchmark.definition.container,
33
+ benchmark.definition.judge.verification,
34
+ );
32
35
  if (
33
36
  candidateFingerprint(
34
37
  benchmark.definition,
@@ -176,7 +176,8 @@ export async function runBenchmark(
176
176
  const selected = options.onlyModels?.map(modelRef);
177
177
  if (selected && !selected.length)
178
178
  throw new Error("Select at least one model");
179
- const runtime = await runtimeFingerprint(definition.container);
179
+ const runtime = await runtimeFingerprint(definition.container, definition.judge.verification);
180
+ if (runtime.verification) definition.judge.verification = runtime.verification;
180
181
  let directory =
181
182
  options.directory ??
182
183
  (!options.fresh
@@ -52,6 +52,20 @@ export async function serveResults(options: {
52
52
  url.searchParams.get("judge") ?? "",
53
53
  ),
54
54
  );
55
+ if (url.pathname === "/api/verification") {
56
+ const result = await reader.verificationEvidence(
57
+ url.searchParams.get("benchmark") ?? "", url.searchParams.get("judge") ?? "",
58
+ url.searchParams.get("id") ?? "", url.searchParams.get("path") ?? undefined,
59
+ );
60
+ if ("receipt" in result) return Response.json(result);
61
+ const mime = /\.png$/i.test(result.path) ? "image/png" : /\.jpe?g$/i.test(result.path) ? "image/jpeg" :
62
+ /\.webp$/i.test(result.path) ? "image/webp" : "text/plain; charset=utf-8";
63
+ return new Response(result.bytes, { headers: {
64
+ "Content-Type": mime, "X-Content-Type-Options": "nosniff",
65
+ "Content-Security-Policy": "default-src 'none'; sandbox",
66
+ "Cache-Control": "no-store",
67
+ } });
68
+ }
55
69
  if (url.pathname === "/api/check-evidence") {
56
70
  const query: EvidenceQuery = {
57
71
  action: (url.searchParams.get("action") ??
package/src/cli.ts CHANGED
@@ -12,6 +12,9 @@ import {
12
12
  serveResults,
13
13
  mergeBenchmarkRuns,
14
14
  snapshotBenchmarkRun,
15
+ buildVerificationImage,
16
+ VERIFICATION_IMAGE,
17
+ prepareInputs,
15
18
  } from "./index";
16
19
  import type { ModelRef } from "./index";
17
20
 
@@ -35,6 +38,7 @@ try {
35
38
  Commands:
36
39
  image Build the candidate container image
37
40
  plan Show missing or changed work without executing it
41
+ prepare Prepare and archive inputs without model calls
38
42
  run Execute missing or changed work in the current result
39
43
  view Serve the results viewer
40
44
  snapshot <run> <name> Export a score snapshot
@@ -55,14 +59,29 @@ Options:
55
59
  --only-repetition <n> Execute only this repetition (repeatable)
56
60
  --max-cost <usd> Scheduling budget for this invocation
57
61
  --final-only Judge only after candidates finish
62
+ --verification Build only the standard verification image (image command)
63
+ --output <dir> New prepared-input directory (prepare command)
58
64
  --port <port> Viewer port (default: 4173)
59
65
 
60
66
  Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
61
67
  } else if (command === "--version") {
62
68
  console.log((await Bun.file(new URL("../package.json", import.meta.url)).json()).version);
63
69
  } else if (command === "image") {
64
- await buildImage((await loadBenchmark(benchmark)).container);
65
- console.log("Candidate image ready");
70
+ const definition = await loadBenchmark(benchmark);
71
+ if (!args.includes("--verification")) {
72
+ await buildImage(definition.container);
73
+ console.log("Candidate image ready");
74
+ }
75
+ const verification = definition.judge.verification ?? (args.includes("--verification")
76
+ ? { image: VERIFICATION_IMAGE, engine: definition.container.engine } : undefined);
77
+ if (verification) {
78
+ await buildVerificationImage(verification);
79
+ console.log("Verification image ready");
80
+ }
81
+ } else if (command === "prepare") {
82
+ if (!option("--output")) throw new Error("prepare requires --output with a new directory");
83
+ const onlyEvals = args.flatMap((arg, index) => arg === "--only-eval" ? [args[index + 1]] : []);
84
+ console.log(JSON.stringify(await prepareInputs(benchmark, { directory: option("--output")!, onlyEvals: onlyEvals.length ? onlyEvals : undefined }), null, 2));
66
85
  } else if (command === "view") {
67
86
  const viewer = await serveResults({
68
87
  resultsPath: resolve(benchmark, "results"),
package/src/index.ts CHANGED
@@ -46,8 +46,16 @@ export type {
46
46
  RecordedEvent,
47
47
  RecordedMessage,
48
48
  RecordedFile,
49
+ CodeScore,
50
+ VerificationEnvironment,
51
+ VerificationRuntime,
52
+ VerificationRequest,
53
+ VerificationResult,
54
+ VerificationArtifact,
49
55
  } from "./judge-context";
56
+ export { buildVerificationImage, VERIFICATION_IMAGE } from "./infra/verification/image";
50
57
  export { loadBenchmark } from "./app/load-benchmark";
58
+ export { prepareInputs } from "./app/prepare-inputs";
51
59
  export {
52
60
  runBenchmark,
53
61
  currentBenchmarkRun,