@hona/openeval 0.4.0 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/JUDGING.md +6 -1
- package/README.md +2 -2
- package/VERIFICATION.md +105 -0
- package/package.json +2 -2
- package/src/app/grade-recording.ts +16 -0
- package/src/app/input-fingerprints.ts +24 -14
- package/src/app/judge-evidence.ts +8 -2
- package/src/app/judge-run.ts +10 -0
- package/src/app/load-benchmark.ts +15 -2
- package/src/app/plan-benchmark.ts +17 -0
- package/src/app/prepare-inputs.ts +40 -0
- package/src/app/read-results.ts +21 -4
- package/src/app/rejudge.ts +5 -0
- package/src/app/retry-run.ts +4 -1
- package/src/app/run-benchmark.ts +2 -1
- package/src/app/serve-results.ts +14 -0
- package/src/cli.ts +21 -2
- package/src/index.ts +8 -0
- package/src/infra/containers/oci.ts +4 -0
- package/src/infra/judging/code-result.ts +21 -3
- package/src/infra/judging/code-source.ts +10 -2
- package/src/infra/judging/code-worker.ts +5 -3
- package/src/infra/judging/code.ts +19 -3
- package/src/infra/judging/contract.ts +10 -2
- package/src/infra/judging/index.ts +1 -0
- package/src/infra/recording/index.ts +14 -2
- package/src/infra/verification/image.ts +33 -0
- package/src/infra/verification/runtime/Dockerfile +18 -0
- package/src/infra/verification/runtime/bun.lock +16 -0
- package/src/infra/verification/runtime/package.json +5 -0
- package/src/infra/verification/session.ts +199 -0
- package/src/judge-context.ts +66 -0
- package/src/types.ts +10 -2
- package/viewer/assets/{abnfDiagram-VCTEODGH-D-idCGaW.js → abnfDiagram-VCTEODGH-4dcAM__t.js} +1 -1
- package/viewer/assets/{angular-html-B-7vkhmj.js → angular-html-DCa1K9Q5.js} +1 -1
- package/viewer/assets/{angular-ts-CqJWLTIZ.js → angular-ts-BwOaP4mi.js} +1 -1
- package/viewer/assets/{apl-DVTjqAhB.js → apl-BDVbqe8d.js} +1 -1
- package/viewer/assets/{arc-Bmz8zsvq.js → arc-vf_TPbdA.js} +1 -1
- package/viewer/assets/architecture-7GRP2DOG-ClUScBK0.js +1 -0
- package/viewer/assets/{architectureDiagram-5GKGNRK7-CxqP2ijS.js → architectureDiagram-5GKGNRK7-D9I5g97H.js} +1 -1
- package/viewer/assets/{astro-Ih6QH4I8.js → astro-C4AVO9I1.js} +1 -1
- package/viewer/assets/{blade-vzkWgA61.js → blade-1K2d7b5P.js} +1 -1
- package/viewer/assets/{blockDiagram-I7D4REHJ-Dm35S95Q.js → blockDiagram-I7D4REHJ-CnebaxAP.js} +1 -1
- package/viewer/assets/{c-092Q-y5e.js → c-CrFxx71c.js} +1 -1
- package/viewer/assets/{c4Diagram-7LVT6UL2-DArgyJNC.js → c4Diagram-7LVT6UL2-CZv4bp92.js} +1 -1
- package/viewer/assets/channel-DOp5Ibzq.js +1 -0
- package/viewer/assets/{chapel-DUOt2X3X.js → chapel-CaV_aD0h.js} +1 -1
- package/viewer/assets/{chunk-4HAMMTFA-Bm3UCwQ6.js → chunk-4HAMMTFA-1oAN0Hqj.js} +1 -1
- package/viewer/assets/{chunk-75Z2AOVW-CpmjsCJX.js → chunk-75Z2AOVW-DG-4lG9j.js} +1 -1
- package/viewer/assets/{chunk-DU6HZSFF-DYT2KEwe.js → chunk-DU6HZSFF-K8r4K98Q.js} +1 -1
- package/viewer/assets/{chunk-F27PBJKO-BFGd2OPj.js → chunk-F27PBJKO-DOLcDIe-.js} +1 -1
- package/viewer/assets/{chunk-GMAD6QVW-D8gzxWqz.js → chunk-GMAD6QVW-fbOr8e3A.js} +1 -1
- package/viewer/assets/{chunk-GVQU2GXP-BZJsi-wS.js → chunk-GVQU2GXP-DGskHf86.js} +1 -1
- package/viewer/assets/{chunk-IMKFNOWR-BKbF8JvM.js → chunk-IMKFNOWR-BkiL77Of.js} +1 -1
- package/viewer/assets/{chunk-L3NEJ4N5-SWKb6tCy.js → chunk-L3NEJ4N5-rHC84FEy.js} +1 -1
- package/viewer/assets/{chunk-OSK3NFVY-BrqoFOmR.js → chunk-OSK3NFVY-CDEFeiSo.js} +1 -1
- package/viewer/assets/{chunk-P2QGCYS3-B10Cyxbx.js → chunk-P2QGCYS3-BHNuVs3t.js} +1 -1
- package/viewer/assets/{chunk-POPQ4Y6H-DXpfBPGQ.js → chunk-POPQ4Y6H-8IU6HL0N.js} +1 -1
- package/viewer/assets/{chunk-PWAF6VOD-DZmaXnfr.js → chunk-PWAF6VOD--FR_J-_V.js} +1 -1
- package/viewer/assets/{chunk-SHT3W25Y-1U6zFGxP.js → chunk-SHT3W25Y-DHfu0YQd.js} +1 -1
- package/viewer/assets/{chunk-SVP7TREG-CaW6KcpM.js → chunk-SVP7TREG-CZxnTp0N.js} +1 -1
- package/viewer/assets/{chunk-TICWLB2K-X4pj7G4S.js → chunk-TICWLB2K-DeYv6m7z.js} +1 -1
- package/viewer/assets/{chunk-XXDRQBXY-0afAM3OP.js → chunk-XXDRQBXY-B-stp2Jg.js} +1 -1
- package/viewer/assets/classDiagram-ZZMXUADV-fi0_kah3.js +1 -0
- package/viewer/assets/classDiagram-v2-VYDZK3BY-fi0_kah3.js +1 -0
- package/viewer/assets/{cobol-CtspwbZ4.js → cobol-Dfqw6e9L.js} +1 -1
- package/viewer/assets/{coffee-qbHB9gj8.js → coffee-B0VyEiZ_.js} +1 -1
- package/viewer/assets/{cose-bilkent-JH36ORCC-7s0vcDw1.js → cose-bilkent-JH36ORCC-DwO-FoMO.js} +1 -1
- package/viewer/assets/{cpp-Dx37X5A_.js → cpp-DtAnLTpm.js} +1 -1
- package/viewer/assets/{crystal-CiWjJvk4.js → crystal-DR276aYT.js} +1 -1
- package/viewer/assets/{css-CLeuyJyb.js → css-jcQdyKCB.js} +1 -1
- package/viewer/assets/{cynefin-OW5HDTMX-B4XFGNNZ.js → cynefin-OW5HDTMX-CkCcLfFW.js} +1 -1
- package/viewer/assets/{cynefinDiagram-5FMLGOSQ-DdNMKbw6.js → cynefinDiagram-5FMLGOSQ-Btlq2-ol.js} +1 -1
- package/viewer/assets/{dagre-GXQ25YYZ-BkDo4A91.js → dagre-GXQ25YYZ-BtEJZ3id.js} +1 -1
- package/viewer/assets/{diagram-S7CK7UJ4-DU8cnnVy.js → diagram-S7CK7UJ4-BdLily-u.js} +1 -1
- package/viewer/assets/{diagram-UQ7AKVKN-Dg2emYKa.js → diagram-UQ7AKVKN-DBfD46yQ.js} +1 -1
- package/viewer/assets/{diagram-VSXAHHWV-C_-YSzcn.js → diagram-VSXAHHWV-BJuAmu9T.js} +1 -1
- package/viewer/assets/{diagram-VX7I27RA-BLCEnxqe.js → diagram-VX7I27RA-BVEEYjj-.js} +1 -1
- package/viewer/assets/{diagram-Z3DM3KII-62CpIiAg.js → diagram-Z3DM3KII-DqYpuagj.js} +1 -1
- package/viewer/assets/{dist-CQIgbdem.js → dist-B0PK9u1q.js} +1 -1
- package/viewer/assets/{ebnfDiagram-PWID7BFC-FkGGETgC.js → ebnfDiagram-PWID7BFC-DafUAY97.js} +1 -1
- package/viewer/assets/{edge-CzmGSXHr.js → edge-CMwzwqao.js} +1 -1
- package/viewer/assets/{elixir-DDfD_-oS.js → elixir-D0m0fb5e.js} +1 -1
- package/viewer/assets/{elm-l7aRfQIq.js → elm-Dm3Vhwyo.js} +1 -1
- package/viewer/assets/{erDiagram-RLTQ6QDP-GlCpsS1v.js → erDiagram-RLTQ6QDP-BjTQFP-n.js} +1 -1
- package/viewer/assets/{erb-qVffN5xM.js → erb-BWrVRqIb.js} +1 -1
- package/viewer/assets/eventmodeling-NTZA5JFV-QswxRol8.js +1 -0
- package/viewer/assets/flowDiagram-HODETNUW-DolIsASs.js +1 -0
- package/viewer/assets/{ganttDiagram-EL5Y4UJY-CH_Tk4Ex.js → ganttDiagram-EL5Y4UJY-cvpUkVwQ.js} +1 -1
- package/viewer/assets/{git-rebase-jEJ8nKk1.js → git-rebase-Bx9WSw2T.js} +1 -1
- package/viewer/assets/{gitGraph-4MIJSDKK-Cc9o1l4Y.js → gitGraph-4MIJSDKK-YF9cEiQc.js} +1 -1
- package/viewer/assets/{gitGraphDiagram-WWUBYQGX-UdfHBATc.js → gitGraphDiagram-WWUBYQGX-Cf7LUlAN.js} +1 -1
- package/viewer/assets/{glimmer-js-BMk2JCW8.js → glimmer-js-jJ5_T2M9.js} +1 -1
- package/viewer/assets/{glimmer-ts-BH7RwQxU.js → glimmer-ts-DqUZ7xnF.js} +1 -1
- package/viewer/assets/{glsl-CAWrZoEq.js → glsl-DHzsK5Yf.js} +1 -1
- package/viewer/assets/{graphql-Bac_hecs.js → graphql-TivQ3Vbm.js} +1 -1
- package/viewer/assets/{hack-1OgpAtm-.js → hack-CXzUKlVo.js} +1 -1
- package/viewer/assets/{haml-CvwG3BzW.js → haml-DB9hsLVR.js} +1 -1
- package/viewer/assets/{handlebars-Cq4oEEp8.js → handlebars-Dfxd-Vhz.js} +1 -1
- package/viewer/assets/{html-D5wZ_JR6.js → html-CwO_HZk6.js} +1 -1
- package/viewer/assets/{html-derivative-ZfzPj0aS.js → html-derivative-Da9QhqUr.js} +1 -1
- package/viewer/assets/{http-BDE2UDvO.js → http-DaNELgRt.js} +1 -1
- package/viewer/assets/{hurl-CQbXNLlk.js → hurl-DCibLrD6.js} +1 -1
- package/viewer/assets/index-Cc4pf8mQ.js +795 -0
- package/viewer/assets/{index-OL_rrWNy.css → index-DVYWRkK6.css} +1 -1
- package/viewer/assets/{info-A6RAGUB7-YD13oMfx.js → info-A6RAGUB7-B7Ru1Oky.js} +1 -1
- package/viewer/assets/{infoDiagram-27XIBGKW-jYKjA_l7.js → infoDiagram-27XIBGKW-RlWd2kYA.js} +1 -1
- package/viewer/assets/{ishikawaDiagram-5VMMS53U-BBPzHiVd.js → ishikawaDiagram-5VMMS53U-CS2-lnGi.js} +1 -1
- package/viewer/assets/{java-Cbpu4oyT.js → java-D2z6ozlO.js} +1 -1
- package/viewer/assets/{javascript-UMuq64YD.js → javascript-CxVsIZNi.js} +1 -1
- package/viewer/assets/{jinja-DVvZtqgU.js → jinja-BSUcBlS_.js} +1 -1
- package/viewer/assets/{jison-CZCXBIV3.js → jison-qcnqXcSy.js} +1 -1
- package/viewer/assets/{journeyDiagram-3NMN7TZE-CMqG5ndb.js → journeyDiagram-3NMN7TZE-DqPSRu4-.js} +1 -1
- package/viewer/assets/{json-CzTvWngu.js → json-eiyev3x0.js} +1 -1
- package/viewer/assets/{jsx-CNJnEGR4.js → jsx-rDHm7tH3.js} +1 -1
- package/viewer/assets/{julia-Bj0q4uoH.js → julia-cFfLusXt.js} +1 -1
- package/viewer/assets/{just-93-MRGUC.js → just-D6wClmab.js} +1 -1
- package/viewer/assets/{kanban-definition-UXKFOSKX-VyFYPq9b.js → kanban-definition-UXKFOSKX-BI_a5cRc.js} +1 -1
- package/viewer/assets/{latex-CXd1tMjA.js → latex-BvnFBRMG.js} +1 -1
- package/viewer/assets/{line-D5nvVDHp.js → line-ChPI9Rz5.js} +1 -1
- package/viewer/assets/{linear-Cq-FJZ_z.js → linear-CBr5H_et.js} +1 -1
- package/viewer/assets/{liquid-Cb-ALXSr.js → liquid-CIOCL1uY.js} +1 -1
- package/viewer/assets/{lua-BxTQiamn.js → lua-Dq6WuAHP.js} +1 -1
- package/viewer/assets/{marko-mFYyU5jn.js → marko-Nl8_Nklj.js} +1 -1
- package/viewer/assets/{mdc-Oryqox7_.js → mdc-obW39E44.js} +1 -1
- package/viewer/assets/{mermaid-parser.core-CCDsanlH.js → mermaid-parser.core-BFOx64cQ.js} +3 -3
- package/viewer/assets/{mermaid.core-Crd1YMgi.js → mermaid.core-BDe6Tfbh.js} +4 -4
- package/viewer/assets/{mindmap-definition-YA3MSWOX-BKLrEFWt.js → mindmap-definition-YA3MSWOX-CjzA1VTL.js} +1 -1
- package/viewer/assets/{nginx-C1PwWP7d.js → nginx-Czz1mxQD.js} +1 -1
- package/viewer/assets/{nim-PszXW-l4.js → nim-Boz13cJ3.js} +1 -1
- package/viewer/assets/{org-DO0CuJlO.js → org-BOlhXpYw.js} +1 -1
- package/viewer/assets/{packet-AYTQ26CC-BNPwLE6g.js → packet-AYTQ26CC-MMkKPhzF.js} +1 -1
- package/viewer/assets/{pegDiagram-XKGWAZYB-DjBg_iPh.js → pegDiagram-XKGWAZYB-q1Rf8_Y9.js} +1 -1
- package/viewer/assets/{perl-CHhAgoDL.js → perl-BA7VmyiG.js} +1 -1
- package/viewer/assets/{php-CA4H6qnu.js → php-dYlaksWx.js} +1 -1
- package/viewer/assets/{pie-WAS4IAKB-BzNmHX-Y.js → pie-WAS4IAKB-CLljEA6l.js} +1 -1
- package/viewer/assets/{pieDiagram-E7YTZNPT-cgMB-fBw.js → pieDiagram-E7YTZNPT-DqzGIO9H.js} +1 -1
- package/viewer/assets/{pug-DDuKTe7C.js → pug-WDsqiRUK.js} +1 -1
- package/viewer/assets/{qml-DSd2VypD.js → qml-B5P0chYn.js} +1 -1
- package/viewer/assets/{quadrantDiagram-AXDQQJYC-Crzhs7Np.js → quadrantDiagram-AXDQQJYC-JV3sYNXt.js} +1 -1
- package/viewer/assets/{r-qf-5qR5Q.js → r-DmZfBDub.js} +1 -1
- package/viewer/assets/{radar-RG4KPBEZ-B3oI70pB.js → radar-RG4KPBEZ-DH4HiGh1.js} +1 -1
- package/viewer/assets/{railroad-74A4TZTK-DOM5Od17.js → railroad-74A4TZTK-BwJg41V9.js} +1 -1
- package/viewer/assets/railroad-abnf-HS5TGJTU-Byryy4k9.js +1 -0
- package/viewer/assets/railroad-ebnf-LZEXJU2U-BABBdJeJ.js +1 -0
- package/viewer/assets/railroad-peg-WCYAUIDC-C7YNbOxe.js +1 -0
- package/viewer/assets/{railroadDiagram-O6MQD6OU-M3SvSYf-.js → railroadDiagram-O6MQD6OU-CrclSzRj.js} +1 -1
- package/viewer/assets/{razor-B9n9DtIL.js → razor-BLQfBppW.js} +1 -1
- package/viewer/assets/{regexp-DhGN0EOR.js → regexp-CePe_vV5.js} +1 -1
- package/viewer/assets/{requirementDiagram-BXWQKSXE-CFisiZTo.js → requirementDiagram-BXWQKSXE-5QlkAVAN.js} +1 -1
- package/viewer/assets/{rst-D1SdhuXd.js → rst-D9pCulha.js} +1 -1
- package/viewer/assets/{ruby-BybsgZgf.js → ruby-r3EAH5Rz.js} +1 -1
- package/viewer/assets/{sankeyDiagram-P5KCCOFB-BKf2TMWg.js → sankeyDiagram-P5KCCOFB-BJzQ5vkn.js} +1 -1
- package/viewer/assets/{sas-Io7QwCtD.js → sas-Dxc2YEvn.js} +1 -1
- package/viewer/assets/{scss-D6XY0yo3.js → scss-rZd3d-ZI.js} +1 -1
- package/viewer/assets/{sequenceDiagram-WJ2MYXX4-CHawNQZ9.js → sequenceDiagram-WJ2MYXX4--tS5gMY4.js} +1 -1
- package/viewer/assets/{shellscript-DSk8kvCh.js → shellscript-B-b5xnkP.js} +1 -1
- package/viewer/assets/{shellsession-Be1CN8DO.js → shellsession-DlWXnhiW.js} +1 -1
- package/viewer/assets/{soy-CGkkqGyT.js → soy-DXWgq6qD.js} +1 -1
- package/viewer/assets/{sql-DIgb716V.js → sql-5vPpIo1-.js} +1 -1
- package/viewer/assets/{src-DP6Z6M4G.js → src-eDHFRM9C.js} +1 -1
- package/viewer/assets/{stata-CS_2tb9p.js → stata-BPQLlmzg.js} +1 -1
- package/viewer/assets/{stateDiagram-D77RDMKH-Cx9Rf6fY.js → stateDiagram-D77RDMKH-xapk0kUZ.js} +1 -1
- package/viewer/assets/stateDiagram-v2-MP3YSRHH-Cb8zYuMi.js +1 -0
- package/viewer/assets/{surrealql-Do4o8w1B.js → surrealql-CTEIAZfm.js} +1 -1
- package/viewer/assets/{svelte-3bKxeV6-.js → svelte-B6EVsDV1.js} +1 -1
- package/viewer/assets/{swimlanes-42K2YHIH-yL2oIw_D.js → swimlanes-42K2YHIH-BLyG_o3-.js} +1 -1
- package/viewer/assets/swimlanesDiagram-VR7AAH4N-RLMdNA3S.js +8 -0
- package/viewer/assets/{templ-BgEYPNGP.js → templ-Bu4quHqE.js} +1 -1
- package/viewer/assets/{tex-D_p6whQw.js → tex-Dv_54fPG.js} +1 -1
- package/viewer/assets/{timeline-definition-24CTP7MA-BJeLq4a5.js → timeline-definition-24CTP7MA-DBaZM3qq.js} +1 -1
- package/viewer/assets/{treeView-Q6P3EWNA-COcnImiE.js → treeView-Q6P3EWNA-C3zH0Rwt.js} +1 -1
- package/viewer/assets/{treemap-WGGIJYW6-DpbxWots.js → treemap-WGGIJYW6-BTho8OjG.js} +1 -1
- package/viewer/assets/{ts-tags-GBmMk7Oh.js → ts-tags-0DHV6Smm.js} +1 -1
- package/viewer/assets/{tsx-BSD-yNhi.js → tsx-CmZ6PB1U.js} +1 -1
- package/viewer/assets/{twig-dwAG5SzX.js → twig-u99yWuG6.js} +1 -1
- package/viewer/assets/{typescript-B8YTPl9v.js → typescript-WF79ZIBQ.js} +1 -1
- package/viewer/assets/{typst-Dpzojbzo.js → typst-DQ7XxvRZ.js} +1 -1
- package/viewer/assets/{vennDiagram-4TSXK5OY-nF34Gjtn.js → vennDiagram-4TSXK5OY-B5Cb69qm.js} +1 -1
- package/viewer/assets/{vue-Djbmk2DC.js → vue-DXac3mhj.js} +1 -1
- package/viewer/assets/{vue-html-BXc5rfco.js → vue-html-FSSTYK94.js} +1 -1
- package/viewer/assets/{vue-vine-DHCfQh17.js → vue-vine-BDF8epqI.js} +1 -1
- package/viewer/assets/{wardley-WFR3VGLG-C1qhMifc.js → wardley-WFR3VGLG-DswcXSmz.js} +1 -1
- package/viewer/assets/{wardleyDiagram-VM6X3IG4-B0BfyYWQ.js → wardleyDiagram-VM6X3IG4-BPgNi0Kp.js} +1 -1
- package/viewer/assets/{xml-CdCEskcV.js → xml-HI856iuk.js} +1 -1
- package/viewer/assets/{xsl-BbNukwXP.js → xsl-Ck-aKGwi.js} +1 -1
- package/viewer/assets/{xychartDiagram-S5SC5T6Z-DJKYce1U.js → xychartDiagram-S5SC5T6Z-B8UAJr6c.js} +1 -1
- package/viewer/assets/{yaml-BYLDET4A.js → yaml-DrPyW5uV.js} +1 -1
- package/viewer/index.html +2 -2
- package/viewer/assets/architecture-7GRP2DOG-D3Kw34KW.js +0 -1
- package/viewer/assets/channel-D2lVv6ST.js +0 -1
- package/viewer/assets/classDiagram-ZZMXUADV-DqBNvABf.js +0 -1
- package/viewer/assets/classDiagram-v2-VYDZK3BY-DqBNvABf.js +0 -1
- package/viewer/assets/eventmodeling-NTZA5JFV-07UfGVCm.js +0 -1
- package/viewer/assets/flowDiagram-HODETNUW-CyIUf7TW.js +0 -1
- package/viewer/assets/index-BezQIu6a.js +0 -795
- package/viewer/assets/railroad-abnf-HS5TGJTU-ScZ2h_5Q.js +0 -1
- package/viewer/assets/railroad-ebnf-LZEXJU2U-_Cl5zxJ-.js +0 -1
- package/viewer/assets/railroad-peg-WCYAUIDC-CRlGg7vV.js +0 -1
- package/viewer/assets/stateDiagram-v2-MP3YSRHH-DRb-6t_S.js +0 -1
- package/viewer/assets/swimlanesDiagram-VR7AAH4N-cDiCT6xi.js +0 -8
package/JUDGING.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Use the [canonical terminology](https://openev.al/docs/terminology/): a criterion
|
|
4
4
|
is a named graded requirement, a score is awarded credit, and a metric is a
|
|
5
|
-
measurement such as token count or cost. OpenEval
|
|
5
|
+
measurement such as token count or cost. OpenEval supports code judges,
|
|
6
6
|
LLM judges, and additive use of both against one recorded EvalRun.
|
|
7
7
|
|
|
8
8
|
## File conventions
|
|
@@ -36,6 +36,9 @@ Only the optional scores object has grading semantics. Each key is a criterion
|
|
|
36
36
|
ID, using lowercase letters, digits, and underscores, starting with a letter.
|
|
37
37
|
Values are booleans, finite numbers from 0 to 1, or null. The host converts true
|
|
38
38
|
to 1 and false to 0. It rejects invalid values rather than clamping them.
|
|
39
|
+
Optional `{ value, reason, evidence, measurements }` objects make code verdicts
|
|
40
|
+
readable without changing their scoring semantics. Declared code criterion IDs
|
|
41
|
+
must match the returned scores. See [artifact verification](VERIFICATION.md).
|
|
39
42
|
response.text is always a string; missing text becomes an empty string. The
|
|
40
43
|
original execution outcome is available separately on context.run.
|
|
41
44
|
|
|
@@ -69,6 +72,8 @@ a JudgeRun error. A hybrid judgment is selected only after both sources succeed.
|
|
|
69
72
|
| recording.export(sessionID?) | Native OpenCode session export |
|
|
70
73
|
| workspace.files/read/text/diff | Verified initial and final file snapshots |
|
|
71
74
|
| workspace.materialize(revision?) | A disposable workspace copy for author-owned verification |
|
|
75
|
+
| verification.run(request) | Bounded commands over a restored artifact in an isolated OCI container |
|
|
76
|
+
| verification.read/text(result, path) | Hash-checked retained output from that verification |
|
|
72
77
|
| native.database() | Read-only SQLite access to a verified database copy |
|
|
73
78
|
| native.sdk() | The pinned OpenCode SDK/API over a separate disposable archive copy |
|
|
74
79
|
| native.schema() | The pinned OpenCode schema module |
|
package/README.md
CHANGED
|
@@ -87,8 +87,8 @@ or does not provide a parameterized query.
|
|
|
87
87
|
| Required recording is unavailable | **null** | **null** |
|
|
88
88
|
|
|
89
89
|
Add an optional `Categories: misalignment` line under a heading to compare models
|
|
90
|
-
by category in the viewer's radar chart and
|
|
91
|
-
rejudging.
|
|
90
|
+
by category in the viewer's radar chart, heatmap, and benchmark scorecard. Export
|
|
91
|
+
the scorecard as a PNG from the results page. Categories never trigger rejudging.
|
|
92
92
|
|
|
93
93
|
→ [Write good rubrics](https://openev.al/docs/rubrics/) · [Download the SQL starter](https://openev.al/starter.zip)
|
|
94
94
|
|
package/VERIFICATION.md
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
# Artifact verification
|
|
2
|
+
|
|
3
|
+
Code judges are still plain functions. When the score depends on delivered code,
|
|
4
|
+
run it in a disposable OCI container rather than on the runner host.
|
|
5
|
+
|
|
6
|
+
```ts
|
|
7
|
+
// benchmark.ts
|
|
8
|
+
import { VERIFICATION_IMAGE, type Benchmark } from "@hona/openeval";
|
|
9
|
+
export default {
|
|
10
|
+
models: ["example/model"],
|
|
11
|
+
judge: { verification: { image: VERIFICATION_IMAGE, cpus: 2, memoryMiB: 4096 } },
|
|
12
|
+
} satisfies Benchmark;
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
Build the candidate and standard verification images with `openeval image`.
|
|
16
|
+
`openeval image --verification` builds only the verification image. Custom images
|
|
17
|
+
must be built separately. Images resolve to immutable IDs before planning or
|
|
18
|
+
collection; a changed verification image invalidates judgment reuse, not the
|
|
19
|
+
candidate's delivered artifacts.
|
|
20
|
+
|
|
21
|
+
```ts
|
|
22
|
+
// judge.ts — trusted script content is an ordinary frozen local import/string.
|
|
23
|
+
import type { JudgeContext } from "@hona/openeval";
|
|
24
|
+
export const criteria = { correct: { name: "Correct output", categories: ["coding"] } };
|
|
25
|
+
export default async (ctx: JudgeContext) => {
|
|
26
|
+
const result = await ctx.verification.run({
|
|
27
|
+
revision: "final", cwd: ".", timeoutMs: 60_000,
|
|
28
|
+
commands: [["bun", "/verification/check.mjs"]],
|
|
29
|
+
files: { "check.mjs": "/* author's outcome checks */" },
|
|
30
|
+
artifacts: ["checks.json", "screenshot.png"],
|
|
31
|
+
});
|
|
32
|
+
return { scores: { correct: {
|
|
33
|
+
value: result.state === "completed" && result.exitCode === 0,
|
|
34
|
+
reason: "Explain what the checks established or why they failed.",
|
|
35
|
+
evidence: [{ kind: "verification", id: result.id }],
|
|
36
|
+
measurements: { elapsedMs: result.elapsedMs },
|
|
37
|
+
} } };
|
|
38
|
+
};
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
The primitive restores an initial or final snapshot, uploads trusted inputs to
|
|
42
|
+
`/verification`, runs argv arrays in order, and stops at a nonzero exit. It does
|
|
43
|
+
not choose scoring thresholds or interpret success. Judge setup/transfer/image
|
|
44
|
+
errors remain JudgeRun errors. A test exit or verification timeout is an
|
|
45
|
+
observation the authored criterion must interpret. Interrupted candidate work
|
|
46
|
+
still needs the rubric's evidence policy; a failed verification of an unfinished
|
|
47
|
+
prefix does not necessarily establish final-task failure.
|
|
48
|
+
|
|
49
|
+
## Isolation and bounds
|
|
50
|
+
|
|
51
|
+
- No host bind mounts, credentials, published ports, or network. Browser/server
|
|
52
|
+
verification can use loopback **inside** the container.
|
|
53
|
+
- Read-only image filesystem, non-root command user, no capabilities, no privilege
|
|
54
|
+
escalation, PID/CPU/memory limits, bounded workspace/temp storage, and a deadline.
|
|
55
|
+
- Trusted check inputs are root-owned and are not writable by delivered code.
|
|
56
|
+
- Commands, exit statuses, image/resource identity, logs, and requested regular
|
|
57
|
+
output files are retained with the JudgeRun. Viewer previews never execute HTML
|
|
58
|
+
or SVG from the delivered artifact.
|
|
59
|
+
- Workspace transfer is limited to 128 MiB; each output archive is limited to
|
|
60
|
+
32 MiB; logs are limited to 1 MiB per stream and are marked when truncated.
|
|
61
|
+
- Containers self-expire. The parent also removes containers bearing only its
|
|
62
|
+
unique execution label after worker interruption. No shared/broad cleanup.
|
|
63
|
+
|
|
64
|
+
The standard image includes Bun 1.4.2, Python, Git, Node, and Playwright 1.63.0
|
|
65
|
+
with Chromium. Import Playwright from
|
|
66
|
+
`/opt/verify/node_modules/playwright/index.mjs`. Dependencies needed by a task
|
|
67
|
+
must be available in the image or supplied as frozen inputs; network installs
|
|
68
|
+
are intentionally unavailable. Materialization alone is not a sandbox.
|
|
69
|
+
`verification.text(result, "stdout" | "stderr")` reads the retained bounded logs;
|
|
70
|
+
those names are reserved and cannot also name output artifacts.
|
|
71
|
+
|
|
72
|
+
The primitive verifies a reconstructed artifact. It does **not** prove what was
|
|
73
|
+
alive in the original candidate container, whether the candidate ran a test,
|
|
74
|
+
or whether original game actions were legal. Those claims need original recording
|
|
75
|
+
evidence. Do not put evaluator scripts in candidate workspaces.
|
|
76
|
+
|
|
77
|
+
## Readable judgments and controls
|
|
78
|
+
|
|
79
|
+
`openeval prepare --output <new-directory>` assembles real candidate inputs and
|
|
80
|
+
runs declared preparation in the candidate image, then archives the prepared
|
|
81
|
+
workspace. `--only-eval` scopes it. This makes zero model calls, creates no
|
|
82
|
+
EvalRun/JudgeRun, and does not change selections. Use it to prove fixture setup
|
|
83
|
+
before collection; it does not prove agent success or human task duration.
|
|
84
|
+
|
|
85
|
+
Existing boolean, numeric, and null scores remain sufficient. Optional structured
|
|
86
|
+
scores add `reason`, `evidence`, and JSON `measurements`. These details appear in
|
|
87
|
+
the normal viewer beside verification receipts; arbitrary returned JSON remains
|
|
88
|
+
inspectable. Declared code criteria must be returned exactly. Missing evidence
|
|
89
|
+
references are judging errors, not candidate zeros.
|
|
90
|
+
Arrays of named check observations render as readable tables, including author
|
|
91
|
+
supplied pass/fail details. They do not introduce additional scores or weights.
|
|
92
|
+
Large tables are explicitly bounded in the display; full returned JSON is retained.
|
|
93
|
+
|
|
94
|
+
`recordEvidence({ workspace: { initial, final }, ... })` can retain constructed
|
|
95
|
+
artifact controls without executing them. `judgeEvidence` then uses the same
|
|
96
|
+
public verification primitive. Constructed controls prove tested boundaries, not
|
|
97
|
+
human-duration calibration or live candidate feasibility.
|
|
98
|
+
|
|
99
|
+
Unused `criteria` labels/categories are removed from executable bundling. Debug
|
|
100
|
+
source maps remain retained but do not affect code identity. If grading reads a
|
|
101
|
+
metadata value, it is executable behavior and remains fingerprinted. Changes to
|
|
102
|
+
criterion IDs, actual grading code, imported inputs, dependencies, or verification
|
|
103
|
+
environment are not reporting-only edits. The verification environment is part
|
|
104
|
+
of judge identity only for evals with `judge.ts`. Rebuilding the image does not
|
|
105
|
+
rejudge Markdown-only evals.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@hona/openeval",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.5.1",
|
|
4
4
|
"description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"repository": {
|
|
@@ -14,7 +14,7 @@
|
|
|
14
14
|
"bin": {
|
|
15
15
|
"openeval": "src/cli.ts"
|
|
16
16
|
},
|
|
17
|
-
"files": ["src", "viewer", "LICENSE", "README.md", "JUDGING.md", "!src/**/*.test.ts"],
|
|
17
|
+
"files": ["src", "viewer", "LICENSE", "README.md", "JUDGING.md", "VERIFICATION.md", "!src/**/*.test.ts"],
|
|
18
18
|
"publishConfig": {
|
|
19
19
|
"access": "public"
|
|
20
20
|
},
|
|
@@ -11,6 +11,8 @@ import { executeCodeJudge } from "../infra/judging/code";
|
|
|
11
11
|
import { codeJudgment, combineJudgments } from "../infra/judging/code-result";
|
|
12
12
|
import { executeJudge } from "../infra/judging";
|
|
13
13
|
import { errorMessage } from "../infra/files";
|
|
14
|
+
import { CandidateEvidence } from "../infra/evidence";
|
|
15
|
+
import { validateCitations } from "../infra/judging/contract";
|
|
14
16
|
|
|
15
17
|
export type GradedRecording = {
|
|
16
18
|
state: "completed" | "failed" | "timed_out";
|
|
@@ -37,10 +39,24 @@ export async function gradeRecording(
|
|
|
37
39
|
recording,
|
|
38
40
|
resolve(directory, "code"),
|
|
39
41
|
input.timeoutMs,
|
|
42
|
+
input.verification,
|
|
40
43
|
);
|
|
41
44
|
if (code.state !== "completed")
|
|
42
45
|
return { state: code.state, code, error: code.error };
|
|
43
46
|
const judged = codeJudgment(code.output!);
|
|
47
|
+
if (input.codeCriteria?.length && (
|
|
48
|
+
Object.keys(judged.scores).length !== input.codeCriteria.length ||
|
|
49
|
+
input.codeCriteria.some(item => !Object.hasOwn(judged.scores, item.id))
|
|
50
|
+
)) throw new Error("judge.ts must return exactly its declared criterion IDs");
|
|
51
|
+
for (const score of Object.values(judged.scores)) for (const citation of score.evidence) {
|
|
52
|
+
if (citation.kind !== "verification") continue;
|
|
53
|
+
const result = code.verifications?.find(item => item.id === citation.id);
|
|
54
|
+
if (!result || (citation.path && !result.artifacts.some(item => item.path === citation.path)))
|
|
55
|
+
throw new Error("Code judgment cites missing verification evidence");
|
|
56
|
+
}
|
|
57
|
+
await validateCitations({ ...judged, scores: Object.fromEntries(Object.entries(judged.scores).map(([id, score]) =>
|
|
58
|
+
[id, { ...score, evidence: score.evidence.filter(citation => citation.kind !== "verification") }])) },
|
|
59
|
+
await CandidateEvidence.open(recording.evidence.directory, recording.evidence.hash));
|
|
44
60
|
for (const id of Object.keys(judged.scores))
|
|
45
61
|
if (input.criteria.some((criterion) => criterion.id === id))
|
|
46
62
|
throw new Error(
|
|
@@ -43,7 +43,8 @@ export function candidateProviders(
|
|
|
43
43
|
: undefined;
|
|
44
44
|
}
|
|
45
45
|
|
|
46
|
-
/** Declared, candidate-visible inputs. Harness implementation hashes are provenance.
|
|
46
|
+
/** Declared, candidate-visible inputs. Harness implementation hashes are provenance.
|
|
47
|
+
* Early stopping is a judge-side policy; the planner re-collects sessions it cut short. */
|
|
47
48
|
export function candidateFingerprint(
|
|
48
49
|
definition: BenchmarkDefinition,
|
|
49
50
|
evalId: string,
|
|
@@ -69,21 +70,24 @@ export function candidateFingerprint(
|
|
|
69
70
|
},
|
|
70
71
|
});
|
|
71
72
|
}
|
|
73
|
+
/** Only judge.ts can run checks, so the verification runtime never changes an LLM-only judgment. */
|
|
72
74
|
export const judgeFingerprint = (
|
|
73
75
|
definition: BenchmarkDefinition,
|
|
74
76
|
evalId: string,
|
|
75
|
-
) =>
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
77
|
+
) => {
|
|
78
|
+
const item = definition.evals.find((item) => item.id === evalId)!;
|
|
79
|
+
const verification = item.code ? definition.judge.verification : undefined;
|
|
80
|
+
return fingerprint({
|
|
81
|
+
rubric: item.judge,
|
|
82
|
+
code: item.code?.hash,
|
|
83
|
+
codeIds: item.codeCriteria?.map((criterion) => criterion.id).sort(),
|
|
84
|
+
agent: item.judge ? JUDGE_AGENT : undefined,
|
|
85
|
+
judge: item.judge
|
|
86
|
+
? { ...definition.judge, verification }
|
|
87
|
+
: { timeoutMs: definition.judge.timeoutMs, verification },
|
|
85
88
|
protocol: JUDGE_PROTOCOL,
|
|
86
89
|
});
|
|
90
|
+
};
|
|
87
91
|
export const savedJudgeFingerprint = (
|
|
88
92
|
input: Pick<
|
|
89
93
|
JudgeRunInput,
|
|
@@ -94,11 +98,15 @@ export const savedJudgeFingerprint = (
|
|
|
94
98
|
| "timeoutMs"
|
|
95
99
|
| "websearch"
|
|
96
100
|
| "code"
|
|
101
|
+
| "codeCriteria"
|
|
102
|
+
| "verification"
|
|
97
103
|
>,
|
|
98
|
-
) =>
|
|
99
|
-
|
|
104
|
+
) => {
|
|
105
|
+
const verification = input.code ? input.verification : undefined;
|
|
106
|
+
return fingerprint({
|
|
100
107
|
rubric: input.rubric,
|
|
101
108
|
code: input.code?.hash,
|
|
109
|
+
codeIds: input.codeCriteria?.map((criterion) => criterion.id).sort(),
|
|
102
110
|
agent: input.agent,
|
|
103
111
|
protocol: input.protocol,
|
|
104
112
|
judge: input.rubric
|
|
@@ -106,6 +114,8 @@ export const savedJudgeFingerprint = (
|
|
|
106
114
|
model: input.model,
|
|
107
115
|
timeoutMs: input.timeoutMs,
|
|
108
116
|
websearch: input.websearch,
|
|
117
|
+
verification,
|
|
109
118
|
}
|
|
110
|
-
: { timeoutMs: input.timeoutMs },
|
|
119
|
+
: { timeoutMs: input.timeoutMs, verification },
|
|
111
120
|
});
|
|
121
|
+
};
|
|
@@ -9,6 +9,8 @@ import { savedJudgeFingerprint } from "./input-fingerprints";
|
|
|
9
9
|
import { mkdir, mkdtemp, rm } from "node:fs/promises";
|
|
10
10
|
import { compileCodeJudge } from "../infra/judging/code-source";
|
|
11
11
|
import { gradeRecording } from "./grade-recording";
|
|
12
|
+
import { readCodeCriteria } from "../infra/judging/code-criteria";
|
|
13
|
+
import { verificationRuntime } from "../infra/verification/image";
|
|
12
14
|
|
|
13
15
|
/** Inspect retained evidence without changing benchmark selections.
|
|
14
16
|
* Useful for calibration and second opinions: https://arxiv.org/abs/2410.12784
|
|
@@ -28,6 +30,7 @@ export async function judgeEvidence(options: {
|
|
|
28
30
|
if (options.rubric && !options.judge?.model)
|
|
29
31
|
throw new Error("A Markdown rubric requires judge.model");
|
|
30
32
|
const code = options.code ? await compileCodeJudge(options.code) : undefined;
|
|
33
|
+
const declared = options.code ? await readCodeCriteria(options.code) : undefined;
|
|
31
34
|
const request: Omit<JudgeRunInput, "judgeHash"> = {
|
|
32
35
|
evalRunId: "retained-evidence",
|
|
33
36
|
evidence: {
|
|
@@ -37,6 +40,8 @@ export async function judgeEvidence(options: {
|
|
|
37
40
|
rubric: options.rubric ?? "",
|
|
38
41
|
kind: code ? (options.rubric ? "hybrid" : "code") : "llm",
|
|
39
42
|
code,
|
|
43
|
+
codeCriteria: declared ? Object.entries(declared as Record<string, { name?: string }>).map(([id, value]) => ({ id, name: value.name ?? id })) : undefined,
|
|
44
|
+
verification: await verificationRuntime(options.judge?.verification),
|
|
40
45
|
agent: options.rubric ? JUDGE_AGENT : undefined,
|
|
41
46
|
model: options.rubric ? options.judge?.model : undefined,
|
|
42
47
|
timeoutMs: options.judge?.timeoutMs ?? 600_000,
|
|
@@ -67,17 +72,18 @@ export async function recordEvidence(options: {
|
|
|
67
72
|
prompt: string;
|
|
68
73
|
response: string;
|
|
69
74
|
tools?: ToolCall[];
|
|
75
|
+
workspace?: { initial?: string; final?: string };
|
|
70
76
|
}): Promise<EvidenceRef> {
|
|
71
77
|
const directory = resolve(options.directory);
|
|
72
78
|
await mkdir(dirname(directory), { recursive: true });
|
|
73
79
|
const workspace = await mkdtemp(resolve(dirname(directory), ".recording-"));
|
|
74
80
|
try {
|
|
75
|
-
const capture = await EvidenceCapture.create(directory);
|
|
81
|
+
const capture = await EvidenceCapture.create(directory, options.workspace?.initial);
|
|
76
82
|
const result = await capture.finish({
|
|
77
83
|
prompt: options.prompt,
|
|
78
84
|
response: { text: options.response },
|
|
79
85
|
tools: options.tools ?? [],
|
|
80
|
-
workspace,
|
|
86
|
+
workspace: options.workspace?.final ?? workspace,
|
|
81
87
|
});
|
|
82
88
|
return { directory, hash: result.sha256 };
|
|
83
89
|
} finally {
|
package/src/app/judge-run.ts
CHANGED
|
@@ -27,6 +27,8 @@ export async function judgeEvalRun(
|
|
|
27
27
|
model: definition.judge ? context.definition.judge.model : undefined,
|
|
28
28
|
kind: definition.code ? (definition.judge ? "hybrid" : "code") : "llm",
|
|
29
29
|
code: definition.code,
|
|
30
|
+
codeCriteria: definition.codeCriteria,
|
|
31
|
+
verification: context.runtime.verification,
|
|
30
32
|
judgeHash: slot.judgeHash,
|
|
31
33
|
timeoutMs: context.definition.judge.timeoutMs,
|
|
32
34
|
websearch: context.definition.judge.websearch,
|
|
@@ -76,6 +78,14 @@ export async function judgeEvalRun(
|
|
|
76
78
|
...graded.code,
|
|
77
79
|
stdout: relative(context.directory, graded.code.stdout),
|
|
78
80
|
stderr: relative(context.directory, graded.code.stderr),
|
|
81
|
+
verifications: graded.code.verifications?.map(result => ({
|
|
82
|
+
...result,
|
|
83
|
+
stdout: relative(context.directory, resolve(directory, "code", result.stdout)),
|
|
84
|
+
stderr: relative(context.directory, resolve(directory, "code", result.stderr)),
|
|
85
|
+
artifacts: result.artifacts.map(artifact => ({ ...artifact,
|
|
86
|
+
file: relative(context.directory, resolve(directory, "code", artifact.file)),
|
|
87
|
+
})),
|
|
88
|
+
})),
|
|
79
89
|
}
|
|
80
90
|
: undefined,
|
|
81
91
|
session: graded.session
|
|
@@ -292,11 +292,11 @@ export async function loadBenchmark(
|
|
|
292
292
|
typeof definition.judge !== "object" ||
|
|
293
293
|
Array.isArray(definition.judge) ||
|
|
294
294
|
Object.keys(definition.judge).some(
|
|
295
|
-
(key) => !["model", "timeoutMs", "websearch"].includes(key),
|
|
295
|
+
(key) => !["model", "timeoutMs", "websearch", "verification"].includes(key),
|
|
296
296
|
))
|
|
297
297
|
)
|
|
298
298
|
throw new Error(
|
|
299
|
-
"judge must be an object with model, timeoutMs, or
|
|
299
|
+
"judge must be an object with model, timeoutMs, websearch, or verification settings",
|
|
300
300
|
);
|
|
301
301
|
if (new Set(models).size !== models.length)
|
|
302
302
|
throw new Error("Benchmark contains duplicate models");
|
|
@@ -337,6 +337,13 @@ export async function loadBenchmark(
|
|
|
337
337
|
const engine = definition.container?.engine ?? "docker";
|
|
338
338
|
if (engine !== "docker" && engine !== "podman")
|
|
339
339
|
throw new Error("Container engine must be docker or podman");
|
|
340
|
+
const verification = definition.judge?.verification;
|
|
341
|
+
if (verification && (
|
|
342
|
+
typeof verification !== "object" || Array.isArray(verification) ||
|
|
343
|
+
Object.keys(verification).some(key => !["image", "engine", "cpus", "memoryMiB"].includes(key)) ||
|
|
344
|
+
typeof verification.image !== "string" || !verification.image.trim() || /\s/.test(verification.image) ||
|
|
345
|
+
!["docker", "podman"].includes(verification.engine ?? engine)
|
|
346
|
+
)) throw new Error("judge.verification requires an image and valid container limits");
|
|
340
347
|
for (const search of [
|
|
341
348
|
definition.candidate?.websearch,
|
|
342
349
|
definition.judge?.websearch,
|
|
@@ -367,6 +374,12 @@ export async function loadBenchmark(
|
|
|
367
374
|
"Judge timeout",
|
|
368
375
|
),
|
|
369
376
|
websearch: definition.judge?.websearch ?? "exa",
|
|
377
|
+
...(verification ? { verification: {
|
|
378
|
+
image: verification.image,
|
|
379
|
+
engine: verification.engine ?? engine,
|
|
380
|
+
cpus: positive(verification.cpus, 2, "Verification CPU count"),
|
|
381
|
+
memoryMiB: positive(verification.memoryMiB, 4096, "Verification memory"),
|
|
382
|
+
} } : {}),
|
|
370
383
|
},
|
|
371
384
|
container: {
|
|
372
385
|
engine,
|
|
@@ -7,6 +7,7 @@ import type {
|
|
|
7
7
|
} from "../types";
|
|
8
8
|
import type { Results } from "../infra/sqlite";
|
|
9
9
|
import { canJudgeEval } from "./eval-state";
|
|
10
|
+
import { monitorPolicy } from "./monitor-policy";
|
|
10
11
|
import {
|
|
11
12
|
candidateFingerprint,
|
|
12
13
|
judgeFingerprint,
|
|
@@ -75,6 +76,22 @@ export function planBenchmark(
|
|
|
75
76
|
action: "failed",
|
|
76
77
|
reason: `Candidate ${execution.state}; explicit retry required`,
|
|
77
78
|
};
|
|
79
|
+
// A session cut short is evidence only under a stopping rule that still applies.
|
|
80
|
+
if (
|
|
81
|
+
execution.state === "stopped" &&
|
|
82
|
+
execution.stop &&
|
|
83
|
+
!monitorPolicy(evalDefinition.settings.earlyStop, model)
|
|
84
|
+
)
|
|
85
|
+
return {
|
|
86
|
+
slot: {
|
|
87
|
+
...slot,
|
|
88
|
+
evalRunId: null,
|
|
89
|
+
judgeRunId: null,
|
|
90
|
+
previousEvalRunId: execution.id,
|
|
91
|
+
},
|
|
92
|
+
action: "candidate",
|
|
93
|
+
reason: "Early stopping is off; the recorded session stopped early",
|
|
94
|
+
};
|
|
78
95
|
if (slot.judgeRunId)
|
|
79
96
|
if (
|
|
80
97
|
results.judgeRun(slot.judgeRunId)?.judgment?.value === null &&
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import { mkdir, mkdtemp, rm } from "node:fs/promises";
|
|
2
|
+
import { resolve } from "node:path";
|
|
3
|
+
import { tmpdir } from "node:os";
|
|
4
|
+
import { loadBenchmark } from "./load-benchmark";
|
|
5
|
+
import { prepareWorkspace } from "../infra/containers/workspace";
|
|
6
|
+
import { CandidateContainer, inspectImage } from "../infra/containers/oci";
|
|
7
|
+
import { createSessionDatabase } from "../infra/opencode/host";
|
|
8
|
+
import { treeHash, writeJson } from "../infra/files";
|
|
9
|
+
|
|
10
|
+
/** Prove preparation without prompting a model, creating EvalRuns, or changing selections. */
|
|
11
|
+
export async function prepareInputs(path: string, options: { directory: string; onlyEvals?: readonly string[] }) {
|
|
12
|
+
const definition = await loadBenchmark(path);
|
|
13
|
+
if (options.onlyEvals?.some(id => !definition.evals.some(item => item.id === id)))
|
|
14
|
+
throw new Error("Select configured evals for preparation");
|
|
15
|
+
const directory = resolve(options.directory);
|
|
16
|
+
if (await Bun.file(resolve(directory, "prepared.json")).exists()) throw new Error("Choose a new prepared-input directory");
|
|
17
|
+
const imageId = await inspectImage(definition.container);
|
|
18
|
+
const scratch = await mkdtemp(resolve(process.platform === "win32" ? "C:/tmp/opencode" : tmpdir(), "openeval-prepare-"));
|
|
19
|
+
const inputs: Array<{ eval: string; directory: string; sourceHash: string; preparedHash: string }> = [];
|
|
20
|
+
try {
|
|
21
|
+
await mkdir(directory, { recursive: true });
|
|
22
|
+
for (const item of definition.evals.filter(item => !options.onlyEvals || options.onlyEvals.includes(item.id))) {
|
|
23
|
+
const staging = resolve(scratch, item.id);
|
|
24
|
+
await mkdir(staging, { recursive: true });
|
|
25
|
+
const workspace = resolve(staging, "workspace");
|
|
26
|
+
await prepareWorkspace(item, workspace);
|
|
27
|
+
const database = resolve(staging, "opencode.db");
|
|
28
|
+
// No credentials or model route are needed to prepare task inputs.
|
|
29
|
+
await createSessionDatabase(database, []);
|
|
30
|
+
await using container = await CandidateContainer.create(definition.container, imageId);
|
|
31
|
+
await container.prepare(workspace, database, definition.candidate.websearch, item.settings.prepare ?? [], staging, definition.candidate.providers);
|
|
32
|
+
const output = resolve(directory, item.id, "workspace");
|
|
33
|
+
await container.snapshot(output, staging);
|
|
34
|
+
inputs.push({ eval: item.id, directory: output, sourceHash: item.sourceHash, preparedHash: await treeHash(output) });
|
|
35
|
+
}
|
|
36
|
+
const result = { imageId, inputs, candidateExecutions: 0, judgeExecutions: 0 };
|
|
37
|
+
await writeJson(resolve(directory, "prepared.json"), result);
|
|
38
|
+
return result;
|
|
39
|
+
} finally { await rm(scratch, { recursive: true, force: true }); }
|
|
40
|
+
}
|
package/src/app/read-results.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { readdir } from "node:fs/promises";
|
|
2
|
-
import { resolve, dirname } from "node:path";
|
|
1
|
+
import { readdir, realpath } from "node:fs/promises";
|
|
2
|
+
import { resolve, dirname, sep } from "node:path";
|
|
3
3
|
import { Results } from "../infra/sqlite";
|
|
4
4
|
import type { BenchmarkRun, EvalRun, JudgeRun, Slot, Cost } from "../types";
|
|
5
5
|
import type {
|
|
@@ -26,7 +26,7 @@ import {
|
|
|
26
26
|
} from "../runtime";
|
|
27
27
|
import { canJudgeEval } from "./eval-state";
|
|
28
28
|
import { CandidateEvidence, type EvidenceQuery } from "../infra/evidence";
|
|
29
|
-
import { contained } from "../infra/files";
|
|
29
|
+
import { contained, hash } from "../infra/files";
|
|
30
30
|
|
|
31
31
|
const scheduled = (slot: Slot, benchmark: BenchmarkRun, now: number) =>
|
|
32
32
|
benchmark.state === "running" &&
|
|
@@ -167,7 +167,7 @@ export class ResultReader {
|
|
|
167
167
|
if (!judge) throw new Error("Unknown judge run");
|
|
168
168
|
const candidate = results.evalRun(judge.input.evalRunId);
|
|
169
169
|
const criteria = new Map(
|
|
170
|
-
judge.input.criteria.map((criterion) => [criterion.id, criterion]),
|
|
170
|
+
[...judge.input.criteria, ...judge.input.codeCriteria ?? []].map((criterion) => [criterion.id, criterion]),
|
|
171
171
|
);
|
|
172
172
|
for (const id of Object.keys(judge.judgment?.scores ?? {}))
|
|
173
173
|
if (!criteria.has(id))
|
|
@@ -225,6 +225,23 @@ export class ResultReader {
|
|
|
225
225
|
})),
|
|
226
226
|
};
|
|
227
227
|
}
|
|
228
|
+
async verificationEvidence(benchmarkId: string, judgeRunId: string, id: string, path?: string) {
|
|
229
|
+
using results = await this.database(benchmarkId);
|
|
230
|
+
const judge = results.judgeRun(judgeRunId);
|
|
231
|
+
const verification = judge?.code?.verifications?.find(item => item.id === id);
|
|
232
|
+
if (!verification) throw new Error("Unknown verification");
|
|
233
|
+
if (!path) return { receipt: verification };
|
|
234
|
+
const artifact = verification.artifacts.find(item => item.path === path);
|
|
235
|
+
const file = path === "stdout" ? verification.stdout : path === "stderr" ? verification.stderr : artifact?.file;
|
|
236
|
+
if (!file) throw new Error("Unknown verification artifact");
|
|
237
|
+
const root = await realpath(dirname(results.path));
|
|
238
|
+
const target = await realpath(contained(root, file));
|
|
239
|
+
if (target !== root && !target.startsWith(root + sep)) throw new Error("Verification artifact leaves its run directory");
|
|
240
|
+
const bytes = await Bun.file(target).bytes();
|
|
241
|
+
if (bytes.length > 32 * 1024 * 1024 || (artifact && hash(bytes) !== artifact.sha256))
|
|
242
|
+
throw new Error("Verification artifact is too large or its hash differs");
|
|
243
|
+
return { bytes, path };
|
|
244
|
+
}
|
|
228
245
|
async checkEvidence(
|
|
229
246
|
benchmarkId: string,
|
|
230
247
|
judgeRunId: string,
|
package/src/app/rejudge.ts
CHANGED
|
@@ -7,6 +7,7 @@ import { judgeEvalRun } from "./judge-run";
|
|
|
7
7
|
import { executeQueue, finishBenchmark } from "./run-benchmark";
|
|
8
8
|
import { judgingFingerprint } from "../infra/judging";
|
|
9
9
|
import { canJudgeEval } from "./eval-state";
|
|
10
|
+
import { verificationRuntime } from "../infra/verification/image";
|
|
10
11
|
|
|
11
12
|
/** Rejudging changes only the selected judgment. The candidate execution is never repeated. */
|
|
12
13
|
export async function judgeRun(
|
|
@@ -59,6 +60,8 @@ export async function judgeRuns(
|
|
|
59
60
|
judgeHash: updated.judgeHash,
|
|
60
61
|
criteria: updated.criteria,
|
|
61
62
|
code: updated.code,
|
|
63
|
+
codeCriteria: updated.codeCriteria,
|
|
64
|
+
categories: updated.categories,
|
|
62
65
|
name: updated.name,
|
|
63
66
|
}
|
|
64
67
|
: evalDefinition;
|
|
@@ -67,7 +70,9 @@ export async function judgeRuns(
|
|
|
67
70
|
const runtime = {
|
|
68
71
|
...previous.runtime,
|
|
69
72
|
judgeHash: await judgingFingerprint(),
|
|
73
|
+
verification: await verificationRuntime(definition.judge.verification),
|
|
70
74
|
};
|
|
75
|
+
if (runtime.verification) definition.judge.verification = runtime.verification;
|
|
71
76
|
const selected = new Set(candidates.map((candidate) => candidate.slotId));
|
|
72
77
|
results.transaction(() => {
|
|
73
78
|
if (results.benchmark?.state === "running")
|
package/src/app/retry-run.ts
CHANGED
|
@@ -28,7 +28,10 @@ export async function retryEvalRun(
|
|
|
28
28
|
const slot = results.slot(previous.slotId)!;
|
|
29
29
|
if (slot.evalRunId !== previous.id)
|
|
30
30
|
throw new Error("Select the current eval run for this repetition");
|
|
31
|
-
const runtime = await runtimeFingerprint(
|
|
31
|
+
const runtime = await runtimeFingerprint(
|
|
32
|
+
benchmark.definition.container,
|
|
33
|
+
benchmark.definition.judge.verification,
|
|
34
|
+
);
|
|
32
35
|
if (
|
|
33
36
|
candidateFingerprint(
|
|
34
37
|
benchmark.definition,
|
package/src/app/run-benchmark.ts
CHANGED
|
@@ -176,7 +176,8 @@ export async function runBenchmark(
|
|
|
176
176
|
const selected = options.onlyModels?.map(modelRef);
|
|
177
177
|
if (selected && !selected.length)
|
|
178
178
|
throw new Error("Select at least one model");
|
|
179
|
-
const runtime = await runtimeFingerprint(definition.container);
|
|
179
|
+
const runtime = await runtimeFingerprint(definition.container, definition.judge.verification);
|
|
180
|
+
if (runtime.verification) definition.judge.verification = runtime.verification;
|
|
180
181
|
let directory =
|
|
181
182
|
options.directory ??
|
|
182
183
|
(!options.fresh
|
package/src/app/serve-results.ts
CHANGED
|
@@ -52,6 +52,20 @@ export async function serveResults(options: {
|
|
|
52
52
|
url.searchParams.get("judge") ?? "",
|
|
53
53
|
),
|
|
54
54
|
);
|
|
55
|
+
if (url.pathname === "/api/verification") {
|
|
56
|
+
const result = await reader.verificationEvidence(
|
|
57
|
+
url.searchParams.get("benchmark") ?? "", url.searchParams.get("judge") ?? "",
|
|
58
|
+
url.searchParams.get("id") ?? "", url.searchParams.get("path") ?? undefined,
|
|
59
|
+
);
|
|
60
|
+
if ("receipt" in result) return Response.json(result);
|
|
61
|
+
const mime = /\.png$/i.test(result.path) ? "image/png" : /\.jpe?g$/i.test(result.path) ? "image/jpeg" :
|
|
62
|
+
/\.webp$/i.test(result.path) ? "image/webp" : "text/plain; charset=utf-8";
|
|
63
|
+
return new Response(result.bytes, { headers: {
|
|
64
|
+
"Content-Type": mime, "X-Content-Type-Options": "nosniff",
|
|
65
|
+
"Content-Security-Policy": "default-src 'none'; sandbox",
|
|
66
|
+
"Cache-Control": "no-store",
|
|
67
|
+
} });
|
|
68
|
+
}
|
|
55
69
|
if (url.pathname === "/api/check-evidence") {
|
|
56
70
|
const query: EvidenceQuery = {
|
|
57
71
|
action: (url.searchParams.get("action") ??
|
package/src/cli.ts
CHANGED
|
@@ -12,6 +12,9 @@ import {
|
|
|
12
12
|
serveResults,
|
|
13
13
|
mergeBenchmarkRuns,
|
|
14
14
|
snapshotBenchmarkRun,
|
|
15
|
+
buildVerificationImage,
|
|
16
|
+
VERIFICATION_IMAGE,
|
|
17
|
+
prepareInputs,
|
|
15
18
|
} from "./index";
|
|
16
19
|
import type { ModelRef } from "./index";
|
|
17
20
|
|
|
@@ -35,6 +38,7 @@ try {
|
|
|
35
38
|
Commands:
|
|
36
39
|
image Build the candidate container image
|
|
37
40
|
plan Show missing or changed work without executing it
|
|
41
|
+
prepare Prepare and archive inputs without model calls
|
|
38
42
|
run Execute missing or changed work in the current result
|
|
39
43
|
view Serve the results viewer
|
|
40
44
|
snapshot <run> <name> Export a score snapshot
|
|
@@ -55,14 +59,29 @@ Options:
|
|
|
55
59
|
--only-repetition <n> Execute only this repetition (repeatable)
|
|
56
60
|
--max-cost <usd> Scheduling budget for this invocation
|
|
57
61
|
--final-only Judge only after candidates finish
|
|
62
|
+
--verification Build only the standard verification image (image command)
|
|
63
|
+
--output <dir> New prepared-input directory (prepare command)
|
|
58
64
|
--port <port> Viewer port (default: 4173)
|
|
59
65
|
|
|
60
66
|
Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
61
67
|
} else if (command === "--version") {
|
|
62
68
|
console.log((await Bun.file(new URL("../package.json", import.meta.url)).json()).version);
|
|
63
69
|
} else if (command === "image") {
|
|
64
|
-
|
|
65
|
-
|
|
70
|
+
const definition = await loadBenchmark(benchmark);
|
|
71
|
+
if (!args.includes("--verification")) {
|
|
72
|
+
await buildImage(definition.container);
|
|
73
|
+
console.log("Candidate image ready");
|
|
74
|
+
}
|
|
75
|
+
const verification = definition.judge.verification ?? (args.includes("--verification")
|
|
76
|
+
? { image: VERIFICATION_IMAGE, engine: definition.container.engine } : undefined);
|
|
77
|
+
if (verification) {
|
|
78
|
+
await buildVerificationImage(verification);
|
|
79
|
+
console.log("Verification image ready");
|
|
80
|
+
}
|
|
81
|
+
} else if (command === "prepare") {
|
|
82
|
+
if (!option("--output")) throw new Error("prepare requires --output with a new directory");
|
|
83
|
+
const onlyEvals = args.flatMap((arg, index) => arg === "--only-eval" ? [args[index + 1]] : []);
|
|
84
|
+
console.log(JSON.stringify(await prepareInputs(benchmark, { directory: option("--output")!, onlyEvals: onlyEvals.length ? onlyEvals : undefined }), null, 2));
|
|
66
85
|
} else if (command === "view") {
|
|
67
86
|
const viewer = await serveResults({
|
|
68
87
|
resultsPath: resolve(benchmark, "results"),
|
package/src/index.ts
CHANGED
|
@@ -46,8 +46,16 @@ export type {
|
|
|
46
46
|
RecordedEvent,
|
|
47
47
|
RecordedMessage,
|
|
48
48
|
RecordedFile,
|
|
49
|
+
CodeScore,
|
|
50
|
+
VerificationEnvironment,
|
|
51
|
+
VerificationRuntime,
|
|
52
|
+
VerificationRequest,
|
|
53
|
+
VerificationResult,
|
|
54
|
+
VerificationArtifact,
|
|
49
55
|
} from "./judge-context";
|
|
56
|
+
export { buildVerificationImage, VERIFICATION_IMAGE } from "./infra/verification/image";
|
|
50
57
|
export { loadBenchmark } from "./app/load-benchmark";
|
|
58
|
+
export { prepareInputs } from "./app/prepare-inputs";
|
|
51
59
|
export {
|
|
52
60
|
runBenchmark,
|
|
53
61
|
currentBenchmarkRun,
|