@hona/openeval 0.3.3 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/JUDGING.md +63 -1
- package/README.md +4 -0
- package/VERIFICATION.md +100 -0
- package/package.json +2 -2
- package/src/app/grade-recording.ts +16 -0
- package/src/app/input-fingerprints.ts +8 -2
- package/src/app/judge-evidence.ts +8 -2
- package/src/app/judge-run.ts +10 -0
- package/src/app/load-benchmark.ts +102 -9
- package/src/app/prepare-inputs.ts +40 -0
- package/src/app/read-results.ts +23 -5
- package/src/app/rejudge.ts +5 -0
- package/src/app/run-benchmark.ts +36 -4
- package/src/app/scores.ts +19 -2
- package/src/app/serve-results.ts +14 -0
- package/src/cli.ts +27 -2
- package/src/criterion-categories.ts +53 -0
- package/src/index.ts +13 -0
- package/src/infra/containers/catalog.ts +77 -21
- package/src/infra/containers/oci.ts +4 -0
- package/src/infra/judging/code-criteria.ts +112 -0
- package/src/infra/judging/code-result.ts +21 -3
- package/src/infra/judging/code-source.ts +10 -2
- package/src/infra/judging/code-worker.ts +5 -3
- package/src/infra/judging/code.ts +19 -3
- package/src/infra/judging/contract.ts +10 -2
- package/src/infra/judging/index.ts +1 -0
- package/src/infra/recording/index.ts +14 -2
- package/src/infra/verification/image.ts +33 -0
- package/src/infra/verification/runtime/Dockerfile +18 -0
- package/src/infra/verification/runtime/bun.lock +16 -0
- package/src/infra/verification/runtime/package.json +5 -0
- package/src/infra/verification/session.ts +199 -0
- package/src/judge-context.ts +66 -0
- package/src/types.ts +22 -2
- package/src/view.ts +59 -0
- package/viewer/assets/{abnfDiagram-VCTEODGH-Cl3HgjMn.js → abnfDiagram-VCTEODGH-tXaAmCdr.js} +1 -1
- package/viewer/assets/{angular-html-DD_Qs90j.js → angular-html-CNhJPCRr.js} +1 -1
- package/viewer/assets/{angular-ts-C6uwolJ0.js → angular-ts-BnGgGecY.js} +1 -1
- package/viewer/assets/{apl-DXXNG1AP.js → apl-ZOvmk1hE.js} +1 -1
- package/viewer/assets/{arc-VljIODVM.js → arc-BalFY-0o.js} +1 -1
- package/viewer/assets/architecture-7GRP2DOG-Dwn_SxkP.js +1 -0
- package/viewer/assets/{architectureDiagram-5GKGNRK7-BBPZQ1uU.js → architectureDiagram-5GKGNRK7-DpSDAEgb.js} +1 -1
- package/viewer/assets/{astro-DZiA762t.js → astro-BJzTFZ3R.js} +1 -1
- package/viewer/assets/{blade-0_CSK9rb.js → blade-CvS8_w1j.js} +1 -1
- package/viewer/assets/{blockDiagram-I7D4REHJ-C3jAen-I.js → blockDiagram-I7D4REHJ-BY3upSR4.js} +1 -1
- package/viewer/assets/{c-Dr4MdHFU.js → c-D5ZSBtUG.js} +1 -1
- package/viewer/assets/{c4Diagram-7LVT6UL2-x9-A2uwu.js → c4Diagram-7LVT6UL2-CA9keRc2.js} +1 -1
- package/viewer/assets/channel-CgM1_6P7.js +1 -0
- package/viewer/assets/{chapel-Bzsv5OV_.js → chapel-1aYDxJri.js} +1 -1
- package/viewer/assets/{chunk-4HAMMTFA-DhJ5vlqv.js → chunk-4HAMMTFA-B0uITxv4.js} +1 -1
- package/viewer/assets/{chunk-75Z2AOVW-BiWuU3A8.js → chunk-75Z2AOVW-CbTHx8Lz.js} +1 -1
- package/viewer/assets/{chunk-DU6HZSFF-DQsE2_KX.js → chunk-DU6HZSFF-C7mZpnch.js} +1 -1
- package/viewer/assets/{chunk-F27PBJKO-eYpvp3yv.js → chunk-F27PBJKO-iUZG9l9x.js} +1 -1
- package/viewer/assets/{chunk-GMAD6QVW-Cuv8q7cp.js → chunk-GMAD6QVW-BYg8qa71.js} +1 -1
- package/viewer/assets/{chunk-GVQU2GXP-C80TX_-9.js → chunk-GVQU2GXP-rnsdTI4Q.js} +1 -1
- package/viewer/assets/{chunk-IMKFNOWR-D6UIvU6O.js → chunk-IMKFNOWR-D0e2NIBw.js} +1 -1
- package/viewer/assets/{chunk-L3NEJ4N5-B97oN3Wz.js → chunk-L3NEJ4N5-Itlw7HBQ.js} +1 -1
- package/viewer/assets/{chunk-OSK3NFVY-Dv9V7Vwg.js → chunk-OSK3NFVY-DjAAQhPw.js} +1 -1
- package/viewer/assets/{chunk-P2QGCYS3-CCWmLNYJ.js → chunk-P2QGCYS3-XOxrUS-o.js} +1 -1
- package/viewer/assets/{chunk-POPQ4Y6H-B1p9ouPt.js → chunk-POPQ4Y6H-SHuwgciA.js} +1 -1
- package/viewer/assets/{chunk-PWAF6VOD-DA8KGNWy.js → chunk-PWAF6VOD-BtrRW-Sl.js} +1 -1
- package/viewer/assets/{chunk-SHT3W25Y-BrG5_y-x.js → chunk-SHT3W25Y-6Pacfssp.js} +1 -1
- package/viewer/assets/{chunk-SVP7TREG-BfR4k1NG.js → chunk-SVP7TREG-eTVS9a20.js} +1 -1
- package/viewer/assets/{chunk-TICWLB2K-Cl_FiqB8.js → chunk-TICWLB2K-CPnAfOZj.js} +1 -1
- package/viewer/assets/{chunk-XXDRQBXY-D4qiShQC.js → chunk-XXDRQBXY-ciXhM2Hb.js} +1 -1
- package/viewer/assets/classDiagram-ZZMXUADV-BtFc7wo5.js +1 -0
- package/viewer/assets/classDiagram-v2-VYDZK3BY-BtFc7wo5.js +1 -0
- package/viewer/assets/{cobol-BnQBc2SA.js → cobol-CEoq_5kf.js} +1 -1
- package/viewer/assets/{coffee-D8jvqPgZ.js → coffee-CZTWsTxw.js} +1 -1
- package/viewer/assets/{cose-bilkent-JH36ORCC-BcmCCZr6.js → cose-bilkent-JH36ORCC-NCkOcFkY.js} +1 -1
- package/viewer/assets/{cpp-MUcJROqQ.js → cpp-Hh-o0ivR.js} +1 -1
- package/viewer/assets/{crystal-CFGJH8of.js → crystal-CL_KTsac.js} +1 -1
- package/viewer/assets/{css-Z5Q-tnUN.js → css-9neLgVLd.js} +1 -1
- package/viewer/assets/{cynefin-OW5HDTMX-Hu_taNlH.js → cynefin-OW5HDTMX-Ba1tlgBB.js} +1 -1
- package/viewer/assets/{cynefinDiagram-5FMLGOSQ-DGCR5Bu-.js → cynefinDiagram-5FMLGOSQ-CMrzWo3p.js} +1 -1
- package/viewer/assets/{dagre-GXQ25YYZ-BONZK_72.js → dagre-GXQ25YYZ-DwjOnUDD.js} +1 -1
- package/viewer/assets/{diagram-S7CK7UJ4-B3PuH0gY.js → diagram-S7CK7UJ4-DA0PXsM6.js} +1 -1
- package/viewer/assets/{diagram-UQ7AKVKN-_nLXEPxU.js → diagram-UQ7AKVKN-C5BBFVvO.js} +1 -1
- package/viewer/assets/{diagram-VSXAHHWV-W8uCnJKR.js → diagram-VSXAHHWV-COgLeOyi.js} +1 -1
- package/viewer/assets/{diagram-VX7I27RA-Dw85ssfk.js → diagram-VX7I27RA-CEu-Pm3B.js} +1 -1
- package/viewer/assets/{diagram-Z3DM3KII-iT_nC6Ci.js → diagram-Z3DM3KII-BGq1_7BE.js} +1 -1
- package/viewer/assets/{dist-CQ0iZX7g.js → dist-CX8-lyNa.js} +1 -1
- package/viewer/assets/{ebnfDiagram-PWID7BFC-CmB3I8gj.js → ebnfDiagram-PWID7BFC-CGoEp9fJ.js} +1 -1
- package/viewer/assets/{edge-ChegMQD3.js → edge-CTz4kCIs.js} +1 -1
- package/viewer/assets/{elixir-BIS6GmoK.js → elixir-Bi8YL7XX.js} +1 -1
- package/viewer/assets/{elm-0b95WpPs.js → elm-C-UxMmSh.js} +1 -1
- package/viewer/assets/{erDiagram-RLTQ6QDP-BdrQ4dDR.js → erDiagram-RLTQ6QDP-Bb0ICmqZ.js} +1 -1
- package/viewer/assets/{erb-B6AHMmNJ.js → erb-ZSq62Y_u.js} +1 -1
- package/viewer/assets/eventmodeling-NTZA5JFV-BCWgnTvv.js +1 -0
- package/viewer/assets/flowDiagram-HODETNUW-B4hhZ1QN.js +1 -0
- package/viewer/assets/{ganttDiagram-EL5Y4UJY-CScNarbN.js → ganttDiagram-EL5Y4UJY-B4KN9ny5.js} +1 -1
- package/viewer/assets/{git-rebase-eLwqxn21.js → git-rebase-Cw9GoTpw.js} +1 -1
- package/viewer/assets/{gitGraph-4MIJSDKK-DBJydFU4.js → gitGraph-4MIJSDKK-CI3xWhLt.js} +1 -1
- package/viewer/assets/{gitGraphDiagram-WWUBYQGX-BNpJTag6.js → gitGraphDiagram-WWUBYQGX-CPv8-MbI.js} +1 -1
- package/viewer/assets/{glimmer-js-BPUd-Hdw.js → glimmer-js-Bb_1emHy.js} +1 -1
- package/viewer/assets/{glimmer-ts-BsY3Gaee.js → glimmer-ts-D2kOXMB0.js} +1 -1
- package/viewer/assets/{glsl-BgZlzkDK.js → glsl-DEqjHGFr.js} +1 -1
- package/viewer/assets/{graphql-BNNbyKev.js → graphql-Bi0OE4k4.js} +1 -1
- package/viewer/assets/{hack-DXqsPyfj.js → hack-aU4VRzRW.js} +1 -1
- package/viewer/assets/{haml-CEIe07b8.js → haml-B-2wwgyn.js} +1 -1
- package/viewer/assets/{handlebars-4TLrUOf9.js → handlebars-Bmq7hbFI.js} +1 -1
- package/viewer/assets/{html-Dirc4JcS.js → html-BBsNbpya.js} +1 -1
- package/viewer/assets/{html-derivative-lYLgmFdz.js → html-derivative-CkVRX-vn.js} +1 -1
- package/viewer/assets/{http-BhrBDOp-.js → http-cQ-nDL4L.js} +1 -1
- package/viewer/assets/{hurl-DWBDJqQR.js → hurl-D5phPwZN.js} +1 -1
- package/viewer/assets/{index-w8_YIakp.css → index-DdaXmlLv.css} +1 -1
- package/viewer/assets/index-Dhnl4xDm.js +795 -0
- package/viewer/assets/{info-A6RAGUB7-Cqp6XvkH.js → info-A6RAGUB7-CRDxOY87.js} +1 -1
- package/viewer/assets/{infoDiagram-27XIBGKW-CCvdaWwj.js → infoDiagram-27XIBGKW-D6y32CfO.js} +1 -1
- package/viewer/assets/{ishikawaDiagram-5VMMS53U-DQ4-h-VU.js → ishikawaDiagram-5VMMS53U-BWFBTCh5.js} +1 -1
- package/viewer/assets/{java-CdPCMwep.js → java-CT3DXSYV.js} +1 -1
- package/viewer/assets/{javascript-CvM774KP.js → javascript-BPmsAXaP.js} +1 -1
- package/viewer/assets/{jinja-BGKKdTiv.js → jinja-Ct_36dEc.js} +1 -1
- package/viewer/assets/{jison-CXj9ikNa.js → jison-BYrmy85c.js} +1 -1
- package/viewer/assets/{journeyDiagram-3NMN7TZE-DUzDc5Z8.js → journeyDiagram-3NMN7TZE-6XCObBEx.js} +1 -1
- package/viewer/assets/{json-NHDEs7sd.js → json-BCFP8udh.js} +1 -1
- package/viewer/assets/{jsx-FNvyvbBd.js → jsx-Dgzhn0Dq.js} +1 -1
- package/viewer/assets/{julia-B3kEyS8t.js → julia-DIddcjVY.js} +1 -1
- package/viewer/assets/{just-CpYQ-9KV.js → just-C5t963St.js} +1 -1
- package/viewer/assets/{kanban-definition-UXKFOSKX-ChHjSmhz.js → kanban-definition-UXKFOSKX-DU1B45HL.js} +1 -1
- package/viewer/assets/{latex-OWDJhn4l.js → latex-BKZooW5w.js} +1 -1
- package/viewer/assets/{line-DA6hWqrZ.js → line-lYgs1iW7.js} +1 -1
- package/viewer/assets/{linear-UNQGDr0C.js → linear-sHqbNFGm.js} +1 -1
- package/viewer/assets/{liquid-PySKINiF.js → liquid-_PLVu-8C.js} +1 -1
- package/viewer/assets/{lua-DVWbE3Qa.js → lua-BWNAOGrF.js} +1 -1
- package/viewer/assets/{marko-9EBuCkFn.js → marko-DC7qz2de.js} +1 -1
- package/viewer/assets/{mdc-BlHUSA6C.js → mdc-Bn_2wB99.js} +1 -1
- package/viewer/assets/{mermaid-parser.core-86Qkc0ID.js → mermaid-parser.core-jzj7JZTj.js} +3 -3
- package/viewer/assets/{mermaid.core-DaVdMmkQ.js → mermaid.core-DpRyWAPv.js} +4 -4
- package/viewer/assets/{mindmap-definition-YA3MSWOX-oloAEnnx.js → mindmap-definition-YA3MSWOX-CIZjqKbE.js} +1 -1
- package/viewer/assets/{nginx-CDJjPEvc.js → nginx-Ce7JLwNg.js} +1 -1
- package/viewer/assets/{nim-DNAK-c_E.js → nim-CP_MVph6.js} +1 -1
- package/viewer/assets/{org-D6PZe7Hw.js → org-BVKft-Jg.js} +1 -1
- package/viewer/assets/{packet-AYTQ26CC-CF1AN50P.js → packet-AYTQ26CC-BwRTCCV1.js} +1 -1
- package/viewer/assets/{pegDiagram-XKGWAZYB-DGz2-yRO.js → pegDiagram-XKGWAZYB-3nilNfrm.js} +1 -1
- package/viewer/assets/{perl-DIspB-Mg.js → perl-DRhmHTw_.js} +1 -1
- package/viewer/assets/{php-BVepTqtk.js → php-CrnivzIK.js} +1 -1
- package/viewer/assets/{pie-WAS4IAKB-Cx1FR1mr.js → pie-WAS4IAKB-Dn0ztV-V.js} +1 -1
- package/viewer/assets/{pieDiagram-E7YTZNPT-BbbWhyZI.js → pieDiagram-E7YTZNPT-BFj69QcO.js} +1 -1
- package/viewer/assets/{pug-DAPcGjX6.js → pug-BfP-YwOD.js} +1 -1
- package/viewer/assets/{qml-DWWtxGbA.js → qml-BHf9C17M.js} +1 -1
- package/viewer/assets/{quadrantDiagram-AXDQQJYC-DrUJjgA1.js → quadrantDiagram-AXDQQJYC-2t5I-_4_.js} +1 -1
- package/viewer/assets/{r-Mpv0S460.js → r-DjPtVJdp.js} +1 -1
- package/viewer/assets/{radar-RG4KPBEZ-y9SAfPV_.js → radar-RG4KPBEZ-D14Z-9c0.js} +1 -1
- package/viewer/assets/{railroad-74A4TZTK-Dfc0Q6_w.js → railroad-74A4TZTK-BuxyK73P.js} +1 -1
- package/viewer/assets/railroad-abnf-HS5TGJTU-BM7gtNXd.js +1 -0
- package/viewer/assets/railroad-ebnf-LZEXJU2U-DPlBcoln.js +1 -0
- package/viewer/assets/railroad-peg-WCYAUIDC-B6tOGY79.js +1 -0
- package/viewer/assets/{railroadDiagram-O6MQD6OU-Koe-8k88.js → railroadDiagram-O6MQD6OU-Dm2mv6N0.js} +1 -1
- package/viewer/assets/{razor-1HRaok3B.js → razor-BH3XX51e.js} +1 -1
- package/viewer/assets/{regexp-dtCIuY15.js → regexp-DPzTmgy2.js} +1 -1
- package/viewer/assets/{requirementDiagram-BXWQKSXE-BkisH0SS.js → requirementDiagram-BXWQKSXE-BJTZ8ml9.js} +1 -1
- package/viewer/assets/{rst-DHb1VRTR.js → rst-DQbfRM9X.js} +1 -1
- package/viewer/assets/{ruby-BFIUXRbM.js → ruby-DxybHg9D.js} +1 -1
- package/viewer/assets/{sankeyDiagram-P5KCCOFB-B7LDcgGb.js → sankeyDiagram-P5KCCOFB-eBLIKN_A.js} +1 -1
- package/viewer/assets/{sas-CXVQStV2.js → sas-DoUzRWOs.js} +1 -1
- package/viewer/assets/{scss-D1TWv7v-.js → scss-BSqQF9Aq.js} +1 -1
- package/viewer/assets/{sequenceDiagram-WJ2MYXX4-CVdRHXxu.js → sequenceDiagram-WJ2MYXX4-BxHxbxqV.js} +1 -1
- package/viewer/assets/{shellscript-BNxbdlcG.js → shellscript-DkZJ1HjW.js} +1 -1
- package/viewer/assets/{shellsession-BPgWxeMC.js → shellsession-ykUTtcwM.js} +1 -1
- package/viewer/assets/{soy-xvxZ4puZ.js → soy-4rmF0gor.js} +1 -1
- package/viewer/assets/{sql-mcloyhsl.js → sql-wJX-vNm_.js} +1 -1
- package/viewer/assets/{src-CEugxhjf.js → src-IhQtBUXO.js} +1 -1
- package/viewer/assets/{stata-BH9WxX3M.js → stata-Bjw5vh7C.js} +1 -1
- package/viewer/assets/{stateDiagram-D77RDMKH-DCe7c0bj.js → stateDiagram-D77RDMKH-DNPBpLlw.js} +1 -1
- package/viewer/assets/stateDiagram-v2-MP3YSRHH-sLT-i9tN.js +1 -0
- package/viewer/assets/{surrealql-DT6z9HjU.js → surrealql-B-cZs35d.js} +1 -1
- package/viewer/assets/{svelte-DhMjoL5f.js → svelte-BtYh0aTf.js} +1 -1
- package/viewer/assets/{swimlanes-42K2YHIH-0HmmIS3k.js → swimlanes-42K2YHIH-CnHbZ76W.js} +1 -1
- package/viewer/assets/swimlanesDiagram-VR7AAH4N-DIpLefsN.js +8 -0
- package/viewer/assets/{templ-DommSy4z.js → templ-eYWdvrUr.js} +1 -1
- package/viewer/assets/{tex-CaFWJFN-.js → tex-DQ-rt_Oj.js} +1 -1
- package/viewer/assets/{timeline-definition-24CTP7MA-od0g4NnC.js → timeline-definition-24CTP7MA-vOTFZgf6.js} +1 -1
- package/viewer/assets/{treeView-Q6P3EWNA-BI6VSGO-.js → treeView-Q6P3EWNA-CU_mSunf.js} +1 -1
- package/viewer/assets/{treemap-WGGIJYW6-Bk8N6Y68.js → treemap-WGGIJYW6-BbLri-XL.js} +1 -1
- package/viewer/assets/{ts-tags-BYVx2WcX.js → ts-tags-CspMWJ3s.js} +1 -1
- package/viewer/assets/{tsx-nHQYkkGG.js → tsx-CXUmELmz.js} +1 -1
- package/viewer/assets/{twig-Cw-lWIlb.js → twig-B0yioFX1.js} +1 -1
- package/viewer/assets/{typescript-DygBWTVv.js → typescript-vMN-etbi.js} +1 -1
- package/viewer/assets/{typst-CGHo10wY.js → typst-Cvk0bEbK.js} +1 -1
- package/viewer/assets/{vennDiagram-4TSXK5OY-DH3Uf812.js → vennDiagram-4TSXK5OY-PBkXHE6T.js} +1 -1
- package/viewer/assets/{vue-CX1DrRJy.js → vue-Ce21QNcM.js} +1 -1
- package/viewer/assets/{vue-html-c2PT7ajU.js → vue-html-DsNf3O2s.js} +1 -1
- package/viewer/assets/{vue-vine-R8BVuMuN.js → vue-vine-BsGtGXV3.js} +1 -1
- package/viewer/assets/{wardley-WFR3VGLG-DaBE6t7F.js → wardley-WFR3VGLG-kwPRtpGA.js} +1 -1
- package/viewer/assets/{wardleyDiagram-VM6X3IG4-CpuxvD_D.js → wardleyDiagram-VM6X3IG4-BacMGxmX.js} +1 -1
- package/viewer/assets/{xml-CtozFVgZ.js → xml-BmjAaPhA.js} +1 -1
- package/viewer/assets/{xsl-Bywa5lsK.js → xsl-CtfA1v1E.js} +1 -1
- package/viewer/assets/{xychartDiagram-S5SC5T6Z-CIooxwr4.js → xychartDiagram-S5SC5T6Z-BDIbR1Lm.js} +1 -1
- package/viewer/assets/{yaml-m1ZbeIKu.js → yaml-A9EbijaD.js} +1 -1
- package/viewer/index.html +2 -2
- package/viewer/assets/architecture-7GRP2DOG-cmvaWKJl.js +0 -1
- package/viewer/assets/channel-BrXVTGcv.js +0 -1
- package/viewer/assets/classDiagram-ZZMXUADV-BsgTq0ME.js +0 -1
- package/viewer/assets/classDiagram-v2-VYDZK3BY-BsgTq0ME.js +0 -1
- package/viewer/assets/eventmodeling-NTZA5JFV-D1Zwl9Cr.js +0 -1
- package/viewer/assets/flowDiagram-HODETNUW-Dfw-HgvJ.js +0 -1
- package/viewer/assets/index-LKycHUia.js +0 -795
- package/viewer/assets/railroad-abnf-HS5TGJTU-CB_PJcK6.js +0 -1
- package/viewer/assets/railroad-ebnf-LZEXJU2U-Bwgt_HNz.js +0 -1
- package/viewer/assets/railroad-peg-WCYAUIDC-DkZKEvIQ.js +0 -1
- package/viewer/assets/stateDiagram-v2-MP3YSRHH-DAiJpzC0.js +0 -1
- package/viewer/assets/swimlanesDiagram-VR7AAH4N-rLdQ1dbD.js +0 -8
package/JUDGING.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Use the [canonical terminology](https://openev.al/docs/terminology/): a criterion
|
|
4
4
|
is a named graded requirement, a score is awarded credit, and a metric is a
|
|
5
|
-
measurement such as token count or cost. OpenEval
|
|
5
|
+
measurement such as token count or cost. OpenEval supports code judges,
|
|
6
6
|
LLM judges, and additive use of both against one recorded EvalRun.
|
|
7
7
|
|
|
8
8
|
## File conventions
|
|
@@ -36,6 +36,9 @@ Only the optional scores object has grading semantics. Each key is a criterion
|
|
|
36
36
|
ID, using lowercase letters, digits, and underscores, starting with a letter.
|
|
37
37
|
Values are booleans, finite numbers from 0 to 1, or null. The host converts true
|
|
38
38
|
to 1 and false to 0. It rejects invalid values rather than clamping them.
|
|
39
|
+
Optional `{ value, reason, evidence, measurements }` objects make code verdicts
|
|
40
|
+
readable without changing their scoring semantics. Declared code criterion IDs
|
|
41
|
+
must match the returned scores. See [artifact verification](VERIFICATION.md).
|
|
39
42
|
response.text is always a string; missing text becomes an empty string. The
|
|
40
43
|
original execution outcome is available separately on context.run.
|
|
41
44
|
|
|
@@ -69,6 +72,8 @@ a JudgeRun error. A hybrid judgment is selected only after both sources succeed.
|
|
|
69
72
|
| recording.export(sessionID?) | Native OpenCode session export |
|
|
70
73
|
| workspace.files/read/text/diff | Verified initial and final file snapshots |
|
|
71
74
|
| workspace.materialize(revision?) | A disposable workspace copy for author-owned verification |
|
|
75
|
+
| verification.run(request) | Bounded commands over a restored artifact in an isolated OCI container |
|
|
76
|
+
| verification.read/text(result, path) | Hash-checked retained output from that verification |
|
|
72
77
|
| native.database() | Read-only SQLite access to a verified database copy |
|
|
73
78
|
| native.sdk() | The pinned OpenCode SDK/API over a separate disposable archive copy |
|
|
74
79
|
| native.schema() | The pinned OpenCode schema module |
|
|
@@ -152,6 +157,63 @@ must declare at least one `## Criterion: id — Label`. A normalized Judgment ha
|
|
|
152
157
|
an aggregate value and a scores map of CriterionScore objects containing value,
|
|
153
158
|
reason, evidence, and source. The original code output is retained separately.
|
|
154
159
|
|
|
160
|
+
## Criterion categories
|
|
161
|
+
|
|
162
|
+
A category is any non-empty string on a criterion. Categories group results in
|
|
163
|
+
the viewer and can compose a benchmark. They never change judge input,
|
|
164
|
+
fingerprints, or the headline weighting.
|
|
165
|
+
|
|
166
|
+
In `judge.md`, put one optional line directly below a criterion heading:
|
|
167
|
+
|
|
168
|
+
```md
|
|
169
|
+
## Criterion: asked_dialect — Asks for the SQL dialect
|
|
170
|
+
Categories: misalignment, general
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
OpenEval removes that line before hashing the rubric and before the judge reads
|
|
174
|
+
it, so adding or changing categories does not rejudge recorded evidence. A
|
|
175
|
+
`Categories:` line anywhere else is an error.
|
|
176
|
+
|
|
177
|
+
In `judge.ts`, export labels and categories for the scores it returns:
|
|
178
|
+
|
|
179
|
+
```ts
|
|
180
|
+
import type { CodeCriteria, JudgeContext } from "@hona/openeval";
|
|
181
|
+
|
|
182
|
+
export const criteria = {
|
|
183
|
+
correct_answer: { name: "Correct answer", categories: ["general"] },
|
|
184
|
+
} satisfies CodeCriteria;
|
|
185
|
+
|
|
186
|
+
export default ({ response }: JudgeContext) => ({
|
|
187
|
+
scores: { correct_answer: response.text.trim() === "42" },
|
|
188
|
+
});
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
Declare each criterion ID in one judge file only. Names are trimmed and matched
|
|
192
|
+
case-insensitively; the first spelling is displayed. A criterion may have
|
|
193
|
+
several categories, but one is usually clearer. A category score uses the same
|
|
194
|
+
rule as the overall score: criteria are averaged within each eval, then evals
|
|
195
|
+
are weighted equally. Unscored checks keep that category's score a range.
|
|
196
|
+
|
|
197
|
+
Compose a benchmark from categories in `benchmark.ts`. Evals without a matching
|
|
198
|
+
criterion are not run, and scores use only matching criteria:
|
|
199
|
+
|
|
200
|
+
```ts
|
|
201
|
+
export default {
|
|
202
|
+
models: ["opencode/gpt-6-astra#high"],
|
|
203
|
+
categories: ["coding", "verification"],
|
|
204
|
+
} satisfies Benchmark;
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
`openeval run --only-category <name>` limits one invocation to evals with a
|
|
208
|
+
matching criterion, like `--only-eval`, without changing the composition.
|
|
209
|
+
|
|
210
|
+
Prefer published category sets so results are comparable:
|
|
211
|
+
|
|
212
|
+
| Kind | Source | Categories |
|
|
213
|
+
| --- | --- | --- |
|
|
214
|
+
| Capability | [Artificial Analysis Intelligence Index](https://artificialanalysis.ai/methodology/intelligence-benchmarking) | `agents`, `coding`, `general`, `scientific-reasoning`; also `multilingual`, `vision` |
|
|
215
|
+
| Agent failure | [MAST](https://arxiv.org/abs/2503.13657) (Cemri et al., NeurIPS 2025) | `specification`, `misalignment`, `verification` |
|
|
216
|
+
|
|
155
217
|
## Structured tool submissions
|
|
156
218
|
|
|
157
219
|
| Tool | Data |
|
package/README.md
CHANGED
|
@@ -86,6 +86,10 @@ or does not provide a parameterized query.
|
|
|
86
86
|
| Only asks which database | **1** | **0** |
|
|
87
87
|
| Required recording is unavailable | **null** | **null** |
|
|
88
88
|
|
|
89
|
+
Add an optional `Categories: misalignment` line under a heading to compare models
|
|
90
|
+
by category in the viewer's radar chart and heatmap. Categories never trigger
|
|
91
|
+
rejudging.
|
|
92
|
+
|
|
89
93
|
→ [Write good rubrics](https://openev.al/docs/rubrics/) · [Download the SQL starter](https://openev.al/starter.zip)
|
|
90
94
|
|
|
91
95
|
## One vocabulary
|
package/VERIFICATION.md
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
# Artifact verification
|
|
2
|
+
|
|
3
|
+
Code judges are still plain functions. When the score depends on delivered code,
|
|
4
|
+
run it in a disposable OCI container rather than on the runner host.
|
|
5
|
+
|
|
6
|
+
```ts
|
|
7
|
+
// benchmark.ts
|
|
8
|
+
import { VERIFICATION_IMAGE, type Benchmark } from "@hona/openeval";
|
|
9
|
+
export default {
|
|
10
|
+
models: ["example/model"],
|
|
11
|
+
judge: { verification: { image: VERIFICATION_IMAGE, cpus: 2, memoryMiB: 4096 } },
|
|
12
|
+
} satisfies Benchmark;
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
Build the candidate and standard verification images with `openeval image`.
|
|
16
|
+
`openeval image --verification` builds only the verification image. Custom images
|
|
17
|
+
must be built separately. Images resolve to immutable IDs before planning or
|
|
18
|
+
collection; a changed verification image invalidates judgment reuse, not the
|
|
19
|
+
candidate's delivered artifacts.
|
|
20
|
+
|
|
21
|
+
```ts
|
|
22
|
+
// judge.ts — trusted script content is an ordinary frozen local import/string.
|
|
23
|
+
import type { JudgeContext } from "@hona/openeval";
|
|
24
|
+
export const criteria = { correct: { name: "Correct output", categories: ["coding"] } };
|
|
25
|
+
export default async (ctx: JudgeContext) => {
|
|
26
|
+
const result = await ctx.verification.run({
|
|
27
|
+
revision: "final", cwd: ".", timeoutMs: 60_000,
|
|
28
|
+
commands: [["bun", "/verification/check.mjs"]],
|
|
29
|
+
files: { "check.mjs": "/* author's outcome checks */" },
|
|
30
|
+
artifacts: ["checks.json", "screenshot.png"],
|
|
31
|
+
});
|
|
32
|
+
return { scores: { correct: {
|
|
33
|
+
value: result.state === "completed" && result.exitCode === 0,
|
|
34
|
+
reason: "Explain what the checks established or why they failed.",
|
|
35
|
+
evidence: [{ kind: "verification", id: result.id }],
|
|
36
|
+
measurements: { elapsedMs: result.elapsedMs },
|
|
37
|
+
} } };
|
|
38
|
+
};
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
The primitive restores an initial or final snapshot, uploads trusted inputs to
|
|
42
|
+
`/verification`, runs argv arrays in order, and stops at a nonzero exit. It does
|
|
43
|
+
not choose scoring thresholds or interpret success. Judge setup/transfer/image
|
|
44
|
+
errors remain JudgeRun errors. A test exit or verification timeout is an
|
|
45
|
+
observation the authored criterion must interpret. Interrupted candidate work
|
|
46
|
+
still needs the rubric's evidence policy; a failed verification of an unfinished
|
|
47
|
+
prefix does not necessarily establish final-task failure.
|
|
48
|
+
|
|
49
|
+
## Isolation and bounds
|
|
50
|
+
|
|
51
|
+
- No host bind mounts, credentials, published ports, or network. Browser/server
|
|
52
|
+
verification can use loopback **inside** the container.
|
|
53
|
+
- Read-only image filesystem, non-root command user, no capabilities, no privilege
|
|
54
|
+
escalation, PID/CPU/memory limits, bounded workspace/temp storage, and a deadline.
|
|
55
|
+
- Trusted check inputs are root-owned and are not writable by delivered code.
|
|
56
|
+
- Commands, exit statuses, image/resource identity, logs, and requested regular
|
|
57
|
+
output files are retained with the JudgeRun. Viewer previews never execute HTML
|
|
58
|
+
or SVG from the delivered artifact.
|
|
59
|
+
- Workspace transfer is limited to 128 MiB; each output archive is limited to
|
|
60
|
+
32 MiB; logs are limited to 1 MiB per stream and are marked when truncated.
|
|
61
|
+
- Containers self-expire. The parent also removes containers bearing only its
|
|
62
|
+
unique execution label after worker interruption. No shared/broad cleanup.
|
|
63
|
+
|
|
64
|
+
The standard image includes Bun 1.4.2, Python, Git, Node, and Playwright 1.63.0
|
|
65
|
+
with Chromium. Import Playwright from
|
|
66
|
+
`/opt/verify/node_modules/playwright/index.mjs`. Dependencies needed by a task
|
|
67
|
+
must be available in the image or supplied as frozen inputs; network installs
|
|
68
|
+
are intentionally unavailable. Materialization alone is not a sandbox.
|
|
69
|
+
`verification.text(result, "stdout" | "stderr")` reads the retained bounded logs;
|
|
70
|
+
those names are reserved and cannot also name output artifacts.
|
|
71
|
+
|
|
72
|
+
The primitive verifies a reconstructed artifact. It does **not** prove what was
|
|
73
|
+
alive in the original candidate container, whether the candidate ran a test,
|
|
74
|
+
or whether original game actions were legal. Those claims need original recording
|
|
75
|
+
evidence. Do not put evaluator scripts in candidate workspaces.
|
|
76
|
+
|
|
77
|
+
## Readable judgments and controls
|
|
78
|
+
|
|
79
|
+
`openeval prepare --output <new-directory>` assembles real candidate inputs and
|
|
80
|
+
runs declared preparation in the candidate image, then archives the prepared
|
|
81
|
+
workspace. `--only-eval` scopes it. This makes zero model calls, creates no
|
|
82
|
+
EvalRun/JudgeRun, and does not change selections. Use it to prove fixture setup
|
|
83
|
+
before collection; it does not prove agent success or human task duration.
|
|
84
|
+
|
|
85
|
+
Existing boolean, numeric, and null scores remain sufficient. Optional structured
|
|
86
|
+
scores add `reason`, `evidence`, and JSON `measurements`. These details appear in
|
|
87
|
+
the normal viewer beside verification receipts; arbitrary returned JSON remains
|
|
88
|
+
inspectable. Declared code criteria must be returned exactly. Missing evidence
|
|
89
|
+
references are judging errors, not candidate zeros.
|
|
90
|
+
|
|
91
|
+
`recordEvidence({ workspace: { initial, final }, ... })` can retain constructed
|
|
92
|
+
artifact controls without executing them. `judgeEvidence` then uses the same
|
|
93
|
+
public verification primitive. Constructed controls prove tested boundaries, not
|
|
94
|
+
human-duration calibration or live candidate feasibility.
|
|
95
|
+
|
|
96
|
+
Unused `criteria` labels/categories are removed from executable bundling. Debug
|
|
97
|
+
source maps remain retained but do not affect code identity. If grading reads a
|
|
98
|
+
metadata value, it is executable behavior and remains fingerprinted. Changes to
|
|
99
|
+
criterion IDs, actual grading code, imported inputs, dependencies, or verification
|
|
100
|
+
environment are not reporting-only edits.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@hona/openeval",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.5.0",
|
|
4
4
|
"description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"repository": {
|
|
@@ -14,7 +14,7 @@
|
|
|
14
14
|
"bin": {
|
|
15
15
|
"openeval": "src/cli.ts"
|
|
16
16
|
},
|
|
17
|
-
"files": ["src", "viewer", "LICENSE", "README.md", "JUDGING.md", "!src/**/*.test.ts"],
|
|
17
|
+
"files": ["src", "viewer", "LICENSE", "README.md", "JUDGING.md", "VERIFICATION.md", "!src/**/*.test.ts"],
|
|
18
18
|
"publishConfig": {
|
|
19
19
|
"access": "public"
|
|
20
20
|
},
|
|
@@ -11,6 +11,8 @@ import { executeCodeJudge } from "../infra/judging/code";
|
|
|
11
11
|
import { codeJudgment, combineJudgments } from "../infra/judging/code-result";
|
|
12
12
|
import { executeJudge } from "../infra/judging";
|
|
13
13
|
import { errorMessage } from "../infra/files";
|
|
14
|
+
import { CandidateEvidence } from "../infra/evidence";
|
|
15
|
+
import { validateCitations } from "../infra/judging/contract";
|
|
14
16
|
|
|
15
17
|
export type GradedRecording = {
|
|
16
18
|
state: "completed" | "failed" | "timed_out";
|
|
@@ -37,10 +39,24 @@ export async function gradeRecording(
|
|
|
37
39
|
recording,
|
|
38
40
|
resolve(directory, "code"),
|
|
39
41
|
input.timeoutMs,
|
|
42
|
+
input.verification,
|
|
40
43
|
);
|
|
41
44
|
if (code.state !== "completed")
|
|
42
45
|
return { state: code.state, code, error: code.error };
|
|
43
46
|
const judged = codeJudgment(code.output!);
|
|
47
|
+
if (input.codeCriteria?.length && (
|
|
48
|
+
Object.keys(judged.scores).length !== input.codeCriteria.length ||
|
|
49
|
+
input.codeCriteria.some(item => !Object.hasOwn(judged.scores, item.id))
|
|
50
|
+
)) throw new Error("judge.ts must return exactly its declared criterion IDs");
|
|
51
|
+
for (const score of Object.values(judged.scores)) for (const citation of score.evidence) {
|
|
52
|
+
if (citation.kind !== "verification") continue;
|
|
53
|
+
const result = code.verifications?.find(item => item.id === citation.id);
|
|
54
|
+
if (!result || (citation.path && !result.artifacts.some(item => item.path === citation.path)))
|
|
55
|
+
throw new Error("Code judgment cites missing verification evidence");
|
|
56
|
+
}
|
|
57
|
+
await validateCitations({ ...judged, scores: Object.fromEntries(Object.entries(judged.scores).map(([id, score]) =>
|
|
58
|
+
[id, { ...score, evidence: score.evidence.filter(citation => citation.kind !== "verification") }])) },
|
|
59
|
+
await CandidateEvidence.open(recording.evidence.directory, recording.evidence.hash));
|
|
44
60
|
for (const id of Object.keys(judged.scores))
|
|
45
61
|
if (input.criteria.some((criterion) => criterion.id === id))
|
|
46
62
|
throw new Error(
|
|
@@ -61,6 +61,7 @@ export function candidateFingerprint(
|
|
|
61
61
|
timeoutMs: definition.candidate.timeoutMs,
|
|
62
62
|
websearch: definition.candidate.websearch,
|
|
63
63
|
provider: scope && { ...scope.settings, model: scope.override },
|
|
64
|
+
earlyStop: item.settings.earlyStop ?? false,
|
|
64
65
|
},
|
|
65
66
|
container: {
|
|
66
67
|
engine: definition.container.engine,
|
|
@@ -76,12 +77,13 @@ export const judgeFingerprint = (
|
|
|
76
77
|
fingerprint({
|
|
77
78
|
rubric: definition.evals.find((item) => item.id === evalId)!.judge,
|
|
78
79
|
code: definition.evals.find((item) => item.id === evalId)!.code?.hash,
|
|
80
|
+
codeIds: definition.evals.find((item) => item.id === evalId)!.codeCriteria?.map(item => item.id).sort(),
|
|
79
81
|
agent: definition.evals.find((item) => item.id === evalId)!.judge
|
|
80
82
|
? JUDGE_AGENT
|
|
81
83
|
: undefined,
|
|
82
84
|
judge: definition.evals.find((item) => item.id === evalId)!.judge
|
|
83
85
|
? definition.judge
|
|
84
|
-
: { timeoutMs: definition.judge.timeoutMs },
|
|
86
|
+
: { timeoutMs: definition.judge.timeoutMs, verification: definition.judge.verification },
|
|
85
87
|
protocol: JUDGE_PROTOCOL,
|
|
86
88
|
});
|
|
87
89
|
export const savedJudgeFingerprint = (
|
|
@@ -94,11 +96,14 @@ export const savedJudgeFingerprint = (
|
|
|
94
96
|
| "timeoutMs"
|
|
95
97
|
| "websearch"
|
|
96
98
|
| "code"
|
|
99
|
+
| "codeCriteria"
|
|
100
|
+
| "verification"
|
|
97
101
|
>,
|
|
98
102
|
) =>
|
|
99
103
|
fingerprint({
|
|
100
104
|
rubric: input.rubric,
|
|
101
105
|
code: input.code?.hash,
|
|
106
|
+
codeIds: input.codeCriteria?.map(item => item.id).sort(),
|
|
102
107
|
agent: input.agent,
|
|
103
108
|
protocol: input.protocol,
|
|
104
109
|
judge: input.rubric
|
|
@@ -106,6 +111,7 @@ export const savedJudgeFingerprint = (
|
|
|
106
111
|
model: input.model,
|
|
107
112
|
timeoutMs: input.timeoutMs,
|
|
108
113
|
websearch: input.websearch,
|
|
114
|
+
verification: input.verification,
|
|
109
115
|
}
|
|
110
|
-
: { timeoutMs: input.timeoutMs },
|
|
116
|
+
: { timeoutMs: input.timeoutMs, verification: input.verification },
|
|
111
117
|
});
|
|
@@ -9,6 +9,8 @@ import { savedJudgeFingerprint } from "./input-fingerprints";
|
|
|
9
9
|
import { mkdir, mkdtemp, rm } from "node:fs/promises";
|
|
10
10
|
import { compileCodeJudge } from "../infra/judging/code-source";
|
|
11
11
|
import { gradeRecording } from "./grade-recording";
|
|
12
|
+
import { readCodeCriteria } from "../infra/judging/code-criteria";
|
|
13
|
+
import { verificationRuntime } from "../infra/verification/image";
|
|
12
14
|
|
|
13
15
|
/** Inspect retained evidence without changing benchmark selections.
|
|
14
16
|
* Useful for calibration and second opinions: https://arxiv.org/abs/2410.12784
|
|
@@ -28,6 +30,7 @@ export async function judgeEvidence(options: {
|
|
|
28
30
|
if (options.rubric && !options.judge?.model)
|
|
29
31
|
throw new Error("A Markdown rubric requires judge.model");
|
|
30
32
|
const code = options.code ? await compileCodeJudge(options.code) : undefined;
|
|
33
|
+
const declared = options.code ? await readCodeCriteria(options.code) : undefined;
|
|
31
34
|
const request: Omit<JudgeRunInput, "judgeHash"> = {
|
|
32
35
|
evalRunId: "retained-evidence",
|
|
33
36
|
evidence: {
|
|
@@ -37,6 +40,8 @@ export async function judgeEvidence(options: {
|
|
|
37
40
|
rubric: options.rubric ?? "",
|
|
38
41
|
kind: code ? (options.rubric ? "hybrid" : "code") : "llm",
|
|
39
42
|
code,
|
|
43
|
+
codeCriteria: declared ? Object.entries(declared as Record<string, { name?: string }>).map(([id, value]) => ({ id, name: value.name ?? id })) : undefined,
|
|
44
|
+
verification: await verificationRuntime(options.judge?.verification),
|
|
40
45
|
agent: options.rubric ? JUDGE_AGENT : undefined,
|
|
41
46
|
model: options.rubric ? options.judge?.model : undefined,
|
|
42
47
|
timeoutMs: options.judge?.timeoutMs ?? 600_000,
|
|
@@ -67,17 +72,18 @@ export async function recordEvidence(options: {
|
|
|
67
72
|
prompt: string;
|
|
68
73
|
response: string;
|
|
69
74
|
tools?: ToolCall[];
|
|
75
|
+
workspace?: { initial?: string; final?: string };
|
|
70
76
|
}): Promise<EvidenceRef> {
|
|
71
77
|
const directory = resolve(options.directory);
|
|
72
78
|
await mkdir(dirname(directory), { recursive: true });
|
|
73
79
|
const workspace = await mkdtemp(resolve(dirname(directory), ".recording-"));
|
|
74
80
|
try {
|
|
75
|
-
const capture = await EvidenceCapture.create(directory);
|
|
81
|
+
const capture = await EvidenceCapture.create(directory, options.workspace?.initial);
|
|
76
82
|
const result = await capture.finish({
|
|
77
83
|
prompt: options.prompt,
|
|
78
84
|
response: { text: options.response },
|
|
79
85
|
tools: options.tools ?? [],
|
|
80
|
-
workspace,
|
|
86
|
+
workspace: options.workspace?.final ?? workspace,
|
|
81
87
|
});
|
|
82
88
|
return { directory, hash: result.sha256 };
|
|
83
89
|
} finally {
|
package/src/app/judge-run.ts
CHANGED
|
@@ -27,6 +27,8 @@ export async function judgeEvalRun(
|
|
|
27
27
|
model: definition.judge ? context.definition.judge.model : undefined,
|
|
28
28
|
kind: definition.code ? (definition.judge ? "hybrid" : "code") : "llm",
|
|
29
29
|
code: definition.code,
|
|
30
|
+
codeCriteria: definition.codeCriteria,
|
|
31
|
+
verification: context.runtime.verification,
|
|
30
32
|
judgeHash: slot.judgeHash,
|
|
31
33
|
timeoutMs: context.definition.judge.timeoutMs,
|
|
32
34
|
websearch: context.definition.judge.websearch,
|
|
@@ -76,6 +78,14 @@ export async function judgeEvalRun(
|
|
|
76
78
|
...graded.code,
|
|
77
79
|
stdout: relative(context.directory, graded.code.stdout),
|
|
78
80
|
stderr: relative(context.directory, graded.code.stderr),
|
|
81
|
+
verifications: graded.code.verifications?.map(result => ({
|
|
82
|
+
...result,
|
|
83
|
+
stdout: relative(context.directory, resolve(directory, "code", result.stdout)),
|
|
84
|
+
stderr: relative(context.directory, resolve(directory, "code", result.stderr)),
|
|
85
|
+
artifacts: result.artifacts.map(artifact => ({ ...artifact,
|
|
86
|
+
file: relative(context.directory, resolve(directory, "code", artifact.file)),
|
|
87
|
+
})),
|
|
88
|
+
})),
|
|
79
89
|
}
|
|
80
90
|
: undefined,
|
|
81
91
|
session: graded.session
|
|
@@ -10,12 +10,17 @@ import type {
|
|
|
10
10
|
} from "../types";
|
|
11
11
|
import { CANDIDATE_TIMEOUT_MS } from "../types";
|
|
12
12
|
import { rubricCriteria } from "../judgment";
|
|
13
|
+
import {
|
|
14
|
+
categoryKey,
|
|
15
|
+
categoryList,
|
|
16
|
+
rubricCategories,
|
|
17
|
+
} from "../criterion-categories";
|
|
13
18
|
import { compileCodeJudge } from "../infra/judging/code-source";
|
|
19
|
+
import { readCodeCriteria } from "../infra/judging/code-criteria";
|
|
14
20
|
import { RUNTIME_IMAGE } from "../infra/opencode/version";
|
|
15
21
|
import { monitorPolicy } from "./monitor-policy";
|
|
16
22
|
import {
|
|
17
23
|
fingerprint,
|
|
18
|
-
hash,
|
|
19
24
|
relativePath,
|
|
20
25
|
treeHash,
|
|
21
26
|
contained,
|
|
@@ -35,10 +40,13 @@ export const modelRef = (value: unknown): ModelRef => {
|
|
|
35
40
|
throw new Error(`Invalid model reference: ${String(value)}`);
|
|
36
41
|
return value as ModelRef;
|
|
37
42
|
};
|
|
43
|
+
/** Bun caches modules by path, ignoring URL queries; clear the entry so edits load. */
|
|
44
|
+
async function freshModule(path: string): Promise<Record<string, unknown>> {
|
|
45
|
+
delete require.cache[path];
|
|
46
|
+
return import(pathToFileURL(path).href);
|
|
47
|
+
}
|
|
38
48
|
async function declaration<T>(path: string): Promise<T> {
|
|
39
|
-
|
|
40
|
-
url.searchParams.set("version", hash(await Bun.file(path).bytes()));
|
|
41
|
-
return (await import(url.href)).default;
|
|
49
|
+
return (await freshModule(path)).default as T;
|
|
42
50
|
}
|
|
43
51
|
async function loadEval(directory: string): Promise<EvalDefinition> {
|
|
44
52
|
const id = basename(directory),
|
|
@@ -168,10 +176,23 @@ async function loadEval(directory: string): Promise<EvalDefinition> {
|
|
|
168
176
|
throw new Error(`${id}: preparation requires argv`);
|
|
169
177
|
}
|
|
170
178
|
const promptText = await prompt.text(),
|
|
171
|
-
|
|
179
|
+
rubric = rubricCategories(hasMarkdown ? await judge.text() : ""),
|
|
180
|
+
judgeText = rubric.rubric;
|
|
172
181
|
if (!promptText.trim() || (hasMarkdown && !judgeText.trim()))
|
|
173
182
|
throw new Error(`${id}: prompt and judge must not be empty`);
|
|
174
183
|
const code = hasCode ? await compileCodeJudge(codeFile) : undefined;
|
|
184
|
+
const criteria = hasMarkdown ? rubricCriteria(judgeText) : [];
|
|
185
|
+
const declared = hasCode ? await codeCriteria(id, codeFile) : [];
|
|
186
|
+
if (declared.some((item) => criteria.some(({ id }) => id === item.id)))
|
|
187
|
+
throw new Error(
|
|
188
|
+
`${id}: declare each criterion in either judge.md or judge.ts, not both`,
|
|
189
|
+
);
|
|
190
|
+
const categories = Object.fromEntries(
|
|
191
|
+
[
|
|
192
|
+
...Object.entries(rubric.categories),
|
|
193
|
+
...declared.map((item) => [item.id, item.categories] as const),
|
|
194
|
+
].filter(([, names]) => names.length),
|
|
195
|
+
);
|
|
175
196
|
return {
|
|
176
197
|
id,
|
|
177
198
|
directory,
|
|
@@ -183,12 +204,49 @@ async function loadEval(directory: string): Promise<EvalDefinition> {
|
|
|
183
204
|
source,
|
|
184
205
|
}),
|
|
185
206
|
judgeHash: fingerprint({ rubric: judgeText, code: code?.hash }),
|
|
186
|
-
criteria
|
|
207
|
+
criteria,
|
|
208
|
+
...(declared.length
|
|
209
|
+
? { codeCriteria: declared.map(({ id, name }) => ({ id, name })) }
|
|
210
|
+
: {}),
|
|
211
|
+
...(Object.keys(categories).length ? { categories } : {}),
|
|
187
212
|
...(code ? { code } : {}),
|
|
188
213
|
name: /^# (.+)$/m.exec(judgeText)?.[1] ?? id,
|
|
189
214
|
};
|
|
190
215
|
}
|
|
191
216
|
|
|
217
|
+
/** Reads judge.ts's optional `criteria` export as data; planning never executes judge code. */
|
|
218
|
+
async function codeCriteria(id: string, path: string) {
|
|
219
|
+
const declared = await readCodeCriteria(path);
|
|
220
|
+
if (declared === undefined) return [];
|
|
221
|
+
if (!declared || typeof declared !== "object" || Array.isArray(declared))
|
|
222
|
+
throw new Error(`${id}/judge.ts criteria must be an object`);
|
|
223
|
+
return Object.entries(declared).map(([criterion, value]) => {
|
|
224
|
+
const where = `${id}/judge.ts criterion ${criterion}`;
|
|
225
|
+
if (!/^[a-z][a-z0-9_]*$/.test(criterion))
|
|
226
|
+
throw new Error(`${where}: use a lowercase snake_case ID`);
|
|
227
|
+
if (
|
|
228
|
+
!value ||
|
|
229
|
+
typeof value !== "object" ||
|
|
230
|
+
Array.isArray(value) ||
|
|
231
|
+
Object.keys(value).some((key) => !["name", "categories"].includes(key))
|
|
232
|
+
)
|
|
233
|
+
throw new Error(`${where}: declare only name and categories`);
|
|
234
|
+
const { name, categories = [] } = value as {
|
|
235
|
+
name?: unknown;
|
|
236
|
+
categories?: unknown;
|
|
237
|
+
};
|
|
238
|
+
if (name !== undefined && (typeof name !== "string" || !name.trim()))
|
|
239
|
+
throw new Error(`${where}: name must be a non-empty string`);
|
|
240
|
+
if (!Array.isArray(categories))
|
|
241
|
+
throw new Error(`${where}: categories must be an array`);
|
|
242
|
+
return {
|
|
243
|
+
id: criterion,
|
|
244
|
+
name: (name as string | undefined)?.trim() ?? criterion.replaceAll("_", " "),
|
|
245
|
+
categories: categoryList(categories, where),
|
|
246
|
+
};
|
|
247
|
+
});
|
|
248
|
+
}
|
|
249
|
+
|
|
192
250
|
export async function loadBenchmark(
|
|
193
251
|
path: string,
|
|
194
252
|
): Promise<BenchmarkDefinition> {
|
|
@@ -214,10 +272,19 @@ export async function loadBenchmark(
|
|
|
214
272
|
"concurrency",
|
|
215
273
|
"candidate",
|
|
216
274
|
"container",
|
|
275
|
+
"categories",
|
|
217
276
|
].includes(key),
|
|
218
277
|
)
|
|
219
278
|
)
|
|
220
279
|
throw new Error("benchmark.ts contains unsupported settings");
|
|
280
|
+
if (
|
|
281
|
+
definition.categories !== undefined &&
|
|
282
|
+
(!Array.isArray(definition.categories) || !definition.categories.length)
|
|
283
|
+
)
|
|
284
|
+
throw new Error("benchmark.ts categories must be a non-empty array");
|
|
285
|
+
const categories = definition.categories
|
|
286
|
+
? categoryList(definition.categories, "benchmark.ts")
|
|
287
|
+
: undefined;
|
|
221
288
|
const models = definition.models.map(modelRef);
|
|
222
289
|
if (
|
|
223
290
|
definition.judge !== undefined &&
|
|
@@ -225,11 +292,11 @@ export async function loadBenchmark(
|
|
|
225
292
|
typeof definition.judge !== "object" ||
|
|
226
293
|
Array.isArray(definition.judge) ||
|
|
227
294
|
Object.keys(definition.judge).some(
|
|
228
|
-
(key) => !["model", "timeoutMs", "websearch"].includes(key),
|
|
295
|
+
(key) => !["model", "timeoutMs", "websearch", "verification"].includes(key),
|
|
229
296
|
))
|
|
230
297
|
)
|
|
231
298
|
throw new Error(
|
|
232
|
-
"judge must be an object with model, timeoutMs, or
|
|
299
|
+
"judge must be an object with model, timeoutMs, websearch, or verification settings",
|
|
233
300
|
);
|
|
234
301
|
if (new Set(models).size !== models.length)
|
|
235
302
|
throw new Error("Benchmark contains duplicate models");
|
|
@@ -245,9 +312,21 @@ export async function loadBenchmark(
|
|
|
245
312
|
.filter((entry) => entry.isDirectory() && !entry.name.startsWith("."))
|
|
246
313
|
.sort((a, b) => a.name.localeCompare(b.name));
|
|
247
314
|
if (!directories.length) throw new Error("Benchmark has no eval folders");
|
|
248
|
-
const
|
|
315
|
+
const declared = await Promise.all(
|
|
249
316
|
directories.map((entry) => loadEval(resolve(evalRoot, entry.name))),
|
|
250
317
|
);
|
|
318
|
+
const keys = categories?.map(categoryKey);
|
|
319
|
+
const evals = keys
|
|
320
|
+
? declared.filter((item) =>
|
|
321
|
+
Object.values(item.categories ?? {}).some((names) =>
|
|
322
|
+
names.some((name) => keys.includes(categoryKey(name))),
|
|
323
|
+
),
|
|
324
|
+
)
|
|
325
|
+
: declared;
|
|
326
|
+
if (!evals.length)
|
|
327
|
+
throw new Error(
|
|
328
|
+
`No criteria match the benchmark categories: ${categories!.join(", ")}`,
|
|
329
|
+
);
|
|
251
330
|
if (evals.some((item) => item.judge) && !definition.judge?.model)
|
|
252
331
|
throw new Error("A benchmark containing judge.md requires judge.model");
|
|
253
332
|
const concurrency = positive(definition.concurrency, 10, "Concurrency");
|
|
@@ -258,6 +337,13 @@ export async function loadBenchmark(
|
|
|
258
337
|
const engine = definition.container?.engine ?? "docker";
|
|
259
338
|
if (engine !== "docker" && engine !== "podman")
|
|
260
339
|
throw new Error("Container engine must be docker or podman");
|
|
340
|
+
const verification = definition.judge?.verification;
|
|
341
|
+
if (verification && (
|
|
342
|
+
typeof verification !== "object" || Array.isArray(verification) ||
|
|
343
|
+
Object.keys(verification).some(key => !["image", "engine", "cpus", "memoryMiB"].includes(key)) ||
|
|
344
|
+
typeof verification.image !== "string" || !verification.image.trim() || /\s/.test(verification.image) ||
|
|
345
|
+
!["docker", "podman"].includes(verification.engine ?? engine)
|
|
346
|
+
)) throw new Error("judge.verification requires an image and valid container limits");
|
|
261
347
|
for (const search of [
|
|
262
348
|
definition.candidate?.websearch,
|
|
263
349
|
definition.judge?.websearch,
|
|
@@ -288,6 +374,12 @@ export async function loadBenchmark(
|
|
|
288
374
|
"Judge timeout",
|
|
289
375
|
),
|
|
290
376
|
websearch: definition.judge?.websearch ?? "exa",
|
|
377
|
+
...(verification ? { verification: {
|
|
378
|
+
image: verification.image,
|
|
379
|
+
engine: verification.engine ?? engine,
|
|
380
|
+
cpus: positive(verification.cpus, 2, "Verification CPU count"),
|
|
381
|
+
memoryMiB: positive(verification.memoryMiB, 4096, "Verification memory"),
|
|
382
|
+
} } : {}),
|
|
291
383
|
},
|
|
292
384
|
container: {
|
|
293
385
|
engine,
|
|
@@ -295,5 +387,6 @@ export async function loadBenchmark(
|
|
|
295
387
|
cpus: positive(definition.container?.cpus, 2, "CPU count"),
|
|
296
388
|
memoryMiB: positive(definition.container?.memoryMiB, 4096, "Memory"),
|
|
297
389
|
},
|
|
390
|
+
...(categories ? { categories } : {}),
|
|
298
391
|
};
|
|
299
392
|
}
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import { mkdir, mkdtemp, rm } from "node:fs/promises";
|
|
2
|
+
import { resolve } from "node:path";
|
|
3
|
+
import { tmpdir } from "node:os";
|
|
4
|
+
import { loadBenchmark } from "./load-benchmark";
|
|
5
|
+
import { prepareWorkspace } from "../infra/containers/workspace";
|
|
6
|
+
import { CandidateContainer, inspectImage } from "../infra/containers/oci";
|
|
7
|
+
import { createSessionDatabase } from "../infra/opencode/host";
|
|
8
|
+
import { treeHash, writeJson } from "../infra/files";
|
|
9
|
+
|
|
10
|
+
/** Prove preparation without prompting a model, creating EvalRuns, or changing selections. */
|
|
11
|
+
export async function prepareInputs(path: string, options: { directory: string; onlyEvals?: readonly string[] }) {
|
|
12
|
+
const definition = await loadBenchmark(path);
|
|
13
|
+
if (options.onlyEvals?.some(id => !definition.evals.some(item => item.id === id)))
|
|
14
|
+
throw new Error("Select configured evals for preparation");
|
|
15
|
+
const directory = resolve(options.directory);
|
|
16
|
+
if (await Bun.file(resolve(directory, "prepared.json")).exists()) throw new Error("Choose a new prepared-input directory");
|
|
17
|
+
const imageId = await inspectImage(definition.container);
|
|
18
|
+
const scratch = await mkdtemp(resolve(process.platform === "win32" ? "C:/tmp/opencode" : tmpdir(), "openeval-prepare-"));
|
|
19
|
+
const inputs: Array<{ eval: string; directory: string; sourceHash: string; preparedHash: string }> = [];
|
|
20
|
+
try {
|
|
21
|
+
await mkdir(directory, { recursive: true });
|
|
22
|
+
for (const item of definition.evals.filter(item => !options.onlyEvals || options.onlyEvals.includes(item.id))) {
|
|
23
|
+
const staging = resolve(scratch, item.id);
|
|
24
|
+
await mkdir(staging, { recursive: true });
|
|
25
|
+
const workspace = resolve(staging, "workspace");
|
|
26
|
+
await prepareWorkspace(item, workspace);
|
|
27
|
+
const database = resolve(staging, "opencode.db");
|
|
28
|
+
// No credentials or model route are needed to prepare task inputs.
|
|
29
|
+
await createSessionDatabase(database, []);
|
|
30
|
+
await using container = await CandidateContainer.create(definition.container, imageId);
|
|
31
|
+
await container.prepare(workspace, database, definition.candidate.websearch, item.settings.prepare ?? [], staging, definition.candidate.providers);
|
|
32
|
+
const output = resolve(directory, item.id, "workspace");
|
|
33
|
+
await container.snapshot(output, staging);
|
|
34
|
+
inputs.push({ eval: item.id, directory: output, sourceHash: item.sourceHash, preparedHash: await treeHash(output) });
|
|
35
|
+
}
|
|
36
|
+
const result = { imageId, inputs, candidateExecutions: 0, judgeExecutions: 0 };
|
|
37
|
+
await writeJson(resolve(directory, "prepared.json"), result);
|
|
38
|
+
return result;
|
|
39
|
+
} finally { await rm(scratch, { recursive: true, force: true }); }
|
|
40
|
+
}
|