@hona/openeval 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/JUDGING.md +6 -1
- package/VERIFICATION.md +100 -0
- package/package.json +2 -2
- package/src/app/grade-recording.ts +16 -0
- package/src/app/input-fingerprints.ts +8 -2
- package/src/app/judge-evidence.ts +8 -2
- package/src/app/judge-run.ts +10 -0
- package/src/app/load-benchmark.ts +15 -2
- package/src/app/prepare-inputs.ts +40 -0
- package/src/app/read-results.ts +21 -4
- package/src/app/rejudge.ts +5 -0
- package/src/app/run-benchmark.ts +2 -1
- package/src/app/serve-results.ts +14 -0
- package/src/cli.ts +21 -2
- package/src/index.ts +8 -0
- package/src/infra/containers/oci.ts +4 -0
- package/src/infra/judging/code-result.ts +21 -3
- package/src/infra/judging/code-source.ts +10 -2
- package/src/infra/judging/code-worker.ts +5 -3
- package/src/infra/judging/code.ts +19 -3
- package/src/infra/judging/contract.ts +10 -2
- package/src/infra/judging/index.ts +1 -0
- package/src/infra/recording/index.ts +14 -2
- package/src/infra/verification/image.ts +33 -0
- package/src/infra/verification/runtime/Dockerfile +18 -0
- package/src/infra/verification/runtime/bun.lock +16 -0
- package/src/infra/verification/runtime/package.json +5 -0
- package/src/infra/verification/session.ts +199 -0
- package/src/judge-context.ts +66 -0
- package/src/types.ts +10 -2
- package/viewer/assets/{abnfDiagram-VCTEODGH-D-idCGaW.js → abnfDiagram-VCTEODGH-tXaAmCdr.js} +1 -1
- package/viewer/assets/{angular-html-B-7vkhmj.js → angular-html-CNhJPCRr.js} +1 -1
- package/viewer/assets/{angular-ts-CqJWLTIZ.js → angular-ts-BnGgGecY.js} +1 -1
- package/viewer/assets/{apl-DVTjqAhB.js → apl-ZOvmk1hE.js} +1 -1
- package/viewer/assets/{arc-Bmz8zsvq.js → arc-BalFY-0o.js} +1 -1
- package/viewer/assets/architecture-7GRP2DOG-Dwn_SxkP.js +1 -0
- package/viewer/assets/{architectureDiagram-5GKGNRK7-CxqP2ijS.js → architectureDiagram-5GKGNRK7-DpSDAEgb.js} +1 -1
- package/viewer/assets/{astro-Ih6QH4I8.js → astro-BJzTFZ3R.js} +1 -1
- package/viewer/assets/{blade-vzkWgA61.js → blade-CvS8_w1j.js} +1 -1
- package/viewer/assets/{blockDiagram-I7D4REHJ-Dm35S95Q.js → blockDiagram-I7D4REHJ-BY3upSR4.js} +1 -1
- package/viewer/assets/{c-092Q-y5e.js → c-D5ZSBtUG.js} +1 -1
- package/viewer/assets/{c4Diagram-7LVT6UL2-DArgyJNC.js → c4Diagram-7LVT6UL2-CA9keRc2.js} +1 -1
- package/viewer/assets/channel-CgM1_6P7.js +1 -0
- package/viewer/assets/{chapel-DUOt2X3X.js → chapel-1aYDxJri.js} +1 -1
- package/viewer/assets/{chunk-4HAMMTFA-Bm3UCwQ6.js → chunk-4HAMMTFA-B0uITxv4.js} +1 -1
- package/viewer/assets/{chunk-75Z2AOVW-CpmjsCJX.js → chunk-75Z2AOVW-CbTHx8Lz.js} +1 -1
- package/viewer/assets/{chunk-DU6HZSFF-DYT2KEwe.js → chunk-DU6HZSFF-C7mZpnch.js} +1 -1
- package/viewer/assets/{chunk-F27PBJKO-BFGd2OPj.js → chunk-F27PBJKO-iUZG9l9x.js} +1 -1
- package/viewer/assets/{chunk-GMAD6QVW-D8gzxWqz.js → chunk-GMAD6QVW-BYg8qa71.js} +1 -1
- package/viewer/assets/{chunk-GVQU2GXP-BZJsi-wS.js → chunk-GVQU2GXP-rnsdTI4Q.js} +1 -1
- package/viewer/assets/{chunk-IMKFNOWR-BKbF8JvM.js → chunk-IMKFNOWR-D0e2NIBw.js} +1 -1
- package/viewer/assets/{chunk-L3NEJ4N5-SWKb6tCy.js → chunk-L3NEJ4N5-Itlw7HBQ.js} +1 -1
- package/viewer/assets/{chunk-OSK3NFVY-BrqoFOmR.js → chunk-OSK3NFVY-DjAAQhPw.js} +1 -1
- package/viewer/assets/{chunk-P2QGCYS3-B10Cyxbx.js → chunk-P2QGCYS3-XOxrUS-o.js} +1 -1
- package/viewer/assets/{chunk-POPQ4Y6H-DXpfBPGQ.js → chunk-POPQ4Y6H-SHuwgciA.js} +1 -1
- package/viewer/assets/{chunk-PWAF6VOD-DZmaXnfr.js → chunk-PWAF6VOD-BtrRW-Sl.js} +1 -1
- package/viewer/assets/{chunk-SHT3W25Y-1U6zFGxP.js → chunk-SHT3W25Y-6Pacfssp.js} +1 -1
- package/viewer/assets/{chunk-SVP7TREG-CaW6KcpM.js → chunk-SVP7TREG-eTVS9a20.js} +1 -1
- package/viewer/assets/{chunk-TICWLB2K-X4pj7G4S.js → chunk-TICWLB2K-CPnAfOZj.js} +1 -1
- package/viewer/assets/{chunk-XXDRQBXY-0afAM3OP.js → chunk-XXDRQBXY-ciXhM2Hb.js} +1 -1
- package/viewer/assets/classDiagram-ZZMXUADV-BtFc7wo5.js +1 -0
- package/viewer/assets/classDiagram-v2-VYDZK3BY-BtFc7wo5.js +1 -0
- package/viewer/assets/{cobol-CtspwbZ4.js → cobol-CEoq_5kf.js} +1 -1
- package/viewer/assets/{coffee-qbHB9gj8.js → coffee-CZTWsTxw.js} +1 -1
- package/viewer/assets/{cose-bilkent-JH36ORCC-7s0vcDw1.js → cose-bilkent-JH36ORCC-NCkOcFkY.js} +1 -1
- package/viewer/assets/{cpp-Dx37X5A_.js → cpp-Hh-o0ivR.js} +1 -1
- package/viewer/assets/{crystal-CiWjJvk4.js → crystal-CL_KTsac.js} +1 -1
- package/viewer/assets/{css-CLeuyJyb.js → css-9neLgVLd.js} +1 -1
- package/viewer/assets/{cynefin-OW5HDTMX-B4XFGNNZ.js → cynefin-OW5HDTMX-Ba1tlgBB.js} +1 -1
- package/viewer/assets/{cynefinDiagram-5FMLGOSQ-DdNMKbw6.js → cynefinDiagram-5FMLGOSQ-CMrzWo3p.js} +1 -1
- package/viewer/assets/{dagre-GXQ25YYZ-BkDo4A91.js → dagre-GXQ25YYZ-DwjOnUDD.js} +1 -1
- package/viewer/assets/{diagram-S7CK7UJ4-DU8cnnVy.js → diagram-S7CK7UJ4-DA0PXsM6.js} +1 -1
- package/viewer/assets/{diagram-UQ7AKVKN-Dg2emYKa.js → diagram-UQ7AKVKN-C5BBFVvO.js} +1 -1
- package/viewer/assets/{diagram-VSXAHHWV-C_-YSzcn.js → diagram-VSXAHHWV-COgLeOyi.js} +1 -1
- package/viewer/assets/{diagram-VX7I27RA-BLCEnxqe.js → diagram-VX7I27RA-CEu-Pm3B.js} +1 -1
- package/viewer/assets/{diagram-Z3DM3KII-62CpIiAg.js → diagram-Z3DM3KII-BGq1_7BE.js} +1 -1
- package/viewer/assets/{dist-CQIgbdem.js → dist-CX8-lyNa.js} +1 -1
- package/viewer/assets/{ebnfDiagram-PWID7BFC-FkGGETgC.js → ebnfDiagram-PWID7BFC-CGoEp9fJ.js} +1 -1
- package/viewer/assets/{edge-CzmGSXHr.js → edge-CTz4kCIs.js} +1 -1
- package/viewer/assets/{elixir-DDfD_-oS.js → elixir-Bi8YL7XX.js} +1 -1
- package/viewer/assets/{elm-l7aRfQIq.js → elm-C-UxMmSh.js} +1 -1
- package/viewer/assets/{erDiagram-RLTQ6QDP-GlCpsS1v.js → erDiagram-RLTQ6QDP-Bb0ICmqZ.js} +1 -1
- package/viewer/assets/{erb-qVffN5xM.js → erb-ZSq62Y_u.js} +1 -1
- package/viewer/assets/eventmodeling-NTZA5JFV-BCWgnTvv.js +1 -0
- package/viewer/assets/flowDiagram-HODETNUW-B4hhZ1QN.js +1 -0
- package/viewer/assets/{ganttDiagram-EL5Y4UJY-CH_Tk4Ex.js → ganttDiagram-EL5Y4UJY-B4KN9ny5.js} +1 -1
- package/viewer/assets/{git-rebase-jEJ8nKk1.js → git-rebase-Cw9GoTpw.js} +1 -1
- package/viewer/assets/{gitGraph-4MIJSDKK-Cc9o1l4Y.js → gitGraph-4MIJSDKK-CI3xWhLt.js} +1 -1
- package/viewer/assets/{gitGraphDiagram-WWUBYQGX-UdfHBATc.js → gitGraphDiagram-WWUBYQGX-CPv8-MbI.js} +1 -1
- package/viewer/assets/{glimmer-js-BMk2JCW8.js → glimmer-js-Bb_1emHy.js} +1 -1
- package/viewer/assets/{glimmer-ts-BH7RwQxU.js → glimmer-ts-D2kOXMB0.js} +1 -1
- package/viewer/assets/{glsl-CAWrZoEq.js → glsl-DEqjHGFr.js} +1 -1
- package/viewer/assets/{graphql-Bac_hecs.js → graphql-Bi0OE4k4.js} +1 -1
- package/viewer/assets/{hack-1OgpAtm-.js → hack-aU4VRzRW.js} +1 -1
- package/viewer/assets/{haml-CvwG3BzW.js → haml-B-2wwgyn.js} +1 -1
- package/viewer/assets/{handlebars-Cq4oEEp8.js → handlebars-Bmq7hbFI.js} +1 -1
- package/viewer/assets/{html-D5wZ_JR6.js → html-BBsNbpya.js} +1 -1
- package/viewer/assets/{html-derivative-ZfzPj0aS.js → html-derivative-CkVRX-vn.js} +1 -1
- package/viewer/assets/{http-BDE2UDvO.js → http-cQ-nDL4L.js} +1 -1
- package/viewer/assets/{hurl-CQbXNLlk.js → hurl-D5phPwZN.js} +1 -1
- package/viewer/assets/{index-OL_rrWNy.css → index-DdaXmlLv.css} +1 -1
- package/viewer/assets/{index-BezQIu6a.js → index-Dhnl4xDm.js} +4 -4
- package/viewer/assets/{info-A6RAGUB7-YD13oMfx.js → info-A6RAGUB7-CRDxOY87.js} +1 -1
- package/viewer/assets/{infoDiagram-27XIBGKW-jYKjA_l7.js → infoDiagram-27XIBGKW-D6y32CfO.js} +1 -1
- package/viewer/assets/{ishikawaDiagram-5VMMS53U-BBPzHiVd.js → ishikawaDiagram-5VMMS53U-BWFBTCh5.js} +1 -1
- package/viewer/assets/{java-Cbpu4oyT.js → java-CT3DXSYV.js} +1 -1
- package/viewer/assets/{javascript-UMuq64YD.js → javascript-BPmsAXaP.js} +1 -1
- package/viewer/assets/{jinja-DVvZtqgU.js → jinja-Ct_36dEc.js} +1 -1
- package/viewer/assets/{jison-CZCXBIV3.js → jison-BYrmy85c.js} +1 -1
- package/viewer/assets/{journeyDiagram-3NMN7TZE-CMqG5ndb.js → journeyDiagram-3NMN7TZE-6XCObBEx.js} +1 -1
- package/viewer/assets/{json-CzTvWngu.js → json-BCFP8udh.js} +1 -1
- package/viewer/assets/{jsx-CNJnEGR4.js → jsx-Dgzhn0Dq.js} +1 -1
- package/viewer/assets/{julia-Bj0q4uoH.js → julia-DIddcjVY.js} +1 -1
- package/viewer/assets/{just-93-MRGUC.js → just-C5t963St.js} +1 -1
- package/viewer/assets/{kanban-definition-UXKFOSKX-VyFYPq9b.js → kanban-definition-UXKFOSKX-DU1B45HL.js} +1 -1
- package/viewer/assets/{latex-CXd1tMjA.js → latex-BKZooW5w.js} +1 -1
- package/viewer/assets/{line-D5nvVDHp.js → line-lYgs1iW7.js} +1 -1
- package/viewer/assets/{linear-Cq-FJZ_z.js → linear-sHqbNFGm.js} +1 -1
- package/viewer/assets/{liquid-Cb-ALXSr.js → liquid-_PLVu-8C.js} +1 -1
- package/viewer/assets/{lua-BxTQiamn.js → lua-BWNAOGrF.js} +1 -1
- package/viewer/assets/{marko-mFYyU5jn.js → marko-DC7qz2de.js} +1 -1
- package/viewer/assets/{mdc-Oryqox7_.js → mdc-Bn_2wB99.js} +1 -1
- package/viewer/assets/{mermaid-parser.core-CCDsanlH.js → mermaid-parser.core-jzj7JZTj.js} +3 -3
- package/viewer/assets/{mermaid.core-Crd1YMgi.js → mermaid.core-DpRyWAPv.js} +4 -4
- package/viewer/assets/{mindmap-definition-YA3MSWOX-BKLrEFWt.js → mindmap-definition-YA3MSWOX-CIZjqKbE.js} +1 -1
- package/viewer/assets/{nginx-C1PwWP7d.js → nginx-Ce7JLwNg.js} +1 -1
- package/viewer/assets/{nim-PszXW-l4.js → nim-CP_MVph6.js} +1 -1
- package/viewer/assets/{org-DO0CuJlO.js → org-BVKft-Jg.js} +1 -1
- package/viewer/assets/{packet-AYTQ26CC-BNPwLE6g.js → packet-AYTQ26CC-BwRTCCV1.js} +1 -1
- package/viewer/assets/{pegDiagram-XKGWAZYB-DjBg_iPh.js → pegDiagram-XKGWAZYB-3nilNfrm.js} +1 -1
- package/viewer/assets/{perl-CHhAgoDL.js → perl-DRhmHTw_.js} +1 -1
- package/viewer/assets/{php-CA4H6qnu.js → php-CrnivzIK.js} +1 -1
- package/viewer/assets/{pie-WAS4IAKB-BzNmHX-Y.js → pie-WAS4IAKB-Dn0ztV-V.js} +1 -1
- package/viewer/assets/{pieDiagram-E7YTZNPT-cgMB-fBw.js → pieDiagram-E7YTZNPT-BFj69QcO.js} +1 -1
- package/viewer/assets/{pug-DDuKTe7C.js → pug-BfP-YwOD.js} +1 -1
- package/viewer/assets/{qml-DSd2VypD.js → qml-BHf9C17M.js} +1 -1
- package/viewer/assets/{quadrantDiagram-AXDQQJYC-Crzhs7Np.js → quadrantDiagram-AXDQQJYC-2t5I-_4_.js} +1 -1
- package/viewer/assets/{r-qf-5qR5Q.js → r-DjPtVJdp.js} +1 -1
- package/viewer/assets/{radar-RG4KPBEZ-B3oI70pB.js → radar-RG4KPBEZ-D14Z-9c0.js} +1 -1
- package/viewer/assets/{railroad-74A4TZTK-DOM5Od17.js → railroad-74A4TZTK-BuxyK73P.js} +1 -1
- package/viewer/assets/railroad-abnf-HS5TGJTU-BM7gtNXd.js +1 -0
- package/viewer/assets/railroad-ebnf-LZEXJU2U-DPlBcoln.js +1 -0
- package/viewer/assets/railroad-peg-WCYAUIDC-B6tOGY79.js +1 -0
- package/viewer/assets/{railroadDiagram-O6MQD6OU-M3SvSYf-.js → railroadDiagram-O6MQD6OU-Dm2mv6N0.js} +1 -1
- package/viewer/assets/{razor-B9n9DtIL.js → razor-BH3XX51e.js} +1 -1
- package/viewer/assets/{regexp-DhGN0EOR.js → regexp-DPzTmgy2.js} +1 -1
- package/viewer/assets/{requirementDiagram-BXWQKSXE-CFisiZTo.js → requirementDiagram-BXWQKSXE-BJTZ8ml9.js} +1 -1
- package/viewer/assets/{rst-D1SdhuXd.js → rst-DQbfRM9X.js} +1 -1
- package/viewer/assets/{ruby-BybsgZgf.js → ruby-DxybHg9D.js} +1 -1
- package/viewer/assets/{sankeyDiagram-P5KCCOFB-BKf2TMWg.js → sankeyDiagram-P5KCCOFB-eBLIKN_A.js} +1 -1
- package/viewer/assets/{sas-Io7QwCtD.js → sas-DoUzRWOs.js} +1 -1
- package/viewer/assets/{scss-D6XY0yo3.js → scss-BSqQF9Aq.js} +1 -1
- package/viewer/assets/{sequenceDiagram-WJ2MYXX4-CHawNQZ9.js → sequenceDiagram-WJ2MYXX4-BxHxbxqV.js} +1 -1
- package/viewer/assets/{shellscript-DSk8kvCh.js → shellscript-DkZJ1HjW.js} +1 -1
- package/viewer/assets/{shellsession-Be1CN8DO.js → shellsession-ykUTtcwM.js} +1 -1
- package/viewer/assets/{soy-CGkkqGyT.js → soy-4rmF0gor.js} +1 -1
- package/viewer/assets/{sql-DIgb716V.js → sql-wJX-vNm_.js} +1 -1
- package/viewer/assets/{src-DP6Z6M4G.js → src-IhQtBUXO.js} +1 -1
- package/viewer/assets/{stata-CS_2tb9p.js → stata-Bjw5vh7C.js} +1 -1
- package/viewer/assets/{stateDiagram-D77RDMKH-Cx9Rf6fY.js → stateDiagram-D77RDMKH-DNPBpLlw.js} +1 -1
- package/viewer/assets/stateDiagram-v2-MP3YSRHH-sLT-i9tN.js +1 -0
- package/viewer/assets/{surrealql-Do4o8w1B.js → surrealql-B-cZs35d.js} +1 -1
- package/viewer/assets/{svelte-3bKxeV6-.js → svelte-BtYh0aTf.js} +1 -1
- package/viewer/assets/{swimlanes-42K2YHIH-yL2oIw_D.js → swimlanes-42K2YHIH-CnHbZ76W.js} +1 -1
- package/viewer/assets/swimlanesDiagram-VR7AAH4N-DIpLefsN.js +8 -0
- package/viewer/assets/{templ-BgEYPNGP.js → templ-eYWdvrUr.js} +1 -1
- package/viewer/assets/{tex-D_p6whQw.js → tex-DQ-rt_Oj.js} +1 -1
- package/viewer/assets/{timeline-definition-24CTP7MA-BJeLq4a5.js → timeline-definition-24CTP7MA-vOTFZgf6.js} +1 -1
- package/viewer/assets/{treeView-Q6P3EWNA-COcnImiE.js → treeView-Q6P3EWNA-CU_mSunf.js} +1 -1
- package/viewer/assets/{treemap-WGGIJYW6-DpbxWots.js → treemap-WGGIJYW6-BbLri-XL.js} +1 -1
- package/viewer/assets/{ts-tags-GBmMk7Oh.js → ts-tags-CspMWJ3s.js} +1 -1
- package/viewer/assets/{tsx-BSD-yNhi.js → tsx-CXUmELmz.js} +1 -1
- package/viewer/assets/{twig-dwAG5SzX.js → twig-B0yioFX1.js} +1 -1
- package/viewer/assets/{typescript-B8YTPl9v.js → typescript-vMN-etbi.js} +1 -1
- package/viewer/assets/{typst-Dpzojbzo.js → typst-Cvk0bEbK.js} +1 -1
- package/viewer/assets/{vennDiagram-4TSXK5OY-nF34Gjtn.js → vennDiagram-4TSXK5OY-PBkXHE6T.js} +1 -1
- package/viewer/assets/{vue-Djbmk2DC.js → vue-Ce21QNcM.js} +1 -1
- package/viewer/assets/{vue-html-BXc5rfco.js → vue-html-DsNf3O2s.js} +1 -1
- package/viewer/assets/{vue-vine-DHCfQh17.js → vue-vine-BsGtGXV3.js} +1 -1
- package/viewer/assets/{wardley-WFR3VGLG-C1qhMifc.js → wardley-WFR3VGLG-kwPRtpGA.js} +1 -1
- package/viewer/assets/{wardleyDiagram-VM6X3IG4-B0BfyYWQ.js → wardleyDiagram-VM6X3IG4-BacMGxmX.js} +1 -1
- package/viewer/assets/{xml-CdCEskcV.js → xml-BmjAaPhA.js} +1 -1
- package/viewer/assets/{xsl-BbNukwXP.js → xsl-CtfA1v1E.js} +1 -1
- package/viewer/assets/{xychartDiagram-S5SC5T6Z-DJKYce1U.js → xychartDiagram-S5SC5T6Z-BDIbR1Lm.js} +1 -1
- package/viewer/assets/{yaml-BYLDET4A.js → yaml-A9EbijaD.js} +1 -1
- package/viewer/index.html +2 -2
- package/viewer/assets/architecture-7GRP2DOG-D3Kw34KW.js +0 -1
- package/viewer/assets/channel-D2lVv6ST.js +0 -1
- package/viewer/assets/classDiagram-ZZMXUADV-DqBNvABf.js +0 -1
- package/viewer/assets/classDiagram-v2-VYDZK3BY-DqBNvABf.js +0 -1
- package/viewer/assets/eventmodeling-NTZA5JFV-07UfGVCm.js +0 -1
- package/viewer/assets/flowDiagram-HODETNUW-CyIUf7TW.js +0 -1
- package/viewer/assets/railroad-abnf-HS5TGJTU-ScZ2h_5Q.js +0 -1
- package/viewer/assets/railroad-ebnf-LZEXJU2U-_Cl5zxJ-.js +0 -1
- package/viewer/assets/railroad-peg-WCYAUIDC-CRlGg7vV.js +0 -1
- package/viewer/assets/stateDiagram-v2-MP3YSRHH-DRb-6t_S.js +0 -1
- package/viewer/assets/swimlanesDiagram-VR7AAH4N-cDiCT6xi.js +0 -8
package/JUDGING.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Use the [canonical terminology](https://openev.al/docs/terminology/): a criterion
|
|
4
4
|
is a named graded requirement, a score is awarded credit, and a metric is a
|
|
5
|
-
measurement such as token count or cost. OpenEval
|
|
5
|
+
measurement such as token count or cost. OpenEval supports code judges,
|
|
6
6
|
LLM judges, and additive use of both against one recorded EvalRun.
|
|
7
7
|
|
|
8
8
|
## File conventions
|
|
@@ -36,6 +36,9 @@ Only the optional scores object has grading semantics. Each key is a criterion
|
|
|
36
36
|
ID, using lowercase letters, digits, and underscores, starting with a letter.
|
|
37
37
|
Values are booleans, finite numbers from 0 to 1, or null. The host converts true
|
|
38
38
|
to 1 and false to 0. It rejects invalid values rather than clamping them.
|
|
39
|
+
Optional `{ value, reason, evidence, measurements }` objects make code verdicts
|
|
40
|
+
readable without changing their scoring semantics. Declared code criterion IDs
|
|
41
|
+
must match the returned scores. See [artifact verification](VERIFICATION.md).
|
|
39
42
|
response.text is always a string; missing text becomes an empty string. The
|
|
40
43
|
original execution outcome is available separately on context.run.
|
|
41
44
|
|
|
@@ -69,6 +72,8 @@ a JudgeRun error. A hybrid judgment is selected only after both sources succeed.
|
|
|
69
72
|
| recording.export(sessionID?) | Native OpenCode session export |
|
|
70
73
|
| workspace.files/read/text/diff | Verified initial and final file snapshots |
|
|
71
74
|
| workspace.materialize(revision?) | A disposable workspace copy for author-owned verification |
|
|
75
|
+
| verification.run(request) | Bounded commands over a restored artifact in an isolated OCI container |
|
|
76
|
+
| verification.read/text(result, path) | Hash-checked retained output from that verification |
|
|
72
77
|
| native.database() | Read-only SQLite access to a verified database copy |
|
|
73
78
|
| native.sdk() | The pinned OpenCode SDK/API over a separate disposable archive copy |
|
|
74
79
|
| native.schema() | The pinned OpenCode schema module |
|
package/VERIFICATION.md
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
# Artifact verification
|
|
2
|
+
|
|
3
|
+
Code judges are still plain functions. When the score depends on delivered code,
|
|
4
|
+
run it in a disposable OCI container rather than on the runner host.
|
|
5
|
+
|
|
6
|
+
```ts
|
|
7
|
+
// benchmark.ts
|
|
8
|
+
import { VERIFICATION_IMAGE, type Benchmark } from "@hona/openeval";
|
|
9
|
+
export default {
|
|
10
|
+
models: ["example/model"],
|
|
11
|
+
judge: { verification: { image: VERIFICATION_IMAGE, cpus: 2, memoryMiB: 4096 } },
|
|
12
|
+
} satisfies Benchmark;
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
Build the candidate and standard verification images with `openeval image`.
|
|
16
|
+
`openeval image --verification` builds only the verification image. Custom images
|
|
17
|
+
must be built separately. Images resolve to immutable IDs before planning or
|
|
18
|
+
collection; a changed verification image invalidates judgment reuse, not the
|
|
19
|
+
candidate's delivered artifacts.
|
|
20
|
+
|
|
21
|
+
```ts
|
|
22
|
+
// judge.ts — trusted script content is an ordinary frozen local import/string.
|
|
23
|
+
import type { JudgeContext } from "@hona/openeval";
|
|
24
|
+
export const criteria = { correct: { name: "Correct output", categories: ["coding"] } };
|
|
25
|
+
export default async (ctx: JudgeContext) => {
|
|
26
|
+
const result = await ctx.verification.run({
|
|
27
|
+
revision: "final", cwd: ".", timeoutMs: 60_000,
|
|
28
|
+
commands: [["bun", "/verification/check.mjs"]],
|
|
29
|
+
files: { "check.mjs": "/* author's outcome checks */" },
|
|
30
|
+
artifacts: ["checks.json", "screenshot.png"],
|
|
31
|
+
});
|
|
32
|
+
return { scores: { correct: {
|
|
33
|
+
value: result.state === "completed" && result.exitCode === 0,
|
|
34
|
+
reason: "Explain what the checks established or why they failed.",
|
|
35
|
+
evidence: [{ kind: "verification", id: result.id }],
|
|
36
|
+
measurements: { elapsedMs: result.elapsedMs },
|
|
37
|
+
} } };
|
|
38
|
+
};
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
The primitive restores an initial or final snapshot, uploads trusted inputs to
|
|
42
|
+
`/verification`, runs argv arrays in order, and stops at a nonzero exit. It does
|
|
43
|
+
not choose scoring thresholds or interpret success. Judge setup/transfer/image
|
|
44
|
+
errors remain JudgeRun errors. A test exit or verification timeout is an
|
|
45
|
+
observation the authored criterion must interpret. Interrupted candidate work
|
|
46
|
+
still needs the rubric's evidence policy; a failed verification of an unfinished
|
|
47
|
+
prefix does not necessarily establish final-task failure.
|
|
48
|
+
|
|
49
|
+
## Isolation and bounds
|
|
50
|
+
|
|
51
|
+
- No host bind mounts, credentials, published ports, or network. Browser/server
|
|
52
|
+
verification can use loopback **inside** the container.
|
|
53
|
+
- Read-only image filesystem, non-root command user, no capabilities, no privilege
|
|
54
|
+
escalation, PID/CPU/memory limits, bounded workspace/temp storage, and a deadline.
|
|
55
|
+
- Trusted check inputs are root-owned and are not writable by delivered code.
|
|
56
|
+
- Commands, exit statuses, image/resource identity, logs, and requested regular
|
|
57
|
+
output files are retained with the JudgeRun. Viewer previews never execute HTML
|
|
58
|
+
or SVG from the delivered artifact.
|
|
59
|
+
- Workspace transfer is limited to 128 MiB; each output archive is limited to
|
|
60
|
+
32 MiB; logs are limited to 1 MiB per stream and are marked when truncated.
|
|
61
|
+
- Containers self-expire. The parent also removes containers bearing only its
|
|
62
|
+
unique execution label after worker interruption. No shared/broad cleanup.
|
|
63
|
+
|
|
64
|
+
The standard image includes Bun 1.4.2, Python, Git, Node, and Playwright 1.63.0
|
|
65
|
+
with Chromium. Import Playwright from
|
|
66
|
+
`/opt/verify/node_modules/playwright/index.mjs`. Dependencies needed by a task
|
|
67
|
+
must be available in the image or supplied as frozen inputs; network installs
|
|
68
|
+
are intentionally unavailable. Materialization alone is not a sandbox.
|
|
69
|
+
`verification.text(result, "stdout" | "stderr")` reads the retained bounded logs;
|
|
70
|
+
those names are reserved and cannot also name output artifacts.
|
|
71
|
+
|
|
72
|
+
The primitive verifies a reconstructed artifact. It does **not** prove what was
|
|
73
|
+
alive in the original candidate container, whether the candidate ran a test,
|
|
74
|
+
or whether original game actions were legal. Those claims need original recording
|
|
75
|
+
evidence. Do not put evaluator scripts in candidate workspaces.
|
|
76
|
+
|
|
77
|
+
## Readable judgments and controls
|
|
78
|
+
|
|
79
|
+
`openeval prepare --output <new-directory>` assembles real candidate inputs and
|
|
80
|
+
runs declared preparation in the candidate image, then archives the prepared
|
|
81
|
+
workspace. `--only-eval` scopes it. This makes zero model calls, creates no
|
|
82
|
+
EvalRun/JudgeRun, and does not change selections. Use it to prove fixture setup
|
|
83
|
+
before collection; it does not prove agent success or human task duration.
|
|
84
|
+
|
|
85
|
+
Existing boolean, numeric, and null scores remain sufficient. Optional structured
|
|
86
|
+
scores add `reason`, `evidence`, and JSON `measurements`. These details appear in
|
|
87
|
+
the normal viewer beside verification receipts; arbitrary returned JSON remains
|
|
88
|
+
inspectable. Declared code criteria must be returned exactly. Missing evidence
|
|
89
|
+
references are judging errors, not candidate zeros.
|
|
90
|
+
|
|
91
|
+
`recordEvidence({ workspace: { initial, final }, ... })` can retain constructed
|
|
92
|
+
artifact controls without executing them. `judgeEvidence` then uses the same
|
|
93
|
+
public verification primitive. Constructed controls prove tested boundaries, not
|
|
94
|
+
human-duration calibration or live candidate feasibility.
|
|
95
|
+
|
|
96
|
+
Unused `criteria` labels/categories are removed from executable bundling. Debug
|
|
97
|
+
source maps remain retained but do not affect code identity. If grading reads a
|
|
98
|
+
metadata value, it is executable behavior and remains fingerprinted. Changes to
|
|
99
|
+
criterion IDs, actual grading code, imported inputs, dependencies, or verification
|
|
100
|
+
environment are not reporting-only edits.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@hona/openeval",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.5.0",
|
|
4
4
|
"description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"repository": {
|
|
@@ -14,7 +14,7 @@
|
|
|
14
14
|
"bin": {
|
|
15
15
|
"openeval": "src/cli.ts"
|
|
16
16
|
},
|
|
17
|
-
"files": ["src", "viewer", "LICENSE", "README.md", "JUDGING.md", "!src/**/*.test.ts"],
|
|
17
|
+
"files": ["src", "viewer", "LICENSE", "README.md", "JUDGING.md", "VERIFICATION.md", "!src/**/*.test.ts"],
|
|
18
18
|
"publishConfig": {
|
|
19
19
|
"access": "public"
|
|
20
20
|
},
|
|
@@ -11,6 +11,8 @@ import { executeCodeJudge } from "../infra/judging/code";
|
|
|
11
11
|
import { codeJudgment, combineJudgments } from "../infra/judging/code-result";
|
|
12
12
|
import { executeJudge } from "../infra/judging";
|
|
13
13
|
import { errorMessage } from "../infra/files";
|
|
14
|
+
import { CandidateEvidence } from "../infra/evidence";
|
|
15
|
+
import { validateCitations } from "../infra/judging/contract";
|
|
14
16
|
|
|
15
17
|
export type GradedRecording = {
|
|
16
18
|
state: "completed" | "failed" | "timed_out";
|
|
@@ -37,10 +39,24 @@ export async function gradeRecording(
|
|
|
37
39
|
recording,
|
|
38
40
|
resolve(directory, "code"),
|
|
39
41
|
input.timeoutMs,
|
|
42
|
+
input.verification,
|
|
40
43
|
);
|
|
41
44
|
if (code.state !== "completed")
|
|
42
45
|
return { state: code.state, code, error: code.error };
|
|
43
46
|
const judged = codeJudgment(code.output!);
|
|
47
|
+
if (input.codeCriteria?.length && (
|
|
48
|
+
Object.keys(judged.scores).length !== input.codeCriteria.length ||
|
|
49
|
+
input.codeCriteria.some(item => !Object.hasOwn(judged.scores, item.id))
|
|
50
|
+
)) throw new Error("judge.ts must return exactly its declared criterion IDs");
|
|
51
|
+
for (const score of Object.values(judged.scores)) for (const citation of score.evidence) {
|
|
52
|
+
if (citation.kind !== "verification") continue;
|
|
53
|
+
const result = code.verifications?.find(item => item.id === citation.id);
|
|
54
|
+
if (!result || (citation.path && !result.artifacts.some(item => item.path === citation.path)))
|
|
55
|
+
throw new Error("Code judgment cites missing verification evidence");
|
|
56
|
+
}
|
|
57
|
+
await validateCitations({ ...judged, scores: Object.fromEntries(Object.entries(judged.scores).map(([id, score]) =>
|
|
58
|
+
[id, { ...score, evidence: score.evidence.filter(citation => citation.kind !== "verification") }])) },
|
|
59
|
+
await CandidateEvidence.open(recording.evidence.directory, recording.evidence.hash));
|
|
44
60
|
for (const id of Object.keys(judged.scores))
|
|
45
61
|
if (input.criteria.some((criterion) => criterion.id === id))
|
|
46
62
|
throw new Error(
|
|
@@ -61,6 +61,7 @@ export function candidateFingerprint(
|
|
|
61
61
|
timeoutMs: definition.candidate.timeoutMs,
|
|
62
62
|
websearch: definition.candidate.websearch,
|
|
63
63
|
provider: scope && { ...scope.settings, model: scope.override },
|
|
64
|
+
earlyStop: item.settings.earlyStop ?? false,
|
|
64
65
|
},
|
|
65
66
|
container: {
|
|
66
67
|
engine: definition.container.engine,
|
|
@@ -76,12 +77,13 @@ export const judgeFingerprint = (
|
|
|
76
77
|
fingerprint({
|
|
77
78
|
rubric: definition.evals.find((item) => item.id === evalId)!.judge,
|
|
78
79
|
code: definition.evals.find((item) => item.id === evalId)!.code?.hash,
|
|
80
|
+
codeIds: definition.evals.find((item) => item.id === evalId)!.codeCriteria?.map(item => item.id).sort(),
|
|
79
81
|
agent: definition.evals.find((item) => item.id === evalId)!.judge
|
|
80
82
|
? JUDGE_AGENT
|
|
81
83
|
: undefined,
|
|
82
84
|
judge: definition.evals.find((item) => item.id === evalId)!.judge
|
|
83
85
|
? definition.judge
|
|
84
|
-
: { timeoutMs: definition.judge.timeoutMs },
|
|
86
|
+
: { timeoutMs: definition.judge.timeoutMs, verification: definition.judge.verification },
|
|
85
87
|
protocol: JUDGE_PROTOCOL,
|
|
86
88
|
});
|
|
87
89
|
export const savedJudgeFingerprint = (
|
|
@@ -94,11 +96,14 @@ export const savedJudgeFingerprint = (
|
|
|
94
96
|
| "timeoutMs"
|
|
95
97
|
| "websearch"
|
|
96
98
|
| "code"
|
|
99
|
+
| "codeCriteria"
|
|
100
|
+
| "verification"
|
|
97
101
|
>,
|
|
98
102
|
) =>
|
|
99
103
|
fingerprint({
|
|
100
104
|
rubric: input.rubric,
|
|
101
105
|
code: input.code?.hash,
|
|
106
|
+
codeIds: input.codeCriteria?.map(item => item.id).sort(),
|
|
102
107
|
agent: input.agent,
|
|
103
108
|
protocol: input.protocol,
|
|
104
109
|
judge: input.rubric
|
|
@@ -106,6 +111,7 @@ export const savedJudgeFingerprint = (
|
|
|
106
111
|
model: input.model,
|
|
107
112
|
timeoutMs: input.timeoutMs,
|
|
108
113
|
websearch: input.websearch,
|
|
114
|
+
verification: input.verification,
|
|
109
115
|
}
|
|
110
|
-
: { timeoutMs: input.timeoutMs },
|
|
116
|
+
: { timeoutMs: input.timeoutMs, verification: input.verification },
|
|
111
117
|
});
|
|
@@ -9,6 +9,8 @@ import { savedJudgeFingerprint } from "./input-fingerprints";
|
|
|
9
9
|
import { mkdir, mkdtemp, rm } from "node:fs/promises";
|
|
10
10
|
import { compileCodeJudge } from "../infra/judging/code-source";
|
|
11
11
|
import { gradeRecording } from "./grade-recording";
|
|
12
|
+
import { readCodeCriteria } from "../infra/judging/code-criteria";
|
|
13
|
+
import { verificationRuntime } from "../infra/verification/image";
|
|
12
14
|
|
|
13
15
|
/** Inspect retained evidence without changing benchmark selections.
|
|
14
16
|
* Useful for calibration and second opinions: https://arxiv.org/abs/2410.12784
|
|
@@ -28,6 +30,7 @@ export async function judgeEvidence(options: {
|
|
|
28
30
|
if (options.rubric && !options.judge?.model)
|
|
29
31
|
throw new Error("A Markdown rubric requires judge.model");
|
|
30
32
|
const code = options.code ? await compileCodeJudge(options.code) : undefined;
|
|
33
|
+
const declared = options.code ? await readCodeCriteria(options.code) : undefined;
|
|
31
34
|
const request: Omit<JudgeRunInput, "judgeHash"> = {
|
|
32
35
|
evalRunId: "retained-evidence",
|
|
33
36
|
evidence: {
|
|
@@ -37,6 +40,8 @@ export async function judgeEvidence(options: {
|
|
|
37
40
|
rubric: options.rubric ?? "",
|
|
38
41
|
kind: code ? (options.rubric ? "hybrid" : "code") : "llm",
|
|
39
42
|
code,
|
|
43
|
+
codeCriteria: declared ? Object.entries(declared as Record<string, { name?: string }>).map(([id, value]) => ({ id, name: value.name ?? id })) : undefined,
|
|
44
|
+
verification: await verificationRuntime(options.judge?.verification),
|
|
40
45
|
agent: options.rubric ? JUDGE_AGENT : undefined,
|
|
41
46
|
model: options.rubric ? options.judge?.model : undefined,
|
|
42
47
|
timeoutMs: options.judge?.timeoutMs ?? 600_000,
|
|
@@ -67,17 +72,18 @@ export async function recordEvidence(options: {
|
|
|
67
72
|
prompt: string;
|
|
68
73
|
response: string;
|
|
69
74
|
tools?: ToolCall[];
|
|
75
|
+
workspace?: { initial?: string; final?: string };
|
|
70
76
|
}): Promise<EvidenceRef> {
|
|
71
77
|
const directory = resolve(options.directory);
|
|
72
78
|
await mkdir(dirname(directory), { recursive: true });
|
|
73
79
|
const workspace = await mkdtemp(resolve(dirname(directory), ".recording-"));
|
|
74
80
|
try {
|
|
75
|
-
const capture = await EvidenceCapture.create(directory);
|
|
81
|
+
const capture = await EvidenceCapture.create(directory, options.workspace?.initial);
|
|
76
82
|
const result = await capture.finish({
|
|
77
83
|
prompt: options.prompt,
|
|
78
84
|
response: { text: options.response },
|
|
79
85
|
tools: options.tools ?? [],
|
|
80
|
-
workspace,
|
|
86
|
+
workspace: options.workspace?.final ?? workspace,
|
|
81
87
|
});
|
|
82
88
|
return { directory, hash: result.sha256 };
|
|
83
89
|
} finally {
|
package/src/app/judge-run.ts
CHANGED
|
@@ -27,6 +27,8 @@ export async function judgeEvalRun(
|
|
|
27
27
|
model: definition.judge ? context.definition.judge.model : undefined,
|
|
28
28
|
kind: definition.code ? (definition.judge ? "hybrid" : "code") : "llm",
|
|
29
29
|
code: definition.code,
|
|
30
|
+
codeCriteria: definition.codeCriteria,
|
|
31
|
+
verification: context.runtime.verification,
|
|
30
32
|
judgeHash: slot.judgeHash,
|
|
31
33
|
timeoutMs: context.definition.judge.timeoutMs,
|
|
32
34
|
websearch: context.definition.judge.websearch,
|
|
@@ -76,6 +78,14 @@ export async function judgeEvalRun(
|
|
|
76
78
|
...graded.code,
|
|
77
79
|
stdout: relative(context.directory, graded.code.stdout),
|
|
78
80
|
stderr: relative(context.directory, graded.code.stderr),
|
|
81
|
+
verifications: graded.code.verifications?.map(result => ({
|
|
82
|
+
...result,
|
|
83
|
+
stdout: relative(context.directory, resolve(directory, "code", result.stdout)),
|
|
84
|
+
stderr: relative(context.directory, resolve(directory, "code", result.stderr)),
|
|
85
|
+
artifacts: result.artifacts.map(artifact => ({ ...artifact,
|
|
86
|
+
file: relative(context.directory, resolve(directory, "code", artifact.file)),
|
|
87
|
+
})),
|
|
88
|
+
})),
|
|
79
89
|
}
|
|
80
90
|
: undefined,
|
|
81
91
|
session: graded.session
|
|
@@ -292,11 +292,11 @@ export async function loadBenchmark(
|
|
|
292
292
|
typeof definition.judge !== "object" ||
|
|
293
293
|
Array.isArray(definition.judge) ||
|
|
294
294
|
Object.keys(definition.judge).some(
|
|
295
|
-
(key) => !["model", "timeoutMs", "websearch"].includes(key),
|
|
295
|
+
(key) => !["model", "timeoutMs", "websearch", "verification"].includes(key),
|
|
296
296
|
))
|
|
297
297
|
)
|
|
298
298
|
throw new Error(
|
|
299
|
-
"judge must be an object with model, timeoutMs, or
|
|
299
|
+
"judge must be an object with model, timeoutMs, websearch, or verification settings",
|
|
300
300
|
);
|
|
301
301
|
if (new Set(models).size !== models.length)
|
|
302
302
|
throw new Error("Benchmark contains duplicate models");
|
|
@@ -337,6 +337,13 @@ export async function loadBenchmark(
|
|
|
337
337
|
const engine = definition.container?.engine ?? "docker";
|
|
338
338
|
if (engine !== "docker" && engine !== "podman")
|
|
339
339
|
throw new Error("Container engine must be docker or podman");
|
|
340
|
+
const verification = definition.judge?.verification;
|
|
341
|
+
if (verification && (
|
|
342
|
+
typeof verification !== "object" || Array.isArray(verification) ||
|
|
343
|
+
Object.keys(verification).some(key => !["image", "engine", "cpus", "memoryMiB"].includes(key)) ||
|
|
344
|
+
typeof verification.image !== "string" || !verification.image.trim() || /\s/.test(verification.image) ||
|
|
345
|
+
!["docker", "podman"].includes(verification.engine ?? engine)
|
|
346
|
+
)) throw new Error("judge.verification requires an image and valid container limits");
|
|
340
347
|
for (const search of [
|
|
341
348
|
definition.candidate?.websearch,
|
|
342
349
|
definition.judge?.websearch,
|
|
@@ -367,6 +374,12 @@ export async function loadBenchmark(
|
|
|
367
374
|
"Judge timeout",
|
|
368
375
|
),
|
|
369
376
|
websearch: definition.judge?.websearch ?? "exa",
|
|
377
|
+
...(verification ? { verification: {
|
|
378
|
+
image: verification.image,
|
|
379
|
+
engine: verification.engine ?? engine,
|
|
380
|
+
cpus: positive(verification.cpus, 2, "Verification CPU count"),
|
|
381
|
+
memoryMiB: positive(verification.memoryMiB, 4096, "Verification memory"),
|
|
382
|
+
} } : {}),
|
|
370
383
|
},
|
|
371
384
|
container: {
|
|
372
385
|
engine,
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import { mkdir, mkdtemp, rm } from "node:fs/promises";
|
|
2
|
+
import { resolve } from "node:path";
|
|
3
|
+
import { tmpdir } from "node:os";
|
|
4
|
+
import { loadBenchmark } from "./load-benchmark";
|
|
5
|
+
import { prepareWorkspace } from "../infra/containers/workspace";
|
|
6
|
+
import { CandidateContainer, inspectImage } from "../infra/containers/oci";
|
|
7
|
+
import { createSessionDatabase } from "../infra/opencode/host";
|
|
8
|
+
import { treeHash, writeJson } from "../infra/files";
|
|
9
|
+
|
|
10
|
+
/** Prove preparation without prompting a model, creating EvalRuns, or changing selections. */
|
|
11
|
+
export async function prepareInputs(path: string, options: { directory: string; onlyEvals?: readonly string[] }) {
|
|
12
|
+
const definition = await loadBenchmark(path);
|
|
13
|
+
if (options.onlyEvals?.some(id => !definition.evals.some(item => item.id === id)))
|
|
14
|
+
throw new Error("Select configured evals for preparation");
|
|
15
|
+
const directory = resolve(options.directory);
|
|
16
|
+
if (await Bun.file(resolve(directory, "prepared.json")).exists()) throw new Error("Choose a new prepared-input directory");
|
|
17
|
+
const imageId = await inspectImage(definition.container);
|
|
18
|
+
const scratch = await mkdtemp(resolve(process.platform === "win32" ? "C:/tmp/opencode" : tmpdir(), "openeval-prepare-"));
|
|
19
|
+
const inputs: Array<{ eval: string; directory: string; sourceHash: string; preparedHash: string }> = [];
|
|
20
|
+
try {
|
|
21
|
+
await mkdir(directory, { recursive: true });
|
|
22
|
+
for (const item of definition.evals.filter(item => !options.onlyEvals || options.onlyEvals.includes(item.id))) {
|
|
23
|
+
const staging = resolve(scratch, item.id);
|
|
24
|
+
await mkdir(staging, { recursive: true });
|
|
25
|
+
const workspace = resolve(staging, "workspace");
|
|
26
|
+
await prepareWorkspace(item, workspace);
|
|
27
|
+
const database = resolve(staging, "opencode.db");
|
|
28
|
+
// No credentials or model route are needed to prepare task inputs.
|
|
29
|
+
await createSessionDatabase(database, []);
|
|
30
|
+
await using container = await CandidateContainer.create(definition.container, imageId);
|
|
31
|
+
await container.prepare(workspace, database, definition.candidate.websearch, item.settings.prepare ?? [], staging, definition.candidate.providers);
|
|
32
|
+
const output = resolve(directory, item.id, "workspace");
|
|
33
|
+
await container.snapshot(output, staging);
|
|
34
|
+
inputs.push({ eval: item.id, directory: output, sourceHash: item.sourceHash, preparedHash: await treeHash(output) });
|
|
35
|
+
}
|
|
36
|
+
const result = { imageId, inputs, candidateExecutions: 0, judgeExecutions: 0 };
|
|
37
|
+
await writeJson(resolve(directory, "prepared.json"), result);
|
|
38
|
+
return result;
|
|
39
|
+
} finally { await rm(scratch, { recursive: true, force: true }); }
|
|
40
|
+
}
|
package/src/app/read-results.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { readdir } from "node:fs/promises";
|
|
2
|
-
import { resolve, dirname } from "node:path";
|
|
1
|
+
import { readdir, realpath } from "node:fs/promises";
|
|
2
|
+
import { resolve, dirname, sep } from "node:path";
|
|
3
3
|
import { Results } from "../infra/sqlite";
|
|
4
4
|
import type { BenchmarkRun, EvalRun, JudgeRun, Slot, Cost } from "../types";
|
|
5
5
|
import type {
|
|
@@ -26,7 +26,7 @@ import {
|
|
|
26
26
|
} from "../runtime";
|
|
27
27
|
import { canJudgeEval } from "./eval-state";
|
|
28
28
|
import { CandidateEvidence, type EvidenceQuery } from "../infra/evidence";
|
|
29
|
-
import { contained } from "../infra/files";
|
|
29
|
+
import { contained, hash } from "../infra/files";
|
|
30
30
|
|
|
31
31
|
const scheduled = (slot: Slot, benchmark: BenchmarkRun, now: number) =>
|
|
32
32
|
benchmark.state === "running" &&
|
|
@@ -167,7 +167,7 @@ export class ResultReader {
|
|
|
167
167
|
if (!judge) throw new Error("Unknown judge run");
|
|
168
168
|
const candidate = results.evalRun(judge.input.evalRunId);
|
|
169
169
|
const criteria = new Map(
|
|
170
|
-
judge.input.criteria.map((criterion) => [criterion.id, criterion]),
|
|
170
|
+
[...judge.input.criteria, ...judge.input.codeCriteria ?? []].map((criterion) => [criterion.id, criterion]),
|
|
171
171
|
);
|
|
172
172
|
for (const id of Object.keys(judge.judgment?.scores ?? {}))
|
|
173
173
|
if (!criteria.has(id))
|
|
@@ -225,6 +225,23 @@ export class ResultReader {
|
|
|
225
225
|
})),
|
|
226
226
|
};
|
|
227
227
|
}
|
|
228
|
+
async verificationEvidence(benchmarkId: string, judgeRunId: string, id: string, path?: string) {
|
|
229
|
+
using results = await this.database(benchmarkId);
|
|
230
|
+
const judge = results.judgeRun(judgeRunId);
|
|
231
|
+
const verification = judge?.code?.verifications?.find(item => item.id === id);
|
|
232
|
+
if (!verification) throw new Error("Unknown verification");
|
|
233
|
+
if (!path) return { receipt: verification };
|
|
234
|
+
const artifact = verification.artifacts.find(item => item.path === path);
|
|
235
|
+
const file = path === "stdout" ? verification.stdout : path === "stderr" ? verification.stderr : artifact?.file;
|
|
236
|
+
if (!file) throw new Error("Unknown verification artifact");
|
|
237
|
+
const root = await realpath(dirname(results.path));
|
|
238
|
+
const target = await realpath(contained(root, file));
|
|
239
|
+
if (target !== root && !target.startsWith(root + sep)) throw new Error("Verification artifact leaves its run directory");
|
|
240
|
+
const bytes = await Bun.file(target).bytes();
|
|
241
|
+
if (bytes.length > 32 * 1024 * 1024 || (artifact && hash(bytes) !== artifact.sha256))
|
|
242
|
+
throw new Error("Verification artifact is too large or its hash differs");
|
|
243
|
+
return { bytes, path };
|
|
244
|
+
}
|
|
228
245
|
async checkEvidence(
|
|
229
246
|
benchmarkId: string,
|
|
230
247
|
judgeRunId: string,
|
package/src/app/rejudge.ts
CHANGED
|
@@ -7,6 +7,7 @@ import { judgeEvalRun } from "./judge-run";
|
|
|
7
7
|
import { executeQueue, finishBenchmark } from "./run-benchmark";
|
|
8
8
|
import { judgingFingerprint } from "../infra/judging";
|
|
9
9
|
import { canJudgeEval } from "./eval-state";
|
|
10
|
+
import { verificationRuntime } from "../infra/verification/image";
|
|
10
11
|
|
|
11
12
|
/** Rejudging changes only the selected judgment. The candidate execution is never repeated. */
|
|
12
13
|
export async function judgeRun(
|
|
@@ -59,6 +60,8 @@ export async function judgeRuns(
|
|
|
59
60
|
judgeHash: updated.judgeHash,
|
|
60
61
|
criteria: updated.criteria,
|
|
61
62
|
code: updated.code,
|
|
63
|
+
codeCriteria: updated.codeCriteria,
|
|
64
|
+
categories: updated.categories,
|
|
62
65
|
name: updated.name,
|
|
63
66
|
}
|
|
64
67
|
: evalDefinition;
|
|
@@ -67,7 +70,9 @@ export async function judgeRuns(
|
|
|
67
70
|
const runtime = {
|
|
68
71
|
...previous.runtime,
|
|
69
72
|
judgeHash: await judgingFingerprint(),
|
|
73
|
+
verification: await verificationRuntime(definition.judge.verification),
|
|
70
74
|
};
|
|
75
|
+
if (runtime.verification) definition.judge.verification = runtime.verification;
|
|
71
76
|
const selected = new Set(candidates.map((candidate) => candidate.slotId));
|
|
72
77
|
results.transaction(() => {
|
|
73
78
|
if (results.benchmark?.state === "running")
|
package/src/app/run-benchmark.ts
CHANGED
|
@@ -176,7 +176,8 @@ export async function runBenchmark(
|
|
|
176
176
|
const selected = options.onlyModels?.map(modelRef);
|
|
177
177
|
if (selected && !selected.length)
|
|
178
178
|
throw new Error("Select at least one model");
|
|
179
|
-
const runtime = await runtimeFingerprint(definition.container);
|
|
179
|
+
const runtime = await runtimeFingerprint(definition.container, definition.judge.verification);
|
|
180
|
+
if (runtime.verification) definition.judge.verification = runtime.verification;
|
|
180
181
|
let directory =
|
|
181
182
|
options.directory ??
|
|
182
183
|
(!options.fresh
|
package/src/app/serve-results.ts
CHANGED
|
@@ -52,6 +52,20 @@ export async function serveResults(options: {
|
|
|
52
52
|
url.searchParams.get("judge") ?? "",
|
|
53
53
|
),
|
|
54
54
|
);
|
|
55
|
+
if (url.pathname === "/api/verification") {
|
|
56
|
+
const result = await reader.verificationEvidence(
|
|
57
|
+
url.searchParams.get("benchmark") ?? "", url.searchParams.get("judge") ?? "",
|
|
58
|
+
url.searchParams.get("id") ?? "", url.searchParams.get("path") ?? undefined,
|
|
59
|
+
);
|
|
60
|
+
if ("receipt" in result) return Response.json(result);
|
|
61
|
+
const mime = /\.png$/i.test(result.path) ? "image/png" : /\.jpe?g$/i.test(result.path) ? "image/jpeg" :
|
|
62
|
+
/\.webp$/i.test(result.path) ? "image/webp" : "text/plain; charset=utf-8";
|
|
63
|
+
return new Response(result.bytes, { headers: {
|
|
64
|
+
"Content-Type": mime, "X-Content-Type-Options": "nosniff",
|
|
65
|
+
"Content-Security-Policy": "default-src 'none'; sandbox",
|
|
66
|
+
"Cache-Control": "no-store",
|
|
67
|
+
} });
|
|
68
|
+
}
|
|
55
69
|
if (url.pathname === "/api/check-evidence") {
|
|
56
70
|
const query: EvidenceQuery = {
|
|
57
71
|
action: (url.searchParams.get("action") ??
|
package/src/cli.ts
CHANGED
|
@@ -12,6 +12,9 @@ import {
|
|
|
12
12
|
serveResults,
|
|
13
13
|
mergeBenchmarkRuns,
|
|
14
14
|
snapshotBenchmarkRun,
|
|
15
|
+
buildVerificationImage,
|
|
16
|
+
VERIFICATION_IMAGE,
|
|
17
|
+
prepareInputs,
|
|
15
18
|
} from "./index";
|
|
16
19
|
import type { ModelRef } from "./index";
|
|
17
20
|
|
|
@@ -35,6 +38,7 @@ try {
|
|
|
35
38
|
Commands:
|
|
36
39
|
image Build the candidate container image
|
|
37
40
|
plan Show missing or changed work without executing it
|
|
41
|
+
prepare Prepare and archive inputs without model calls
|
|
38
42
|
run Execute missing or changed work in the current result
|
|
39
43
|
view Serve the results viewer
|
|
40
44
|
snapshot <run> <name> Export a score snapshot
|
|
@@ -55,14 +59,29 @@ Options:
|
|
|
55
59
|
--only-repetition <n> Execute only this repetition (repeatable)
|
|
56
60
|
--max-cost <usd> Scheduling budget for this invocation
|
|
57
61
|
--final-only Judge only after candidates finish
|
|
62
|
+
--verification Build only the standard verification image (image command)
|
|
63
|
+
--output <dir> New prepared-input directory (prepare command)
|
|
58
64
|
--port <port> Viewer port (default: 4173)
|
|
59
65
|
|
|
60
66
|
Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
61
67
|
} else if (command === "--version") {
|
|
62
68
|
console.log((await Bun.file(new URL("../package.json", import.meta.url)).json()).version);
|
|
63
69
|
} else if (command === "image") {
|
|
64
|
-
|
|
65
|
-
|
|
70
|
+
const definition = await loadBenchmark(benchmark);
|
|
71
|
+
if (!args.includes("--verification")) {
|
|
72
|
+
await buildImage(definition.container);
|
|
73
|
+
console.log("Candidate image ready");
|
|
74
|
+
}
|
|
75
|
+
const verification = definition.judge.verification ?? (args.includes("--verification")
|
|
76
|
+
? { image: VERIFICATION_IMAGE, engine: definition.container.engine } : undefined);
|
|
77
|
+
if (verification) {
|
|
78
|
+
await buildVerificationImage(verification);
|
|
79
|
+
console.log("Verification image ready");
|
|
80
|
+
}
|
|
81
|
+
} else if (command === "prepare") {
|
|
82
|
+
if (!option("--output")) throw new Error("prepare requires --output with a new directory");
|
|
83
|
+
const onlyEvals = args.flatMap((arg, index) => arg === "--only-eval" ? [args[index + 1]] : []);
|
|
84
|
+
console.log(JSON.stringify(await prepareInputs(benchmark, { directory: option("--output")!, onlyEvals: onlyEvals.length ? onlyEvals : undefined }), null, 2));
|
|
66
85
|
} else if (command === "view") {
|
|
67
86
|
const viewer = await serveResults({
|
|
68
87
|
resultsPath: resolve(benchmark, "results"),
|
package/src/index.ts
CHANGED
|
@@ -46,8 +46,16 @@ export type {
|
|
|
46
46
|
RecordedEvent,
|
|
47
47
|
RecordedMessage,
|
|
48
48
|
RecordedFile,
|
|
49
|
+
CodeScore,
|
|
50
|
+
VerificationEnvironment,
|
|
51
|
+
VerificationRuntime,
|
|
52
|
+
VerificationRequest,
|
|
53
|
+
VerificationResult,
|
|
54
|
+
VerificationArtifact,
|
|
49
55
|
} from "./judge-context";
|
|
56
|
+
export { buildVerificationImage, VERIFICATION_IMAGE } from "./infra/verification/image";
|
|
50
57
|
export { loadBenchmark } from "./app/load-benchmark";
|
|
58
|
+
export { prepareInputs } from "./app/prepare-inputs";
|
|
51
59
|
export {
|
|
52
60
|
runBenchmark,
|
|
53
61
|
currentBenchmarkRun,
|
|
@@ -12,6 +12,7 @@ import { extractWorkspaceArchive } from "./transfer";
|
|
|
12
12
|
import { treeHash, fingerprint, writeJson } from "../files";
|
|
13
13
|
import { OPENCODE_VERSION } from "../opencode/host";
|
|
14
14
|
import { judgingFingerprint } from "../judging";
|
|
15
|
+
import { verificationRuntime } from "../verification/image";
|
|
15
16
|
|
|
16
17
|
const runtimeDirectory = fileURLToPath(new URL("./runtime/", import.meta.url));
|
|
17
18
|
export async function buildImage(config: BenchmarkDefinition["container"]) {
|
|
@@ -46,8 +47,10 @@ export async function inspectImage(config: BenchmarkDefinition["container"]) {
|
|
|
46
47
|
}
|
|
47
48
|
export async function runtimeFingerprint(
|
|
48
49
|
config: BenchmarkDefinition["container"],
|
|
50
|
+
verification?: BenchmarkDefinition["judge"]["verification"],
|
|
49
51
|
) {
|
|
50
52
|
const imageId = await inspectImage(config);
|
|
53
|
+
const verified = await verificationRuntime(verification);
|
|
51
54
|
return {
|
|
52
55
|
imageId,
|
|
53
56
|
candidateHash: fingerprint({
|
|
@@ -80,6 +83,7 @@ export async function runtimeFingerprint(
|
|
|
80
83
|
),
|
|
81
84
|
}),
|
|
82
85
|
judgeHash: await judgingFingerprint(),
|
|
86
|
+
...(verified ? { verification: verified } : {}),
|
|
83
87
|
};
|
|
84
88
|
}
|
|
85
89
|
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import type { JsonValue } from "../../judge-context";
|
|
2
2
|
import type { CriterionScore, Judgment } from "../../types";
|
|
3
3
|
import { criterionMean, isScored } from "../../judgment";
|
|
4
|
+
import { parseCitation } from "./contract";
|
|
4
5
|
|
|
5
6
|
/** Preserve author JSON exactly; reject values JSON would silently discard or coerce. */
|
|
6
7
|
export function jsonOutput(
|
|
@@ -53,7 +54,12 @@ export function codeJudgment(output: JsonValue): Judgment {
|
|
|
53
54
|
Object.entries(values).map(([id, supplied]): [string, CriterionScore] => {
|
|
54
55
|
if (!/^[a-z][a-z0-9_]*$/.test(id))
|
|
55
56
|
throw new Error(`Invalid criterion ID: ${id}`);
|
|
56
|
-
const
|
|
57
|
+
const detailed = supplied !== null && typeof supplied === "object" && !Array.isArray(supplied) ? supplied : undefined;
|
|
58
|
+
if (detailed && (Object.keys(detailed).some(key => !["value", "reason", "evidence", "measurements"].includes(key)) ||
|
|
59
|
+
typeof detailed.reason !== "string" || !detailed.reason.trim()))
|
|
60
|
+
throw new Error(`scores.${id} requires a non-empty reason and only value, reason, evidence, and measurements`);
|
|
61
|
+
const raw = detailed ? detailed.value : supplied;
|
|
62
|
+
const value = typeof raw === "boolean" ? Number(raw) : raw;
|
|
57
63
|
if (value !== null && !isScored(value))
|
|
58
64
|
throw new Error(
|
|
59
65
|
`scores.${id} must be a boolean, a finite number from 0 to 1, or null`,
|
|
@@ -62,11 +68,23 @@ export function codeJudgment(output: JsonValue): Judgment {
|
|
|
62
68
|
id,
|
|
63
69
|
{
|
|
64
70
|
value: value as number | null,
|
|
65
|
-
reason:
|
|
71
|
+
reason: detailed ? detailed.reason as string :
|
|
66
72
|
value === null
|
|
67
73
|
? "judge.ts returned an unresolved score."
|
|
68
74
|
: `judge.ts returned ${String(supplied)}.`,
|
|
69
|
-
evidence:
|
|
75
|
+
evidence: detailed?.evidence !== undefined
|
|
76
|
+
? (() => {
|
|
77
|
+
if (!Array.isArray(detailed.evidence) || detailed.evidence.length > 32)
|
|
78
|
+
throw new Error(`scores.${id}.evidence must contain at most 32 citations`);
|
|
79
|
+
return detailed.evidence.map(parseCitation);
|
|
80
|
+
})()
|
|
81
|
+
: value === null ? [] : [{ kind: "recording" }],
|
|
82
|
+
...(detailed?.measurements !== undefined ? { measurements: (() => {
|
|
83
|
+
const values = detailed.measurements;
|
|
84
|
+
if (!values || typeof values !== "object" || Array.isArray(values))
|
|
85
|
+
throw new Error(`scores.${id}.measurements must be an object`);
|
|
86
|
+
return values;
|
|
87
|
+
})() } : {}),
|
|
70
88
|
source: "judge.ts",
|
|
71
89
|
},
|
|
72
90
|
];
|