@hona/openeval 0.2.2 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/JUDGING.md +135 -23
- package/README.md +71 -9
- package/package.json +7 -6
- package/src/app/cost-plan.ts +20 -16
- package/src/app/eval-state.ts +2 -1
- package/src/app/grade-recording.ts +72 -0
- package/src/app/input-fingerprints.ts +22 -8
- package/src/app/judge-evidence.ts +26 -11
- package/src/app/judge-run.ts +36 -19
- package/src/app/load-benchmark.ts +48 -16
- package/src/app/read-results.ts +35 -2
- package/src/app/read-run.ts +23 -2
- package/src/app/rejudge.ts +2 -1
- package/src/app/run-benchmark.ts +8 -1
- package/src/app/run-eval-pipeline.ts +2 -1
- package/src/app/run-eval.ts +13 -0
- package/src/app/scores.ts +76 -34
- package/src/app/serve-results.ts +2 -0
- package/src/evidence.ts +4 -1
- package/src/index.ts +15 -3
- package/src/infra/containers/runtime/package.json +2 -2
- package/src/infra/containers/runtime/server.mjs +1 -1
- package/src/infra/evidence/index.ts +39 -2
- package/src/infra/evidence/tool-calls.ts +5 -2
- package/src/infra/judging/agent.ts +3 -2
- package/src/infra/judging/code-result.ts +102 -0
- package/src/infra/judging/code-source.ts +96 -0
- package/src/infra/judging/code-worker.ts +26 -0
- package/src/infra/judging/code.ts +97 -0
- package/src/infra/judging/contract.ts +82 -64
- package/src/infra/judging/evidence-tool.ts +3 -1
- package/src/infra/judging/index.ts +14 -0
- package/src/infra/judging/judge-agent.md +24 -18
- package/src/infra/judging/observer.ts +8 -7
- package/src/infra/judging/submission.ts +8 -8
- package/src/infra/judging/tools.ts +9 -9
- package/src/infra/opencode/host.ts +1 -1
- package/src/infra/opencode/read-recording.ts +6 -1
- package/src/infra/opencode/version.ts +2 -0
- package/src/infra/recording/index.ts +240 -0
- package/src/infra/recording/metrics.ts +193 -0
- package/src/infra/sqlite/index.ts +29 -19
- package/src/judge-context.ts +141 -0
- package/src/judgment.ts +26 -17
- package/src/types.ts +36 -17
- package/src/view.ts +40 -14
- package/viewer/THIRD_PARTY_LICENSES.md +2 -2
- package/viewer/assets/{abnfDiagram-VCTEODGH-B1kcYgQG.js → abnfDiagram-VCTEODGH-Ck7reUId.js} +1 -1
- package/viewer/assets/{angular-html-DrYCUiZv.js → angular-html-DzRW817u.js} +1 -1
- package/viewer/assets/{angular-ts-BgjwQn-Z.js → angular-ts-Dwpo5JSv.js} +1 -1
- package/viewer/assets/{apl-DlHqe8o4.js → apl-bt59SDvg.js} +1 -1
- package/viewer/assets/{arc-C4nf7ZoY.js → arc-Dwbl2Pkh.js} +1 -1
- package/viewer/assets/architecture-7GRP2DOG-BZpt0zSq.js +1 -0
- package/viewer/assets/{architectureDiagram-5GKGNRK7-BIWoe0fO.js → architectureDiagram-5GKGNRK7-BcmO-Qqp.js} +1 -1
- package/viewer/assets/{astro-BrlHnIh9.js → astro-B3siF1pJ.js} +1 -1
- package/viewer/assets/{blade-DOcerOiW.js → blade-BevF_cYv.js} +1 -1
- package/viewer/assets/{blockDiagram-I7D4REHJ-C3x13a9m.js → blockDiagram-I7D4REHJ-MzX9_xf9.js} +1 -1
- package/viewer/assets/{c-BMbThxEo.js → c-gavBjOcY.js} +1 -1
- package/viewer/assets/{c4Diagram-7LVT6UL2-BeWktchx.js → c4Diagram-7LVT6UL2-BbHHPJGg.js} +1 -1
- package/viewer/assets/channel-BnnBJr7O.js +1 -0
- package/viewer/assets/{chapel-k2cjkSmc.js → chapel-DgshBe1K.js} +1 -1
- package/viewer/assets/{chunk-4HAMMTFA-o6YtJsqE.js → chunk-4HAMMTFA-D1Gsg_qO.js} +1 -1
- package/viewer/assets/{chunk-75Z2AOVW-KohQa9Nw.js → chunk-75Z2AOVW-EfPG3FOY.js} +1 -1
- package/viewer/assets/{chunk-DU6HZSFF-C88wxc6m.js → chunk-DU6HZSFF-BxB0lceI.js} +1 -1
- package/viewer/assets/{chunk-F27PBJKO-mO4AP2as.js → chunk-F27PBJKO-DRBDr_-Q.js} +1 -1
- package/viewer/assets/{chunk-GMAD6QVW-DzQX2Cld.js → chunk-GMAD6QVW-AbIWvPs8.js} +1 -1
- package/viewer/assets/{chunk-GVQU2GXP-CS78RfJD.js → chunk-GVQU2GXP-aIYNUxo9.js} +1 -1
- package/viewer/assets/{chunk-IMKFNOWR-DxS42GUo.js → chunk-IMKFNOWR-BwpTTnl_.js} +1 -1
- package/viewer/assets/{chunk-L3NEJ4N5-DGo-uYLy.js → chunk-L3NEJ4N5-KtxD1mfg.js} +1 -1
- package/viewer/assets/{chunk-OSK3NFVY-BOcXQG16.js → chunk-OSK3NFVY-CinbAXbX.js} +1 -1
- package/viewer/assets/{chunk-P2QGCYS3-B7GnZdIQ.js → chunk-P2QGCYS3-Dst8DZSR.js} +1 -1
- package/viewer/assets/{chunk-POPQ4Y6H-DgBcYHog.js → chunk-POPQ4Y6H-DqvBdud7.js} +1 -1
- package/viewer/assets/{chunk-PWAF6VOD-DpEq6qHA.js → chunk-PWAF6VOD-Cl2eG05g.js} +1 -1
- package/viewer/assets/{chunk-SHT3W25Y-CrOGxKM2.js → chunk-SHT3W25Y-D9FbZ836.js} +1 -1
- package/viewer/assets/{chunk-SVP7TREG-4HY7Fljj.js → chunk-SVP7TREG-_4n_lRpO.js} +1 -1
- package/viewer/assets/{chunk-TICWLB2K-B1Rl31EC.js → chunk-TICWLB2K-DUXjjJlY.js} +1 -1
- package/viewer/assets/{chunk-XXDRQBXY-CK8giEW-.js → chunk-XXDRQBXY-BdRoSapX.js} +1 -1
- package/viewer/assets/classDiagram-ZZMXUADV-F9J3BKRU.js +1 -0
- package/viewer/assets/classDiagram-v2-VYDZK3BY-F9J3BKRU.js +1 -0
- package/viewer/assets/{cobol-CupUwvw9.js → cobol-D5zqvXzY.js} +1 -1
- package/viewer/assets/{coffee-C17pkozF.js → coffee-SCAyVn9N.js} +1 -1
- package/viewer/assets/{cose-bilkent-JH36ORCC-V1OmrmCv.js → cose-bilkent-JH36ORCC-B5uDWB-J.js} +1 -1
- package/viewer/assets/{cpp-BJwVQVXg.js → cpp-RUQ97EK5.js} +1 -1
- package/viewer/assets/{crystal-BF-oCN-L.js → crystal-FauNVdkE.js} +1 -1
- package/viewer/assets/{css-BMgkVI3c.js → css-DuzOk7di.js} +1 -1
- package/viewer/assets/{cynefin-OW5HDTMX-BEjY_gsc.js → cynefin-OW5HDTMX-D0N8Dm6P.js} +1 -1
- package/viewer/assets/{cynefinDiagram-5FMLGOSQ-bfSYgr6z.js → cynefinDiagram-5FMLGOSQ-IGGiqH_Y.js} +1 -1
- package/viewer/assets/{dagre-GXQ25YYZ-VtwTgGfa.js → dagre-GXQ25YYZ-PoopsFcY.js} +1 -1
- package/viewer/assets/{diagram-S7CK7UJ4-CHaAAyW7.js → diagram-S7CK7UJ4-D6GuIlRD.js} +1 -1
- package/viewer/assets/{diagram-UQ7AKVKN-CjL65-As.js → diagram-UQ7AKVKN-Cj8lu7Hi.js} +1 -1
- package/viewer/assets/{diagram-VSXAHHWV-Cnxrk1TJ.js → diagram-VSXAHHWV-bI4-SjAp.js} +1 -1
- package/viewer/assets/{diagram-VX7I27RA-D78jPXr7.js → diagram-VX7I27RA-jWvTT618.js} +1 -1
- package/viewer/assets/{diagram-Z3DM3KII-DKjnOaF2.js → diagram-Z3DM3KII-D6Vcvcr8.js} +1 -1
- package/viewer/assets/{dist-BTYPW0dz.js → dist-Cn65YNrn.js} +1 -1
- package/viewer/assets/{ebnfDiagram-PWID7BFC-7XiXh5Lx.js → ebnfDiagram-PWID7BFC-y-ZxfP2H.js} +1 -1
- package/viewer/assets/{edge-B34N_HjN.js → edge-Bc0EBQ7y.js} +1 -1
- package/viewer/assets/{elixir-CQzWMyKD.js → elixir-BmxsT0oc.js} +1 -1
- package/viewer/assets/{elm-BNJiG-wg.js → elm-BrTAZ55m.js} +1 -1
- package/viewer/assets/{erDiagram-RLTQ6QDP-YPFEgZxU.js → erDiagram-RLTQ6QDP-Wa1tikbb.js} +1 -1
- package/viewer/assets/{erb-D5Rm3UNa.js → erb-B60HRxmg.js} +1 -1
- package/viewer/assets/eventmodeling-NTZA5JFV-BNvBHnQf.js +1 -0
- package/viewer/assets/flowDiagram-HODETNUW-zGgqOVuO.js +1 -0
- package/viewer/assets/{ganttDiagram-EL5Y4UJY-ergSGtTP.js → ganttDiagram-EL5Y4UJY-DjskXB68.js} +1 -1
- package/viewer/assets/{git-rebase-BpIMZIRG.js → git-rebase-CK9f68l3.js} +1 -1
- package/viewer/assets/{gitGraph-4MIJSDKK-CZt4REVj.js → gitGraph-4MIJSDKK-CvRhO7Ae.js} +1 -1
- package/viewer/assets/{gitGraphDiagram-WWUBYQGX-h9QMzC2r.js → gitGraphDiagram-WWUBYQGX-Bg_OLNL1.js} +1 -1
- package/viewer/assets/{glimmer-js-SeSwns5_.js → glimmer-js-CVc7keaR.js} +1 -1
- package/viewer/assets/{glimmer-ts-BExI5tTz.js → glimmer-ts-DYraWlmH.js} +1 -1
- package/viewer/assets/{glsl-CXikKw2b.js → glsl-BpYswgfh.js} +1 -1
- package/viewer/assets/{graphql-6GpQIInt.js → graphql-C3iDzxap.js} +1 -1
- package/viewer/assets/{hack-i7rizvpB.js → hack-D8oEoipZ.js} +1 -1
- package/viewer/assets/{haml-C68gmkTf.js → haml-BJb-BQAN.js} +1 -1
- package/viewer/assets/{handlebars-BBn_c0Xt.js → handlebars-wB54lHaB.js} +1 -1
- package/viewer/assets/{html-BObmAZQA.js → html-HIvKkSZ3.js} +1 -1
- package/viewer/assets/{html-derivative-BjMuR1LM.js → html-derivative-CDavbmdn.js} +1 -1
- package/viewer/assets/{http-BJBk8gt0.js → http-DEgShh0-.js} +1 -1
- package/viewer/assets/{hurl-D7obCYfg.js → hurl-D8wvJcyh.js} +1 -1
- package/viewer/assets/{index-FMtYXcX7.js → index-Ck3NBCVJ.js} +9 -7
- package/viewer/assets/{index-Wz0qQyIk.css → index-DAuO1BM4.css} +1 -1
- package/viewer/assets/{info-A6RAGUB7-CrzitQIU.js → info-A6RAGUB7-DL3H6Iqn.js} +1 -1
- package/viewer/assets/{infoDiagram-27XIBGKW-D2CgRDxu.js → infoDiagram-27XIBGKW-DSLzhrL8.js} +1 -1
- package/viewer/assets/{ishikawaDiagram-5VMMS53U-BhnQMHOo.js → ishikawaDiagram-5VMMS53U-B-ycXUrd.js} +1 -1
- package/viewer/assets/{java-5VJU_EW8.js → java-C5iWdXOl.js} +1 -1
- package/viewer/assets/{javascript-r8UgndxQ.js → javascript-PMSXGhX1.js} +1 -1
- package/viewer/assets/{jinja-BY2ibHG5.js → jinja-CJ_l63gX.js} +1 -1
- package/viewer/assets/{jison-C_7YxymZ.js → jison-DGaZyJAT.js} +1 -1
- package/viewer/assets/{journeyDiagram-3NMN7TZE-DSBzKQEa.js → journeyDiagram-3NMN7TZE-2p56tZPm.js} +1 -1
- package/viewer/assets/{json-8e1WlJYD.js → json-CCQ1eo9o.js} +1 -1
- package/viewer/assets/{jsx-BYp5GD4Y.js → jsx-DeLT99JL.js} +1 -1
- package/viewer/assets/{julia-Bkob1uVX.js → julia-DODLsulx.js} +1 -1
- package/viewer/assets/{just-CUnSjty-.js → just-B_Nvjvhv.js} +1 -1
- package/viewer/assets/{kanban-definition-UXKFOSKX-CotbrQue.js → kanban-definition-UXKFOSKX-B42fPx4K.js} +1 -1
- package/viewer/assets/{latex-UdE3ivqk.js → latex-CCjy2I9z.js} +1 -1
- package/viewer/assets/{line-DDh6a6hz.js → line-DGK8UZYe.js} +1 -1
- package/viewer/assets/{linear-BxMsCbKb.js → linear-LxXu6wOl.js} +1 -1
- package/viewer/assets/{liquid-WL4MvEIT.js → liquid-C2tknZP0.js} +1 -1
- package/viewer/assets/{lua-CtOubZta.js → lua-D6UzQw5Z.js} +1 -1
- package/viewer/assets/{marko-C0GSQJi3.js → marko-CGd5W4Hn.js} +1 -1
- package/viewer/assets/{mdc-DUivddFc.js → mdc-Bp5z92Vo.js} +1 -1
- package/viewer/assets/{mermaid-parser.core-FhKE_uv2.js → mermaid-parser.core-xGBt_sNt.js} +3 -3
- package/viewer/assets/{mermaid.core-BSOWwyvd.js → mermaid.core-D1TUMBuT.js} +4 -4
- package/viewer/assets/{mindmap-definition-YA3MSWOX-tsLZ_Opn.js → mindmap-definition-YA3MSWOX-BY9cVMtI.js} +1 -1
- package/viewer/assets/{nginx-B__INAtC.js → nginx-DAsqAd3h.js} +1 -1
- package/viewer/assets/{nim-UEY12GJF.js → nim-COOJz623.js} +1 -1
- package/viewer/assets/{org-DDIZsz95.js → org-CBV1qqrA.js} +1 -1
- package/viewer/assets/{packet-AYTQ26CC-CqdJnqfo.js → packet-AYTQ26CC-61q5HcB3.js} +1 -1
- package/viewer/assets/{pegDiagram-XKGWAZYB-BJs7qu7m.js → pegDiagram-XKGWAZYB-DLEkmimO.js} +1 -1
- package/viewer/assets/{perl-D5KwNZt7.js → perl-6SV_ayGd.js} +1 -1
- package/viewer/assets/{php-BEs7d_Dk.js → php-BUb-Io9m.js} +1 -1
- package/viewer/assets/{pie-WAS4IAKB-lgzqBWqy.js → pie-WAS4IAKB-DgijP3YA.js} +1 -1
- package/viewer/assets/{pieDiagram-E7YTZNPT-Cxe9LnnE.js → pieDiagram-E7YTZNPT-Cp20J80-.js} +1 -1
- package/viewer/assets/{pug-B0dsfQOV.js → pug-D5s94d1b.js} +1 -1
- package/viewer/assets/{qml-XPIhrIzf.js → qml-D8g4NGVo.js} +1 -1
- package/viewer/assets/{quadrantDiagram-AXDQQJYC-D5JGwC11.js → quadrantDiagram-AXDQQJYC-Djsc9xpm.js} +1 -1
- package/viewer/assets/{r-C6hQdpIM.js → r-DXvz5K4A.js} +1 -1
- package/viewer/assets/{radar-RG4KPBEZ-B7JRLhUA.js → radar-RG4KPBEZ-BCmoUpe8.js} +1 -1
- package/viewer/assets/{railroad-74A4TZTK-Df7ZQBLn.js → railroad-74A4TZTK-DZ0SYESf.js} +1 -1
- package/viewer/assets/railroad-abnf-HS5TGJTU-C2CQDbu6.js +1 -0
- package/viewer/assets/railroad-ebnf-LZEXJU2U-6ssJubZr.js +1 -0
- package/viewer/assets/railroad-peg-WCYAUIDC-UOxDGPyg.js +1 -0
- package/viewer/assets/{railroadDiagram-O6MQD6OU-DZJzpmP8.js → railroadDiagram-O6MQD6OU-DmD4JR_t.js} +1 -1
- package/viewer/assets/{razor-C3KSUuOw.js → razor-D9CjUAvE.js} +1 -1
- package/viewer/assets/{regexp-n-T936Rk.js → regexp-CnuLvLnC.js} +1 -1
- package/viewer/assets/{requirementDiagram-BXWQKSXE-Be9lEFrk.js → requirementDiagram-BXWQKSXE-BTci7X6J.js} +1 -1
- package/viewer/assets/{rst-fVnHDyXD.js → rst-Chdv5DuC.js} +1 -1
- package/viewer/assets/{ruby-DuRgxlXF.js → ruby-DXQkNv8k.js} +1 -1
- package/viewer/assets/{sankeyDiagram-P5KCCOFB-BsgKAEBN.js → sankeyDiagram-P5KCCOFB-BaHSMANQ.js} +1 -1
- package/viewer/assets/{sas-C0WA2OFa.js → sas-B6S0Oq1H.js} +1 -1
- package/viewer/assets/{scss-CQDZ4smR.js → scss-BnFPXH54.js} +1 -1
- package/viewer/assets/{sequenceDiagram-WJ2MYXX4-BkUDWL0D.js → sequenceDiagram-WJ2MYXX4-BUk21g1t.js} +1 -1
- package/viewer/assets/{shellscript-CE84G0GP.js → shellscript-Brdg8vyE.js} +1 -1
- package/viewer/assets/{shellsession-BdJvdaqL.js → shellsession-Df33UGC7.js} +1 -1
- package/viewer/assets/{soy-BSxOREC0.js → soy-DE83qari.js} +1 -1
- package/viewer/assets/{sql-D4MN9uRI.js → sql-C838IanB.js} +1 -1
- package/viewer/assets/{src-BT_kESTA.js → src-tvfSkCy9.js} +1 -1
- package/viewer/assets/{stata-B6nXAaGV.js → stata-BjrFc6Ob.js} +1 -1
- package/viewer/assets/{stateDiagram-D77RDMKH-C80wp8xG.js → stateDiagram-D77RDMKH-B7yRJv9-.js} +1 -1
- package/viewer/assets/stateDiagram-v2-MP3YSRHH-BCwhllSg.js +1 -0
- package/viewer/assets/{surrealql-fVjZmhV3.js → surrealql-BQl2P64c.js} +1 -1
- package/viewer/assets/{svelte-DnBQ5p7d.js → svelte-WhVuP9VM.js} +1 -1
- package/viewer/assets/{swimlanes-42K2YHIH-Cff6yT8e.js → swimlanes-42K2YHIH-on6wyUF5.js} +1 -1
- package/viewer/assets/swimlanesDiagram-VR7AAH4N-D0QyhouX.js +8 -0
- package/viewer/assets/{templ-TCx51Uts.js → templ-GlyHWfam.js} +1 -1
- package/viewer/assets/{tex-CF6SMAQ_.js → tex-Wu25UU_Z.js} +1 -1
- package/viewer/assets/{timeline-definition-24CTP7MA-Cje7BvaN.js → timeline-definition-24CTP7MA-DsgJtBig.js} +1 -1
- package/viewer/assets/{treeView-Q6P3EWNA-D-QLY3IO.js → treeView-Q6P3EWNA-Bae2q_db.js} +1 -1
- package/viewer/assets/{treemap-WGGIJYW6-BCHcTa4N.js → treemap-WGGIJYW6-BJsxvsBb.js} +1 -1
- package/viewer/assets/{ts-tags-BqLRnADX.js → ts-tags-C6TgXLV6.js} +1 -1
- package/viewer/assets/{tsx-D7KpDX0b.js → tsx-DVoPHges.js} +1 -1
- package/viewer/assets/{twig-CEZLObpN.js → twig-CKzx4A42.js} +1 -1
- package/viewer/assets/{typescript-DAbIDLbf.js → typescript-DvXz8u_k.js} +1 -1
- package/viewer/assets/{typst-CqN7-whv.js → typst-BcepwfpV.js} +1 -1
- package/viewer/assets/{vennDiagram-4TSXK5OY-Bz20rx02.js → vennDiagram-4TSXK5OY-CKUR4uY_.js} +1 -1
- package/viewer/assets/{vue-kTxN6wdc.js → vue-Bqzme_T-.js} +1 -1
- package/viewer/assets/{vue-html-DQmvvKbd.js → vue-html-DZHKNOUy.js} +1 -1
- package/viewer/assets/{vue-vine-Did3qTVW.js → vue-vine-azvFs0P6.js} +1 -1
- package/viewer/assets/{wardley-WFR3VGLG-CpXaqOit.js → wardley-WFR3VGLG-Ddsu3qLB.js} +1 -1
- package/viewer/assets/{wardleyDiagram-VM6X3IG4-Cz06jHpA.js → wardleyDiagram-VM6X3IG4-CCHx9UX4.js} +1 -1
- package/viewer/assets/{xml-te2hFcAA.js → xml-BZNC4g25.js} +1 -1
- package/viewer/assets/{xsl-X4fuKSVl.js → xsl-DXEWaWc5.js} +1 -1
- package/viewer/assets/{xychartDiagram-S5SC5T6Z-BXCw0sfu.js → xychartDiagram-S5SC5T6Z-CkoTq405.js} +1 -1
- package/viewer/assets/{yaml-NHtDSZab.js → yaml-DLY3wDEk.js} +1 -1
- package/viewer/index.html +2 -2
- package/viewer/assets/architecture-7GRP2DOG-Be250VKZ.js +0 -1
- package/viewer/assets/channel-DtULXRee.js +0 -1
- package/viewer/assets/classDiagram-ZZMXUADV-Bmqu01Xf.js +0 -1
- package/viewer/assets/classDiagram-v2-VYDZK3BY-Bmqu01Xf.js +0 -1
- package/viewer/assets/eventmodeling-NTZA5JFV-CJCg5f6p.js +0 -1
- package/viewer/assets/flowDiagram-HODETNUW-BFzq4Zvs.js +0 -1
- package/viewer/assets/railroad-abnf-HS5TGJTU-Z8q3QVTf.js +0 -1
- package/viewer/assets/railroad-ebnf-LZEXJU2U-BWoRkG3W.js +0 -1
- package/viewer/assets/railroad-peg-WCYAUIDC-BzoV1cN7.js +0 -1
- package/viewer/assets/stateDiagram-v2-MP3YSRHH-CYdsKKlz.js +0 -1
- package/viewer/assets/swimlanesDiagram-VR7AAH4N-niTUGjMT.js +0 -8
package/JUDGING.md
CHANGED
|
@@ -1,8 +1,116 @@
|
|
|
1
1
|
# Judging recorded work
|
|
2
2
|
|
|
3
|
+
Use the [canonical terminology](https://openev.al/docs/terminology/): a criterion
|
|
4
|
+
is a named graded requirement, a score is awarded credit, and a metric is a
|
|
5
|
+
measurement such as token count or cost. OpenEval 0.3.0 supports code judges,
|
|
6
|
+
LLM judges, and additive use of both against one recorded EvalRun.
|
|
7
|
+
|
|
8
|
+
## File conventions
|
|
9
|
+
|
|
10
|
+
Every eval has prompt.md and at least one judge file:
|
|
11
|
+
|
|
12
|
+
| File | Role |
|
|
13
|
+
| --- | --- |
|
|
14
|
+
| judge.md | An LLM rubric with named criteria |
|
|
15
|
+
| judge.ts | An ordinary default-exported function receiving JudgeContext |
|
|
16
|
+
| Both | Both contribute distinct criterion scores; duplicate IDs are errors |
|
|
17
|
+
|
|
18
|
+
Only benchmarks containing judge.md need judge.model. Code judges run in a
|
|
19
|
+
separate Bun process on finalized evidence, under judge.timeoutMs (default ten
|
|
20
|
+
minutes). Code and hybrid evals use final grading; earlyStop is available to
|
|
21
|
+
Markdown-only evals. Candidate execution remains isolated from all judge code.
|
|
22
|
+
|
|
23
|
+
## Plain code judges
|
|
24
|
+
|
|
25
|
+
```ts
|
|
26
|
+
import type { JudgeContext } from "@hona/openeval";
|
|
27
|
+
|
|
28
|
+
export default ({ response }: JudgeContext) => ({
|
|
29
|
+
scores: { correct_answer: response.text === "APPLE" },
|
|
30
|
+
observed: response.text,
|
|
31
|
+
});
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
The function may be asynchronous and may return any JSON-compatible value.
|
|
35
|
+
Only the optional scores object has grading semantics. Each key is a criterion
|
|
36
|
+
ID, using lowercase letters, digits, and underscores, starting with a letter.
|
|
37
|
+
Values are booleans, finite numbers from 0 to 1, or null. The host converts true
|
|
38
|
+
to 1 and false to 0. It rejects invalid values rather than clamping them.
|
|
39
|
+
response.text is always a string; missing text becomes an empty string. The
|
|
40
|
+
original execution outcome is available separately on context.run.
|
|
41
|
+
|
|
42
|
+
Custom output is retained verbatim. An output without scores is unscored and
|
|
43
|
+
cannot silently disappear from the benchmark denominator. Use consistent score
|
|
44
|
+
IDs across models and repetitions; a missing required score stays unresolved.
|
|
45
|
+
There are no built-in task-specific scorers, registration steps, or builder APIs.
|
|
46
|
+
|
|
47
|
+
The host bundles local imports before execution, captures source maps and
|
|
48
|
+
dependency manifests/lockfiles, and records their fingerprint. Code, imported
|
|
49
|
+
references, or dependency changes schedule rejudging rather than candidate
|
|
50
|
+
execution. Use static imports for reference data; make runtime network and file
|
|
51
|
+
inputs reproducible when an author-owned judge uses them.
|
|
52
|
+
|
|
53
|
+
The viewer shows normalized criterion scores, original returned JSON, frozen
|
|
54
|
+
source, process logs, and recorded candidate metrics. A synchronous loop can be
|
|
55
|
+
terminated by the host deadline. A code exception, invalid result, or timeout is
|
|
56
|
+
a JudgeRun error. A hybrid judgment is selected only after both sources succeed.
|
|
57
|
+
|
|
58
|
+
## JudgeContext and recorded data
|
|
59
|
+
|
|
60
|
+
| Primitive | Data |
|
|
61
|
+
| --- | --- |
|
|
62
|
+
| response / prompt | Final root answer and the exact task prompt |
|
|
63
|
+
| run | Recorded EvalRun and runtime inputs; null for constructed controls |
|
|
64
|
+
| metrics | Candidate-only usage, cost, tool reliability, compactions, and timing |
|
|
65
|
+
| recording.events(filter?) | Complete retained native events with their sequence and time |
|
|
66
|
+
| recording.tools(filter?) | Inputs, outputs, states, and timing of recorded invocations |
|
|
67
|
+
| recording.sessions() | All sessions in the candidate's isolated native archive |
|
|
68
|
+
| recording.messages(sessionID?) | Full paginated native history, including before compaction |
|
|
69
|
+
| recording.export(sessionID?) | Native OpenCode session export |
|
|
70
|
+
| workspace.files/read/text/diff | Verified initial and final file snapshots |
|
|
71
|
+
| workspace.materialize(revision?) | A disposable workspace copy for author-owned verification |
|
|
72
|
+
| native.database() | Read-only SQLite access to a verified database copy |
|
|
73
|
+
| native.sdk() | The pinned OpenCode SDK/API over a separate disposable archive copy |
|
|
74
|
+
| native.schema() | The pinned OpenCode schema module |
|
|
75
|
+
|
|
76
|
+
Native readers initialize lazily and are disposed by the runner. SDK operations
|
|
77
|
+
and native schema upgrades affect only their disposable copy. The recorded
|
|
78
|
+
OpenCode version remains available separately from the reader version. Files and
|
|
79
|
+
database copies are checked against recorded hashes.
|
|
80
|
+
These APIs expose recorded data, not the user's live OpenCode service.
|
|
81
|
+
|
|
82
|
+
For independent inspection:
|
|
83
|
+
|
|
84
|
+
```ts
|
|
85
|
+
import { readRecording } from "@hona/openeval";
|
|
86
|
+
|
|
87
|
+
await using context = await readRecording("./results/RUN", "eval_ID");
|
|
88
|
+
console.log(context.metrics.tools.errorRate);
|
|
89
|
+
const history = await context.recording.messages();
|
|
90
|
+
const database = await context.native.database();
|
|
91
|
+
console.log(database.query("SELECT name FROM sqlite_master").all());
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
Metrics use deduplicated durable events across the candidate execution's
|
|
95
|
+
sessions. Usage includes recorded model requests and auxiliary usage such as
|
|
96
|
+
compaction. Cost is reported OpenCode usage, not an invoice or a promise of free
|
|
97
|
+
service. Missing usage is unavailable. Code that calls external services
|
|
98
|
+
directly can return its own accounting as metadata.
|
|
99
|
+
|
|
100
|
+
Tool error rate is failed / (succeeded + failed), with null when there are no
|
|
101
|
+
terminal calls. Unfinished calls are reported separately. A successful shell
|
|
102
|
+
tool reporting failed tests is not a native tool failure. Counts describe
|
|
103
|
+
recorded native invocations; do not infer uncaptured work inside a batched call.
|
|
104
|
+
|
|
105
|
+
Timing uses the union of closed recorded intervals. modelActiveMs includes
|
|
106
|
+
model-step and compaction spans, including time within those steps such as
|
|
107
|
+
retries. outputTokensPerSecond is reported output tokens per model-active second.
|
|
108
|
+
Token categories retain their native meanings; do not blindly add overlapping
|
|
109
|
+
reasoning, output, or cache categories into a new total.
|
|
110
|
+
|
|
3
111
|
## Shared judge agent
|
|
4
112
|
|
|
5
|
-
|
|
113
|
+
The Markdown judge uses the native OpenCode V2 primary agent `openeval-judge` for final
|
|
6
114
|
grading, live checks, rejudging, calibration, and independent audits. Its
|
|
7
115
|
[base system prompt](src/infra/judging/judge-agent.md) owns the common evidence,
|
|
8
116
|
citation, uncertainty, early-decision, and output rules. See
|
|
@@ -13,43 +121,44 @@ the eval rubric. It remains present across compaction. User turns identify the
|
|
|
13
121
|
phase; structured context and decisions move through registered Code Mode tools.
|
|
14
122
|
The tools use native Effect schemas for argument validation and catalog types.
|
|
15
123
|
|
|
16
|
-
|
|
124
|
+
An LLM JudgeRun records its shared profile in `input.agent`, explicit criteria,
|
|
17
125
|
mode, runtime hash, and protocol. The effective
|
|
18
126
|
native configuration is archived at `configuration/opencode.json` in its judge
|
|
19
127
|
directory. The profile contributes to the judge input fingerprint, so a shared
|
|
20
128
|
prompt change schedules rejudging using saved evidence. Candidate input
|
|
21
129
|
fingerprints do not include the judge profile.
|
|
22
130
|
|
|
23
|
-
## Eval
|
|
131
|
+
## Eval rubrics
|
|
24
132
|
|
|
25
|
-
Declare one or more
|
|
133
|
+
Declare one or more criteria in `judge.md`:
|
|
26
134
|
|
|
27
135
|
```md
|
|
28
136
|
# Advice quality
|
|
29
137
|
|
|
30
|
-
##
|
|
138
|
+
## Criterion: current_advice — Advice for the current situation
|
|
31
139
|
Pass when the requested advice is present and its recommendations are available now.
|
|
32
140
|
Fail when a recommendation is unavailable now, or the requested advice is omitted.
|
|
33
141
|
|
|
34
|
-
##
|
|
142
|
+
## Criterion: later_advice — Advice for later stages
|
|
35
143
|
Judge later recommendations at their explicitly stated stage.
|
|
36
144
|
```
|
|
37
145
|
|
|
38
|
-
Keep
|
|
39
|
-
The LLM applies these rules. The SDK validates the declared IDs,
|
|
146
|
+
Keep criterion-specific rules, accepted alternatives, and domain facts in that file.
|
|
147
|
+
The LLM applies these rules. The SDK validates the declared IDs, normalized criterion
|
|
40
148
|
values, recorded evidence references, and exact optional quotes. It calculates
|
|
41
|
-
the equal-weight
|
|
42
|
-
protocol error; missing source evidence can produce a null
|
|
43
|
-
must declare at least one `##
|
|
44
|
-
|
|
149
|
+
the equal-weight mean of criterion scores. A missing criterion or invalid citation is a judge
|
|
150
|
+
protocol error; missing source evidence can produce a null criterion score. Every rubric
|
|
151
|
+
must declare at least one `## Criterion: id — Label`. A normalized Judgment has
|
|
152
|
+
an aggregate value and a scores map of CriterionScore objects containing value,
|
|
153
|
+
reason, evidence, and source. The original code output is retained separately.
|
|
45
154
|
|
|
46
155
|
## Structured tool submissions
|
|
47
156
|
|
|
48
157
|
| Tool | Data |
|
|
49
158
|
| --- | --- |
|
|
50
|
-
| `judge_context` | Current request ID, phase,
|
|
51
|
-
| `candidate_evidence` | Recorded responses, tools, events, messages, and
|
|
52
|
-
| `submit_judgment` | Scores keyed by
|
|
159
|
+
| `judge_context` | Current request ID, phase, criterion declarations, and evidence index |
|
|
160
|
+
| `candidate_evidence` | Recorded responses, tools, events, messages, artifacts, and metrics |
|
|
161
|
+
| `submit_judgment` | Scores keyed by criterion ID, with reasons and citations |
|
|
53
162
|
| `continue_judging` | An early check's reason for needing more evidence |
|
|
54
163
|
|
|
55
164
|
Native argument validation and citation validation return errors directly to the
|
|
@@ -61,14 +170,14 @@ submission fails the JudgeRun rather than assigning a candidate zero.
|
|
|
61
170
|
Tool schemas stay stable across checks. Each request ID is bound to one fixed
|
|
62
171
|
evidence view. Only the first valid submission is accepted; stale, concurrent, or
|
|
63
172
|
cancelled submissions cannot overwrite it or affect a later check. Early
|
|
64
|
-
submissions require every
|
|
65
|
-
permits null
|
|
173
|
+
submissions require every criterion score to be non-null and irreversible. Final grading
|
|
174
|
+
permits null criterion scores and rejects `continue_judging`.
|
|
66
175
|
|
|
67
176
|
Accepted submissions and semantic rejections are recorded in
|
|
68
177
|
`judgment-submissions.json`, with their request/check IDs and checkpoints. Native
|
|
69
178
|
schema failures are retained in the native tool transcript.
|
|
70
179
|
|
|
71
|
-
Every decided
|
|
180
|
+
Every decided criterion score cites recorded evidence. Citations can name a response,
|
|
72
181
|
message ID, tool-call ID, event sequence, or initial/final artifact path. The
|
|
73
182
|
evidence reference/checkpoint binds those citations to the exact recording.
|
|
74
183
|
Source URLs are supplementary domain references, not substitutes for citations
|
|
@@ -80,7 +189,7 @@ For example, the judge submits this from Code Mode after inspecting the recordin
|
|
|
80
189
|
const context = await tools.judge_context({});
|
|
81
190
|
await tools.submit_judgment({
|
|
82
191
|
requestId: context.requestId,
|
|
83
|
-
|
|
192
|
+
scores: {
|
|
84
193
|
current_advice: {
|
|
85
194
|
value: 1,
|
|
86
195
|
reason: "The current-stage recommendation satisfies the rubric.",
|
|
@@ -95,7 +204,7 @@ await tools.submit_judgment({
|
|
|
95
204
|
});
|
|
96
205
|
```
|
|
97
206
|
|
|
98
|
-
The host stores the
|
|
207
|
+
The host stores the criterion-score map and aggregate value `0.5`. A citation's optional `quote` must occur
|
|
99
208
|
literally in the referenced response/tool/message/event/artifact. An artifact
|
|
100
209
|
citation must specify `path` and `revision` (`initial` or `final`); other record
|
|
101
210
|
citations specify `id`. Semantic interpretation remains the judge's job.
|
|
@@ -103,7 +212,9 @@ citations specify `id`. Semantic interpretation remains the judge's job.
|
|
|
103
212
|
`recordEvidence` creates a calibration recording from supplied text/tool records
|
|
104
213
|
without executing a model or tool. `judgeEvidence` grades retained evidence into
|
|
105
214
|
a new standalone audit directory without changing a benchmark's active scores.
|
|
106
|
-
Use `judgeRuns` when an explicit rejudge should update active selections.
|
|
215
|
+
Use `judgeRuns` when an explicit rejudge should update active selections. For a
|
|
216
|
+
code-only control, pass code: "./path/to/judge.ts" to judgeEvidence; no judge model
|
|
217
|
+
is needed. Pass both rubric and code for additive grading.
|
|
107
218
|
|
|
108
219
|
Runtime uses recorded work intervals. Judge checks record `executionStartedAt`
|
|
109
220
|
when they obtain a worker, separately from their queue-admission `startedAt`.
|
|
@@ -119,5 +230,6 @@ creation/completion dates remain provenance, not elapsed-runtime measurements.
|
|
|
119
230
|
- [JudgeBench](https://arxiv.org/abs/2410.12784): validate factual and logical
|
|
120
231
|
judging ability with known examples, rather than assuming model strength.
|
|
121
232
|
|
|
122
|
-
Keep task-specific decisions in
|
|
123
|
-
|
|
233
|
+
Keep task-specific decisions in the author-owned judge.md and judge.ts files.
|
|
234
|
+
The SDK handles recording, execution, validation, and equal-weight aggregation.
|
|
235
|
+
OpenEval 0.3.0 uses results schema 5 and the canonical API only.
|
package/README.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
<div align="center">
|
|
2
2
|
<h1>OpenEval</h1>
|
|
3
3
|
<p><strong>Write the task. Judge the evidence.</strong></p>
|
|
4
|
-
<p>
|
|
4
|
+
<p>Code and LLM judges for agents. Plain functions, isolated runs, inspectable scores.</p>
|
|
5
5
|
<p>
|
|
6
6
|
<a href="https://openev.al">Website</a> ·
|
|
7
7
|
<a href="https://openeval.pages.dev">Live preview</a> ·
|
|
@@ -16,18 +16,46 @@
|
|
|
16
16
|
</p>
|
|
17
17
|
</div>
|
|
18
18
|
|
|
19
|
-

|
|
20
20
|
|
|
21
21
|
*Interactive documentation example. Viewer screenshots use illustrative data and fictional model labels.*
|
|
22
22
|
|
|
23
|
-
##
|
|
23
|
+
## A prompt and a judge
|
|
24
24
|
|
|
25
25
|
| File | What you write | Who reads it |
|
|
26
26
|
| --- | --- | --- |
|
|
27
27
|
| `prompt.md` | A natural, focused task | Candidate agent |
|
|
28
|
-
| `judge.md` |
|
|
28
|
+
| `judge.md` | A rubric with named criteria and scoring rules | LLM judge |
|
|
29
|
+
| `judge.ts` | A plain function returning scores and custom JSON | Host-side Bun process |
|
|
29
30
|
| `eval.ts` *(optional)* | Workspace preparation and early stopping | Host |
|
|
30
31
|
|
|
32
|
+
Use `judge.md`, `judge.ts`, or both. Both judge files contribute distinct criteria
|
|
33
|
+
from the same recorded candidate execution.
|
|
34
|
+
|
|
35
|
+
### Deterministic: an ordinary function
|
|
36
|
+
|
|
37
|
+
For a task that asks the candidate to reply with exactly `APPLE`:
|
|
38
|
+
|
|
39
|
+
```ts
|
|
40
|
+
// evals/exact-answer/judge.ts
|
|
41
|
+
import type { JudgeContext } from "@hona/openeval";
|
|
42
|
+
|
|
43
|
+
export default ({ response }: JudgeContext) => ({
|
|
44
|
+
scores: { correct_answer: response.text === "APPLE" },
|
|
45
|
+
});
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Booleans become 0 or 1. Numeric scores can be any finite value from 0 to 1; null
|
|
49
|
+
is unresolved. Other JSON is retained as author-defined data. Cost, tokens, tool
|
|
50
|
+
reliability, timing, source identity, and recording links are supplied by the
|
|
51
|
+
runner. Code-only benchmarks do not need a judge model.
|
|
52
|
+
|
|
53
|
+
The context also exposes native events, complete message history, tool calls,
|
|
54
|
+
workspace snapshots, and lazy access to the recorded OpenCode SDK, schema, and
|
|
55
|
+
read-only database. See [code judges and data access](https://openev.al/docs/code-judges/).
|
|
56
|
+
|
|
57
|
+
### Model-based: write a rubric
|
|
58
|
+
|
|
31
59
|
**`evals/ask-dialect/prompt.md`**
|
|
32
60
|
|
|
33
61
|
```md
|
|
@@ -39,12 +67,12 @@ Write a SQL query for the ten most recent orders for a customer.
|
|
|
39
67
|
```md
|
|
40
68
|
# Requests the SQL dialect
|
|
41
69
|
|
|
42
|
-
##
|
|
70
|
+
## Criterion: asked_dialect — Asks for the SQL dialect
|
|
43
71
|
|
|
44
72
|
Pass when the agent asks which database or SQL dialect is in use.
|
|
45
73
|
Fail when it assumes a dialect without asking. Asking alongside a draft counts.
|
|
46
74
|
|
|
47
|
-
##
|
|
75
|
+
## Criterion: safe_parameters — Uses bound parameters
|
|
48
76
|
|
|
49
77
|
Pass when the proposed query uses a bound customer-ID parameter and explains
|
|
50
78
|
how to supply its value. Fail when it interpolates customer input into SQL
|
|
@@ -60,10 +88,30 @@ or does not provide a parameterized query.
|
|
|
60
88
|
|
|
61
89
|
→ [Write good rubrics](https://openev.al/docs/rubrics/) · [Download the SQL starter](https://openev.al/starter.zip)
|
|
62
90
|
|
|
91
|
+
## One vocabulary
|
|
92
|
+
|
|
93
|
+
A **benchmark** contains **evals**. Each eval defines a task and a **rubric**.
|
|
94
|
+
**Judges** produce **scores** for the rubric's **criteria**. Runs also record
|
|
95
|
+
**metrics** such as cost, tokens, and tool reliability.
|
|
96
|
+
|
|
97
|
+
- A **criterion** is a named requirement being graded, such as `safe_parameters`.
|
|
98
|
+
- A **score** is awarded credit, normalized from 0 to 1, or an aggregate of it.
|
|
99
|
+
- A **metric** is an observed or calculated measurement. A criterion must
|
|
100
|
+
explicitly use that measurement for it to affect the grade.
|
|
101
|
+
- A **judgment** is the judge's output. **BenchmarkRun**, **EvalRun**, and
|
|
102
|
+
**JudgeRun** name recorded executions, rather than reusable definitions.
|
|
103
|
+
|
|
104
|
+
See the [canonical terminology](https://openev.al/docs/terminology/) and the
|
|
105
|
+
[website glossary](https://openev.al/docs/terminology/).
|
|
106
|
+
|
|
63
107
|
## Choose models. Run. Inspect.
|
|
64
108
|
|
|
65
109
|
Requires **Bun 1.4.2+**, **Docker**, and connected models in **OpenCode**.
|
|
66
110
|
|
|
111
|
+
OpenEval 0.3.1 pins the production OpenCode packages at **2.0.3**. Run the
|
|
112
|
+
`image` command after upgrading to build `openeval-runtime:2.0.3`. Recorded runs
|
|
113
|
+
retain the OpenCode version that actually executed them.
|
|
114
|
+
|
|
67
115
|
```sh
|
|
68
116
|
bun add --exact @hona/openeval
|
|
69
117
|
```
|
|
@@ -80,6 +128,9 @@ export default {
|
|
|
80
128
|
} satisfies Benchmark;
|
|
81
129
|
```
|
|
82
130
|
|
|
131
|
+
The judge model is required when any eval contains judge.md. A code-only
|
|
132
|
+
benchmark can omit the judge setting, or set only judge.timeoutMs.
|
|
133
|
+
|
|
83
134
|
```sh
|
|
84
135
|
bunx --bun @hona/openeval image
|
|
85
136
|
bunx --bun @hona/openeval plan --only-eval ask-dialect
|
|
@@ -106,26 +157,37 @@ Models still declared in `benchmark.ts` can be added back by a later `run`.
|
|
|
106
157
|
flowchart LR
|
|
107
158
|
P["prompt.md"] --> C["Isolated candidate"] --> E["Recording"]
|
|
108
159
|
J["judge.md"] --> G["Judge + citations"]
|
|
109
|
-
|
|
160
|
+
T["judge.ts"] --> F["Code + recorded metrics"]
|
|
161
|
+
E --> G
|
|
162
|
+
E --> F
|
|
163
|
+
G --> S["Criterion scores"]
|
|
164
|
+
F --> S
|
|
165
|
+
S --> V["Results viewer"]
|
|
110
166
|
```
|
|
111
167
|
|
|
112
168
|
## See what earned the score
|
|
113
169
|
|
|
170
|
+

|
|
171
|
+
|
|
172
|
+
*Illustrative label-reading task. The code judge ran on constructed responses.*
|
|
173
|
+
|
|
114
174
|

|
|
115
175
|
|
|
116
176
|
| Capability | What you get | Guide |
|
|
117
177
|
| --- | --- | --- |
|
|
118
|
-
| Multiple
|
|
178
|
+
| Multiple criteria | Independent scores from one recording | [Rubrics](https://openev.al/docs/rubrics/) |
|
|
179
|
+
| Code and hybrid judges | Plain functions, booleans, fractional credit, custom JSON | [Code judges](https://openev.al/docs/code-judges/) |
|
|
119
180
|
| Controlled workspaces | Readable files, pinned Git inputs, preparation | [Workspaces](https://openev.al/docs/workspaces/) |
|
|
120
181
|
| Small batches | Eval, model, repetition, and cost controls | [Running](https://openev.al/docs/running/) |
|
|
121
182
|
| Transparent scores | Equal eval weights; bounds for unresolved checks | [Scoring](https://openev.al/docs/scoring/) |
|
|
122
183
|
| Evidence inspection | Sessions, tool results, artifacts, and citations | [Evidence](https://openev.al/docs/evidence/) |
|
|
184
|
+
| Native data primitives | Full history, event traces, SDK, schema, and read-only SQL | [Recorded data](https://openev.al/docs/code-judges/#context) |
|
|
123
185
|
| Rejudging | New judgments from retained, immutable recordings | [Evidence](https://openev.al/docs/evidence/#revise) |
|
|
124
186
|
|
|
125
187
|
<details>
|
|
126
188
|
<summary><strong>Inspect a judgment and its evidence</strong></summary>
|
|
127
189
|
|
|
128
|
-

|
|
129
191
|
|
|
130
192
|
</details>
|
|
131
193
|
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@hona/openeval",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "
|
|
3
|
+
"version": "0.3.1",
|
|
4
|
+
"description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"repository": {
|
|
7
7
|
"type": "git",
|
|
@@ -28,10 +28,11 @@
|
|
|
28
28
|
},
|
|
29
29
|
"dependencies": {
|
|
30
30
|
"@types/tar-stream": "3.1.4",
|
|
31
|
-
"@opencode/client": "
|
|
32
|
-
"@opencode/core": "
|
|
33
|
-
"@opencode/sdk": "
|
|
34
|
-
"@opencode/
|
|
31
|
+
"@opencode/client": "2.0.3",
|
|
32
|
+
"@opencode/core": "2.0.3",
|
|
33
|
+
"@opencode/sdk": "2.0.3",
|
|
34
|
+
"@opencode/schema": "2.0.3",
|
|
35
|
+
"@opencode/util": "2.0.3",
|
|
35
36
|
"drizzle-orm": "0.45.2",
|
|
36
37
|
"effect": "4.0.0-rc.112",
|
|
37
38
|
"tar-stream": "3.1.7"
|
package/src/app/cost-plan.ts
CHANGED
|
@@ -43,6 +43,9 @@ export function estimateWork(
|
|
|
43
43
|
(item) => item.action === "candidate" || item.action === "judge",
|
|
44
44
|
);
|
|
45
45
|
const items = ready.map((item) => {
|
|
46
|
+
const hasLlm = !!definition.evals.find(
|
|
47
|
+
(evalDefinition) => evalDefinition.id === item.slot.evalId,
|
|
48
|
+
)!.judge;
|
|
46
49
|
const candidate =
|
|
47
50
|
cost(
|
|
48
51
|
evals.filter(
|
|
@@ -53,22 +56,23 @@ export function estimateWork(
|
|
|
53
56
|
) ??
|
|
54
57
|
cost(evals.filter((run) => run.input.model === item.slot.model)) ??
|
|
55
58
|
cost(evals);
|
|
56
|
-
const judge =
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
(
|
|
60
|
-
run
|
|
61
|
-
|
|
62
|
-
(
|
|
63
|
-
e
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
59
|
+
const judge = !hasLlm
|
|
60
|
+
? 0
|
|
61
|
+
: (cost(
|
|
62
|
+
judges.filter(
|
|
63
|
+
(run) =>
|
|
64
|
+
run.input.model === definition.judge.model &&
|
|
65
|
+
evals.some(
|
|
66
|
+
(e) =>
|
|
67
|
+
e.id === run.input.evalRunId &&
|
|
68
|
+
e.input.evalId === item.slot.evalId,
|
|
69
|
+
),
|
|
70
|
+
),
|
|
71
|
+
) ??
|
|
72
|
+
cost(
|
|
73
|
+
judges.filter((run) => run.input.model === definition.judge.model),
|
|
74
|
+
) ??
|
|
75
|
+
cost(judges));
|
|
72
76
|
return {
|
|
73
77
|
slotId: item.slot.id,
|
|
74
78
|
estimatedUSD:
|
package/src/app/eval-state.ts
CHANGED
|
@@ -5,6 +5,7 @@ export function canJudgeEval(
|
|
|
5
5
|
run: EvalRun | undefined,
|
|
6
6
|
): run is EvalRun & { evidence: EvidenceRef } {
|
|
7
7
|
return (
|
|
8
|
-
!!run?.evidence &&
|
|
8
|
+
!!run?.evidence &&
|
|
9
|
+
["completed", "stopped", "timed_out", "failed"].includes(run.state)
|
|
9
10
|
);
|
|
10
11
|
}
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
import { resolve } from "node:path";
|
|
2
|
+
import type {
|
|
3
|
+
JudgeRunInput,
|
|
4
|
+
Judgment,
|
|
5
|
+
OpenCodeStreamEvent,
|
|
6
|
+
SessionArchive,
|
|
7
|
+
} from "../types";
|
|
8
|
+
import type { CodeJudgeExecution } from "../judge-context";
|
|
9
|
+
import type { RecordingInput } from "../infra/recording";
|
|
10
|
+
import { executeCodeJudge } from "../infra/judging/code";
|
|
11
|
+
import { codeJudgment, combineJudgments } from "../infra/judging/code-result";
|
|
12
|
+
import { executeJudge } from "../infra/judging";
|
|
13
|
+
import { errorMessage } from "../infra/files";
|
|
14
|
+
|
|
15
|
+
export type GradedRecording = {
|
|
16
|
+
state: "completed" | "failed" | "timed_out";
|
|
17
|
+
judgment?: Judgment;
|
|
18
|
+
session?: SessionArchive;
|
|
19
|
+
code?: CodeJudgeExecution;
|
|
20
|
+
error?: string;
|
|
21
|
+
};
|
|
22
|
+
|
|
23
|
+
/** Both source files contribute to one judgment over the same finalized recording. */
|
|
24
|
+
export async function gradeRecording(
|
|
25
|
+
input: JudgeRunInput,
|
|
26
|
+
recording: RecordingInput,
|
|
27
|
+
directory: string,
|
|
28
|
+
onEvent: (event: OpenCodeStreamEvent) => void,
|
|
29
|
+
evaluateLlm: typeof executeJudge = executeJudge,
|
|
30
|
+
): Promise<GradedRecording> {
|
|
31
|
+
let code: CodeJudgeExecution | undefined, session: SessionArchive | undefined;
|
|
32
|
+
try {
|
|
33
|
+
const judgments: Judgment[] = [];
|
|
34
|
+
if (input.code) {
|
|
35
|
+
code = await executeCodeJudge(
|
|
36
|
+
input.code,
|
|
37
|
+
recording,
|
|
38
|
+
resolve(directory, "code"),
|
|
39
|
+
input.timeoutMs,
|
|
40
|
+
);
|
|
41
|
+
if (code.state !== "completed")
|
|
42
|
+
return { state: code.state, code, error: code.error };
|
|
43
|
+
const judged = codeJudgment(code.output!);
|
|
44
|
+
for (const id of Object.keys(judged.scores))
|
|
45
|
+
if (input.criteria.some((criterion) => criterion.id === id))
|
|
46
|
+
throw new Error(
|
|
47
|
+
`Duplicate criterion ID from judge.md and judge.ts: ${id}`,
|
|
48
|
+
);
|
|
49
|
+
judgments.push(judged);
|
|
50
|
+
}
|
|
51
|
+
if (input.rubric) {
|
|
52
|
+
const graded = await evaluateLlm(input, directory, onEvent);
|
|
53
|
+
session = graded.session;
|
|
54
|
+
if (graded.result.state !== "completed" || !graded.judgment)
|
|
55
|
+
return {
|
|
56
|
+
state: "failed",
|
|
57
|
+
code,
|
|
58
|
+
session,
|
|
59
|
+
error: graded.result.error ?? "LLM judge did not submit scores",
|
|
60
|
+
};
|
|
61
|
+
judgments.push(graded.judgment);
|
|
62
|
+
}
|
|
63
|
+
return {
|
|
64
|
+
state: "completed",
|
|
65
|
+
code,
|
|
66
|
+
session,
|
|
67
|
+
judgment: combineJudgments(...judgments),
|
|
68
|
+
};
|
|
69
|
+
} catch (error) {
|
|
70
|
+
return { state: "failed", code, session, error: errorMessage(error) };
|
|
71
|
+
}
|
|
72
|
+
}
|
|
@@ -45,23 +45,37 @@ export const judgeFingerprint = (
|
|
|
45
45
|
) =>
|
|
46
46
|
fingerprint({
|
|
47
47
|
rubric: definition.evals.find((item) => item.id === evalId)!.judge,
|
|
48
|
-
|
|
49
|
-
|
|
48
|
+
code: definition.evals.find((item) => item.id === evalId)!.code?.hash,
|
|
49
|
+
agent: definition.evals.find((item) => item.id === evalId)!.judge
|
|
50
|
+
? JUDGE_AGENT
|
|
51
|
+
: undefined,
|
|
52
|
+
judge: definition.evals.find((item) => item.id === evalId)!.judge
|
|
53
|
+
? definition.judge
|
|
54
|
+
: { timeoutMs: definition.judge.timeoutMs },
|
|
50
55
|
protocol: JUDGE_PROTOCOL,
|
|
51
56
|
});
|
|
52
57
|
export const savedJudgeFingerprint = (
|
|
53
58
|
input: Pick<
|
|
54
59
|
JudgeRunInput,
|
|
55
|
-
|
|
60
|
+
| "rubric"
|
|
61
|
+
| "agent"
|
|
62
|
+
| "protocol"
|
|
63
|
+
| "model"
|
|
64
|
+
| "timeoutMs"
|
|
65
|
+
| "websearch"
|
|
66
|
+
| "code"
|
|
56
67
|
>,
|
|
57
68
|
) =>
|
|
58
69
|
fingerprint({
|
|
59
70
|
rubric: input.rubric,
|
|
71
|
+
code: input.code?.hash,
|
|
60
72
|
agent: input.agent,
|
|
61
73
|
protocol: input.protocol,
|
|
62
|
-
judge:
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
74
|
+
judge: input.rubric
|
|
75
|
+
? {
|
|
76
|
+
model: input.model,
|
|
77
|
+
timeoutMs: input.timeoutMs,
|
|
78
|
+
websearch: input.websearch,
|
|
79
|
+
}
|
|
80
|
+
: { timeoutMs: input.timeoutMs },
|
|
67
81
|
});
|
|
@@ -1,37 +1,47 @@
|
|
|
1
1
|
import { resolve, dirname } from "node:path";
|
|
2
2
|
import type { EvidenceRef, Judge, JudgeRunInput, ToolCall } from "../types";
|
|
3
|
-
import {
|
|
3
|
+
import { judgingFingerprint } from "../infra/judging";
|
|
4
4
|
import { EvidenceCapture } from "../infra/evidence";
|
|
5
|
-
import { JUDGE_PROTOCOL,
|
|
5
|
+
import { JUDGE_PROTOCOL, rubricCriteria } from "../judgment";
|
|
6
6
|
import { JUDGE_AGENT } from "../infra/judging/agent";
|
|
7
7
|
import { writeJson } from "../infra/files";
|
|
8
8
|
import { savedJudgeFingerprint } from "./input-fingerprints";
|
|
9
9
|
import { mkdir, mkdtemp, rm } from "node:fs/promises";
|
|
10
|
+
import { compileCodeJudge } from "../infra/judging/code-source";
|
|
11
|
+
import { gradeRecording } from "./grade-recording";
|
|
10
12
|
|
|
11
13
|
/** Inspect retained evidence without changing benchmark selections.
|
|
12
14
|
* Useful for calibration and second opinions: https://arxiv.org/abs/2410.12784
|
|
13
15
|
*/
|
|
14
16
|
export async function judgeEvidence(options: {
|
|
15
17
|
evidence: EvidenceRef;
|
|
16
|
-
rubric
|
|
17
|
-
|
|
18
|
+
rubric?: string;
|
|
19
|
+
code?: string;
|
|
20
|
+
judge?: Judge;
|
|
18
21
|
directory: string;
|
|
19
22
|
}) {
|
|
20
23
|
const directory = resolve(options.directory);
|
|
21
24
|
if (await Bun.file(resolve(directory, "input.json")).exists())
|
|
22
25
|
throw new Error("Choose a new judge evidence directory");
|
|
26
|
+
if (!options.rubric?.trim() && !options.code)
|
|
27
|
+
throw new Error("Provide rubric text, a code judge path, or both");
|
|
28
|
+
if (options.rubric && !options.judge?.model)
|
|
29
|
+
throw new Error("A Markdown rubric requires judge.model");
|
|
30
|
+
const code = options.code ? await compileCodeJudge(options.code) : undefined;
|
|
23
31
|
const request: Omit<JudgeRunInput, "judgeHash"> = {
|
|
24
32
|
evalRunId: "retained-evidence",
|
|
25
33
|
evidence: {
|
|
26
34
|
...options.evidence,
|
|
27
35
|
directory: resolve(options.evidence.directory),
|
|
28
36
|
},
|
|
29
|
-
rubric: options.rubric,
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
37
|
+
rubric: options.rubric ?? "",
|
|
38
|
+
kind: code ? (options.rubric ? "hybrid" : "code") : "llm",
|
|
39
|
+
code,
|
|
40
|
+
agent: options.rubric ? JUDGE_AGENT : undefined,
|
|
41
|
+
model: options.rubric ? options.judge?.model : undefined,
|
|
42
|
+
timeoutMs: options.judge?.timeoutMs ?? 600_000,
|
|
43
|
+
websearch: options.judge?.websearch ?? false,
|
|
44
|
+
criteria: options.rubric ? rubricCriteria(options.rubric) : [],
|
|
35
45
|
protocol: JUDGE_PROTOCOL,
|
|
36
46
|
mode: "final" as const,
|
|
37
47
|
runtimeHash: await judgingFingerprint(),
|
|
@@ -41,7 +51,12 @@ export async function judgeEvidence(options: {
|
|
|
41
51
|
judgeHash: savedJudgeFingerprint(request),
|
|
42
52
|
};
|
|
43
53
|
await writeJson(resolve(directory, "input.json"), input);
|
|
44
|
-
const result = await
|
|
54
|
+
const result = await gradeRecording(
|
|
55
|
+
input,
|
|
56
|
+
{ evidence: input.evidence },
|
|
57
|
+
directory,
|
|
58
|
+
() => {},
|
|
59
|
+
);
|
|
45
60
|
await writeJson(resolve(directory, "result.json"), result);
|
|
46
61
|
return result;
|
|
47
62
|
}
|