@hona/openeval 0.3.2 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/JUDGING.md +57 -0
- package/README.md +13 -1
- package/package.json +1 -1
- package/src/app/input-fingerprints.ts +37 -7
- package/src/app/load-benchmark.ts +87 -7
- package/src/app/merge-runs.ts +1 -0
- package/src/app/read-results.ts +3 -1
- package/src/app/run-benchmark.ts +43 -4
- package/src/app/run-eval.ts +5 -1
- package/src/app/scores.ts +19 -2
- package/src/cli.ts +6 -0
- package/src/criterion-categories.ts +53 -0
- package/src/index.ts +5 -0
- package/src/infra/containers/catalog.ts +126 -0
- package/src/infra/judging/code-criteria.ts +112 -0
- package/src/infra/opencode/auth.ts +6 -2
- package/src/types.ts +16 -0
- package/src/view.ts +62 -0
- package/viewer/assets/{abnfDiagram-VCTEODGH-BD2a0nVg.js → abnfDiagram-VCTEODGH-D-idCGaW.js} +1 -1
- package/viewer/assets/{angular-html-DdVCwIK9.js → angular-html-B-7vkhmj.js} +1 -1
- package/viewer/assets/{angular-ts-CuqPbRgQ.js → angular-ts-CqJWLTIZ.js} +1 -1
- package/viewer/assets/{apl-CtDG5Cje.js → apl-DVTjqAhB.js} +1 -1
- package/viewer/assets/{arc-wyqSIjyo.js → arc-Bmz8zsvq.js} +1 -1
- package/viewer/assets/architecture-7GRP2DOG-D3Kw34KW.js +1 -0
- package/viewer/assets/{architectureDiagram-5GKGNRK7-BmPFNQC8.js → architectureDiagram-5GKGNRK7-CxqP2ijS.js} +1 -1
- package/viewer/assets/{astro-D6G6kLSd.js → astro-Ih6QH4I8.js} +1 -1
- package/viewer/assets/{blade-lg78flvd.js → blade-vzkWgA61.js} +1 -1
- package/viewer/assets/{blockDiagram-I7D4REHJ-uc_S4CYt.js → blockDiagram-I7D4REHJ-Dm35S95Q.js} +1 -1
- package/viewer/assets/{c-DnJ04u1N.js → c-092Q-y5e.js} +1 -1
- package/viewer/assets/{c4Diagram-7LVT6UL2-C6MKduOJ.js → c4Diagram-7LVT6UL2-DArgyJNC.js} +1 -1
- package/viewer/assets/channel-D2lVv6ST.js +1 -0
- package/viewer/assets/{chapel-BXrX4Xy8.js → chapel-DUOt2X3X.js} +1 -1
- package/viewer/assets/{chunk-4HAMMTFA-sUmS_B9_.js → chunk-4HAMMTFA-Bm3UCwQ6.js} +1 -1
- package/viewer/assets/{chunk-75Z2AOVW-C-WwWkXY.js → chunk-75Z2AOVW-CpmjsCJX.js} +1 -1
- package/viewer/assets/{chunk-DU6HZSFF-BMoyWhsq.js → chunk-DU6HZSFF-DYT2KEwe.js} +1 -1
- package/viewer/assets/{chunk-F27PBJKO-CP5YX4kg.js → chunk-F27PBJKO-BFGd2OPj.js} +1 -1
- package/viewer/assets/{chunk-GMAD6QVW-DD4A7Z3L.js → chunk-GMAD6QVW-D8gzxWqz.js} +1 -1
- package/viewer/assets/{chunk-GVQU2GXP-HCwJTvMX.js → chunk-GVQU2GXP-BZJsi-wS.js} +1 -1
- package/viewer/assets/{chunk-IMKFNOWR-B-IMCCL2.js → chunk-IMKFNOWR-BKbF8JvM.js} +1 -1
- package/viewer/assets/{chunk-L3NEJ4N5-DggkuyxL.js → chunk-L3NEJ4N5-SWKb6tCy.js} +1 -1
- package/viewer/assets/{chunk-OSK3NFVY-DVlD89z7.js → chunk-OSK3NFVY-BrqoFOmR.js} +1 -1
- package/viewer/assets/{chunk-P2QGCYS3-BxVX7Csy.js → chunk-P2QGCYS3-B10Cyxbx.js} +1 -1
- package/viewer/assets/{chunk-POPQ4Y6H-BAU1NKsx.js → chunk-POPQ4Y6H-DXpfBPGQ.js} +1 -1
- package/viewer/assets/{chunk-PWAF6VOD-DMn6Wi7h.js → chunk-PWAF6VOD-DZmaXnfr.js} +1 -1
- package/viewer/assets/{chunk-SHT3W25Y-CgZcCWnM.js → chunk-SHT3W25Y-1U6zFGxP.js} +1 -1
- package/viewer/assets/{chunk-SVP7TREG-Bkhb6FAV.js → chunk-SVP7TREG-CaW6KcpM.js} +1 -1
- package/viewer/assets/{chunk-TICWLB2K-D2tOhmdF.js → chunk-TICWLB2K-X4pj7G4S.js} +1 -1
- package/viewer/assets/{chunk-XXDRQBXY-fcL8PDvO.js → chunk-XXDRQBXY-0afAM3OP.js} +1 -1
- package/viewer/assets/classDiagram-ZZMXUADV-DqBNvABf.js +1 -0
- package/viewer/assets/classDiagram-v2-VYDZK3BY-DqBNvABf.js +1 -0
- package/viewer/assets/{cobol-CI6TrDxj.js → cobol-CtspwbZ4.js} +1 -1
- package/viewer/assets/{coffee-Ed3tsq0W.js → coffee-qbHB9gj8.js} +1 -1
- package/viewer/assets/{cose-bilkent-JH36ORCC-BLw4c4lj.js → cose-bilkent-JH36ORCC-7s0vcDw1.js} +1 -1
- package/viewer/assets/{cpp-NAHlcPQh.js → cpp-Dx37X5A_.js} +1 -1
- package/viewer/assets/{crystal-BXGK-sOE.js → crystal-CiWjJvk4.js} +1 -1
- package/viewer/assets/{css-BFp-qE9r.js → css-CLeuyJyb.js} +1 -1
- package/viewer/assets/{cynefin-OW5HDTMX-gzIP1VDo.js → cynefin-OW5HDTMX-B4XFGNNZ.js} +1 -1
- package/viewer/assets/{cynefinDiagram-5FMLGOSQ-Czu3f68h.js → cynefinDiagram-5FMLGOSQ-DdNMKbw6.js} +1 -1
- package/viewer/assets/{dagre-GXQ25YYZ-BAzeXsit.js → dagre-GXQ25YYZ-BkDo4A91.js} +1 -1
- package/viewer/assets/{diagram-S7CK7UJ4-CxrIUY6C.js → diagram-S7CK7UJ4-DU8cnnVy.js} +1 -1
- package/viewer/assets/{diagram-UQ7AKVKN-D_JFHKr_.js → diagram-UQ7AKVKN-Dg2emYKa.js} +1 -1
- package/viewer/assets/{diagram-VSXAHHWV-PWTXzum8.js → diagram-VSXAHHWV-C_-YSzcn.js} +1 -1
- package/viewer/assets/{diagram-VX7I27RA-C_daaExL.js → diagram-VX7I27RA-BLCEnxqe.js} +1 -1
- package/viewer/assets/{diagram-Z3DM3KII-B-fVF8bU.js → diagram-Z3DM3KII-62CpIiAg.js} +1 -1
- package/viewer/assets/{dist-BsTox645.js → dist-CQIgbdem.js} +1 -1
- package/viewer/assets/{ebnfDiagram-PWID7BFC-AvlNOJnV.js → ebnfDiagram-PWID7BFC-FkGGETgC.js} +1 -1
- package/viewer/assets/{edge-CJ-xiaut.js → edge-CzmGSXHr.js} +1 -1
- package/viewer/assets/{elixir-DwyPuR-j.js → elixir-DDfD_-oS.js} +1 -1
- package/viewer/assets/{elm-Dq236SU3.js → elm-l7aRfQIq.js} +1 -1
- package/viewer/assets/{erDiagram-RLTQ6QDP-CgD6jWsh.js → erDiagram-RLTQ6QDP-GlCpsS1v.js} +1 -1
- package/viewer/assets/{erb-Bp546jL9.js → erb-qVffN5xM.js} +1 -1
- package/viewer/assets/eventmodeling-NTZA5JFV-07UfGVCm.js +1 -0
- package/viewer/assets/flowDiagram-HODETNUW-CyIUf7TW.js +1 -0
- package/viewer/assets/{ganttDiagram-EL5Y4UJY-VPFwa9x1.js → ganttDiagram-EL5Y4UJY-CH_Tk4Ex.js} +1 -1
- package/viewer/assets/{git-rebase-D2PaLLH_.js → git-rebase-jEJ8nKk1.js} +1 -1
- package/viewer/assets/{gitGraph-4MIJSDKK-xi6qI7Z0.js → gitGraph-4MIJSDKK-Cc9o1l4Y.js} +1 -1
- package/viewer/assets/{gitGraphDiagram-WWUBYQGX-BlZkClbr.js → gitGraphDiagram-WWUBYQGX-UdfHBATc.js} +1 -1
- package/viewer/assets/{glimmer-js-yehmMKoY.js → glimmer-js-BMk2JCW8.js} +1 -1
- package/viewer/assets/{glimmer-ts-CcjA_qmM.js → glimmer-ts-BH7RwQxU.js} +1 -1
- package/viewer/assets/{glsl-Cp9eQ-AE.js → glsl-CAWrZoEq.js} +1 -1
- package/viewer/assets/{graphql-Cq1lEM8j.js → graphql-Bac_hecs.js} +1 -1
- package/viewer/assets/{hack-DPGfOK3S.js → hack-1OgpAtm-.js} +1 -1
- package/viewer/assets/{haml-DkQK82Gj.js → haml-CvwG3BzW.js} +1 -1
- package/viewer/assets/{handlebars-DTZJdhbW.js → handlebars-Cq4oEEp8.js} +1 -1
- package/viewer/assets/{html-DpTpgMiG.js → html-D5wZ_JR6.js} +1 -1
- package/viewer/assets/{html-derivative-uhUUy5oC.js → html-derivative-ZfzPj0aS.js} +1 -1
- package/viewer/assets/{http-B92AnJJ4.js → http-BDE2UDvO.js} +1 -1
- package/viewer/assets/{hurl-Dol2Yi-b.js → hurl-CQbXNLlk.js} +1 -1
- package/viewer/assets/index-BezQIu6a.js +795 -0
- package/viewer/assets/{index-DAuO1BM4.css → index-OL_rrWNy.css} +1 -1
- package/viewer/assets/{info-A6RAGUB7-B4V2dCg0.js → info-A6RAGUB7-YD13oMfx.js} +1 -1
- package/viewer/assets/{infoDiagram-27XIBGKW-DEVrKQCK.js → infoDiagram-27XIBGKW-jYKjA_l7.js} +1 -1
- package/viewer/assets/{ishikawaDiagram-5VMMS53U-BHff_Dap.js → ishikawaDiagram-5VMMS53U-BBPzHiVd.js} +1 -1
- package/viewer/assets/{java--cIN42Lc.js → java-Cbpu4oyT.js} +1 -1
- package/viewer/assets/{javascript-DyvSzR0T.js → javascript-UMuq64YD.js} +1 -1
- package/viewer/assets/{jinja-BGYQ1xJZ.js → jinja-DVvZtqgU.js} +1 -1
- package/viewer/assets/{jison-ClYldt1W.js → jison-CZCXBIV3.js} +1 -1
- package/viewer/assets/{journeyDiagram-3NMN7TZE-tlsIEFNa.js → journeyDiagram-3NMN7TZE-CMqG5ndb.js} +1 -1
- package/viewer/assets/{json-DEW4gudM.js → json-CzTvWngu.js} +1 -1
- package/viewer/assets/{jsx-lokPhZBE.js → jsx-CNJnEGR4.js} +1 -1
- package/viewer/assets/{julia-Bo9paQzm.js → julia-Bj0q4uoH.js} +1 -1
- package/viewer/assets/{just-DqqAsrxT.js → just-93-MRGUC.js} +1 -1
- package/viewer/assets/{kanban-definition-UXKFOSKX-CP8oZ9Q5.js → kanban-definition-UXKFOSKX-VyFYPq9b.js} +1 -1
- package/viewer/assets/{latex-CO3oZ8pK.js → latex-CXd1tMjA.js} +1 -1
- package/viewer/assets/{line-CWhya-77.js → line-D5nvVDHp.js} +1 -1
- package/viewer/assets/{linear-BJzBzMzU.js → linear-Cq-FJZ_z.js} +1 -1
- package/viewer/assets/{liquid-CUbRj59y.js → liquid-Cb-ALXSr.js} +1 -1
- package/viewer/assets/{lua-CXig6-_n.js → lua-BxTQiamn.js} +1 -1
- package/viewer/assets/{marko-DkQFFLti.js → marko-mFYyU5jn.js} +1 -1
- package/viewer/assets/{mdc-BFGAQDvL.js → mdc-Oryqox7_.js} +1 -1
- package/viewer/assets/{mermaid-parser.core-DqMtuveo.js → mermaid-parser.core-CCDsanlH.js} +3 -3
- package/viewer/assets/{mermaid.core-C6ANPM_a.js → mermaid.core-Crd1YMgi.js} +4 -4
- package/viewer/assets/{mindmap-definition-YA3MSWOX-CMZQVws2.js → mindmap-definition-YA3MSWOX-BKLrEFWt.js} +1 -1
- package/viewer/assets/{nginx-0b9CGIHO.js → nginx-C1PwWP7d.js} +1 -1
- package/viewer/assets/{nim-DplMSFX0.js → nim-PszXW-l4.js} +1 -1
- package/viewer/assets/{org-DqGSm_ke.js → org-DO0CuJlO.js} +1 -1
- package/viewer/assets/{packet-AYTQ26CC-DDCDKPAj.js → packet-AYTQ26CC-BNPwLE6g.js} +1 -1
- package/viewer/assets/{pegDiagram-XKGWAZYB-DjPb-guT.js → pegDiagram-XKGWAZYB-DjBg_iPh.js} +1 -1
- package/viewer/assets/{perl-D3nVgYGh.js → perl-CHhAgoDL.js} +1 -1
- package/viewer/assets/{php-CbHf38Cn.js → php-CA4H6qnu.js} +1 -1
- package/viewer/assets/{pie-WAS4IAKB-GN2C-_ew.js → pie-WAS4IAKB-BzNmHX-Y.js} +1 -1
- package/viewer/assets/{pieDiagram-E7YTZNPT-BCi2gkKC.js → pieDiagram-E7YTZNPT-cgMB-fBw.js} +1 -1
- package/viewer/assets/{pug-CZT1s20p.js → pug-DDuKTe7C.js} +1 -1
- package/viewer/assets/{qml-BkHbAs5e.js → qml-DSd2VypD.js} +1 -1
- package/viewer/assets/{quadrantDiagram-AXDQQJYC-BnQPKfTn.js → quadrantDiagram-AXDQQJYC-Crzhs7Np.js} +1 -1
- package/viewer/assets/{r-Bs6brPiR.js → r-qf-5qR5Q.js} +1 -1
- package/viewer/assets/{radar-RG4KPBEZ-DX0KNp6-.js → radar-RG4KPBEZ-B3oI70pB.js} +1 -1
- package/viewer/assets/{railroad-74A4TZTK-6asimRo7.js → railroad-74A4TZTK-DOM5Od17.js} +1 -1
- package/viewer/assets/railroad-abnf-HS5TGJTU-ScZ2h_5Q.js +1 -0
- package/viewer/assets/railroad-ebnf-LZEXJU2U-_Cl5zxJ-.js +1 -0
- package/viewer/assets/railroad-peg-WCYAUIDC-CRlGg7vV.js +1 -0
- package/viewer/assets/{railroadDiagram-O6MQD6OU-BMqn2tiZ.js → railroadDiagram-O6MQD6OU-M3SvSYf-.js} +1 -1
- package/viewer/assets/{razor-C2n7Xtyr.js → razor-B9n9DtIL.js} +1 -1
- package/viewer/assets/{regexp-Bo6Pl_fZ.js → regexp-DhGN0EOR.js} +1 -1
- package/viewer/assets/{requirementDiagram-BXWQKSXE-BUDP5axK.js → requirementDiagram-BXWQKSXE-CFisiZTo.js} +1 -1
- package/viewer/assets/{rst-xn9jHLsv.js → rst-D1SdhuXd.js} +1 -1
- package/viewer/assets/{ruby-uezBOJQA.js → ruby-BybsgZgf.js} +1 -1
- package/viewer/assets/{sankeyDiagram-P5KCCOFB-BqF__G69.js → sankeyDiagram-P5KCCOFB-BKf2TMWg.js} +1 -1
- package/viewer/assets/{sas-BFNcactC.js → sas-Io7QwCtD.js} +1 -1
- package/viewer/assets/{scss-DQctmtN5.js → scss-D6XY0yo3.js} +1 -1
- package/viewer/assets/{sequenceDiagram-WJ2MYXX4-COTwvRM0.js → sequenceDiagram-WJ2MYXX4-CHawNQZ9.js} +1 -1
- package/viewer/assets/{shellscript-b4scok3x.js → shellscript-DSk8kvCh.js} +1 -1
- package/viewer/assets/{shellsession-DhHx2Tve.js → shellsession-Be1CN8DO.js} +1 -1
- package/viewer/assets/{soy-CKEQyp4E.js → soy-CGkkqGyT.js} +1 -1
- package/viewer/assets/{sql-BJIx5HBg.js → sql-DIgb716V.js} +1 -1
- package/viewer/assets/{src-D9BxzP51.js → src-DP6Z6M4G.js} +1 -1
- package/viewer/assets/{stata-BxUsTnsL.js → stata-CS_2tb9p.js} +1 -1
- package/viewer/assets/{stateDiagram-D77RDMKH-CxPAUQG9.js → stateDiagram-D77RDMKH-Cx9Rf6fY.js} +1 -1
- package/viewer/assets/stateDiagram-v2-MP3YSRHH-DRb-6t_S.js +1 -0
- package/viewer/assets/{surrealql-Db9C_syE.js → surrealql-Do4o8w1B.js} +1 -1
- package/viewer/assets/{svelte-Bdum9H2e.js → svelte-3bKxeV6-.js} +1 -1
- package/viewer/assets/{swimlanes-42K2YHIH-DgVSbM2N.js → swimlanes-42K2YHIH-yL2oIw_D.js} +1 -1
- package/viewer/assets/swimlanesDiagram-VR7AAH4N-cDiCT6xi.js +8 -0
- package/viewer/assets/{templ-CPd8XvAU.js → templ-BgEYPNGP.js} +1 -1
- package/viewer/assets/{tex-ugtfHkTM.js → tex-D_p6whQw.js} +1 -1
- package/viewer/assets/{timeline-definition-24CTP7MA-C-uRF487.js → timeline-definition-24CTP7MA-BJeLq4a5.js} +1 -1
- package/viewer/assets/{treeView-Q6P3EWNA-BmAs0zGR.js → treeView-Q6P3EWNA-COcnImiE.js} +1 -1
- package/viewer/assets/{treemap-WGGIJYW6-BbMlmxPC.js → treemap-WGGIJYW6-DpbxWots.js} +1 -1
- package/viewer/assets/{ts-tags-CVfSxSrq.js → ts-tags-GBmMk7Oh.js} +1 -1
- package/viewer/assets/{tsx-DziBqK0e.js → tsx-BSD-yNhi.js} +1 -1
- package/viewer/assets/{twig-XWBVDqQT.js → twig-dwAG5SzX.js} +1 -1
- package/viewer/assets/{typescript-BQaQFWJj.js → typescript-B8YTPl9v.js} +1 -1
- package/viewer/assets/{typst-CKAqlUvq.js → typst-Dpzojbzo.js} +1 -1
- package/viewer/assets/{vennDiagram-4TSXK5OY-f5gjsolQ.js → vennDiagram-4TSXK5OY-nF34Gjtn.js} +1 -1
- package/viewer/assets/{vue-Bi21UrIJ.js → vue-Djbmk2DC.js} +1 -1
- package/viewer/assets/{vue-html-BXbrkthZ.js → vue-html-BXc5rfco.js} +1 -1
- package/viewer/assets/{vue-vine-BizJO_Ju.js → vue-vine-DHCfQh17.js} +1 -1
- package/viewer/assets/{wardley-WFR3VGLG-C1G-q39u.js → wardley-WFR3VGLG-C1qhMifc.js} +1 -1
- package/viewer/assets/{wardleyDiagram-VM6X3IG4-4Okmn6Ky.js → wardleyDiagram-VM6X3IG4-B0BfyYWQ.js} +1 -1
- package/viewer/assets/{xml-f1bGPKhD.js → xml-CdCEskcV.js} +1 -1
- package/viewer/assets/{xsl-BiaU5ESx.js → xsl-BbNukwXP.js} +1 -1
- package/viewer/assets/{xychartDiagram-S5SC5T6Z-B5TiTNm-.js → xychartDiagram-S5SC5T6Z-DJKYce1U.js} +1 -1
- package/viewer/assets/{yaml-ChWybZ9g.js → yaml-BYLDET4A.js} +1 -1
- package/viewer/index.html +3 -3
- package/viewer/assets/architecture-7GRP2DOG-BIR2L0CZ.js +0 -1
- package/viewer/assets/channel-BtIqrdXz.js +0 -1
- package/viewer/assets/classDiagram-ZZMXUADV-Fm1xUxjv.js +0 -1
- package/viewer/assets/classDiagram-v2-VYDZK3BY-Fm1xUxjv.js +0 -1
- package/viewer/assets/eventmodeling-NTZA5JFV-DjF7f3nw.js +0 -1
- package/viewer/assets/flowDiagram-HODETNUW-COsUd-GC.js +0 -1
- package/viewer/assets/index-tgc7CLl2.js +0 -795
- package/viewer/assets/railroad-abnf-HS5TGJTU-DYq0sKms.js +0 -1
- package/viewer/assets/railroad-ebnf-LZEXJU2U-Ua2mHJlY.js +0 -1
- package/viewer/assets/railroad-peg-WCYAUIDC-D39Fr79_.js +0 -1
- package/viewer/assets/stateDiagram-v2-MP3YSRHH-Cd1sSziE.js +0 -1
- package/viewer/assets/swimlanesDiagram-VR7AAH4N-D4gI-oVS.js +0 -8
package/JUDGING.md
CHANGED
|
@@ -152,6 +152,63 @@ must declare at least one `## Criterion: id — Label`. A normalized Judgment ha
|
|
|
152
152
|
an aggregate value and a scores map of CriterionScore objects containing value,
|
|
153
153
|
reason, evidence, and source. The original code output is retained separately.
|
|
154
154
|
|
|
155
|
+
## Criterion categories
|
|
156
|
+
|
|
157
|
+
A category is any non-empty string on a criterion. Categories group results in
|
|
158
|
+
the viewer and can compose a benchmark. They never change judge input,
|
|
159
|
+
fingerprints, or the headline weighting.
|
|
160
|
+
|
|
161
|
+
In `judge.md`, put one optional line directly below a criterion heading:
|
|
162
|
+
|
|
163
|
+
```md
|
|
164
|
+
## Criterion: asked_dialect — Asks for the SQL dialect
|
|
165
|
+
Categories: misalignment, general
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
OpenEval removes that line before hashing the rubric and before the judge reads
|
|
169
|
+
it, so adding or changing categories does not rejudge recorded evidence. A
|
|
170
|
+
`Categories:` line anywhere else is an error.
|
|
171
|
+
|
|
172
|
+
In `judge.ts`, export labels and categories for the scores it returns:
|
|
173
|
+
|
|
174
|
+
```ts
|
|
175
|
+
import type { CodeCriteria, JudgeContext } from "@hona/openeval";
|
|
176
|
+
|
|
177
|
+
export const criteria = {
|
|
178
|
+
correct_answer: { name: "Correct answer", categories: ["general"] },
|
|
179
|
+
} satisfies CodeCriteria;
|
|
180
|
+
|
|
181
|
+
export default ({ response }: JudgeContext) => ({
|
|
182
|
+
scores: { correct_answer: response.text.trim() === "42" },
|
|
183
|
+
});
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
Declare each criterion ID in one judge file only. Names are trimmed and matched
|
|
187
|
+
case-insensitively; the first spelling is displayed. A criterion may have
|
|
188
|
+
several categories, but one is usually clearer. A category score uses the same
|
|
189
|
+
rule as the overall score: criteria are averaged within each eval, then evals
|
|
190
|
+
are weighted equally. Unscored checks keep that category's score a range.
|
|
191
|
+
|
|
192
|
+
Compose a benchmark from categories in `benchmark.ts`. Evals without a matching
|
|
193
|
+
criterion are not run, and scores use only matching criteria:
|
|
194
|
+
|
|
195
|
+
```ts
|
|
196
|
+
export default {
|
|
197
|
+
models: ["opencode/gpt-6-astra#high"],
|
|
198
|
+
categories: ["coding", "verification"],
|
|
199
|
+
} satisfies Benchmark;
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
`openeval run --only-category <name>` limits one invocation to evals with a
|
|
203
|
+
matching criterion, like `--only-eval`, without changing the composition.
|
|
204
|
+
|
|
205
|
+
Prefer published category sets so results are comparable:
|
|
206
|
+
|
|
207
|
+
| Kind | Source | Categories |
|
|
208
|
+
| --- | --- | --- |
|
|
209
|
+
| Capability | [Artificial Analysis Intelligence Index](https://artificialanalysis.ai/methodology/intelligence-benchmarking) | `agents`, `coding`, `general`, `scientific-reasoning`; also `multilingual`, `vision` |
|
|
210
|
+
| Agent failure | [MAST](https://arxiv.org/abs/2503.13657) (Cemri et al., NeurIPS 2025) | `specification`, `misalignment`, `verification` |
|
|
211
|
+
|
|
155
212
|
## Structured tool submissions
|
|
156
213
|
|
|
157
214
|
| Tool | Data |
|
package/README.md
CHANGED
|
@@ -86,6 +86,10 @@ or does not provide a parameterized query.
|
|
|
86
86
|
| Only asks which database | **1** | **0** |
|
|
87
87
|
| Required recording is unavailable | **null** | **null** |
|
|
88
88
|
|
|
89
|
+
Add an optional `Categories: misalignment` line under a heading to compare models
|
|
90
|
+
by category in the viewer's radar chart and heatmap. Categories never trigger
|
|
91
|
+
rejudging.
|
|
92
|
+
|
|
89
93
|
→ [Write good rubrics](https://openev.al/docs/rubrics/) · [Download the SQL starter](https://openev.al/starter.zip)
|
|
90
94
|
|
|
91
95
|
## One vocabulary
|
|
@@ -211,13 +215,21 @@ await runBenchmark("./my-benchmark", {
|
|
|
211
215
|
|
|
212
216
|
## Write evals with an agent
|
|
213
217
|
|
|
218
|
+
Copy the **Agent prompt** at [openev.al](https://openev.al), or use
|
|
219
|
+
[agent-start.md](https://github.com/Hona/openeval/blob/main/agent-start.md). It walks through project selection,
|
|
220
|
+
prerequisites, the writing skill, one eval, models, and an optional first run.
|
|
221
|
+
The prompt starts from [llms.txt](https://openev.al/llms.txt); documentation pages
|
|
222
|
+
also support Markdown fetches and direct `index.md` URLs.
|
|
223
|
+
|
|
214
224
|
Use the public [Eval Writing skill](https://github.com/Hona/openeval/tree/main/.opencode/skills/eval-writing)
|
|
215
225
|
to turn a real failure into an eval, review a rubric, or investigate misleading
|
|
216
226
|
scores. It guides an agent through concrete false-pass/false-failure examples,
|
|
217
227
|
accepted alternatives, evidence requirements, and human-reviewed calibration.
|
|
218
228
|
|
|
219
229
|
Copy the whole `.opencode/skills/eval-writing/` directory, including `references/`,
|
|
220
|
-
into the same path in your project
|
|
230
|
+
into the same path in your project, or extract the
|
|
231
|
+
[project skill ZIP](https://openev.al/eval-writing.zip) into your project root.
|
|
232
|
+
For global use, copy it to
|
|
221
233
|
`~/.config/opencode/skills/eval-writing/`. Then run **`/eval-writing`** in OpenCode.
|
|
222
234
|
|
|
223
235
|
> Use eval-writing to review this task and rubric. Show me the strongest false
|
package/package.json
CHANGED
|
@@ -3,11 +3,46 @@ import type {
|
|
|
3
3
|
BenchmarkRun,
|
|
4
4
|
ModelRef,
|
|
5
5
|
JudgeRunInput,
|
|
6
|
+
ProviderDefinitions,
|
|
6
7
|
} from "../types";
|
|
7
8
|
import { fingerprint } from "../infra/files";
|
|
8
9
|
import { JUDGE_PROTOCOL } from "../judgment";
|
|
9
10
|
import { JUDGE_AGENT } from "../infra/judging/agent";
|
|
10
11
|
|
|
12
|
+
/** The candidate's own provider settings and model override; other entries do not apply to it. */
|
|
13
|
+
function providerScope(
|
|
14
|
+
providers: ProviderDefinitions | undefined,
|
|
15
|
+
model: ModelRef,
|
|
16
|
+
) {
|
|
17
|
+
const [name] = model.split("#"),
|
|
18
|
+
slash = name.indexOf("/"),
|
|
19
|
+
id = name.slice(0, slash),
|
|
20
|
+
modelId = name.slice(slash + 1);
|
|
21
|
+
const { models, ...settings } = providers?.[id] ?? {};
|
|
22
|
+
const override = models?.[modelId];
|
|
23
|
+
return override || Object.keys(settings).length
|
|
24
|
+
? { id, modelId, settings, override }
|
|
25
|
+
: undefined;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** Provider configuration installed in one candidate container. */
|
|
29
|
+
export function candidateProviders(
|
|
30
|
+
providers: ProviderDefinitions | undefined,
|
|
31
|
+
model: ModelRef,
|
|
32
|
+
): ProviderDefinitions | undefined {
|
|
33
|
+
const scope = providerScope(providers, model);
|
|
34
|
+
return scope
|
|
35
|
+
? {
|
|
36
|
+
[scope.id]: {
|
|
37
|
+
...scope.settings,
|
|
38
|
+
...(scope.override
|
|
39
|
+
? { models: { [scope.modelId]: scope.override } }
|
|
40
|
+
: {}),
|
|
41
|
+
},
|
|
42
|
+
}
|
|
43
|
+
: undefined;
|
|
44
|
+
}
|
|
45
|
+
|
|
11
46
|
/** Declared, candidate-visible inputs. Harness implementation hashes are provenance. */
|
|
12
47
|
export function candidateFingerprint(
|
|
13
48
|
definition: BenchmarkDefinition,
|
|
@@ -16,10 +51,7 @@ export function candidateFingerprint(
|
|
|
16
51
|
runtime: BenchmarkRun["runtime"],
|
|
17
52
|
) {
|
|
18
53
|
const item = definition.evals.find((item) => item.id === evalId)!;
|
|
19
|
-
const
|
|
20
|
-
slash = name.indexOf("/");
|
|
21
|
-
const provider = definition.candidate.providers?.[name.slice(0, slash)];
|
|
22
|
-
const { models, ...settings } = provider ?? {};
|
|
54
|
+
const scope = providerScope(definition.candidate.providers, model);
|
|
23
55
|
return fingerprint({
|
|
24
56
|
prompt: item.prompt,
|
|
25
57
|
source: item.sourceHash,
|
|
@@ -28,9 +60,7 @@ export function candidateFingerprint(
|
|
|
28
60
|
candidate: {
|
|
29
61
|
timeoutMs: definition.candidate.timeoutMs,
|
|
30
62
|
websearch: definition.candidate.websearch,
|
|
31
|
-
provider:
|
|
32
|
-
? { ...settings, model: models?.[name.slice(slash + 1)] }
|
|
33
|
-
: undefined,
|
|
63
|
+
provider: scope && { ...scope.settings, model: scope.override },
|
|
34
64
|
},
|
|
35
65
|
container: {
|
|
36
66
|
engine: definition.container.engine,
|
|
@@ -10,12 +10,17 @@ import type {
|
|
|
10
10
|
} from "../types";
|
|
11
11
|
import { CANDIDATE_TIMEOUT_MS } from "../types";
|
|
12
12
|
import { rubricCriteria } from "../judgment";
|
|
13
|
+
import {
|
|
14
|
+
categoryKey,
|
|
15
|
+
categoryList,
|
|
16
|
+
rubricCategories,
|
|
17
|
+
} from "../criterion-categories";
|
|
13
18
|
import { compileCodeJudge } from "../infra/judging/code-source";
|
|
19
|
+
import { readCodeCriteria } from "../infra/judging/code-criteria";
|
|
14
20
|
import { RUNTIME_IMAGE } from "../infra/opencode/version";
|
|
15
21
|
import { monitorPolicy } from "./monitor-policy";
|
|
16
22
|
import {
|
|
17
23
|
fingerprint,
|
|
18
|
-
hash,
|
|
19
24
|
relativePath,
|
|
20
25
|
treeHash,
|
|
21
26
|
contained,
|
|
@@ -35,10 +40,13 @@ export const modelRef = (value: unknown): ModelRef => {
|
|
|
35
40
|
throw new Error(`Invalid model reference: ${String(value)}`);
|
|
36
41
|
return value as ModelRef;
|
|
37
42
|
};
|
|
43
|
+
/** Bun caches modules by path, ignoring URL queries; clear the entry so edits load. */
|
|
44
|
+
async function freshModule(path: string): Promise<Record<string, unknown>> {
|
|
45
|
+
delete require.cache[path];
|
|
46
|
+
return import(pathToFileURL(path).href);
|
|
47
|
+
}
|
|
38
48
|
async function declaration<T>(path: string): Promise<T> {
|
|
39
|
-
|
|
40
|
-
url.searchParams.set("version", hash(await Bun.file(path).bytes()));
|
|
41
|
-
return (await import(url.href)).default;
|
|
49
|
+
return (await freshModule(path)).default as T;
|
|
42
50
|
}
|
|
43
51
|
async function loadEval(directory: string): Promise<EvalDefinition> {
|
|
44
52
|
const id = basename(directory),
|
|
@@ -168,10 +176,23 @@ async function loadEval(directory: string): Promise<EvalDefinition> {
|
|
|
168
176
|
throw new Error(`${id}: preparation requires argv`);
|
|
169
177
|
}
|
|
170
178
|
const promptText = await prompt.text(),
|
|
171
|
-
|
|
179
|
+
rubric = rubricCategories(hasMarkdown ? await judge.text() : ""),
|
|
180
|
+
judgeText = rubric.rubric;
|
|
172
181
|
if (!promptText.trim() || (hasMarkdown && !judgeText.trim()))
|
|
173
182
|
throw new Error(`${id}: prompt and judge must not be empty`);
|
|
174
183
|
const code = hasCode ? await compileCodeJudge(codeFile) : undefined;
|
|
184
|
+
const criteria = hasMarkdown ? rubricCriteria(judgeText) : [];
|
|
185
|
+
const declared = hasCode ? await codeCriteria(id, codeFile) : [];
|
|
186
|
+
if (declared.some((item) => criteria.some(({ id }) => id === item.id)))
|
|
187
|
+
throw new Error(
|
|
188
|
+
`${id}: declare each criterion in either judge.md or judge.ts, not both`,
|
|
189
|
+
);
|
|
190
|
+
const categories = Object.fromEntries(
|
|
191
|
+
[
|
|
192
|
+
...Object.entries(rubric.categories),
|
|
193
|
+
...declared.map((item) => [item.id, item.categories] as const),
|
|
194
|
+
].filter(([, names]) => names.length),
|
|
195
|
+
);
|
|
175
196
|
return {
|
|
176
197
|
id,
|
|
177
198
|
directory,
|
|
@@ -183,12 +204,49 @@ async function loadEval(directory: string): Promise<EvalDefinition> {
|
|
|
183
204
|
source,
|
|
184
205
|
}),
|
|
185
206
|
judgeHash: fingerprint({ rubric: judgeText, code: code?.hash }),
|
|
186
|
-
criteria
|
|
207
|
+
criteria,
|
|
208
|
+
...(declared.length
|
|
209
|
+
? { codeCriteria: declared.map(({ id, name }) => ({ id, name })) }
|
|
210
|
+
: {}),
|
|
211
|
+
...(Object.keys(categories).length ? { categories } : {}),
|
|
187
212
|
...(code ? { code } : {}),
|
|
188
213
|
name: /^# (.+)$/m.exec(judgeText)?.[1] ?? id,
|
|
189
214
|
};
|
|
190
215
|
}
|
|
191
216
|
|
|
217
|
+
/** Reads judge.ts's optional `criteria` export as data; planning never executes judge code. */
|
|
218
|
+
async function codeCriteria(id: string, path: string) {
|
|
219
|
+
const declared = await readCodeCriteria(path);
|
|
220
|
+
if (declared === undefined) return [];
|
|
221
|
+
if (!declared || typeof declared !== "object" || Array.isArray(declared))
|
|
222
|
+
throw new Error(`${id}/judge.ts criteria must be an object`);
|
|
223
|
+
return Object.entries(declared).map(([criterion, value]) => {
|
|
224
|
+
const where = `${id}/judge.ts criterion ${criterion}`;
|
|
225
|
+
if (!/^[a-z][a-z0-9_]*$/.test(criterion))
|
|
226
|
+
throw new Error(`${where}: use a lowercase snake_case ID`);
|
|
227
|
+
if (
|
|
228
|
+
!value ||
|
|
229
|
+
typeof value !== "object" ||
|
|
230
|
+
Array.isArray(value) ||
|
|
231
|
+
Object.keys(value).some((key) => !["name", "categories"].includes(key))
|
|
232
|
+
)
|
|
233
|
+
throw new Error(`${where}: declare only name and categories`);
|
|
234
|
+
const { name, categories = [] } = value as {
|
|
235
|
+
name?: unknown;
|
|
236
|
+
categories?: unknown;
|
|
237
|
+
};
|
|
238
|
+
if (name !== undefined && (typeof name !== "string" || !name.trim()))
|
|
239
|
+
throw new Error(`${where}: name must be a non-empty string`);
|
|
240
|
+
if (!Array.isArray(categories))
|
|
241
|
+
throw new Error(`${where}: categories must be an array`);
|
|
242
|
+
return {
|
|
243
|
+
id: criterion,
|
|
244
|
+
name: (name as string | undefined)?.trim() ?? criterion.replaceAll("_", " "),
|
|
245
|
+
categories: categoryList(categories, where),
|
|
246
|
+
};
|
|
247
|
+
});
|
|
248
|
+
}
|
|
249
|
+
|
|
192
250
|
export async function loadBenchmark(
|
|
193
251
|
path: string,
|
|
194
252
|
): Promise<BenchmarkDefinition> {
|
|
@@ -214,10 +272,19 @@ export async function loadBenchmark(
|
|
|
214
272
|
"concurrency",
|
|
215
273
|
"candidate",
|
|
216
274
|
"container",
|
|
275
|
+
"categories",
|
|
217
276
|
].includes(key),
|
|
218
277
|
)
|
|
219
278
|
)
|
|
220
279
|
throw new Error("benchmark.ts contains unsupported settings");
|
|
280
|
+
if (
|
|
281
|
+
definition.categories !== undefined &&
|
|
282
|
+
(!Array.isArray(definition.categories) || !definition.categories.length)
|
|
283
|
+
)
|
|
284
|
+
throw new Error("benchmark.ts categories must be a non-empty array");
|
|
285
|
+
const categories = definition.categories
|
|
286
|
+
? categoryList(definition.categories, "benchmark.ts")
|
|
287
|
+
: undefined;
|
|
221
288
|
const models = definition.models.map(modelRef);
|
|
222
289
|
if (
|
|
223
290
|
definition.judge !== undefined &&
|
|
@@ -245,9 +312,21 @@ export async function loadBenchmark(
|
|
|
245
312
|
.filter((entry) => entry.isDirectory() && !entry.name.startsWith("."))
|
|
246
313
|
.sort((a, b) => a.name.localeCompare(b.name));
|
|
247
314
|
if (!directories.length) throw new Error("Benchmark has no eval folders");
|
|
248
|
-
const
|
|
315
|
+
const declared = await Promise.all(
|
|
249
316
|
directories.map((entry) => loadEval(resolve(evalRoot, entry.name))),
|
|
250
317
|
);
|
|
318
|
+
const keys = categories?.map(categoryKey);
|
|
319
|
+
const evals = keys
|
|
320
|
+
? declared.filter((item) =>
|
|
321
|
+
Object.values(item.categories ?? {}).some((names) =>
|
|
322
|
+
names.some((name) => keys.includes(categoryKey(name))),
|
|
323
|
+
),
|
|
324
|
+
)
|
|
325
|
+
: declared;
|
|
326
|
+
if (!evals.length)
|
|
327
|
+
throw new Error(
|
|
328
|
+
`No criteria match the benchmark categories: ${categories!.join(", ")}`,
|
|
329
|
+
);
|
|
251
330
|
if (evals.some((item) => item.judge) && !definition.judge?.model)
|
|
252
331
|
throw new Error("A benchmark containing judge.md requires judge.model");
|
|
253
332
|
const concurrency = positive(definition.concurrency, 10, "Concurrency");
|
|
@@ -295,5 +374,6 @@ export async function loadBenchmark(
|
|
|
295
374
|
cpus: positive(definition.container?.cpus, 2, "CPU count"),
|
|
296
375
|
memoryMiB: positive(definition.container?.memoryMiB, 4096, "Memory"),
|
|
297
376
|
},
|
|
377
|
+
...(categories ? { categories } : {}),
|
|
298
378
|
};
|
|
299
379
|
}
|
package/src/app/merge-runs.ts
CHANGED
package/src/app/read-results.ts
CHANGED
|
@@ -14,7 +14,7 @@ import type {
|
|
|
14
14
|
StageStatus,
|
|
15
15
|
JudgeAudit,
|
|
16
16
|
} from "../view";
|
|
17
|
-
import { sumCosts } from "../view";
|
|
17
|
+
import { categoryCatalog, sumCosts } from "../view";
|
|
18
18
|
import { benchmarkScores } from "./scores";
|
|
19
19
|
import { executionRuntime } from "./execution-runtime";
|
|
20
20
|
import {
|
|
@@ -315,6 +315,8 @@ export class ResultReader {
|
|
|
315
315
|
evalNames: Object.fromEntries(
|
|
316
316
|
evals.map((item) => [item.id, item.name]),
|
|
317
317
|
),
|
|
318
|
+
modelNames: run.modelNames ?? {},
|
|
319
|
+
categories: categoryCatalog(scores[0]?.components ?? []),
|
|
318
320
|
cost: evalId ? evalCosts[evalId] : sumCosts(Object.values(evalCosts)),
|
|
319
321
|
evalCosts,
|
|
320
322
|
runtime,
|
package/src/app/run-benchmark.ts
CHANGED
|
@@ -10,6 +10,7 @@ import type {
|
|
|
10
10
|
} from "../types";
|
|
11
11
|
import { Results } from "../infra/sqlite";
|
|
12
12
|
import { runtimeFingerprint } from "../infra/containers/oci";
|
|
13
|
+
import { readModelNames } from "../infra/containers/catalog";
|
|
13
14
|
import { prepareWorkspace } from "../infra/containers/workspace";
|
|
14
15
|
import { fingerprint, errorMessage } from "../infra/files";
|
|
15
16
|
import { loadBenchmark, modelRef } from "./load-benchmark";
|
|
@@ -22,6 +23,11 @@ import { canJudgeEval } from "./eval-state";
|
|
|
22
23
|
import { isScored } from "../judgment";
|
|
23
24
|
import { benchmarkScores } from "./scores";
|
|
24
25
|
import { CostBudget, estimateWork } from "./cost-plan";
|
|
26
|
+
import {
|
|
27
|
+
categoryKey,
|
|
28
|
+
categoryList,
|
|
29
|
+
inCategories,
|
|
30
|
+
} from "../criterion-categories";
|
|
25
31
|
|
|
26
32
|
export type RunBenchmarkOptions = {
|
|
27
33
|
directory?: string;
|
|
@@ -31,6 +37,8 @@ export type RunBenchmarkOptions = {
|
|
|
31
37
|
/** Execute work only for these models while retaining the full aggregate. */
|
|
32
38
|
onlyModels?: readonly ModelRef[];
|
|
33
39
|
onlyEvals?: readonly string[];
|
|
40
|
+
/** Execute only evals with a criterion in any of these categories. */
|
|
41
|
+
onlyCategories?: readonly string[];
|
|
34
42
|
onlyRepetitions?: readonly number[];
|
|
35
43
|
/** Stop admitting work when reported spend plus reservations reaches this amount. */
|
|
36
44
|
maxCostUSD?: number;
|
|
@@ -38,6 +46,29 @@ export type RunBenchmarkOptions = {
|
|
|
38
46
|
onEvent?: ExecutionObserver;
|
|
39
47
|
};
|
|
40
48
|
|
|
49
|
+
/** Intersect --only-eval with evals that have a criterion in any --only-category. */
|
|
50
|
+
function scopedEvals(
|
|
51
|
+
definition: BenchmarkDefinition,
|
|
52
|
+
options: Pick<RunBenchmarkOptions, "onlyEvals" | "onlyCategories">,
|
|
53
|
+
) {
|
|
54
|
+
if (!options.onlyCategories) return options.onlyEvals;
|
|
55
|
+
const keys = categoryList(options.onlyCategories, "--only-category").map(
|
|
56
|
+
categoryKey,
|
|
57
|
+
);
|
|
58
|
+
if (!keys.length) throw new Error("Select at least one category");
|
|
59
|
+
const matching = definition.evals
|
|
60
|
+
.filter((item) =>
|
|
61
|
+
Object.values(item.categories ?? {}).some((names) =>
|
|
62
|
+
inCategories({ categories: names }, keys),
|
|
63
|
+
),
|
|
64
|
+
)
|
|
65
|
+
.map((item) => item.id)
|
|
66
|
+
.filter((id) => !options.onlyEvals || options.onlyEvals.includes(id));
|
|
67
|
+
if (!matching.length)
|
|
68
|
+
throw new Error("No selected evals have criteria in these categories");
|
|
69
|
+
return matching;
|
|
70
|
+
}
|
|
71
|
+
|
|
41
72
|
export async function currentBenchmarkRun(directory: string) {
|
|
42
73
|
const root = resolve(directory, "results");
|
|
43
74
|
const dirs = (await readdir(root, { withFileTypes: true }).catch(() => []))
|
|
@@ -140,6 +171,7 @@ export async function runBenchmark(
|
|
|
140
171
|
))
|
|
141
172
|
)
|
|
142
173
|
throw new Error("Select configured eval IDs with --only-eval");
|
|
174
|
+
const onlyEvals = scopedEvals(definition, options);
|
|
143
175
|
new CostBudget(options.maxCostUSD, () => 0);
|
|
144
176
|
const selected = options.onlyModels?.map(modelRef);
|
|
145
177
|
if (selected && !selected.length)
|
|
@@ -191,11 +223,11 @@ export async function runBenchmark(
|
|
|
191
223
|
results.slots().map((slot) => [slot.id, slot]),
|
|
192
224
|
);
|
|
193
225
|
let plan = planBenchmark(definition, runtime, results);
|
|
194
|
-
if (selected ||
|
|
226
|
+
if (selected || onlyEvals || options.onlyRepetitions) {
|
|
195
227
|
const retained = new Map(results.slots().map((slot) => [slot.id, slot]));
|
|
196
228
|
plan = plan.map((item) =>
|
|
197
229
|
((!selected || selected.includes(item.slot.model)) &&
|
|
198
|
-
(!
|
|
230
|
+
(!onlyEvals || onlyEvals.includes(item.slot.evalId)) &&
|
|
199
231
|
(!options.onlyRepetitions ||
|
|
200
232
|
options.onlyRepetitions.includes(item.slot.repetition))) ||
|
|
201
233
|
item.action === "reuse"
|
|
@@ -215,11 +247,17 @@ export async function runBenchmark(
|
|
|
215
247
|
(item) => item.action === "candidate" || item.action === "judge",
|
|
216
248
|
);
|
|
217
249
|
const previous = results.benchmark;
|
|
250
|
+
const modelNames = {
|
|
251
|
+
...previous?.modelNames,
|
|
252
|
+
// Names are display metadata; an unavailable catalog keeps the recorded names.
|
|
253
|
+
...(await readModelNames(definition, runtime.imageId).catch(() => ({}))),
|
|
254
|
+
};
|
|
218
255
|
if (
|
|
219
256
|
!work.length &&
|
|
220
257
|
previous &&
|
|
221
258
|
fingerprint(previous.definition) === fingerprint(definition) &&
|
|
222
|
-
fingerprint(previous.runtime) === fingerprint(runtime)
|
|
259
|
+
fingerprint(previous.runtime) === fingerprint(runtime) &&
|
|
260
|
+
fingerprint(previous.modelNames ?? {}) === fingerprint(modelNames)
|
|
223
261
|
)
|
|
224
262
|
return { directory, plan, estimate, benchmark: previous };
|
|
225
263
|
const benchmark: BenchmarkRun = {
|
|
@@ -232,11 +270,12 @@ export async function runBenchmark(
|
|
|
232
270
|
scheduledSlotIds: work.map((item) => item.slot.id),
|
|
233
271
|
definition,
|
|
234
272
|
runtime,
|
|
273
|
+
modelNames,
|
|
235
274
|
sources: previous?.sources,
|
|
236
275
|
execution: {
|
|
237
276
|
startedAt: now,
|
|
238
277
|
onlyModels: selected,
|
|
239
|
-
onlyEvals
|
|
278
|
+
onlyEvals,
|
|
240
279
|
onlyRepetitions: options.onlyRepetitions,
|
|
241
280
|
budgetUSD: options.maxCostUSD,
|
|
242
281
|
estimatedUSD: estimate.estimatedUSD,
|
package/src/app/run-eval.ts
CHANGED
|
@@ -5,6 +5,7 @@ import { executeCandidate } from "../infra/containers/candidate";
|
|
|
5
5
|
import type { EvidenceFeed } from "../evidence";
|
|
6
6
|
import { CandidateEvidence } from "../infra/evidence";
|
|
7
7
|
import { measureRecording } from "../infra/recording/metrics";
|
|
8
|
+
import { candidateProviders } from "./input-fingerprints";
|
|
8
9
|
|
|
9
10
|
/** The normal candidate use case: record its inputs, execute, finalize its evidence. */
|
|
10
11
|
export async function runEval(
|
|
@@ -42,7 +43,10 @@ export async function runEval(
|
|
|
42
43
|
websearch: context.definition.candidate.websearch,
|
|
43
44
|
container: context.definition.container,
|
|
44
45
|
imageId: context.runtime.imageId,
|
|
45
|
-
providers:
|
|
46
|
+
providers: candidateProviders(
|
|
47
|
+
context.definition.candidate.providers,
|
|
48
|
+
slot.model,
|
|
49
|
+
),
|
|
46
50
|
},
|
|
47
51
|
directory,
|
|
48
52
|
(event) => {
|
package/src/app/scores.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import type { BenchmarkDefinition, JudgeRun, Slot } from "../types";
|
|
2
2
|
import { modelScore } from "../view";
|
|
3
3
|
import { isScored } from "../judgment";
|
|
4
|
+
import { categoryKey, inCategories } from "../criterion-categories";
|
|
4
5
|
|
|
5
6
|
/** Scores are derived from a single selection snapshot; every eval has equal weight. */
|
|
6
7
|
export function benchmarkScores(
|
|
@@ -13,10 +14,14 @@ export function benchmarkScores(
|
|
|
13
14
|
const evals = definition.evals.filter(
|
|
14
15
|
(item) => !evalId || item.id === evalId,
|
|
15
16
|
);
|
|
17
|
+
const keys = definition.categories?.map(categoryKey);
|
|
16
18
|
const definitions = new Map(
|
|
17
19
|
evals.map((item) => {
|
|
18
20
|
const criteria = new Map(
|
|
19
|
-
item.criteria.map((criterion) => [
|
|
21
|
+
[...item.criteria, ...(item.codeCriteria ?? [])].map((criterion) => [
|
|
22
|
+
criterion.id,
|
|
23
|
+
criterion,
|
|
24
|
+
]),
|
|
20
25
|
);
|
|
21
26
|
for (const slot of slots.filter(
|
|
22
27
|
(slot) => slot.active && slot.evalId === item.id,
|
|
@@ -28,7 +33,16 @@ export function benchmarkScores(
|
|
|
28
33
|
if (!criteria.has(id))
|
|
29
34
|
criteria.set(id, { id, name: id.replaceAll("_", " ") });
|
|
30
35
|
}
|
|
31
|
-
|
|
36
|
+
const categorized = [...criteria.values()].map((criterion) => ({
|
|
37
|
+
...criterion,
|
|
38
|
+
categories: item.categories?.[criterion.id] ?? [],
|
|
39
|
+
}));
|
|
40
|
+
return [
|
|
41
|
+
item.id,
|
|
42
|
+
keys
|
|
43
|
+
? categorized.filter((criterion) => inCategories(criterion, keys))
|
|
44
|
+
: categorized,
|
|
45
|
+
];
|
|
32
46
|
}),
|
|
33
47
|
);
|
|
34
48
|
return definition.models.map((model) =>
|
|
@@ -57,6 +71,9 @@ export function benchmarkScores(
|
|
|
57
71
|
eval: item.id,
|
|
58
72
|
criterion: criterion.id,
|
|
59
73
|
name: criterion.name,
|
|
74
|
+
...(criterion.categories.length
|
|
75
|
+
? { categories: criterion.categories }
|
|
76
|
+
: {}),
|
|
60
77
|
value:
|
|
61
78
|
values.length === definition.repetitions
|
|
62
79
|
? scoredSum / definition.repetitions
|
package/src/cli.ts
CHANGED
|
@@ -51,6 +51,7 @@ Options:
|
|
|
51
51
|
--model <provider/id> Model to add or remove (repeatable)
|
|
52
52
|
--only-model <ref> Execute only this model (repeatable)
|
|
53
53
|
--only-eval <id> Execute only this eval (repeatable)
|
|
54
|
+
--only-category <name> Execute only evals with criteria in this category (repeatable)
|
|
54
55
|
--only-repetition <n> Execute only this repetition (repeatable)
|
|
55
56
|
--max-cost <usd> Scheduling budget for this invocation
|
|
56
57
|
--final-only Judge only after candidates finish
|
|
@@ -84,6 +85,11 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
|
84
85
|
arg === "--only-eval" ? [args[index + 1]] : [],
|
|
85
86
|
)
|
|
86
87
|
: undefined,
|
|
88
|
+
onlyCategories: args.includes("--only-category")
|
|
89
|
+
? args.flatMap((arg, index) =>
|
|
90
|
+
arg === "--only-category" ? [args[index + 1]] : [],
|
|
91
|
+
)
|
|
92
|
+
: undefined,
|
|
87
93
|
maxCostUSD: args.includes("--max-cost")
|
|
88
94
|
? Number(option("--max-cost"))
|
|
89
95
|
: undefined,
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
/** Categories are reporting metadata: any non-empty string, matched case-insensitively. */
|
|
2
|
+
export const categoryKey = (name: string) => name.trim().toLowerCase();
|
|
3
|
+
|
|
4
|
+
/** Trim names and drop case-insensitive duplicates, keeping the first spelling. */
|
|
5
|
+
export function categoryList(values: readonly unknown[], where: string) {
|
|
6
|
+
const names: string[] = [],
|
|
7
|
+
keys = new Set<string>();
|
|
8
|
+
for (const value of values) {
|
|
9
|
+
if (typeof value !== "string" || !value.trim())
|
|
10
|
+
throw new Error(`${where}: categories must be non-empty strings`);
|
|
11
|
+
const name = value.trim();
|
|
12
|
+
if (keys.has(categoryKey(name))) continue;
|
|
13
|
+
keys.add(categoryKey(name));
|
|
14
|
+
names.push(name);
|
|
15
|
+
}
|
|
16
|
+
return names;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
const CATEGORY_LINE = /^Categor(?:y|ies):[ \t]*(.*?)\r?$/i;
|
|
20
|
+
const HEADING = /^## Criterion: ([a-z][a-z0-9_]*)[ \t]*[—–-]/;
|
|
21
|
+
|
|
22
|
+
/** Reads the optional `Categories:` line directly below each `## Criterion:` heading.
|
|
23
|
+
* The returned rubric omits those lines, so adding or changing categories never
|
|
24
|
+
* changes judge input or judge fingerprints.
|
|
25
|
+
*/
|
|
26
|
+
export function rubricCategories(rubric: string) {
|
|
27
|
+
const lines = rubric.split("\n"),
|
|
28
|
+
kept: string[] = [],
|
|
29
|
+
categories: Record<string, string[]> = {};
|
|
30
|
+
for (let index = 0; index < lines.length; index++) {
|
|
31
|
+
const line = lines[index]!;
|
|
32
|
+
if (CATEGORY_LINE.test(line))
|
|
33
|
+
throw new Error(
|
|
34
|
+
"Put a Categories: line directly below its ## Criterion: heading",
|
|
35
|
+
);
|
|
36
|
+
kept.push(line);
|
|
37
|
+
const heading = HEADING.exec(line),
|
|
38
|
+
next = heading ? CATEGORY_LINE.exec(lines[index + 1] ?? "") : null;
|
|
39
|
+
if (!heading || !next) continue;
|
|
40
|
+
categories[heading[1]!] = categoryList(
|
|
41
|
+
next[1]!.split(","),
|
|
42
|
+
`Criterion ${heading[1]}`,
|
|
43
|
+
);
|
|
44
|
+
index++;
|
|
45
|
+
}
|
|
46
|
+
return { rubric: kept.join("\n"), categories };
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/** Whether a scored criterion belongs to any of the selected category keys. */
|
|
50
|
+
export const inCategories = (
|
|
51
|
+
part: { categories?: readonly string[] },
|
|
52
|
+
keys: readonly string[],
|
|
53
|
+
) => !!part.categories?.some((name) => keys.includes(categoryKey(name)));
|
package/src/index.ts
CHANGED
|
@@ -9,6 +9,7 @@ export type {
|
|
|
9
9
|
JudgeRun,
|
|
10
10
|
Judgment,
|
|
11
11
|
CriterionDefinition,
|
|
12
|
+
CodeCriteria,
|
|
12
13
|
CriterionScore,
|
|
13
14
|
EvidenceCitation,
|
|
14
15
|
ToolCall,
|
|
@@ -31,6 +32,10 @@ export type {
|
|
|
31
32
|
} from "./evidence";
|
|
32
33
|
export { CANDIDATE_TIMEOUT_MS } from "./types";
|
|
33
34
|
export { rubricCriteria, criterionMean, isScored } from "./judgment";
|
|
35
|
+
export {
|
|
36
|
+
categoryKey,
|
|
37
|
+
rubricCategories,
|
|
38
|
+
} from "./criterion-categories";
|
|
34
39
|
export type {
|
|
35
40
|
JudgeContext,
|
|
36
41
|
JudgeFunction,
|