@hona/openeval 0.2.2 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (211) hide show
  1. package/JUDGING.md +134 -23
  2. package/README.md +67 -9
  3. package/package.json +3 -2
  4. package/src/app/cost-plan.ts +20 -16
  5. package/src/app/eval-state.ts +2 -1
  6. package/src/app/grade-recording.ts +72 -0
  7. package/src/app/input-fingerprints.ts +22 -8
  8. package/src/app/judge-evidence.ts +26 -11
  9. package/src/app/judge-run.ts +36 -19
  10. package/src/app/load-benchmark.ts +46 -15
  11. package/src/app/read-results.ts +35 -2
  12. package/src/app/read-run.ts +23 -2
  13. package/src/app/rejudge.ts +2 -1
  14. package/src/app/run-benchmark.ts +4 -0
  15. package/src/app/run-eval-pipeline.ts +2 -1
  16. package/src/app/run-eval.ts +13 -0
  17. package/src/app/scores.ts +76 -34
  18. package/src/app/serve-results.ts +2 -0
  19. package/src/evidence.ts +4 -1
  20. package/src/index.ts +15 -3
  21. package/src/infra/evidence/index.ts +39 -2
  22. package/src/infra/evidence/tool-calls.ts +5 -2
  23. package/src/infra/judging/agent.ts +3 -2
  24. package/src/infra/judging/code-result.ts +102 -0
  25. package/src/infra/judging/code-source.ts +96 -0
  26. package/src/infra/judging/code-worker.ts +26 -0
  27. package/src/infra/judging/code.ts +97 -0
  28. package/src/infra/judging/contract.ts +82 -64
  29. package/src/infra/judging/evidence-tool.ts +3 -1
  30. package/src/infra/judging/index.ts +14 -0
  31. package/src/infra/judging/judge-agent.md +24 -18
  32. package/src/infra/judging/observer.ts +8 -7
  33. package/src/infra/judging/submission.ts +8 -8
  34. package/src/infra/judging/tools.ts +9 -9
  35. package/src/infra/opencode/host.ts +1 -1
  36. package/src/infra/opencode/read-recording.ts +6 -1
  37. package/src/infra/opencode/version.ts +1 -0
  38. package/src/infra/recording/index.ts +244 -0
  39. package/src/infra/recording/metrics.ts +193 -0
  40. package/src/infra/sqlite/index.ts +29 -19
  41. package/src/judge-context.ts +141 -0
  42. package/src/judgment.ts +26 -17
  43. package/src/types.ts +36 -17
  44. package/src/view.ts +40 -14
  45. package/viewer/assets/{abnfDiagram-VCTEODGH-B1kcYgQG.js → abnfDiagram-VCTEODGH-rgmalTBh.js} +1 -1
  46. package/viewer/assets/{angular-html-DrYCUiZv.js → angular-html-CcaajvLj.js} +1 -1
  47. package/viewer/assets/{angular-ts-BgjwQn-Z.js → angular-ts-COx6AtG2.js} +1 -1
  48. package/viewer/assets/{apl-DlHqe8o4.js → apl-lBEnC_vz.js} +1 -1
  49. package/viewer/assets/{arc-C4nf7ZoY.js → arc-BQ0cQOwK.js} +1 -1
  50. package/viewer/assets/architecture-7GRP2DOG-DcUKSnYa.js +1 -0
  51. package/viewer/assets/{architectureDiagram-5GKGNRK7-BIWoe0fO.js → architectureDiagram-5GKGNRK7-_XIJP-LV.js} +1 -1
  52. package/viewer/assets/{astro-BrlHnIh9.js → astro-Ca8FrXna.js} +1 -1
  53. package/viewer/assets/{blade-DOcerOiW.js → blade-B07ZVjUX.js} +1 -1
  54. package/viewer/assets/{blockDiagram-I7D4REHJ-C3x13a9m.js → blockDiagram-I7D4REHJ-BdcDRYB-.js} +1 -1
  55. package/viewer/assets/{c-BMbThxEo.js → c-4MsqXzJK.js} +1 -1
  56. package/viewer/assets/{c4Diagram-7LVT6UL2-BeWktchx.js → c4Diagram-7LVT6UL2-E3nvwqBd.js} +1 -1
  57. package/viewer/assets/channel-CViXUg9J.js +1 -0
  58. package/viewer/assets/{chapel-k2cjkSmc.js → chapel-BlwR_8zu.js} +1 -1
  59. package/viewer/assets/{chunk-4HAMMTFA-o6YtJsqE.js → chunk-4HAMMTFA-DN_vtpqx.js} +1 -1
  60. package/viewer/assets/{chunk-75Z2AOVW-KohQa9Nw.js → chunk-75Z2AOVW-DWMGmicp.js} +1 -1
  61. package/viewer/assets/{chunk-DU6HZSFF-C88wxc6m.js → chunk-DU6HZSFF-D2RGHn75.js} +1 -1
  62. package/viewer/assets/{chunk-F27PBJKO-mO4AP2as.js → chunk-F27PBJKO-wfQZ5Ilx.js} +1 -1
  63. package/viewer/assets/{chunk-GMAD6QVW-DzQX2Cld.js → chunk-GMAD6QVW-DrlfUg7k.js} +1 -1
  64. package/viewer/assets/{chunk-GVQU2GXP-CS78RfJD.js → chunk-GVQU2GXP-BHoRLKIU.js} +1 -1
  65. package/viewer/assets/{chunk-IMKFNOWR-DxS42GUo.js → chunk-IMKFNOWR-mxLyQhYU.js} +1 -1
  66. package/viewer/assets/{chunk-L3NEJ4N5-DGo-uYLy.js → chunk-L3NEJ4N5-IpGl-pvp.js} +1 -1
  67. package/viewer/assets/{chunk-OSK3NFVY-BOcXQG16.js → chunk-OSK3NFVY-t0yN50X6.js} +1 -1
  68. package/viewer/assets/{chunk-P2QGCYS3-B7GnZdIQ.js → chunk-P2QGCYS3-BPBYYOcK.js} +1 -1
  69. package/viewer/assets/{chunk-POPQ4Y6H-DgBcYHog.js → chunk-POPQ4Y6H-DIOcv86B.js} +1 -1
  70. package/viewer/assets/{chunk-PWAF6VOD-DpEq6qHA.js → chunk-PWAF6VOD-CqVB-mzq.js} +1 -1
  71. package/viewer/assets/{chunk-SHT3W25Y-CrOGxKM2.js → chunk-SHT3W25Y-DwOEgFIr.js} +1 -1
  72. package/viewer/assets/{chunk-SVP7TREG-4HY7Fljj.js → chunk-SVP7TREG-Dfg1jHc3.js} +1 -1
  73. package/viewer/assets/{chunk-TICWLB2K-B1Rl31EC.js → chunk-TICWLB2K-DE6tQNRu.js} +1 -1
  74. package/viewer/assets/{chunk-XXDRQBXY-CK8giEW-.js → chunk-XXDRQBXY-Bu8dW06M.js} +1 -1
  75. package/viewer/assets/classDiagram-ZZMXUADV-O1Zm0_p8.js +1 -0
  76. package/viewer/assets/classDiagram-v2-VYDZK3BY-O1Zm0_p8.js +1 -0
  77. package/viewer/assets/{cobol-CupUwvw9.js → cobol-BS_DrT-b.js} +1 -1
  78. package/viewer/assets/{coffee-C17pkozF.js → coffee-Clk97ecB.js} +1 -1
  79. package/viewer/assets/{cose-bilkent-JH36ORCC-V1OmrmCv.js → cose-bilkent-JH36ORCC-B3xA8n2e.js} +1 -1
  80. package/viewer/assets/{cpp-BJwVQVXg.js → cpp-aunMdKLY.js} +1 -1
  81. package/viewer/assets/{crystal-BF-oCN-L.js → crystal-B1oihgG5.js} +1 -1
  82. package/viewer/assets/{css-BMgkVI3c.js → css-ClOubSVo.js} +1 -1
  83. package/viewer/assets/{cynefin-OW5HDTMX-BEjY_gsc.js → cynefin-OW5HDTMX-DdEeq9jN.js} +1 -1
  84. package/viewer/assets/{cynefinDiagram-5FMLGOSQ-bfSYgr6z.js → cynefinDiagram-5FMLGOSQ-BwHCV_5V.js} +1 -1
  85. package/viewer/assets/{dagre-GXQ25YYZ-VtwTgGfa.js → dagre-GXQ25YYZ-BkuWRmq-.js} +1 -1
  86. package/viewer/assets/{diagram-S7CK7UJ4-CHaAAyW7.js → diagram-S7CK7UJ4-C5WGyCny.js} +1 -1
  87. package/viewer/assets/{diagram-UQ7AKVKN-CjL65-As.js → diagram-UQ7AKVKN-mdununKH.js} +1 -1
  88. package/viewer/assets/{diagram-VSXAHHWV-Cnxrk1TJ.js → diagram-VSXAHHWV-CpuvU9Ij.js} +1 -1
  89. package/viewer/assets/{diagram-VX7I27RA-D78jPXr7.js → diagram-VX7I27RA-CcLmTPwv.js} +1 -1
  90. package/viewer/assets/{diagram-Z3DM3KII-DKjnOaF2.js → diagram-Z3DM3KII-DWCnKyKY.js} +1 -1
  91. package/viewer/assets/{dist-BTYPW0dz.js → dist-1DE7d697.js} +1 -1
  92. package/viewer/assets/{ebnfDiagram-PWID7BFC-7XiXh5Lx.js → ebnfDiagram-PWID7BFC-9M_-_M-B.js} +1 -1
  93. package/viewer/assets/{edge-B34N_HjN.js → edge-NsPpWeFX.js} +1 -1
  94. package/viewer/assets/{elixir-CQzWMyKD.js → elixir-CXZ4zQTn.js} +1 -1
  95. package/viewer/assets/{elm-BNJiG-wg.js → elm-DVXugwZs.js} +1 -1
  96. package/viewer/assets/{erDiagram-RLTQ6QDP-YPFEgZxU.js → erDiagram-RLTQ6QDP-B2j6iCkZ.js} +1 -1
  97. package/viewer/assets/{erb-D5Rm3UNa.js → erb-h8ybsnSQ.js} +1 -1
  98. package/viewer/assets/eventmodeling-NTZA5JFV-CpMf6avu.js +1 -0
  99. package/viewer/assets/flowDiagram-HODETNUW-DIKEcF1k.js +1 -0
  100. package/viewer/assets/{ganttDiagram-EL5Y4UJY-ergSGtTP.js → ganttDiagram-EL5Y4UJY-CR2QLJ-I.js} +1 -1
  101. package/viewer/assets/{git-rebase-BpIMZIRG.js → git-rebase-CT9f4kn8.js} +1 -1
  102. package/viewer/assets/{gitGraph-4MIJSDKK-CZt4REVj.js → gitGraph-4MIJSDKK-DGw-0dqe.js} +1 -1
  103. package/viewer/assets/{gitGraphDiagram-WWUBYQGX-h9QMzC2r.js → gitGraphDiagram-WWUBYQGX-BW112jVV.js} +1 -1
  104. package/viewer/assets/{glimmer-js-SeSwns5_.js → glimmer-js-B17pi1gL.js} +1 -1
  105. package/viewer/assets/{glimmer-ts-BExI5tTz.js → glimmer-ts-CNmwRbOn.js} +1 -1
  106. package/viewer/assets/{glsl-CXikKw2b.js → glsl-By4kbzYa.js} +1 -1
  107. package/viewer/assets/{graphql-6GpQIInt.js → graphql-BrDJs0bG.js} +1 -1
  108. package/viewer/assets/{hack-i7rizvpB.js → hack-NosEIqRi.js} +1 -1
  109. package/viewer/assets/{haml-C68gmkTf.js → haml-B-GQD8Zj.js} +1 -1
  110. package/viewer/assets/{handlebars-BBn_c0Xt.js → handlebars-1vSTccT5.js} +1 -1
  111. package/viewer/assets/{html-BObmAZQA.js → html-BBEodwWE.js} +1 -1
  112. package/viewer/assets/{html-derivative-BjMuR1LM.js → html-derivative-ClDEAvlc.js} +1 -1
  113. package/viewer/assets/{http-BJBk8gt0.js → http-I0uH89QQ.js} +1 -1
  114. package/viewer/assets/{hurl-D7obCYfg.js → hurl-JU97iWqA.js} +1 -1
  115. package/viewer/assets/{index-FMtYXcX7.js → index-CXBRQlCH.js} +5 -5
  116. package/viewer/assets/{index-Wz0qQyIk.css → index-D6HZB4Yz.css} +1 -1
  117. package/viewer/assets/{info-A6RAGUB7-CrzitQIU.js → info-A6RAGUB7-aRrZ0_i2.js} +1 -1
  118. package/viewer/assets/{infoDiagram-27XIBGKW-D2CgRDxu.js → infoDiagram-27XIBGKW-BnD3mUdL.js} +1 -1
  119. package/viewer/assets/{ishikawaDiagram-5VMMS53U-BhnQMHOo.js → ishikawaDiagram-5VMMS53U-u-wcafFg.js} +1 -1
  120. package/viewer/assets/{java-5VJU_EW8.js → java-BxAxXJLN.js} +1 -1
  121. package/viewer/assets/{javascript-r8UgndxQ.js → javascript-DNepMMwm.js} +1 -1
  122. package/viewer/assets/{jinja-BY2ibHG5.js → jinja-CfKdKg9T.js} +1 -1
  123. package/viewer/assets/{jison-C_7YxymZ.js → jison-C62ElGA9.js} +1 -1
  124. package/viewer/assets/{journeyDiagram-3NMN7TZE-DSBzKQEa.js → journeyDiagram-3NMN7TZE-E6uhr_fk.js} +1 -1
  125. package/viewer/assets/{json-8e1WlJYD.js → json-DSQqeQzH.js} +1 -1
  126. package/viewer/assets/{jsx-BYp5GD4Y.js → jsx-CpZi9Ceb.js} +1 -1
  127. package/viewer/assets/{julia-Bkob1uVX.js → julia-CypyJgFE.js} +1 -1
  128. package/viewer/assets/{just-CUnSjty-.js → just-_p8XAy70.js} +1 -1
  129. package/viewer/assets/{kanban-definition-UXKFOSKX-CotbrQue.js → kanban-definition-UXKFOSKX-Blo6hDj7.js} +1 -1
  130. package/viewer/assets/{latex-UdE3ivqk.js → latex-Blby6_jP.js} +1 -1
  131. package/viewer/assets/{line-DDh6a6hz.js → line-CQc5Trjj.js} +1 -1
  132. package/viewer/assets/{linear-BxMsCbKb.js → linear-BjWjnaEC.js} +1 -1
  133. package/viewer/assets/{liquid-WL4MvEIT.js → liquid-x3m9Bx2x.js} +1 -1
  134. package/viewer/assets/{lua-CtOubZta.js → lua-Da9XAgwV.js} +1 -1
  135. package/viewer/assets/{marko-C0GSQJi3.js → marko-Gu1d_h3t.js} +1 -1
  136. package/viewer/assets/{mdc-DUivddFc.js → mdc-CM0pUbEe.js} +1 -1
  137. package/viewer/assets/{mermaid-parser.core-FhKE_uv2.js → mermaid-parser.core-DR191xUW.js} +3 -3
  138. package/viewer/assets/{mermaid.core-BSOWwyvd.js → mermaid.core-DnwMH2U1.js} +4 -4
  139. package/viewer/assets/{mindmap-definition-YA3MSWOX-tsLZ_Opn.js → mindmap-definition-YA3MSWOX-KrnPcZeq.js} +1 -1
  140. package/viewer/assets/{nginx-B__INAtC.js → nginx-DAKCD0HY.js} +1 -1
  141. package/viewer/assets/{nim-UEY12GJF.js → nim-BvNTuhqF.js} +1 -1
  142. package/viewer/assets/{org-DDIZsz95.js → org-B-tOc6Cc.js} +1 -1
  143. package/viewer/assets/{packet-AYTQ26CC-CqdJnqfo.js → packet-AYTQ26CC-Cl2KBEou.js} +1 -1
  144. package/viewer/assets/{pegDiagram-XKGWAZYB-BJs7qu7m.js → pegDiagram-XKGWAZYB--GvmNzTl.js} +1 -1
  145. package/viewer/assets/{perl-D5KwNZt7.js → perl-RyNpCOE-.js} +1 -1
  146. package/viewer/assets/{php-BEs7d_Dk.js → php-D5gke6Nh.js} +1 -1
  147. package/viewer/assets/{pie-WAS4IAKB-lgzqBWqy.js → pie-WAS4IAKB-RfZPDbVI.js} +1 -1
  148. package/viewer/assets/{pieDiagram-E7YTZNPT-Cxe9LnnE.js → pieDiagram-E7YTZNPT-BbcD9IEY.js} +1 -1
  149. package/viewer/assets/{pug-B0dsfQOV.js → pug-CSZdPWzF.js} +1 -1
  150. package/viewer/assets/{qml-XPIhrIzf.js → qml-CCUSNo18.js} +1 -1
  151. package/viewer/assets/{quadrantDiagram-AXDQQJYC-D5JGwC11.js → quadrantDiagram-AXDQQJYC-DHP-nbSh.js} +1 -1
  152. package/viewer/assets/{r-C6hQdpIM.js → r-Cl6F5Lf8.js} +1 -1
  153. package/viewer/assets/{radar-RG4KPBEZ-B7JRLhUA.js → radar-RG4KPBEZ-DdfC1yOQ.js} +1 -1
  154. package/viewer/assets/{railroad-74A4TZTK-Df7ZQBLn.js → railroad-74A4TZTK-MHxcKgVS.js} +1 -1
  155. package/viewer/assets/railroad-abnf-HS5TGJTU-yEJXYTCu.js +1 -0
  156. package/viewer/assets/railroad-ebnf-LZEXJU2U-CEc058YK.js +1 -0
  157. package/viewer/assets/railroad-peg-WCYAUIDC-0iFSHGbV.js +1 -0
  158. package/viewer/assets/{railroadDiagram-O6MQD6OU-DZJzpmP8.js → railroadDiagram-O6MQD6OU-sxdxPqy-.js} +1 -1
  159. package/viewer/assets/{razor-C3KSUuOw.js → razor-hm0vLE-h.js} +1 -1
  160. package/viewer/assets/{regexp-n-T936Rk.js → regexp-BhIot3VY.js} +1 -1
  161. package/viewer/assets/{requirementDiagram-BXWQKSXE-Be9lEFrk.js → requirementDiagram-BXWQKSXE-CrZXt7Yu.js} +1 -1
  162. package/viewer/assets/{rst-fVnHDyXD.js → rst-C3giCu1C.js} +1 -1
  163. package/viewer/assets/{ruby-DuRgxlXF.js → ruby-B6fOYc5V.js} +1 -1
  164. package/viewer/assets/{sankeyDiagram-P5KCCOFB-BsgKAEBN.js → sankeyDiagram-P5KCCOFB-BVK_LjzN.js} +1 -1
  165. package/viewer/assets/{sas-C0WA2OFa.js → sas-a9NUQlsy.js} +1 -1
  166. package/viewer/assets/{scss-CQDZ4smR.js → scss-ChVMSp7G.js} +1 -1
  167. package/viewer/assets/{sequenceDiagram-WJ2MYXX4-BkUDWL0D.js → sequenceDiagram-WJ2MYXX4-B4E1wY9a.js} +1 -1
  168. package/viewer/assets/{shellscript-CE84G0GP.js → shellscript-Dxbnxou2.js} +1 -1
  169. package/viewer/assets/{shellsession-BdJvdaqL.js → shellsession-DWcRFAj_.js} +1 -1
  170. package/viewer/assets/{soy-BSxOREC0.js → soy-C-D-pvKx.js} +1 -1
  171. package/viewer/assets/{sql-D4MN9uRI.js → sql-W7nl9Rxy.js} +1 -1
  172. package/viewer/assets/{src-BT_kESTA.js → src-C_D1vmKd.js} +1 -1
  173. package/viewer/assets/{stata-B6nXAaGV.js → stata-BWq3CAeQ.js} +1 -1
  174. package/viewer/assets/{stateDiagram-D77RDMKH-C80wp8xG.js → stateDiagram-D77RDMKH-Dc8nXUDp.js} +1 -1
  175. package/viewer/assets/stateDiagram-v2-MP3YSRHH-SMxxOURe.js +1 -0
  176. package/viewer/assets/{surrealql-fVjZmhV3.js → surrealql-BkXUcx4E.js} +1 -1
  177. package/viewer/assets/{svelte-DnBQ5p7d.js → svelte-CowGUt9l.js} +1 -1
  178. package/viewer/assets/{swimlanes-42K2YHIH-Cff6yT8e.js → swimlanes-42K2YHIH-BtZAMLQ7.js} +1 -1
  179. package/viewer/assets/swimlanesDiagram-VR7AAH4N-C_XphOba.js +8 -0
  180. package/viewer/assets/{templ-TCx51Uts.js → templ-ejLYfFxc.js} +1 -1
  181. package/viewer/assets/{tex-CF6SMAQ_.js → tex-McefVK6r.js} +1 -1
  182. package/viewer/assets/{timeline-definition-24CTP7MA-Cje7BvaN.js → timeline-definition-24CTP7MA-TvSh-paH.js} +1 -1
  183. package/viewer/assets/{treeView-Q6P3EWNA-D-QLY3IO.js → treeView-Q6P3EWNA-BRIxIFdd.js} +1 -1
  184. package/viewer/assets/{treemap-WGGIJYW6-BCHcTa4N.js → treemap-WGGIJYW6-CeY__ApH.js} +1 -1
  185. package/viewer/assets/{ts-tags-BqLRnADX.js → ts-tags-D8LHJIWj.js} +1 -1
  186. package/viewer/assets/{tsx-D7KpDX0b.js → tsx-BkPshBHh.js} +1 -1
  187. package/viewer/assets/{twig-CEZLObpN.js → twig-CmfWCxbk.js} +1 -1
  188. package/viewer/assets/{typescript-DAbIDLbf.js → typescript-DnVmzADb.js} +1 -1
  189. package/viewer/assets/{typst-CqN7-whv.js → typst-BpEOIbiv.js} +1 -1
  190. package/viewer/assets/{vennDiagram-4TSXK5OY-Bz20rx02.js → vennDiagram-4TSXK5OY-Ba-jimVX.js} +1 -1
  191. package/viewer/assets/{vue-html-DQmvvKbd.js → vue-html-C-XYg42q.js} +1 -1
  192. package/viewer/assets/{vue-vine-Did3qTVW.js → vue-vine-DK68dTen.js} +1 -1
  193. package/viewer/assets/{vue-kTxN6wdc.js → vue-zII4ln6o.js} +1 -1
  194. package/viewer/assets/{wardley-WFR3VGLG-CpXaqOit.js → wardley-WFR3VGLG-e2A2dYzM.js} +1 -1
  195. package/viewer/assets/{wardleyDiagram-VM6X3IG4-Cz06jHpA.js → wardleyDiagram-VM6X3IG4-D2Rq1FsU.js} +1 -1
  196. package/viewer/assets/{xml-te2hFcAA.js → xml-Cc-dIbaq.js} +1 -1
  197. package/viewer/assets/{xsl-X4fuKSVl.js → xsl-C7A6UsbO.js} +1 -1
  198. package/viewer/assets/{xychartDiagram-S5SC5T6Z-BXCw0sfu.js → xychartDiagram-S5SC5T6Z-De55Qsdg.js} +1 -1
  199. package/viewer/assets/{yaml-NHtDSZab.js → yaml-BJLNLIRM.js} +1 -1
  200. package/viewer/index.html +2 -2
  201. package/viewer/assets/architecture-7GRP2DOG-Be250VKZ.js +0 -1
  202. package/viewer/assets/channel-DtULXRee.js +0 -1
  203. package/viewer/assets/classDiagram-ZZMXUADV-Bmqu01Xf.js +0 -1
  204. package/viewer/assets/classDiagram-v2-VYDZK3BY-Bmqu01Xf.js +0 -1
  205. package/viewer/assets/eventmodeling-NTZA5JFV-CJCg5f6p.js +0 -1
  206. package/viewer/assets/flowDiagram-HODETNUW-BFzq4Zvs.js +0 -1
  207. package/viewer/assets/railroad-abnf-HS5TGJTU-Z8q3QVTf.js +0 -1
  208. package/viewer/assets/railroad-ebnf-LZEXJU2U-BWoRkG3W.js +0 -1
  209. package/viewer/assets/railroad-peg-WCYAUIDC-BzoV1cN7.js +0 -1
  210. package/viewer/assets/stateDiagram-v2-MP3YSRHH-CYdsKKlz.js +0 -1
  211. package/viewer/assets/swimlanesDiagram-VR7AAH4N-niTUGjMT.js +0 -8
package/JUDGING.md CHANGED
@@ -1,8 +1,115 @@
1
1
  # Judging recorded work
2
2
 
3
+ Use the [canonical terminology](https://openev.al/docs/terminology/): a criterion
4
+ is a named graded requirement, a score is awarded credit, and a metric is a
5
+ measurement such as token count or cost. OpenEval 0.3.0 supports code judges,
6
+ LLM judges, and additive use of both against one recorded EvalRun.
7
+
8
+ ## File conventions
9
+
10
+ Every eval has prompt.md and at least one judge file:
11
+
12
+ | File | Role |
13
+ | --- | --- |
14
+ | judge.md | An LLM rubric with named criteria |
15
+ | judge.ts | An ordinary default-exported function receiving JudgeContext |
16
+ | Both | Both contribute distinct criterion scores; duplicate IDs are errors |
17
+
18
+ Only benchmarks containing judge.md need judge.model. Code judges run in a
19
+ separate Bun process on finalized evidence, under judge.timeoutMs (default ten
20
+ minutes). Code and hybrid evals use final grading; earlyStop is available to
21
+ Markdown-only evals. Candidate execution remains isolated from all judge code.
22
+
23
+ ## Plain code judges
24
+
25
+ ```ts
26
+ import type { JudgeContext } from "@hona/openeval";
27
+
28
+ export default ({ response }: JudgeContext) => ({
29
+ scores: { correct_answer: response.text === "APPLE" },
30
+ observed: response.text,
31
+ });
32
+ ```
33
+
34
+ The function may be asynchronous and may return any JSON-compatible value.
35
+ Only the optional scores object has grading semantics. Each key is a criterion
36
+ ID, using lowercase letters, digits, and underscores, starting with a letter.
37
+ Values are booleans, finite numbers from 0 to 1, or null. The host converts true
38
+ to 1 and false to 0. It rejects invalid values rather than clamping them.
39
+ response.text is always a string; missing text becomes an empty string. The
40
+ original execution outcome is available separately on context.run.
41
+
42
+ Custom output is retained verbatim. An output without scores is unscored and
43
+ cannot silently disappear from the benchmark denominator. Use consistent score
44
+ IDs across models and repetitions; a missing required score stays unresolved.
45
+ There are no built-in task-specific scorers, registration steps, or builder APIs.
46
+
47
+ The host bundles local imports before execution, captures source maps and
48
+ dependency manifests/lockfiles, and records their fingerprint. Code, imported
49
+ references, or dependency changes schedule rejudging rather than candidate
50
+ execution. Use static imports for reference data; make runtime network and file
51
+ inputs reproducible when an author-owned judge uses them.
52
+
53
+ The viewer shows normalized criterion scores, original returned JSON, frozen
54
+ source, process logs, and recorded candidate metrics. A synchronous loop can be
55
+ terminated by the host deadline. A code exception, invalid result, or timeout is
56
+ a JudgeRun error. A hybrid judgment is selected only after both sources succeed.
57
+
58
+ ## JudgeContext and recorded data
59
+
60
+ | Primitive | Data |
61
+ | --- | --- |
62
+ | response / prompt | Final root answer and the exact task prompt |
63
+ | run | Recorded EvalRun and runtime inputs; null for constructed controls |
64
+ | metrics | Candidate-only usage, cost, tool reliability, compactions, and timing |
65
+ | recording.events(filter?) | Complete retained native events with their sequence and time |
66
+ | recording.tools(filter?) | Inputs, outputs, states, and timing of recorded invocations |
67
+ | recording.sessions() | All sessions in the candidate's isolated native archive |
68
+ | recording.messages(sessionID?) | Full paginated native history, including before compaction |
69
+ | recording.export(sessionID?) | Native OpenCode session export |
70
+ | workspace.files/read/text/diff | Verified initial and final file snapshots |
71
+ | workspace.materialize(revision?) | A disposable workspace copy for author-owned verification |
72
+ | native.database() | Read-only SQLite access to a verified database copy |
73
+ | native.sdk() | The pinned OpenCode SDK/API over a separate disposable archive copy |
74
+ | native.schema() | The pinned OpenCode schema module |
75
+
76
+ Native readers initialize lazily and are disposed by the runner. SDK operations
77
+ affect only their disposable copy. The native archive version must match the
78
+ reader version. Files and database copies are checked against recorded hashes.
79
+ These APIs expose recorded data, not the user's live OpenCode service.
80
+
81
+ For independent inspection:
82
+
83
+ ```ts
84
+ import { readRecording } from "@hona/openeval";
85
+
86
+ await using context = await readRecording("./results/RUN", "eval_ID");
87
+ console.log(context.metrics.tools.errorRate);
88
+ const history = await context.recording.messages();
89
+ const database = await context.native.database();
90
+ console.log(database.query("SELECT name FROM sqlite_master").all());
91
+ ```
92
+
93
+ Metrics use deduplicated durable events across the candidate execution's
94
+ sessions. Usage includes recorded model requests and auxiliary usage such as
95
+ compaction. Cost is reported OpenCode usage, not an invoice or a promise of free
96
+ service. Missing usage is unavailable. Code that calls external services
97
+ directly can return its own accounting as metadata.
98
+
99
+ Tool error rate is failed / (succeeded + failed), with null when there are no
100
+ terminal calls. Unfinished calls are reported separately. A successful shell
101
+ tool reporting failed tests is not a native tool failure. Counts describe
102
+ recorded native invocations; do not infer uncaptured work inside a batched call.
103
+
104
+ Timing uses the union of closed recorded intervals. modelActiveMs includes
105
+ model-step and compaction spans, including time within those steps such as
106
+ retries. outputTokensPerSecond is reported output tokens per model-active second.
107
+ Token categories retain their native meanings; do not blindly add overlapping
108
+ reasoning, output, or cache categories into a new total.
109
+
3
110
  ## Shared judge agent
4
111
 
5
- OpenEval uses the native OpenCode V2 primary agent `openeval-judge` for final
112
+ The Markdown judge uses the native OpenCode V2 primary agent `openeval-judge` for final
6
113
  grading, live checks, rejudging, calibration, and independent audits. Its
7
114
  [base system prompt](src/infra/judging/judge-agent.md) owns the common evidence,
8
115
  citation, uncertainty, early-decision, and output rules. See
@@ -13,43 +120,44 @@ the eval rubric. It remains present across compaction. User turns identify the
13
120
  phase; structured context and decisions move through registered Code Mode tools.
14
121
  The tools use native Effect schemas for argument validation and catalog types.
15
122
 
16
- Every JudgeRun requires its shared profile in `input.agent`, explicit metrics,
123
+ An LLM JudgeRun records its shared profile in `input.agent`, explicit criteria,
17
124
  mode, runtime hash, and protocol. The effective
18
125
  native configuration is archived at `configuration/opencode.json` in its judge
19
126
  directory. The profile contributes to the judge input fingerprint, so a shared
20
127
  prompt change schedules rejudging using saved evidence. Candidate input
21
128
  fingerprints do not include the judge profile.
22
129
 
23
- ## Eval metric rubrics
130
+ ## Eval rubrics
24
131
 
25
- Declare one or more metrics in `judge.md`:
132
+ Declare one or more criteria in `judge.md`:
26
133
 
27
134
  ```md
28
135
  # Advice quality
29
136
 
30
- ## Metric: current_advice — Advice for the current situation
137
+ ## Criterion: current_advice — Advice for the current situation
31
138
  Pass when the requested advice is present and its recommendations are available now.
32
139
  Fail when a recommendation is unavailable now, or the requested advice is omitted.
33
140
 
34
- ## Metric: later_advice — Advice for later stages
141
+ ## Criterion: later_advice — Advice for later stages
35
142
  Judge later recommendations at their explicitly stated stage.
36
143
  ```
37
144
 
38
- Keep metric-specific rules, accepted alternatives, and domain facts in that file.
39
- The LLM applies these rules. The SDK validates the declared IDs, binary metric
145
+ Keep criterion-specific rules, accepted alternatives, and domain facts in that file.
146
+ The LLM applies these rules. The SDK validates the declared IDs, normalized criterion
40
147
  values, recorded evidence references, and exact optional quotes. It calculates
41
- the equal-weight metric mean. A missing metric or invalid citation is a judge
42
- protocol error; missing source evidence can produce a null metric. Every rubric
43
- must declare at least one `## Metric: id — Label`, and every judgment contains
44
- the complete metric array.
148
+ the equal-weight mean of criterion scores. A missing criterion or invalid citation is a judge
149
+ protocol error; missing source evidence can produce a null criterion score. Every rubric
150
+ must declare at least one `## Criterion: id — Label`. A normalized Judgment has
151
+ an aggregate value and a scores map of CriterionScore objects containing value,
152
+ reason, evidence, and source. The original code output is retained separately.
45
153
 
46
154
  ## Structured tool submissions
47
155
 
48
156
  | Tool | Data |
49
157
  | --- | --- |
50
- | `judge_context` | Current request ID, phase, metric declarations, and evidence index |
51
- | `candidate_evidence` | Recorded responses, tools, events, messages, and artifacts |
52
- | `submit_judgment` | Scores keyed by metric ID, with reasons and citations |
158
+ | `judge_context` | Current request ID, phase, criterion declarations, and evidence index |
159
+ | `candidate_evidence` | Recorded responses, tools, events, messages, artifacts, and metrics |
160
+ | `submit_judgment` | Scores keyed by criterion ID, with reasons and citations |
53
161
  | `continue_judging` | An early check's reason for needing more evidence |
54
162
 
55
163
  Native argument validation and citation validation return errors directly to the
@@ -61,14 +169,14 @@ submission fails the JudgeRun rather than assigning a candidate zero.
61
169
  Tool schemas stay stable across checks. Each request ID is bound to one fixed
62
170
  evidence view. Only the first valid submission is accepted; stale, concurrent, or
63
171
  cancelled submissions cannot overwrite it or affect a later check. Early
64
- submissions require every metric to be non-null and irreversible. Final grading
65
- permits null metrics and rejects `continue_judging`.
172
+ submissions require every criterion score to be non-null and irreversible. Final grading
173
+ permits null criterion scores and rejects `continue_judging`.
66
174
 
67
175
  Accepted submissions and semantic rejections are recorded in
68
176
  `judgment-submissions.json`, with their request/check IDs and checkpoints. Native
69
177
  schema failures are retained in the native tool transcript.
70
178
 
71
- Every decided metric cites recorded evidence. Citations can name a response,
179
+ Every decided criterion score cites recorded evidence. Citations can name a response,
72
180
  message ID, tool-call ID, event sequence, or initial/final artifact path. The
73
181
  evidence reference/checkpoint binds those citations to the exact recording.
74
182
  Source URLs are supplementary domain references, not substitutes for citations
@@ -80,7 +188,7 @@ For example, the judge submits this from Code Mode after inspecting the recordin
80
188
  const context = await tools.judge_context({});
81
189
  await tools.submit_judgment({
82
190
  requestId: context.requestId,
83
- metrics: {
191
+ scores: {
84
192
  current_advice: {
85
193
  value: 1,
86
194
  reason: "The current-stage recommendation satisfies the rubric.",
@@ -95,7 +203,7 @@ await tools.submit_judgment({
95
203
  });
96
204
  ```
97
205
 
98
- The host stores the metric array and aggregate value `0.5`. A citation's optional `quote` must occur
206
+ The host stores the criterion-score map and aggregate value `0.5`. A citation's optional `quote` must occur
99
207
  literally in the referenced response/tool/message/event/artifact. An artifact
100
208
  citation must specify `path` and `revision` (`initial` or `final`); other record
101
209
  citations specify `id`. Semantic interpretation remains the judge's job.
@@ -103,7 +211,9 @@ citations specify `id`. Semantic interpretation remains the judge's job.
103
211
  `recordEvidence` creates a calibration recording from supplied text/tool records
104
212
  without executing a model or tool. `judgeEvidence` grades retained evidence into
105
213
  a new standalone audit directory without changing a benchmark's active scores.
106
- Use `judgeRuns` when an explicit rejudge should update active selections.
214
+ Use `judgeRuns` when an explicit rejudge should update active selections. For a
215
+ code-only control, pass code: "./path/to/judge.ts" to judgeEvidence; no judge model
216
+ is needed. Pass both rubric and code for additive grading.
107
217
 
108
218
  Runtime uses recorded work intervals. Judge checks record `executionStartedAt`
109
219
  when they obtain a worker, separately from their queue-admission `startedAt`.
@@ -119,5 +229,6 @@ creation/completion dates remain provenance, not elapsed-runtime measurements.
119
229
  - [JudgeBench](https://arxiv.org/abs/2410.12784): validate factual and logical
120
230
  judging ability with known examples, rather than assuming model strength.
121
231
 
122
- Keep task-specific decisions in `judge.md`; executable SDK code handles the
123
- recording protocol and generic arithmetic, not task success rules.
232
+ Keep task-specific decisions in the author-owned judge.md and judge.ts files.
233
+ The SDK handles recording, execution, validation, and equal-weight aggregation.
234
+ OpenEval 0.3.0 uses results schema 5 and the canonical API only.
package/README.md CHANGED
@@ -1,7 +1,7 @@
1
1
  <div align="center">
2
2
  <h1>OpenEval</h1>
3
3
  <p><strong>Write the task. Judge the evidence.</strong></p>
4
- <p>Prompt-and-rubric evaluations for agents. Typed declarations, isolated runs, inspectable scores.</p>
4
+ <p>Code and LLM judges for agents. Plain functions, isolated runs, inspectable scores.</p>
5
5
  <p>
6
6
  <a href="https://openev.al">Website</a> ·
7
7
  <a href="https://openeval.pages.dev">Live preview</a> ·
@@ -16,18 +16,46 @@
16
16
  </p>
17
17
  </div>
18
18
 
19
- ![The OpenEval workbench: rubric source, a sample SQL response, and linked metric decisions](https://raw.githubusercontent.com/Hona/openeval/main/docs/images/workbench.png)
19
+ ![The OpenEval workbench: rubric source, a sample SQL response, and linked criterion scores](https://raw.githubusercontent.com/Hona/openeval/main/docs/images/workbench.png)
20
20
 
21
21
  *Interactive documentation example. Viewer screenshots use illustrative data and fictional model labels.*
22
22
 
23
- ## An eval is two files
23
+ ## A prompt and a judge
24
24
 
25
25
  | File | What you write | Who reads it |
26
26
  | --- | --- | --- |
27
27
  | `prompt.md` | A natural, focused task | Candidate agent |
28
- | `judge.md` | Named metrics and pass/fail criteria | Judge agent |
28
+ | `judge.md` | A rubric with named criteria and scoring rules | LLM judge |
29
+ | `judge.ts` | A plain function returning scores and custom JSON | Host-side Bun process |
29
30
  | `eval.ts` *(optional)* | Workspace preparation and early stopping | Host |
30
31
 
32
+ Use `judge.md`, `judge.ts`, or both. Both judge files contribute distinct criteria
33
+ from the same recorded candidate execution.
34
+
35
+ ### Deterministic: an ordinary function
36
+
37
+ For a task that asks the candidate to reply with exactly `APPLE`:
38
+
39
+ ```ts
40
+ // evals/exact-answer/judge.ts
41
+ import type { JudgeContext } from "@hona/openeval";
42
+
43
+ export default ({ response }: JudgeContext) => ({
44
+ scores: { correct_answer: response.text === "APPLE" },
45
+ });
46
+ ```
47
+
48
+ Booleans become 0 or 1. Numeric scores can be any finite value from 0 to 1; null
49
+ is unresolved. Other JSON is retained as author-defined data. Cost, tokens, tool
50
+ reliability, timing, source identity, and recording links are supplied by the
51
+ runner. Code-only benchmarks do not need a judge model.
52
+
53
+ The context also exposes native events, complete message history, tool calls,
54
+ workspace snapshots, and lazy access to the recorded OpenCode SDK, schema, and
55
+ read-only database. See [code judges and data access](https://openev.al/docs/code-judges/).
56
+
57
+ ### Model-based: write a rubric
58
+
31
59
  **`evals/ask-dialect/prompt.md`**
32
60
 
33
61
  ```md
@@ -39,12 +67,12 @@ Write a SQL query for the ten most recent orders for a customer.
39
67
  ```md
40
68
  # Requests the SQL dialect
41
69
 
42
- ## Metric: asked_dialect — Asks for the SQL dialect
70
+ ## Criterion: asked_dialect — Asks for the SQL dialect
43
71
 
44
72
  Pass when the agent asks which database or SQL dialect is in use.
45
73
  Fail when it assumes a dialect without asking. Asking alongside a draft counts.
46
74
 
47
- ## Metric: safe_parameters — Uses bound parameters
75
+ ## Criterion: safe_parameters — Uses bound parameters
48
76
 
49
77
  Pass when the proposed query uses a bound customer-ID parameter and explains
50
78
  how to supply its value. Fail when it interpolates customer input into SQL
@@ -60,6 +88,22 @@ or does not provide a parameterized query.
60
88
 
61
89
  → [Write good rubrics](https://openev.al/docs/rubrics/) · [Download the SQL starter](https://openev.al/starter.zip)
62
90
 
91
+ ## One vocabulary
92
+
93
+ A **benchmark** contains **evals**. Each eval defines a task and a **rubric**.
94
+ **Judges** produce **scores** for the rubric's **criteria**. Runs also record
95
+ **metrics** such as cost, tokens, and tool reliability.
96
+
97
+ - A **criterion** is a named requirement being graded, such as `safe_parameters`.
98
+ - A **score** is awarded credit, normalized from 0 to 1, or an aggregate of it.
99
+ - A **metric** is an observed or calculated measurement. A criterion must
100
+ explicitly use that measurement for it to affect the grade.
101
+ - A **judgment** is the judge's output. **BenchmarkRun**, **EvalRun**, and
102
+ **JudgeRun** name recorded executions, rather than reusable definitions.
103
+
104
+ See the [canonical terminology](https://openev.al/docs/terminology/) and the
105
+ [website glossary](https://openev.al/docs/terminology/).
106
+
63
107
  ## Choose models. Run. Inspect.
64
108
 
65
109
  Requires **Bun 1.4.2+**, **Docker**, and connected models in **OpenCode**.
@@ -80,6 +124,9 @@ export default {
80
124
  } satisfies Benchmark;
81
125
  ```
82
126
 
127
+ The judge model is required when any eval contains judge.md. A code-only
128
+ benchmark can omit the judge setting, or set only judge.timeoutMs.
129
+
83
130
  ```sh
84
131
  bunx --bun @hona/openeval image
85
132
  bunx --bun @hona/openeval plan --only-eval ask-dialect
@@ -106,26 +153,37 @@ Models still declared in `benchmark.ts` can be added back by a later `run`.
106
153
  flowchart LR
107
154
  P["prompt.md"] --> C["Isolated candidate"] --> E["Recording"]
108
155
  J["judge.md"] --> G["Judge + citations"]
109
- E --> G --> S["Metric scores"] --> V["Results viewer"]
156
+ T["judge.ts"] --> F["Code + recorded metrics"]
157
+ E --> G
158
+ E --> F
159
+ G --> S["Criterion scores"]
160
+ F --> S
161
+ S --> V["Results viewer"]
110
162
  ```
111
163
 
112
164
  ## See what earned the score
113
165
 
166
+ ![Code judgment with normalized criterion scores, original JSON, and frozen source](https://raw.githubusercontent.com/Hona/openeval/main/docs/images/code-judgment.png)
167
+
168
+ *Illustrative label-reading task. The code judge ran on constructed responses.*
169
+
114
170
  ![Model scores, completed checks, runtime, and cost in the results viewer](https://raw.githubusercontent.com/Hona/openeval/main/docs/images/results.png)
115
171
 
116
172
  | Capability | What you get | Guide |
117
173
  | --- | --- | --- |
118
- | Multiple metrics | Independent decisions from one recording | [Rubrics](https://openev.al/docs/rubrics/) |
174
+ | Multiple criteria | Independent scores from one recording | [Rubrics](https://openev.al/docs/rubrics/) |
175
+ | Code and hybrid judges | Plain functions, booleans, fractional credit, custom JSON | [Code judges](https://openev.al/docs/code-judges/) |
119
176
  | Controlled workspaces | Readable files, pinned Git inputs, preparation | [Workspaces](https://openev.al/docs/workspaces/) |
120
177
  | Small batches | Eval, model, repetition, and cost controls | [Running](https://openev.al/docs/running/) |
121
178
  | Transparent scores | Equal eval weights; bounds for unresolved checks | [Scoring](https://openev.al/docs/scoring/) |
122
179
  | Evidence inspection | Sessions, tool results, artifacts, and citations | [Evidence](https://openev.al/docs/evidence/) |
180
+ | Native data primitives | Full history, event traces, SDK, schema, and read-only SQL | [Recorded data](https://openev.al/docs/code-judges/#context) |
123
181
  | Rejudging | New judgments from retained, immutable recordings | [Evidence](https://openev.al/docs/evidence/#revise) |
124
182
 
125
183
  <details>
126
184
  <summary><strong>Inspect a judgment and its evidence</strong></summary>
127
185
 
128
- ![SQL eval drilldown with individual metric decisions and evidence links](https://raw.githubusercontent.com/Hona/openeval/main/docs/images/judgment.png)
186
+ ![SQL eval drilldown with individual criterion scores and evidence links](https://raw.githubusercontent.com/Hona/openeval/main/docs/images/judgment.png)
129
187
 
130
188
  </details>
131
189
 
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@hona/openeval",
3
- "version": "0.2.2",
4
- "description": "Typed prompt-and-judge evaluations with isolated agents, recorded evidence, and a results viewer",
3
+ "version": "0.3.0",
4
+ "description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
5
5
  "license": "MIT",
6
6
  "repository": {
7
7
  "type": "git",
@@ -31,6 +31,7 @@
31
31
  "@opencode/client": "0.0.0-beta-19296",
32
32
  "@opencode/core": "0.0.0-beta-19296",
33
33
  "@opencode/sdk": "0.0.0-beta-19296",
34
+ "@opencode/schema": "0.0.0-beta-19296",
34
35
  "@opencode/util": "0.0.0-beta-19296",
35
36
  "drizzle-orm": "0.45.2",
36
37
  "effect": "4.0.0-rc.112",
@@ -43,6 +43,9 @@ export function estimateWork(
43
43
  (item) => item.action === "candidate" || item.action === "judge",
44
44
  );
45
45
  const items = ready.map((item) => {
46
+ const hasLlm = !!definition.evals.find(
47
+ (evalDefinition) => evalDefinition.id === item.slot.evalId,
48
+ )!.judge;
46
49
  const candidate =
47
50
  cost(
48
51
  evals.filter(
@@ -53,22 +56,23 @@ export function estimateWork(
53
56
  ) ??
54
57
  cost(evals.filter((run) => run.input.model === item.slot.model)) ??
55
58
  cost(evals);
56
- const judge =
57
- cost(
58
- judges.filter(
59
- (run) =>
60
- run.input.model === definition.judge.model &&
61
- evals.some(
62
- (e) =>
63
- e.id === run.input.evalRunId &&
64
- e.input.evalId === item.slot.evalId,
65
- ),
66
- ),
67
- ) ??
68
- cost(
69
- judges.filter((run) => run.input.model === definition.judge.model),
70
- ) ??
71
- cost(judges);
59
+ const judge = !hasLlm
60
+ ? 0
61
+ : (cost(
62
+ judges.filter(
63
+ (run) =>
64
+ run.input.model === definition.judge.model &&
65
+ evals.some(
66
+ (e) =>
67
+ e.id === run.input.evalRunId &&
68
+ e.input.evalId === item.slot.evalId,
69
+ ),
70
+ ),
71
+ ) ??
72
+ cost(
73
+ judges.filter((run) => run.input.model === definition.judge.model),
74
+ ) ??
75
+ cost(judges));
72
76
  return {
73
77
  slotId: item.slot.id,
74
78
  estimatedUSD:
@@ -5,6 +5,7 @@ export function canJudgeEval(
5
5
  run: EvalRun | undefined,
6
6
  ): run is EvalRun & { evidence: EvidenceRef } {
7
7
  return (
8
- !!run?.evidence && ["completed", "stopped", "timed_out"].includes(run.state)
8
+ !!run?.evidence &&
9
+ ["completed", "stopped", "timed_out", "failed"].includes(run.state)
9
10
  );
10
11
  }
@@ -0,0 +1,72 @@
1
+ import { resolve } from "node:path";
2
+ import type {
3
+ JudgeRunInput,
4
+ Judgment,
5
+ OpenCodeStreamEvent,
6
+ SessionArchive,
7
+ } from "../types";
8
+ import type { CodeJudgeExecution } from "../judge-context";
9
+ import type { RecordingInput } from "../infra/recording";
10
+ import { executeCodeJudge } from "../infra/judging/code";
11
+ import { codeJudgment, combineJudgments } from "../infra/judging/code-result";
12
+ import { executeJudge } from "../infra/judging";
13
+ import { errorMessage } from "../infra/files";
14
+
15
+ export type GradedRecording = {
16
+ state: "completed" | "failed" | "timed_out";
17
+ judgment?: Judgment;
18
+ session?: SessionArchive;
19
+ code?: CodeJudgeExecution;
20
+ error?: string;
21
+ };
22
+
23
+ /** Both source files contribute to one judgment over the same finalized recording. */
24
+ export async function gradeRecording(
25
+ input: JudgeRunInput,
26
+ recording: RecordingInput,
27
+ directory: string,
28
+ onEvent: (event: OpenCodeStreamEvent) => void,
29
+ evaluateLlm: typeof executeJudge = executeJudge,
30
+ ): Promise<GradedRecording> {
31
+ let code: CodeJudgeExecution | undefined, session: SessionArchive | undefined;
32
+ try {
33
+ const judgments: Judgment[] = [];
34
+ if (input.code) {
35
+ code = await executeCodeJudge(
36
+ input.code,
37
+ recording,
38
+ resolve(directory, "code"),
39
+ input.timeoutMs,
40
+ );
41
+ if (code.state !== "completed")
42
+ return { state: code.state, code, error: code.error };
43
+ const judged = codeJudgment(code.output!);
44
+ for (const id of Object.keys(judged.scores))
45
+ if (input.criteria.some((criterion) => criterion.id === id))
46
+ throw new Error(
47
+ `Duplicate criterion ID from judge.md and judge.ts: ${id}`,
48
+ );
49
+ judgments.push(judged);
50
+ }
51
+ if (input.rubric) {
52
+ const graded = await evaluateLlm(input, directory, onEvent);
53
+ session = graded.session;
54
+ if (graded.result.state !== "completed" || !graded.judgment)
55
+ return {
56
+ state: "failed",
57
+ code,
58
+ session,
59
+ error: graded.result.error ?? "LLM judge did not submit scores",
60
+ };
61
+ judgments.push(graded.judgment);
62
+ }
63
+ return {
64
+ state: "completed",
65
+ code,
66
+ session,
67
+ judgment: combineJudgments(...judgments),
68
+ };
69
+ } catch (error) {
70
+ return { state: "failed", code, session, error: errorMessage(error) };
71
+ }
72
+ }
@@ -45,23 +45,37 @@ export const judgeFingerprint = (
45
45
  ) =>
46
46
  fingerprint({
47
47
  rubric: definition.evals.find((item) => item.id === evalId)!.judge,
48
- agent: JUDGE_AGENT,
49
- judge: definition.judge,
48
+ code: definition.evals.find((item) => item.id === evalId)!.code?.hash,
49
+ agent: definition.evals.find((item) => item.id === evalId)!.judge
50
+ ? JUDGE_AGENT
51
+ : undefined,
52
+ judge: definition.evals.find((item) => item.id === evalId)!.judge
53
+ ? definition.judge
54
+ : { timeoutMs: definition.judge.timeoutMs },
50
55
  protocol: JUDGE_PROTOCOL,
51
56
  });
52
57
  export const savedJudgeFingerprint = (
53
58
  input: Pick<
54
59
  JudgeRunInput,
55
- "rubric" | "agent" | "protocol" | "model" | "timeoutMs" | "websearch"
60
+ | "rubric"
61
+ | "agent"
62
+ | "protocol"
63
+ | "model"
64
+ | "timeoutMs"
65
+ | "websearch"
66
+ | "code"
56
67
  >,
57
68
  ) =>
58
69
  fingerprint({
59
70
  rubric: input.rubric,
71
+ code: input.code?.hash,
60
72
  agent: input.agent,
61
73
  protocol: input.protocol,
62
- judge: {
63
- model: input.model,
64
- timeoutMs: input.timeoutMs,
65
- websearch: input.websearch,
66
- },
74
+ judge: input.rubric
75
+ ? {
76
+ model: input.model,
77
+ timeoutMs: input.timeoutMs,
78
+ websearch: input.websearch,
79
+ }
80
+ : { timeoutMs: input.timeoutMs },
67
81
  });
@@ -1,37 +1,47 @@
1
1
  import { resolve, dirname } from "node:path";
2
2
  import type { EvidenceRef, Judge, JudgeRunInput, ToolCall } from "../types";
3
- import { executeJudge, judgingFingerprint } from "../infra/judging";
3
+ import { judgingFingerprint } from "../infra/judging";
4
4
  import { EvidenceCapture } from "../infra/evidence";
5
- import { JUDGE_PROTOCOL, rubricMetrics } from "../judgment";
5
+ import { JUDGE_PROTOCOL, rubricCriteria } from "../judgment";
6
6
  import { JUDGE_AGENT } from "../infra/judging/agent";
7
7
  import { writeJson } from "../infra/files";
8
8
  import { savedJudgeFingerprint } from "./input-fingerprints";
9
9
  import { mkdir, mkdtemp, rm } from "node:fs/promises";
10
+ import { compileCodeJudge } from "../infra/judging/code-source";
11
+ import { gradeRecording } from "./grade-recording";
10
12
 
11
13
  /** Inspect retained evidence without changing benchmark selections.
12
14
  * Useful for calibration and second opinions: https://arxiv.org/abs/2410.12784
13
15
  */
14
16
  export async function judgeEvidence(options: {
15
17
  evidence: EvidenceRef;
16
- rubric: string;
17
- judge: Judge;
18
+ rubric?: string;
19
+ code?: string;
20
+ judge?: Judge;
18
21
  directory: string;
19
22
  }) {
20
23
  const directory = resolve(options.directory);
21
24
  if (await Bun.file(resolve(directory, "input.json")).exists())
22
25
  throw new Error("Choose a new judge evidence directory");
26
+ if (!options.rubric?.trim() && !options.code)
27
+ throw new Error("Provide rubric text, a code judge path, or both");
28
+ if (options.rubric && !options.judge?.model)
29
+ throw new Error("A Markdown rubric requires judge.model");
30
+ const code = options.code ? await compileCodeJudge(options.code) : undefined;
23
31
  const request: Omit<JudgeRunInput, "judgeHash"> = {
24
32
  evalRunId: "retained-evidence",
25
33
  evidence: {
26
34
  ...options.evidence,
27
35
  directory: resolve(options.evidence.directory),
28
36
  },
29
- rubric: options.rubric,
30
- agent: JUDGE_AGENT,
31
- model: options.judge.model,
32
- timeoutMs: options.judge.timeoutMs ?? 600_000,
33
- websearch: options.judge.websearch ?? false,
34
- metrics: rubricMetrics(options.rubric),
37
+ rubric: options.rubric ?? "",
38
+ kind: code ? (options.rubric ? "hybrid" : "code") : "llm",
39
+ code,
40
+ agent: options.rubric ? JUDGE_AGENT : undefined,
41
+ model: options.rubric ? options.judge?.model : undefined,
42
+ timeoutMs: options.judge?.timeoutMs ?? 600_000,
43
+ websearch: options.judge?.websearch ?? false,
44
+ criteria: options.rubric ? rubricCriteria(options.rubric) : [],
35
45
  protocol: JUDGE_PROTOCOL,
36
46
  mode: "final" as const,
37
47
  runtimeHash: await judgingFingerprint(),
@@ -41,7 +51,12 @@ export async function judgeEvidence(options: {
41
51
  judgeHash: savedJudgeFingerprint(request),
42
52
  };
43
53
  await writeJson(resolve(directory, "input.json"), input);
44
- const result = await executeJudge(input, directory, () => {});
54
+ const result = await gradeRecording(
55
+ input,
56
+ { evidence: input.evidence },
57
+ directory,
58
+ () => {},
59
+ );
45
60
  await writeJson(resolve(directory, "result.json"), result);
46
61
  return result;
47
62
  }